From 651026317c682957298dc53206886ec3c89604ab Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Thu, 17 Sep 2026 23:49:58 -0400 Subject: [PATCH 01/69] The by-CPU tools rank by CPU, not elapsed time (#3523) (#3552) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit get_top_queries_by_cpu and get_top_procedures_by_cpu ordered by summed delta_elapsed_time in both SKUs, so on a wait-bound server the real CPU consumers could be missing from the page entirely and attributed_cpu_ratio read as "hidden CPU" when it meant "wrong sort key". All eight ranking sites now key on CPU: Darling's TopQueriesSql, TopQueriesByHostObjectSql, and TopProceduresSql (CTE cut + post-WAITFOR-trim outer sort) and Lite's twin reads. The viewer's Duration grids keep their elapsed ranking by design. Tests: SQL-text ordering pins for all three Darling consts (plus the rollup const joins the Postgres-dialect theory), and behavior tests in both SKUs seeding a CPU king whose elapsed time ranks last with more groups than the over-fetch admits — under the old key it was cut, not merely mis-sorted. Red-watched: the Lite cut test fails against the old ordering. Fixes #3523 Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx Co-authored-by: Claude Fable 5 --- Darling/Darling.Tests/CpuRankingLiveTests.cs | 148 ++++++++++++++++ .../Darling.Tests/DarlingMcpDataToolsTests.cs | 25 ++- .../Mcp/DarlingDataReader.cs | 15 +- Lite.Tests/QueryStatsCpuRankingReaderTests.cs | 164 ++++++++++++++++++ .../QueryStatsModuleAttributionReaderTests.cs | 5 +- Lite/Services/LocalDataService.QueryStats.cs | 6 +- 6 files changed, 350 insertions(+), 13 deletions(-) create mode 100644 Darling/Darling.Tests/CpuRankingLiveTests.cs create mode 100644 Lite.Tests/QueryStatsCpuRankingReaderTests.cs diff --git a/Darling/Darling.Tests/CpuRankingLiveTests.cs b/Darling/Darling.Tests/CpuRankingLiveTests.cs new file mode 100644 index 000000000..170f23265 --- /dev/null +++ b/Darling/Darling.Tests/CpuRankingLiveTests.cs @@ -0,0 +1,148 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Linq; +using System.Threading; +using System.Threading.Tasks; +using Darling.Tests; +using Npgsql; +using PerformanceMonitor.Collectors; +using PerformanceMonitor.Common; +using PerformanceMonitor.Darling.Service.Mcp; +using PerformanceMonitor.Darling.Storage; +using Xunit; + +namespace PerformanceMonitor.Darling.Tests; + +/// +/// #3523: get_top_queries_by_cpu / get_top_procedures_by_cpu ranked by summed ELAPSED time. +/// +/// The defect. On a wait-bound server, elapsed and CPU disagree wildly — a query that sleeps on +/// locks for minutes ranks above one that burns a core — so the real CPU consumers could be absent from the +/// page entirely, while attributed_cpu_ratio (page CPU / measured process CPU) read as "hidden or +/// evicted CPU" when it actually meant "wrong sort key". Every CPU investigation starts at this tool. +/// +/// Why a LIVE test on top of the SQL-text pins. The queries reads have TWO ordering sites — the +/// ranking CTE's ORDER BY ... LIMIT top + 5 decides which groups SURVIVE at all, and the outer +/// post-WAITFOR-trim sort decides the returned order. A text pin restates each clause; only real rows through +/// real Postgres prove the cut. The seed makes the CPU king the WORST group by elapsed time with more +/// competing groups than the over-fetch admits, so under the old key it was not merely mis-sorted — it was +/// cut before the outer sort could see it. +/// +[Collection("live-postgres")] +public sealed class CpuRankingLiveTests +{ + private const string ServerName = "darling-cpu-ranking-e2e"; + private static readonly int ServerId = ServerIdHelper.GetDeterministicHashCode(ServerName); + private const string Db = "waitbound"; + + [Fact] + public async Task ByCpuReads_RankAndCutByWorkerTime_WhenCpuAndElapsedDisagree_AgainstDevPostgres() + { + var connectionString = Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + Assert.SkipWhen(string.IsNullOrEmpty(connectionString), + "Set DARLING_TEST_PG to a Postgres connection string to run the live CPU-ranking test."); + + var ct = TestContext.Current.CancellationToken; + + using var connection = new NpgsqlConnection(connectionString); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + await CleanupAsync(connection, ct); + + await using var postgres = NpgsqlDataSource.Create(connectionString!); + var succeeded = false; + try + { + await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); + var now = DarlingMcpTestData.Naive(DateTime.UtcNow); + + /* The wait-bound shape: one CPU king whose elapsed time is the SMALLEST on the box, and six + lock-sleepers whose elapsed dwarfs it while their CPU is noise. Seven groups against + top: 1 (over-fetch 6) means an elapsed-keyed CTE cuts the king before the outer sort. */ + await PlantQueryAsync(connection, ct, now.AddMinutes(-9), "0xCPUKING", + "SELECT CpuBurner FROM Numbers", cpuUs: 900_000L, elapsedUs: 1_000L); + for (var i = 1; i <= 6; i++) + { + await PlantQueryAsync(connection, ct, now.AddMinutes(-8), $"0xSLEEPER{i}", + $"SELECT Blocked{i} FROM Locked WHERE Id = {i}", cpuUs: 1_000L + i, elapsedUs: 800_000L + i * 10_000L); + } + + /* ---- the cut: top 1 must be the CPU king, in BOTH groupings (the rollup const has its own + copies of both ordering sites). All rows are ad-hoc, so the rollup grouping degenerates to + per-hash and exercises purely its ordering keys. */ + foreach (var rollUp in new[] { false, true }) + { + var top1 = await DarlingDataReader.GetTopQueriesByCpuAsync( + postgres, ServerId, now.AddHours(-1), now.AddMinutes(5), top: 1, databaseName: null, + rollUpByHostObject: rollUp, cancellationToken: ct); + var king = Assert.Single(top1); + Assert.Equal("0xCPUKING", king.QueryHash); + } + + /* ---- the order: the full page comes back in descending CPU order, king first. */ + var page = await DarlingDataReader.GetTopQueriesByCpuAsync( + postgres, ServerId, now.AddHours(-1), now.AddMinutes(5), top: 20, databaseName: null, + rollUpByHostObject: false, cancellationToken: ct); + Assert.Equal(7, page.Count); + Assert.Equal("0xCPUKING", page[0].QueryHash); + Assert.Equal(page.Select(r => r.TotalCpuUs).OrderByDescending(v => v), page.Select(r => r.TotalCpuUs)); + + /* ---- procedures: same disagreement, same promise (single ordering site, no over-fetch). */ + await PlantProcedureAsync(connection, ct, now.AddMinutes(-7), "usp_CpuHog", cpuUs: 900_000L, elapsedUs: 1_000L); + await PlantProcedureAsync(connection, ct, now.AddMinutes(-6), "usp_WaitBound", cpuUs: 5_000L, elapsedUs: 900_000L); + + var topProc = await DarlingDataReader.GetTopProceduresByCpuAsync( + postgres, ServerId, now.AddHours(-1), now.AddMinutes(5), top: 1, databaseName: null, cancellationToken: ct); + Assert.Equal("usp_CpuHog", Assert.Single(topProc).ObjectName); + + succeeded = true; + } + finally + { + /* #1902: teardown on its OWN connection, never the body's — see HostObjectRollupLiveTests. */ + await LiveStoreCleanup.RunAsync(connectionString!, succeeded, async (cleanup, cleanupCt) => + await CleanupAsync(cleanup, cleanupCt)); + } + } + + private static async Task PlantQueryAsync( + NpgsqlConnection connection, CancellationToken ct, DateTime at, + string queryHash, string queryText, long cpuUs, long elapsedUs) + { + var sqlHandle = "0xSQLH" + Convert.ToHexString( + System.Security.Cryptography.SHA256.HashData(System.Text.Encoding.UTF8.GetBytes(queryText)))[..12]; + var digest = System.Security.Cryptography.SHA256.HashData(System.Text.Encoding.UTF8.GetBytes(queryText)); + + await DarlingMcpTestData.ExecAsync(connection, ct, + @"INSERT INTO query_stats (collection_id, collection_time, server_id, server_name, database_name, + query_hash, query_plan_hash, sql_handle, plan_handle, query_text, + query_text_digest, delta_execution_count, delta_worker_time, + delta_elapsed_time, delta_logical_reads, min_dop, max_dop) + VALUES ($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,$11,$12,$13,$14,$15,$16,$17)", + CollectionIdGenerator.Next(), at, ServerId, ServerName, Db, + queryHash, "0xPLANHASH", sqlHandle, "0xPLANH", queryText, + digest, 10L, cpuUs, elapsedUs, 100L, 1, 1); + } + + private static async Task PlantProcedureAsync( + NpgsqlConnection connection, CancellationToken ct, DateTime at, + string objectName, long cpuUs, long elapsedUs) => + await DarlingMcpTestData.ExecAsync(connection, ct, + @"INSERT INTO procedure_stats (collection_id, collection_time, server_id, server_name, database_name, + schema_name, object_name, object_type, delta_execution_count, + delta_worker_time, delta_elapsed_time) + VALUES ($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,$11)", + CollectionIdGenerator.Next(), at, ServerId, ServerName, Db, + "dbo", objectName, "SQL_STORED_PROCEDURE", 10L, cpuUs, elapsedUs); + + private static async Task CleanupAsync(NpgsqlConnection connection, CancellationToken ct) => + await DarlingMcpTestData.ExecAsync(connection, ct, + $"DELETE FROM query_stats WHERE server_id = {ServerId}; DELETE FROM procedure_stats WHERE server_id = {ServerId}; DELETE FROM servers WHERE server_id = {ServerId}"); +} diff --git a/Darling/Darling.Tests/DarlingMcpDataToolsTests.cs b/Darling/Darling.Tests/DarlingMcpDataToolsTests.cs index 6aaf61d39..991063237 100644 --- a/Darling/Darling.Tests/DarlingMcpDataToolsTests.cs +++ b/Darling/Darling.Tests/DarlingMcpDataToolsTests.cs @@ -648,7 +648,28 @@ public void TopProceduresSql_AggregatesDeltas_OptionalDbFilter_ReadsBaseTable() Assert.Contains("FROM procedure_stats", sql, StringComparison.Ordinal); Assert.Contains("GROUP BY database_name, schema_name, object_name, object_type", sql, StringComparison.Ordinal); Assert.Contains("$5::text IS NULL OR database_name = $5", sql, StringComparison.Ordinal); - Assert.Contains("SUM(delta_elapsed_time) DESC", sql, StringComparison.Ordinal); + Assert.Contains("SUM(delta_worker_time) DESC", sql, StringComparison.Ordinal); + } + + /* #3523: every by-CPU read RANKED by summed elapsed time — on a wait-bound server the real CPU + consumers could be absent from the page entirely, and attributed_cpu_ratio then read as "hidden + CPU" when it meant "wrong sort key". Both the ranking cut (the CTE's ORDER BY ... LIMIT) and the + post-WAITFOR-trim final ordering must key on CPU; the viewer's Duration grids keep their elapsed + ranking by design and are pinned separately in ViewerQueriesTests. */ + [Theory] + [InlineData(nameof(DarlingDataReader.TopQueriesSql))] + [InlineData(nameof(DarlingDataReader.TopQueriesByHostObjectSql))] + [InlineData(nameof(DarlingDataReader.TopProceduresSql))] + public void ByCpuReads_RankByWorkerTime_NeverElapsed(string sqlName) + { + var sql = SqlByName(sqlName); + Assert.Contains("ORDER BY SUM(delta_worker_time) DESC", sql, StringComparison.Ordinal); + Assert.DoesNotContain("SUM(delta_elapsed_time) DESC", sql, StringComparison.Ordinal); + Assert.DoesNotContain("total_elapsed_us DESC", sql, StringComparison.Ordinal); + if (sqlName != nameof(DarlingDataReader.TopProceduresSql)) + { + Assert.Contains("ORDER BY r.total_cpu_us DESC", sql, StringComparison.Ordinal); + } } [Fact] @@ -727,6 +748,7 @@ exactly like the viewer's UTC-offset read. */ [InlineData(nameof(DarlingDataReader.TempDbTrendSql))] [InlineData(nameof(DarlingDataReader.LatestPerfmonStatsSql))] [InlineData(nameof(DarlingDataReader.TopQueriesSql))] + [InlineData(nameof(DarlingDataReader.TopQueriesByHostObjectSql))] [InlineData(nameof(DarlingDataReader.TopProceduresSql))] [InlineData(nameof(DarlingDataReader.QueryStoreTopSql))] [InlineData(nameof(DarlingDataReader.ServerListSql))] @@ -757,6 +779,7 @@ public void Reads_ArePostgresDialect_NoTsqlIsms(string sqlName) nameof(DarlingDataReader.TempDbTrendSql) => DarlingDataReader.TempDbTrendSql, nameof(DarlingDataReader.LatestPerfmonStatsSql) => DarlingDataReader.LatestPerfmonStatsSql, nameof(DarlingDataReader.TopQueriesSql) => DarlingDataReader.TopQueriesSql, + nameof(DarlingDataReader.TopQueriesByHostObjectSql) => DarlingDataReader.TopQueriesByHostObjectSql, nameof(DarlingDataReader.TopProceduresSql) => DarlingDataReader.TopProceduresSql, nameof(DarlingDataReader.QueryStoreTopSql) => DarlingDataReader.QueryStoreTopSql, nameof(DarlingDataReader.ServerListSql) => DarlingDataReader.ServerListSql, diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingDataReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingDataReader.cs index 85a523d02..6d1657c60 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingDataReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingDataReader.cs @@ -706,7 +706,8 @@ public static async Task> GetLatestPerfmonStatsAsync( /// /// Top query-stats groups over the window — a focused projection of the viewer's TopQueriesSql /// (the columns Lite's get_top_queries_by_cpu returns): group by (database, query_hash), sum the - /// deltas + carry min/max spreads, rank by summed delta_elapsed_time descending, over-fetch by + /// deltas + carry min/max spreads, rank by summed delta_worker_time (CPU — the tool's promise; + /// #3523, the viewer's duration grid keeps its elapsed ranking) descending, over-fetch by /// 5 to drop WAITFOR shells via the latest-text LATERAL, cap at top. Summed bigints CAST back to bigint /// for the typed reader. The aggregate reads the base query_stats table (it projects no text); /// the text LATERAL reads v_query_stats, which resolves the #1767 payload dimension — the plan @@ -753,7 +754,7 @@ FROM query_stats NULL and keep collapsing into one group per hash exactly as before. */ GROUP BY database_name, query_hash, host_object_name HAVING SUM(delta_execution_count) > 0 OR SUM(delta_elapsed_time) > 0 - ORDER BY SUM(delta_elapsed_time) DESC + ORDER BY SUM(delta_worker_time) DESC LIMIT $4 + 5 ) SELECT @@ -795,7 +796,7 @@ ORDER BY collection_time DESC LIMIT 1 ) AS t ON TRUE WHERE t.query_text IS NULL OR t.query_text NOT LIKE 'WAITFOR%' - ORDER BY r.total_elapsed_us DESC + ORDER BY r.total_cpu_us DESC LIMIT $4 """; @@ -864,7 +865,7 @@ without that arm every unrelated ad-hoc statement in a database would pool into GROUP BY database_name, host_object_name, CASE WHEN host_object_name IS NULL THEN query_hash END HAVING SUM(delta_execution_count) > 0 OR SUM(delta_elapsed_time) > 0 - ORDER BY SUM(delta_elapsed_time) DESC + ORDER BY SUM(delta_worker_time) DESC LIMIT $4 + 5 ) SELECT @@ -907,7 +908,7 @@ ORDER BY collection_time DESC LIMIT 1 ) AS t ON TRUE WHERE t.query_text IS NULL OR t.query_text NOT LIKE 'WAITFOR%' - ORDER BY r.total_elapsed_us DESC + ORDER BY r.total_cpu_us DESC LIMIT $4 """; @@ -960,7 +961,7 @@ public static async Task> GetTopQueriesByCpuAsync( /// Top procedure-stats groups over the window — a focused projection of the viewer's /// TopProceduresSql (the columns Lite's get_top_procedures_by_cpu returns): group by /// (database, schema, object, type), sum the deltas + carry min/max spreads, rank by summed - /// delta_elapsed_time descending, cap at top. Reads the base procedure_stats table + /// delta_worker_time (CPU — the tool's promise; #3523) descending, cap at top. Reads the base procedure_stats table /// (no v_ view). $1 server_id, $2/$3 window (naive UTC), $4 top. /// public const string TopProceduresSql = """ @@ -989,7 +990,7 @@ FROM procedure_stats AND ($5::text IS NULL OR database_name = $5) GROUP BY database_name, schema_name, object_name, object_type HAVING SUM(delta_execution_count) > 0 OR SUM(delta_elapsed_time) > 0 - ORDER BY SUM(delta_elapsed_time) DESC + ORDER BY SUM(delta_worker_time) DESC LIMIT $4 """; diff --git a/Lite.Tests/QueryStatsCpuRankingReaderTests.cs b/Lite.Tests/QueryStatsCpuRankingReaderTests.cs new file mode 100644 index 000000000..6591218f0 --- /dev/null +++ b/Lite.Tests/QueryStatsCpuRankingReaderTests.cs @@ -0,0 +1,164 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor Lite. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Linq; +using System.Threading.Tasks; +using DuckDB.NET.Data; +using PerformanceMonitorLite.Database; +using PerformanceMonitorLite.Services; +using PerformanceMonitorLite.Tests; +using Xunit; + +namespace Lite.Tests; + +/// +/// #3523: Lite's GetTopQueriesByCpuAsync / GetTopProceduresByCpuAsync — the reads behind +/// get_top_queries_by_cpu / get_top_procedures_by_cpu and the Query Performance grids — +/// ranked by summed ELAPSED time. On a wait-bound server the two disagree wildly, so the real CPU +/// consumers could be absent from the page entirely. The queries read has TWO ordering sites: the ranking +/// CTE's ORDER BY ... LIMIT top + 5 decides which groups SURVIVE, and the outer post-WAITFOR-trim +/// sort decides the returned order — so the seed makes the CPU king the WORST group by elapsed with more +/// competing groups than the over-fetch admits: under the old key it was cut, not merely mis-sorted. +/// Real-DuckDB round-trip in the shared-fixture harness, like . +/// +public sealed class QueryStatsCpuRankingReaderTests : IClassFixture, IDisposable +{ + private const int ServerId = 8815; + + private readonly DuckDbInitializer _duckDb; + private DuckDBConnection? _seedConn; + private long _nextId = 1; + + /* Anchor in the recent past so the default 24h window includes it; last_execution_time must be >= the + window start (the reader's staleness filter, utc offset 0). */ + private static readonly DateTime Collected = + DateTime.SpecifyKind(DateTime.UtcNow.AddMinutes(-90), DateTimeKind.Unspecified); + + public QueryStatsCpuRankingReaderTests(SharedDuckDbFixture fixture) + { + fixture.ResetData(); + _duckDb = fixture.DuckDb; + } + + public void Dispose() => _seedConn?.Dispose(); + + private async Task SeedConnectionAsync() + { + if (_seedConn is null) + { + _seedConn = _duckDb.CreateConnection(); + await _seedConn.OpenAsync(); + } + return _seedConn; + } + + /* The wait-bound shape: one CPU king whose elapsed time is the smallest on the box, six lock-sleepers + whose elapsed dwarfs it while their CPU is noise. */ + private async Task SeedDisagreeingQueriesAsync() + { + await SeedQueryStatsAsync("0xCPUKING", "SELECT CpuBurner FROM Numbers", cpuUs: 900_000, elapsedUs: 1_000); + for (var i = 1; i <= 6; i++) + await SeedQueryStatsAsync($"0xSLEEPER{i}", $"SELECT Blocked{i} FROM Locked", cpuUs: 1_000 + i, elapsedUs: 800_000 + i * 10_000); + } + + [Fact] + public async Task TopQueries_Top1_SurvivesTheCteCut_ByCpuNotElapsed() + { + var service = new LocalDataService(_duckDb); + await SeedDisagreeingQueriesAsync(); + + /* Seven groups against top: 1 (CTE over-fetch = 6): an elapsed-keyed CTE cuts the CPU king + before the outer sort ever sees it. */ + var rows = await service.GetTopQueriesByCpuAsync(ServerId, hoursBack: 24, top: 1); + + Assert.Equal("0xCPUKING", Assert.Single(rows).QueryHash); + } + + [Fact] + public async Task TopQueries_ReturnsThePage_InDescendingCpuOrder() + { + var service = new LocalDataService(_duckDb); + await SeedDisagreeingQueriesAsync(); + + var rows = await service.GetTopQueriesByCpuAsync(ServerId); + + Assert.Equal(7, rows.Count); + Assert.Equal("0xCPUKING", rows[0].QueryHash); + Assert.Equal(rows.Select(r => r.TotalCpuUs).OrderByDescending(v => v), rows.Select(r => r.TotalCpuUs)); + } + + [Fact] + public async Task TopProcedures_RankByCpu_WhenCpuAndElapsedDisagree() + { + var service = new LocalDataService(_duckDb); + + await SeedProcedureStatsAsync("usp_CpuHog", cpuUs: 900_000, elapsedUs: 1_000); + await SeedProcedureStatsAsync("usp_WaitBound", cpuUs: 5_000, elapsedUs: 900_000); + + var top1 = await service.GetTopProceduresByCpuAsync(ServerId, hoursBack: 24, top: 1); + Assert.Equal("usp_CpuHog", Assert.Single(top1).ObjectName); + + var page = await service.GetTopProceduresByCpuAsync(ServerId); + Assert.Equal(new[] { "usp_CpuHog", "usp_WaitBound" }, page.Select(r => r.ObjectName)); + } + + // ── Seeding helpers (column-list INSERT; unset columns default to NULL) ── + + private async Task SeedQueryStatsAsync(string queryHash, string queryText, long cpuUs, long elapsedUs) + { + using var readLock = _duckDb.AcquireReadLock(); + var connection = await SeedConnectionAsync(); + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +INSERT INTO query_stats + (collection_id, collection_time, server_id, server_name, database_name, + query_hash, sql_handle, last_execution_time, delta_execution_count, + delta_worker_time, delta_elapsed_time, query_text) +VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11, $12)"; + cmd.Parameters.Add(new DuckDBParameter { Value = _nextId++ }); + cmd.Parameters.Add(new DuckDBParameter { Value = Collected }); + cmd.Parameters.Add(new DuckDBParameter { Value = ServerId }); + cmd.Parameters.Add(new DuckDBParameter { Value = "TestSrv" }); + cmd.Parameters.Add(new DuckDBParameter { Value = "TestDb" }); + cmd.Parameters.Add(new DuckDBParameter { Value = queryHash }); + cmd.Parameters.Add(new DuckDBParameter { Value = "0xH" + queryHash }); + cmd.Parameters.Add(new DuckDBParameter { Value = Collected }); + cmd.Parameters.Add(new DuckDBParameter { Value = 10L }); + cmd.Parameters.Add(new DuckDBParameter { Value = cpuUs }); + cmd.Parameters.Add(new DuckDBParameter { Value = elapsedUs }); + cmd.Parameters.Add(new DuckDBParameter { Value = queryText }); + await cmd.ExecuteNonQueryAsync(); + } + + private async Task SeedProcedureStatsAsync(string objectName, long cpuUs, long elapsedUs) + { + using var readLock = _duckDb.AcquireReadLock(); + var connection = await SeedConnectionAsync(); + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +INSERT INTO procedure_stats + (collection_id, collection_time, server_id, server_name, database_name, + schema_name, object_name, object_type, last_execution_time, + delta_execution_count, delta_worker_time, delta_elapsed_time) +VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11, $12)"; + cmd.Parameters.Add(new DuckDBParameter { Value = _nextId++ }); + cmd.Parameters.Add(new DuckDBParameter { Value = Collected }); + cmd.Parameters.Add(new DuckDBParameter { Value = ServerId }); + cmd.Parameters.Add(new DuckDBParameter { Value = "TestSrv" }); + cmd.Parameters.Add(new DuckDBParameter { Value = "TestDb" }); + cmd.Parameters.Add(new DuckDBParameter { Value = "dbo" }); + cmd.Parameters.Add(new DuckDBParameter { Value = objectName }); + cmd.Parameters.Add(new DuckDBParameter { Value = "SQL_STORED_PROCEDURE" }); + cmd.Parameters.Add(new DuckDBParameter { Value = Collected }); + cmd.Parameters.Add(new DuckDBParameter { Value = 10L }); + cmd.Parameters.Add(new DuckDBParameter { Value = cpuUs }); + cmd.Parameters.Add(new DuckDBParameter { Value = elapsedUs }); + await cmd.ExecuteNonQueryAsync(); + } +} diff --git a/Lite.Tests/QueryStatsModuleAttributionReaderTests.cs b/Lite.Tests/QueryStatsModuleAttributionReaderTests.cs index afaec36a0..c897ef86e 100644 --- a/Lite.Tests/QueryStatsModuleAttributionReaderTests.cs +++ b/Lite.Tests/QueryStatsModuleAttributionReaderTests.cs @@ -18,7 +18,7 @@ namespace Lite.Tests; /// -/// Real-DuckDB round-trip pins for the #1568 module attribution in Top Queries by Duration +/// Real-DuckDB round-trip pins for the #1568 module attribution in Top Queries by CPU /// (): the read-time LEFT JOIN of /// query_stats.sql_handle to procedure_stats.sql_handle (both stores persist the SAME /// normalized CONVERT(varchar(130), ..., 1) handle text). A statement whose handle matches a cached @@ -69,7 +69,8 @@ public async Task TopQueries_AttributesMatchedHandleToModule_UnmatchedIsAdHoc() { var service = new LocalDataService(_duckDb); - /* Attributed query: sql_handle matches a cached procedure — bigger elapsed so it ranks first. */ + /* Attributed query: sql_handle matches a cached procedure. (Lookups below are by hash — the + #3523 CPU ranking never matters here, and these seeds carry no worker time.) */ await SeedQueryStatsAsync("TestDb", "0xATTRIB", "0xMOD", deltaExec: 5, deltaElapsedUs: 300_000, "SELECT attributed"); await SeedProcedureStatsAsync("TestDb", "dbo", "usp_Thing", "PROCEDURE", "0xMOD"); diff --git a/Lite/Services/LocalDataService.QueryStats.cs b/Lite/Services/LocalDataService.QueryStats.cs index bb4b0e760..3026a8aa7 100644 --- a/Lite/Services/LocalDataService.QueryStats.cs +++ b/Lite/Services/LocalDataService.QueryStats.cs @@ -158,7 +158,7 @@ FROM v_query_stats AND last_execution_time >= $2 + $5 * INTERVAL '1' MINUTE" + dbClause + @" GROUP BY database_name, query_hash, host_object_name HAVING SUM(delta_execution_count) > 0 OR SUM(delta_elapsed_time) > 0 - ORDER BY SUM(delta_elapsed_time) DESC + ORDER BY SUM(delta_worker_time) DESC LIMIT $4 + 5 ), module AS ( @@ -210,7 +210,7 @@ LIMIT 1 ) t ON TRUE LEFT JOIN module m ON m.sql_handle = r.sql_handle WHERE t.query_text IS NULL OR t.query_text NOT LIKE 'WAITFOR%' -ORDER BY r.total_elapsed_us DESC +ORDER BY r.total_cpu_us DESC LIMIT $4"; command.Parameters.Add(new DuckDBParameter { Value = serverId }); @@ -916,7 +916,7 @@ FROM v_procedure_stats AND last_execution_time >= $2 + $5 * INTERVAL '1' MINUTE" + dbClause + @" GROUP BY database_name, schema_name, object_name, object_type HAVING SUM(delta_execution_count) > 0 OR SUM(delta_elapsed_time) > 0 -ORDER BY SUM(delta_elapsed_time) DESC +ORDER BY SUM(delta_worker_time) DESC LIMIT $4"; command.Parameters.Add(new DuckDBParameter { Value = serverId }); From 19009676dd8bec60996c67cdcbbeb21d422a0ee0 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Thu, 17 Sep 2026 23:50:21 -0400 Subject: [PATCH 02/69] analyze_server: an empty analysis window is "unavailable", not an all-clear (#3524) (#3553) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The analysis pipeline's data-span gate measures LIFETIME history, so a server whose collection died (broken credential, unreachable target) still passes it, collects zero facts over the empty window, and the MCP tool rendered that as status "empty" with "All metrics are within normal ranges" — an affirmative all-clear for a dead collector. Both SKUs' analysis services now set WindowEmptyMessage (beside InsufficientDataMessage) at the zero-facts branch, and both analyze tools return the same "unavailable" envelope get_analysis_facts already uses for this state, pointing the caller at get_collection_health. The true-negative all-clear is unchanged for runs that collected facts and found nothing. The unavailable envelope keeps the #2506 persistence hints, so anchored empty-window runs still disclose non-persistence. Tests, both SKUs, on the REAL 24h gate: 25h of history whose newest row is five hours old -> unavailable; the same server with one benign wait inside the window -> the existing all-clear. Darling's is gated (DARLING_TEST_PG) and ran green against a live PG 18.4 + TimescaleDB store; Lite's runs on the shared DuckDB fixture. The viewer Recommendations tabs render the same false all-clear from the same state and only branch on InsufficientDataMessage; that half is #3551 (out of this fix's file scope). Fixes #3524 Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx Co-authored-by: Claude Fable 5 --- .../AnalyzeWindowEmptyLivePostgresTests.cs | 149 ++++++++++++++++++ .../DarlingAnalysisService.cs | 25 +++ .../Mcp/DarlingMcpTools.cs | 25 ++- Lite.Tests/AnalysisWindowEmptyTests.cs | 148 +++++++++++++++++ Lite/Analysis/AnalysisService.cs | 25 +++ Lite/Mcp/McpAnalysisTools.cs | 25 ++- 6 files changed, 395 insertions(+), 2 deletions(-) create mode 100644 Darling/Darling.Tests/AnalyzeWindowEmptyLivePostgresTests.cs create mode 100644 Lite.Tests/AnalysisWindowEmptyTests.cs diff --git a/Darling/Darling.Tests/AnalyzeWindowEmptyLivePostgresTests.cs b/Darling/Darling.Tests/AnalyzeWindowEmptyLivePostgresTests.cs new file mode 100644 index 000000000..5ac44c3fc --- /dev/null +++ b/Darling/Darling.Tests/AnalyzeWindowEmptyLivePostgresTests.cs @@ -0,0 +1,149 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Text.Json; +using System.Threading; +using System.Threading.Tasks; +using Npgsql; +using PerformanceMonitor.Collectors; +using PerformanceMonitor.Common; +using PerformanceMonitor.Darling.Analysis; +using PerformanceMonitor.Darling.Service.Mcp; +using PerformanceMonitor.Darling.Storage; +using Xunit; + +namespace Darling.Tests; + +/// +/// Gated (DARLING_TEST_PG) proof of #3524: analyze_server must not answer "all metrics are +/// within normal ranges" when the analysis window collected NOTHING. The pipeline's data-span gate +/// measures LIFETIME history, so a server whose collection died (broken credential, unreachable +/// target) still passes it — and the zero-facts pass used to return a bare [] that the tool +/// rendered as a true-negative all-clear. Both sides of the distinction are asserted, on the REAL +/// 24h gate rather than a zeroed one, because the bug lives precisely in the gap between "enough +/// history" and "an empty window". +/// +[Collection("live-postgres")] +public sealed class AnalyzeWindowEmptyLivePostgresTests +{ + private const string ServerName = "darling-analyze-window-empty-e2e"; + private const string StaleWait = "WE3524_STALE_WAIT"; + private const string BenignWait = "WE3524_BENIGN_WAIT"; + private static readonly int ServerId = ServerIdHelper.GetDeterministicHashCode(ServerName); + + private static string? ConnectionString => Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + + [Fact] + public async Task ADeadCollectorWindow_IsUnavailable_AndAWindowWithFactsKeepsTheAllClear() + { + var cs = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(cs), "Set DARLING_TEST_PG to a Postgres connection string to run the live window-empty analysis test."); + + var ct = TestContext.Current.CancellationToken; + using var connection = new NpgsqlConnection(cs); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + await DeleteRowsAsync(connection, ct); + + await using var postgres = NpgsqlDataSource.Create(cs!); + + var bodySucceeded = false; + try + { + await RegisterServerAsync(connection, ct); + var service = new DarlingAnalysisService(postgres); + + /* ── the dead-collector shape: 25 hours of history (the real 24h gate PASSES) whose + newest row is five hours old, so the tool's default 4h window holds nothing. */ + await PlantWaitAsync(connection, Naive(DateTime.UtcNow.AddHours(-30)), StaleWait, 60_000L, ct); + await PlantWaitAsync(connection, Naive(DateTime.UtcNow.AddHours(-5)), StaleWait, 60_000L, ct); + + var deadWindow = await DarlingMcpTools.AnalyzeServer(service, postgres, ServerName); + using (var doc = JsonDocument.Parse(deadWindow)) + { + Assert.Equal("unavailable", doc.RootElement.GetProperty("status").GetString()); + + /* The #2506 persistence disclosure survives the new envelope — an anchored + empty-window run still owes the caller that context, and this unanchored one + reports the ordinary answer. */ + Assert.True(doc.RootElement.GetProperty("hints").GetProperty("persisted").GetBoolean()); + } + Assert.Contains("get_collection_health", deadWindow, StringComparison.Ordinal); + Assert.Contains("NOT an all-clear", deadWindow, StringComparison.Ordinal); + Assert.DoesNotContain("within normal ranges", deadWindow, StringComparison.Ordinal); + + /* The service says WHICH kind of nothing this was: the window, not the lifetime span. */ + Assert.NotNull(service.WindowEmptyMessage); + Assert.Null(service.InsufficientDataMessage); + + /* ── the control, and the reason the fix is a distinction rather than a rewording: the + SAME server with one benign wait INSIDE the window has facts to score, finds nothing + wrong, and keeps the genuine true-negative all-clear. */ + await PlantWaitAsync(connection, Naive(DateTime.UtcNow.AddMinutes(-30)), BenignWait, 100L, ct); + + var healthyWindow = await DarlingMcpTools.AnalyzeServer(service, postgres, ServerName); + using (var doc = JsonDocument.Parse(healthyWindow)) + { + Assert.Equal("empty", doc.RootElement.GetProperty("status").GetString()); + } + Assert.Contains("All metrics are within normal ranges", healthyWindow, StringComparison.Ordinal); + Assert.Null(service.WindowEmptyMessage); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, async (cleanup, cleanupCt) => + await DeleteRowsAsync(cleanup, cleanupCt)); + } + } + + /// Naive-UTC, the Kind every timestamp column in this store is bound with. + private static DateTime Naive(DateTime value) => DateTime.SpecifyKind(value, DateTimeKind.Unspecified); + + private static async Task RegisterServerAsync(NpgsqlConnection connection, CancellationToken ct) + { + using var command = new NpgsqlCommand(@" +INSERT INTO servers (server_id, server_name, display_name, is_enabled, sql_major_version, created_date, modified_date) +VALUES ($1, $2, $3, TRUE, 15, $4, $4) +ON CONFLICT (server_id) DO UPDATE SET is_enabled = TRUE, sql_major_version = 15;", connection); + command.Parameters.AddWithValue(ServerId); + command.Parameters.AddWithValue(ServerName); + command.Parameters.AddWithValue(ServerName); + command.Parameters.AddWithValue(Naive(DateTime.UtcNow)); + await command.ExecuteNonQueryAsync(ct); + } + + private static async Task PlantWaitAsync( + NpgsqlConnection connection, DateTime at, string waitType, long waitMs, CancellationToken ct) + { + using var command = new NpgsqlCommand(@" +INSERT INTO wait_stats + (collection_id, collection_time, server_id, server_name, wait_type, delta_waiting_tasks, delta_wait_time_ms) +VALUES ($1, $2, $3, $4, $5, $6, $7)", connection); + command.Parameters.AddWithValue(CollectionIdGenerator.Next()); + command.Parameters.AddWithValue(at); + command.Parameters.AddWithValue(ServerId); + command.Parameters.AddWithValue(ServerName); + command.Parameters.AddWithValue(waitType); + command.Parameters.AddWithValue(50L); + command.Parameters.AddWithValue(waitMs); + await command.ExecuteNonQueryAsync(ct); + } + + private static async Task DeleteRowsAsync(NpgsqlConnection connection, CancellationToken ct) + { + using var cleanup = new NpgsqlCommand( + $"DELETE FROM wait_stats WHERE server_id = {ServerId}; " + + $"DELETE FROM analysis_findings WHERE server_id = {ServerId}; " + + $"DELETE FROM analysis_muted WHERE server_id = {ServerId}; " + + $"DELETE FROM servers WHERE server_id = {ServerId};", connection); + await cleanup.ExecuteNonQueryAsync(ct); + } +} diff --git a/Darling/PerformanceMonitor.Darling.Analysis/DarlingAnalysisService.cs b/Darling/PerformanceMonitor.Darling.Analysis/DarlingAnalysisService.cs index 714fd086e..263d1e47c 100644 --- a/Darling/PerformanceMonitor.Darling.Analysis/DarlingAnalysisService.cs +++ b/Darling/PerformanceMonitor.Darling.Analysis/DarlingAnalysisService.cs @@ -133,6 +133,15 @@ public sealed class DarlingAnalysisService /// public string? InsufficientDataMessage { get; private set; } + /// + /// Set after AnalyzeAsync when the server PASSED the data-span gate but the analysis window itself + /// produced zero facts (#3524). Null otherwise. The gate measures TOTAL history, so a server whose + /// collection died still sails through it and lands on an empty window — which is a dead collector + /// or an unreachable target, not a healthy server. Callers must not render an empty findings list + /// as an all-clear while this is set; nothing was measured. + /// + public string? WindowEmptyMessage { get; private set; } + /// /// How the last pass ended EARLY, or null when it ran through (#2430). Set inside the pass's own /// catch, so here means a genuine fault: the pass reached the @@ -216,6 +225,7 @@ public async Task> AnalyzeAsync(AnalysisContext context) IsAnalyzing = true; InsufficientDataMessage = null; + WindowEmptyMessage = null; EndedEarlyAs = null; try @@ -257,6 +267,21 @@ cannot unwind from inside it — these boundary checks are what turn the token i if (facts.Count == 0) { + /* #3524: the span gate above passed on LIFETIME history, so an empty WINDOW here means + collection stopped producing rows for it — not that the server is healthy. Say so, + instead of returning a bare [] that reads exactly like "analyzed and found nothing". */ + WindowEmptyMessage = + $"No facts were collected in the analysis window " + + $"({context.TimeRangeStart:yyyy-MM-dd HH:mm} to {context.TimeRangeEnd:yyyy-MM-dd HH:mm} UTC) " + + $"even though this server has {dataSpanHours:F1} hours of total collected history. " + + "Collection appears to have stopped or broken for this window, so nothing was measured " + + "— this is NOT an all-clear."; + + _logger?.LogWarning( + "[DarlingAnalysisService] No facts in the analysis window for {Server} ({Start} to {End}) " + + "despite {Span:F1}h of total history — collection may be down", + context.ServerName, context.TimeRangeStart, context.TimeRangeEnd, dataSpanHours); + LastAnalysisTime = DateTime.UtcNow; return []; } diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTools.cs index e4addaa31..2f83130b7 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTools.cs @@ -80,10 +80,33 @@ make every ordinary run look anchored to the engine — which is exactly the set ? null : "as_of was supplied, so this analysis ran over a PAST window and is exploratory: the findings below are complete but were NOT written to the store. A finding row carries the time the analysis RAN, and the reads that consume those rows (get_analysis_findings, the viewer's Recommendations tab) treat the newest analysis_time as this server's CURRENT state — so persisting a backdated run would make last week's findings today's headline and would inflate the occurrence stats of any live incident sharing a story path. Re-run without as_of to analyze and persist the present."; + if (analysisService.WindowEmptyMessage != null) + { + /* #3524: zero facts in the window is a DEAD-COLLECTOR shape, not a clean bill of + health — the data-span gate passes on lifetime history, so a server whose + collection broke last week lands here, and the "empty" all-clear below would tell + the caller in prose that all metrics are normal when nothing was measured at all. + Same miss vocabulary as get_analysis_facts' zero-facts case; same hints block as + the all-clear, because an anchored empty-window run still owes the caller the + persistence disclosure. */ + return McpHelpers.Status( + "unavailable", + analysisService.WindowEmptyMessage + + " Check get_collection_health to see when collectors last succeeded and why they stopped.", + new + { + analysis_time = analysisService.LastAnalysisTime?.ToString("o"), + persisted = anchor is null, + persistence_note = persistenceNote + }); + } + if (findings.Count == 0) { /* A successful analysis that found nothing wrong: a true negative ("all clear"), - surfaced with the shared miss vocabulary so callers branch on it uniformly. */ + surfaced with the shared miss vocabulary so callers branch on it uniformly. Facts + WERE collected and scored this time — the window-collected-nothing case returned + above as unavailable instead (#3524). */ return McpHelpers.Status( "empty", "No significant findings. All metrics are within normal ranges.", diff --git a/Lite.Tests/AnalysisWindowEmptyTests.cs b/Lite.Tests/AnalysisWindowEmptyTests.cs new file mode 100644 index 000000000..8b8d8b87b --- /dev/null +++ b/Lite.Tests/AnalysisWindowEmptyTests.cs @@ -0,0 +1,148 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor Lite. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.IO; +using System.Text.Json; +using System.Threading.Tasks; +using DuckDB.NET.Data; +using PerformanceMonitorLite.Analysis; +using PerformanceMonitorLite.Database; +using PerformanceMonitorLite.Mcp; +using PerformanceMonitorLite.Models; +using PerformanceMonitorLite.Services; +using Xunit; + +namespace PerformanceMonitorLite.Tests; + +/// +/// #3524: analyze_server must not answer "all metrics are within normal ranges" when the +/// analysis window collected NOTHING. The pipeline's data-span gate measures LIFETIME history, so a +/// server whose collection died (broken credential, unreachable target) still passes it — and the +/// zero-facts pass used to return a bare [] that the tool rendered as a true-negative +/// all-clear. Both sides of the distinction are asserted, on the REAL 24h gate rather than a zeroed +/// one, because the bug lives precisely in the gap between "enough history" and "an empty window". +/// +public sealed class AnalysisWindowEmptyTests : IClassFixture, IDisposable +{ + private const string StaleWait = "WE3524_STALE_WAIT"; + private const string BenignWait = "WE3524_BENIGN_WAIT"; + + private readonly string _tempDir; + private readonly DuckDbInitializer _duckDb; + private readonly ServerManager _serverManager; + private readonly int _serverId; + private long _nextId = -3_524_000; + private DuckDBConnection? _seedConn; + + public AnalysisWindowEmptyTests(SharedDuckDbFixture fixture) + { + fixture.ResetData(); + _duckDb = fixture.DuckDb; + + _tempDir = Path.Combine(Path.GetTempPath(), "AnalysisWindowEmptyTests_" + Guid.NewGuid().ToString("N")[..8]); + var configDir = Path.Combine(_tempDir, "config"); + Directory.CreateDirectory(configDir); + + /* Windows auth so AddServer never touches the credential store — no DPAPI side effects. */ + _serverManager = new ServerManager(configDir); + var server = new ServerConnection { ServerName = "TestServer", DisplayName = "TestServer" }; + _serverManager.AddServer(server); + + _serverId = RemoteCollectorService.GetDeterministicHashCode( + RemoteCollectorService.GetServerNameForStorage(server)); + } + + public void Dispose() + { + _seedConn?.Dispose(); + try { if (Directory.Exists(_tempDir)) Directory.Delete(_tempDir, recursive: true); } + catch { /* best-effort cleanup */ } + } + + /// + /// The dead-collector shape: 25 hours of history (the real 24h gate PASSES) whose newest row is + /// five hours old, so the tool's default 4h window holds nothing. That must come back as + /// unavailable with the pointer at collection health — never as the all-clear prose an + /// operator (or their agent) would read as "this server is fine". + /// + [Fact] + public async Task ADeadCollectorWindow_IsUnavailable_NotAnAllClear() + { + await PlantWaitAsync(DateTime.UtcNow.AddHours(-30), StaleWait, 60_000L); + await PlantWaitAsync(DateTime.UtcNow.AddHours(-5), StaleWait, 60_000L); + + var service = new AnalysisService(_duckDb); + var result = await McpAnalysisTools.AnalyzeServer(service, _serverManager); + + using var doc = JsonDocument.Parse(result); + Assert.Equal("unavailable", doc.RootElement.GetProperty("status").GetString()); + Assert.Contains("get_collection_health", result, StringComparison.Ordinal); + Assert.Contains("NOT an all-clear", result, StringComparison.Ordinal); + Assert.DoesNotContain("within normal ranges", result, StringComparison.Ordinal); + + /* The service says WHICH kind of nothing this was: the window, not the lifetime span. */ + Assert.NotNull(service.WindowEmptyMessage); + Assert.Null(service.InsufficientDataMessage); + + /* The #2506 persistence disclosure survives the new envelope — an anchored empty-window run + still owes the caller that context, and this unanchored one reports the ordinary answer. */ + Assert.True(doc.RootElement.GetProperty("hints").GetProperty("persisted").GetBoolean()); + } + + /// + /// The control, and the reason the fix is a distinction rather than a rewording: the SAME server + /// with one benign wait INSIDE the window has facts to score, finds nothing wrong, and keeps the + /// genuine true-negative all-clear. + /// + [Fact] + public async Task AWindowWithFactsButNoFindings_KeepsTheAllClear() + { + await PlantWaitAsync(DateTime.UtcNow.AddHours(-30), StaleWait, 60_000L); + await PlantWaitAsync(DateTime.UtcNow.AddHours(-5), StaleWait, 60_000L); + await PlantWaitAsync(DateTime.UtcNow.AddMinutes(-30), BenignWait, 100L); + + var service = new AnalysisService(_duckDb); + var result = await McpAnalysisTools.AnalyzeServer(service, _serverManager); + + using var doc = JsonDocument.Parse(result); + Assert.Equal("empty", doc.RootElement.GetProperty("status").GetString()); + Assert.Contains("All metrics are within normal ranges", result, StringComparison.Ordinal); + Assert.Null(service.WindowEmptyMessage); + } + + private async Task SeedConnectionAsync() + { + if (_seedConn is null) + { + _seedConn = _duckDb.CreateConnection(); + await _seedConn.OpenAsync(); + } + return _seedConn; + } + + private async Task PlantWaitAsync(DateTime at, string waitType, long deltaWaitMs) + { + using var readLock = _duckDb.AcquireReadLock(); + var conn = await SeedConnectionAsync(); + using var cmd = conn.CreateCommand(); + cmd.CommandText = @" +INSERT INTO wait_stats + (collection_id, collection_time, server_id, server_name, wait_type, + waiting_tasks_count, wait_time_ms, signal_wait_time_ms, + delta_waiting_tasks, delta_wait_time_ms, delta_signal_wait_time_ms) +VALUES ($1, $2, $3, 'TestServer', $4, 50, $5, 0, 50, $5, 0)"; + void P(object v) => cmd.Parameters.Add(new DuckDBParameter { Value = v }); + P(_nextId--); + P(at); + P(_serverId); + P(waitType); + P(deltaWaitMs); + await cmd.ExecuteNonQueryAsync(); + } +} diff --git a/Lite/Analysis/AnalysisService.cs b/Lite/Analysis/AnalysisService.cs index 731fe2248..6cdb191c6 100644 --- a/Lite/Analysis/AnalysisService.cs +++ b/Lite/Analysis/AnalysisService.cs @@ -55,6 +55,15 @@ public class AnalysisService /// public string? InsufficientDataMessage { get; private set; } + /// + /// Set after AnalyzeAsync when the server PASSED the data-span gate but the analysis window itself + /// produced zero facts (#3524). Null otherwise. The gate measures TOTAL history, so a server whose + /// collection died still sails through it and lands on an empty window — which is a dead collector + /// or an unreachable target, not a healthy server. Callers must not render an empty findings list + /// as an all-clear while this is set; nothing was measured. + /// + public string? WindowEmptyMessage { get; private set; } + /// #1757: resolves a collector's configured retention so the /// baseline provider can warn when a source table is retained for less than the baseline window. Optional /// — null simply disables that warning, which is why every existing caller keeps working unchanged. @@ -123,6 +132,7 @@ public async Task> AnalyzeAsync(AnalysisContext context) IsAnalyzing = true; InsufficientDataMessage = null; + WindowEmptyMessage = null; try { @@ -167,6 +177,21 @@ context arrives here whenever the budget is very short or the pass queued behind if (facts.Count == 0) { + /* #3524: the span gate above passed on LIFETIME history, so an empty WINDOW here means + collection stopped producing rows for it — not that the server is healthy. Say so, + instead of returning a bare [] that reads exactly like "analyzed and found nothing". */ + WindowEmptyMessage = + $"No facts were collected in the analysis window " + + $"({context.TimeRangeStart:yyyy-MM-dd HH:mm} to {context.TimeRangeEnd:yyyy-MM-dd HH:mm} UTC) " + + $"even though this server has {dataSpanHours:F1} hours of total collected history. " + + "Collection appears to have stopped or broken for this window, so nothing was measured " + + "— this is NOT an all-clear."; + + AppLogger.Warn("AnalysisService", + $"No facts in the analysis window for {context.ServerName} " + + $"({context.TimeRangeStart:o} to {context.TimeRangeEnd:o}) " + + $"despite {dataSpanHours:F1}h of total history — collection may be down"); + LastAnalysisTime = DateTime.UtcNow; return []; } diff --git a/Lite/Mcp/McpAnalysisTools.cs b/Lite/Mcp/McpAnalysisTools.cs index 4109fff1f..9697e6128 100644 --- a/Lite/Mcp/McpAnalysisTools.cs +++ b/Lite/Mcp/McpAnalysisTools.cs @@ -55,10 +55,33 @@ make every ordinary run look anchored to the engine — which is exactly the set ? null : "as_of was supplied, so this analysis ran over a PAST window and is exploratory: the findings below are complete but were NOT written to the store. A finding row carries the time the analysis RAN, and the reads that consume those rows (get_analysis_findings, the viewer's Recommendations tab) treat the newest analysis_time as this server's CURRENT state — so persisting a backdated run would make last week's findings today's headline and would inflate the occurrence stats of any live incident sharing a story path. Re-run without as_of to analyze and persist the present."; + if (analysisService.WindowEmptyMessage != null) + { + /* #3524: zero facts in the window is a DEAD-COLLECTOR shape, not a clean bill of + health — the data-span gate passes on lifetime history, so a server whose + collection broke last week lands here, and the "empty" all-clear below would tell + the caller in prose that all metrics are normal when nothing was measured at all. + Same miss vocabulary as get_analysis_facts' zero-facts case; same hints block as + the all-clear, because an anchored empty-window run still owes the caller the + persistence disclosure. */ + return McpHelpers.Status( + "unavailable", + analysisService.WindowEmptyMessage + + " Check get_collection_health to see when collectors last succeeded and why they stopped.", + new + { + analysis_time = analysisService.LastAnalysisTime?.ToString("o"), + persisted = anchor is null, + persistence_note = persistenceNote + }); + } + if (findings.Count == 0) { /* A successful analysis that found nothing wrong: a true negative ("all clear"), - surfaced with the shared miss vocabulary so callers branch on it uniformly. */ + surfaced with the shared miss vocabulary so callers branch on it uniformly. Facts + WERE collected and scored this time — the window-collected-nothing case returned + above as unavailable instead (#3524). */ return McpHelpers.Status( "empty", "No significant findings. All metrics are within normal ranges.", From 3541662a6dd6f80a98bc62aed1dd5a8864656208 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Thu, 17 Sep 2026 23:52:59 -0400 Subject: [PATCH 03/69] PG MCP verdicts: classify autovacuum from the ranking's GREATEST, pick worst slot by severity, and stop spelling unknown growth as measured zero (#3554) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit get_pg_autovacuum_health (#3534): the reader ranks by GREATEST(dead ratio, insert ratio) but severity came from the dead ratio alone, so an append-only worst_table ten times past its INSERT threshold read "ok". Severity now comes from the same GREATEST the ranking uses, the insert-side ratio is published as insert_threshold_ratio (mirroring the SQL CASE's -1 sentinel handling), a one-sample window nulls dead_tuples_growing and dead_tuple_change instead of claiming "flat" (first_seen_at published so the caller can see the window), and the page-scoped summary counts gain the sibling limit_reached discriminator with past_threshold_count reading both axes. get_pg_replication_slots (#3535): worst_slot was the fattest slot, not the worst-classified one — an active 45 GB keeping-pace slot ("ok") headlined over an inactive 2->8 GB grower ("critical_orphan_filling_disk"). Slots now order by a severity Rank (the wraparound tool's pattern) with size only as the tiebreak; growth is null when either endpoint is the -1 sentinel or the window holds one sample, with unknown-growth Classify arms (warning_retaining_wal_growth_unknown / info_inactive_growth_unknown) so a sentinel can never read as stable; and raw retained_wal_bytes nulls the sentinel as its _gb sibling always has. Tests: Classify unknown-growth arms and the Rank ladder pinned (including that every emittable severity holds a rung above ok); two new gated live classes drive both tools end to end against a real store — the severity-vs-size worst pick, the insert-driven worst_table's severity, sentinel and one-sample growth reading unknown, and limit_reached. Fixes #3534, fixes #3535 Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx Co-authored-by: Claude Fable 5 --- .../DarlingPgAutovacuumReaderTests.cs | 14 ++ .../DarlingPgAutovacuumVerdictLiveTests.cs | 177 ++++++++++++++++++ .../Darling.Tests/DarlingPgSlotReaderTests.cs | 89 +++++++++ .../DarlingPgSlotVerdictLiveTests.cs | 165 ++++++++++++++++ .../Mcp/DarlingMcpPgAutovacuumTools.cs | 45 ++++- .../Mcp/DarlingMcpPgSlotTools.cs | 63 +++++-- 6 files changed, 534 insertions(+), 19 deletions(-) create mode 100644 Darling/Darling.Tests/DarlingPgAutovacuumVerdictLiveTests.cs create mode 100644 Darling/Darling.Tests/DarlingPgSlotVerdictLiveTests.cs diff --git a/Darling/Darling.Tests/DarlingPgAutovacuumReaderTests.cs b/Darling/Darling.Tests/DarlingPgAutovacuumReaderTests.cs index 1c1d230e2..55e41bb86 100644 --- a/Darling/Darling.Tests/DarlingPgAutovacuumReaderTests.cs +++ b/Darling/Darling.Tests/DarlingPgAutovacuumReaderTests.cs @@ -191,6 +191,20 @@ public void FarPastThresholdIsCriticalEvenWhenFlat() Assert.Equal("critical_far_past_threshold", DarlingMcpPgAutovacuumTools.Classify(false, 25.0, true)); } + /// + /// Growth null — a one-sample window — never escalates to the growing verdict and never reads as + /// flat's "blocked or not running" implication either: the band's plain label carries it, and the + /// bands growth does not refine are untouched by it. + /// + [Fact] + public void UnknownGrowthNeverEscalatesAndNeverReadsFlat() + { + Assert.Equal("warning_past_threshold", DarlingMcpPgAutovacuumTools.Classify(false, 3.0, null)); + Assert.Equal("critical_far_past_threshold", DarlingMcpPgAutovacuumTools.Classify(false, 25.0, null)); + Assert.Equal("info_at_threshold", DarlingMcpPgAutovacuumTools.Classify(false, 1.5, null)); + Assert.Equal("ok", DarlingMcpPgAutovacuumTools.Classify(false, 0.5, null)); + } + /// Every severity is a distinct string, so a caller can switch on it. [Fact] public void SeveritiesAreDistinct() diff --git a/Darling/Darling.Tests/DarlingPgAutovacuumVerdictLiveTests.cs b/Darling/Darling.Tests/DarlingPgAutovacuumVerdictLiveTests.cs new file mode 100644 index 000000000..cf5390132 --- /dev/null +++ b/Darling/Darling.Tests/DarlingPgAutovacuumVerdictLiveTests.cs @@ -0,0 +1,177 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Linq; +using System.Text.Json; +using System.Threading; +using System.Threading.Tasks; +using Npgsql; +using PerformanceMonitor.Collectors; +using PerformanceMonitor.Darling.Service.Mcp; +using PerformanceMonitor.Darling.Storage; +using Xunit; + +namespace Darling.Tests; + +/// +/// get_pg_autovacuum_health end to end against a real store — the #3534 verdict seam. +/// +/// The test that carries this class is the append-only table. The read ranks by +/// GREATEST(dead ratio, insert ratio), so a table ten times past its INSERT threshold with zero dead +/// tuples arrives as worst_table — and the tool used to classify it from the dead ratio alone, handing +/// the #1-ranked table severity "ok" with a 0 threshold_ratio. An agent reads that as "worst thing here +/// is fine" and moves on from the classic wraparound route the collector gathers inserts_since_vacuum +/// for in the first place. The projection arithmetic lives inline in the tool, so only a real round-trip +/// exercises it. +/// +/// And the growth claim a single sample cannot make. One reading means first == latest, and +/// dead_tuples_growing = false from it converts into "autovacuum blocked or not running" territory the +/// window cannot support. Null, with first_seen_at published so the caller can see how much history +/// stands behind the claim. +/// +[Collection("live-postgres")] +public sealed class DarlingPgAutovacuumVerdictLiveTests +{ + private const int ServerId = -853917; + private const string ServerName = "pg-autovacuum-verdict-e2e"; + + private static string? ConnectionString => Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + + [Fact] + public async Task InsertDrivenWorstTableCarriesItsRankSeverity_AndOneSampleGrowthIsUnknown() + { + var cs = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(cs), + "Set DARLING_TEST_PG to a Postgres connection string to run the live autovacuum verdict test."); + + var ct = TestContext.Current.CancellationToken; + using var connection = new NpgsqlConnection(cs); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + await DeleteRowsAsync(connection, ct); + + await using var dataSource = NpgsqlDataSource.Create(cs!); + var bodySucceeded = false; + + try + { + await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); + + var t0 = MinutesAgo(30); + var t1 = t0.AddMinutes(10); + + /* appendonly: zero dead tuples, 10x past its INSERT threshold. GREATEST ranks it #1; the + dead-only severity called it "ok" (#3534's scenario, verbatim). */ + await SeedAsync(connection, ct, t0, "appendonly", deadTuples: 0, vacuumThreshold: 1_050, + insertsSinceVacuum: 100_000, insertVacuumThreshold: 10_000); + await SeedAsync(connection, ct, t1, "appendonly", deadTuples: 0, vacuumThreshold: 1_050, + insertsSinceVacuum: 100_000, insertVacuumThreshold: 10_000); + + /* churny: past its dead threshold and climbing — the case the dead axis already handled, + kept as the control that the insert axis widens the verdict rather than replacing it. */ + await SeedAsync(connection, ct, t0, "churny", deadTuples: 1_500, vacuumThreshold: 1_000, + insertsSinceVacuum: 0, insertVacuumThreshold: 10_000); + await SeedAsync(connection, ct, t1, "churny", deadTuples: 2_500, vacuumThreshold: 1_000, + insertsSinceVacuum: 0, insertVacuumThreshold: 10_000); + + /* onesample: a single reading. Growth is unmeasurable, which is not the same as "not + growing". */ + await SeedAsync(connection, ct, t1, "onesample", deadTuples: 3_000, vacuumThreshold: 1_000, + insertsSinceVacuum: 0, insertVacuumThreshold: 10_000); + + var payload = JsonDocument.Parse( + await DarlingMcpPgAutovacuumTools.GetPgAutovacuumHealth(dataSource, ServerName, 4)).RootElement; + + Assert.Equal("tables_with_pending_maintenance", payload.GetProperty("status").GetString()); + + var tables = payload.GetProperty("tables").EnumerateArray().ToArray(); + + /* THE assertion: worst_table and its severity come from the SAME axis. Ten times past the + insert threshold is critical, and the insert-side ratio is published beside the dead one + so the figure the verdict came from is visible. */ + Assert.Equal("public.appendonly", payload.GetProperty("worst_table").GetString()); + Assert.Equal("critical_far_past_threshold", payload.GetProperty("worst_severity").GetString()); + + var append = Row(tables, "public.appendonly"); + Assert.Equal("critical_far_past_threshold", append.GetProperty("severity").GetString()); + Assert.Equal(10.0, append.GetProperty("insert_threshold_ratio").GetDouble(), 2); + Assert.Equal(0.0, append.GetProperty("threshold_ratio").GetDouble(), 2); + + /* The control: a dead-driven table classifies exactly as it always did. */ + var churny = Row(tables, "public.churny"); + Assert.Equal("warning_past_threshold_and_growing", churny.GetProperty("severity").GetString()); + Assert.True(churny.GetProperty("dead_tuples_growing").GetBoolean()); + Assert.Equal(1_000, churny.GetProperty("dead_tuple_change").GetInt64()); + + /* One sample: growth and the change figure are NULL, and first_seen_at == measured_at says + why — the window holds one reading, not a flat line. The band's plain label carries it. */ + var oneSample = Row(tables, "public.onesample"); + Assert.Equal(JsonValueKind.Null, oneSample.GetProperty("dead_tuples_growing").ValueKind); + Assert.Equal(JsonValueKind.Null, oneSample.GetProperty("dead_tuple_change").ValueKind); + Assert.Equal( + oneSample.GetProperty("measured_at").GetDateTime(), + oneSample.GetProperty("first_seen_at").GetDateTime()); + Assert.Equal("warning_past_threshold", oneSample.GetProperty("severity").GetString()); + + /* The summary counts read both axes, growing counts only MEASURED growth, and the + undersized call discloses its page scope (the get_pg_database_stats pattern). */ + Assert.Equal(3, payload.GetProperty("past_threshold_count").GetInt32()); + Assert.Equal(1, payload.GetProperty("growing_count").GetInt32()); + Assert.False(payload.GetProperty("limit_reached").GetBoolean()); + + var truncated = JsonDocument.Parse( + await DarlingMcpPgAutovacuumTools.GetPgAutovacuumHealth(dataSource, ServerName, 4, 2)).RootElement; + Assert.Equal(2, truncated.GetProperty("table_count").GetInt32()); + Assert.True(truncated.GetProperty("limit_reached").GetBoolean()); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, async (cleanup, cleanupCt) => + await DeleteRowsAsync(cleanup, cleanupCt)); + } + } + + /// The row for one table, asserted present with a message naming what DID come back. + private static JsonElement Row(JsonElement[] rows, string tableName) + { + var row = rows.FirstOrDefault(r => r.GetProperty("table_name").GetString() == tableName); + + Assert.True( + row.ValueKind == JsonValueKind.Object, + $"no row for '{tableName}' — the read returned [{string.Join(", ", rows.Select(r => r.GetProperty("table_name").GetString()))}]"); + + return row; + } + + private static DateTime MinutesAgo(int minutes) => + DarlingMcpTestData.TruncateToSeconds(DateTime.UtcNow.AddMinutes(-minutes)); + + private static async Task SeedAsync( + NpgsqlConnection connection, CancellationToken ct, DateTime collectionTimeUtc, string tableName, + long deadTuples, long vacuumThreshold, long insertsSinceVacuum, long insertVacuumThreshold) => + await DarlingMcpTestData.ExecAsync(connection, ct, @" +INSERT INTO pg_autovacuum_stats + (collection_id, collection_time, server_id, server_name, database_name, schema_name, table_name, + live_tuples, dead_tuples, vacuum_threshold, mods_since_analyze, analyze_threshold, + inserts_since_vacuum, insert_vacuum_threshold, autovacuum_disabled, total_bytes, autovacuum_count) +VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11, $12, $13, $14, $15, $16, $17)", + CollectionIdGenerator.Next(), DarlingMcpTestData.Naive(collectionTimeUtc), + ServerId, ServerName, "appdb", "public", tableName, + 10_000L, deadTuples, vacuumThreshold, 0L, 500L, + insertsSinceVacuum, insertVacuumThreshold, false, 1_000_000L, 1L); + + private static async Task DeleteRowsAsync(NpgsqlConnection connection, CancellationToken ct) + { + await DarlingMcpTestData.ExecAsync(connection, ct, "DELETE FROM pg_autovacuum_stats WHERE server_id = $1", ServerId); + await DarlingMcpTestData.ExecAsync(connection, ct, "DELETE FROM servers WHERE server_id = $1", ServerId); + await DarlingMcpTestData.ExecAsync(connection, ct, "DELETE FROM config_monitored_servers WHERE server_id = $1", ServerId); + } +} diff --git a/Darling/Darling.Tests/DarlingPgSlotReaderTests.cs b/Darling/Darling.Tests/DarlingPgSlotReaderTests.cs index a62df35ce..a14d175cc 100644 --- a/Darling/Darling.Tests/DarlingPgSlotReaderTests.cs +++ b/Darling/Darling.Tests/DarlingPgSlotReaderTests.cs @@ -88,6 +88,7 @@ public void TerminalWalStatesAreCriticalRegardlessOfActivityOrGrowth(string walS { Assert.Equal(expected, DarlingMcpPgSlotTools.Classify(walStatus, isActive: true, retainedWalGrowing: false)); Assert.Equal(expected, DarlingMcpPgSlotTools.Classify(walStatus, isActive: false, retainedWalGrowing: true)); + Assert.Equal(expected, DarlingMcpPgSlotTools.Classify(walStatus, isActive: false, retainedWalGrowing: null)); } /// @@ -137,6 +138,92 @@ public void ActiveReservedSlotIsOk() Assert.Equal("ok", DarlingMcpPgSlotTools.Classify("reserved", isActive: true, retainedWalGrowing: false)); } + /// + /// Unknown growth — the collector's -1 sentinel, or a one-sample window — is its own verdict, never + /// the flat one. "Flat" is a measured claim, and it is the exact claim that separates a consumer + /// between polls from a volume filling; a sentinel spelled as measured zero read every unmeasurable + /// slot as stable (#3535). + /// + [Fact] + public void UnknownGrowthIsItsOwnVerdictNeverFlat() + { + /* The orphan-candidate shape with its discriminator unmeasured: a warning that says so, not + the measured-flat warning and not a fabricated critical. */ + Assert.Equal( + "warning_retaining_wal_growth_unknown", + DarlingMcpPgSlotTools.Classify("extended", isActive: false, retainedWalGrowing: null)); + + Assert.Equal( + "info_inactive_growth_unknown", + DarlingMcpPgSlotTools.Classify("reserved", isActive: false, retainedWalGrowing: null)); + Assert.Equal( + "info_inactive_growth_unknown", + DarlingMcpPgSlotTools.Classify(null, isActive: false, retainedWalGrowing: null)); + + /* Where measured growth would not escalate anyway, unknown growth manufactures nothing: an + active consumer stays ok, and an active extended slot stays the plain retaining warning. */ + Assert.Equal("ok", DarlingMcpPgSlotTools.Classify("reserved", isActive: true, retainedWalGrowing: null)); + Assert.Equal( + "warning_retaining_wal", + DarlingMcpPgSlotTools.Classify("extended", isActive: true, retainedWalGrowing: null)); + } + + /// + /// worst_slot is picked by severity rank with size only as the tiebreak, so the ladder itself is + /// pinned: strictly descending, criticals above warnings above infos above ok, and a label Rank does + /// not know sorts with ok rather than above anything it does. + /// + [Fact] + public void RankOrdersEverySeverityWorstFirst() + { + var ladder = new[] + { + "critical_slot_lost", + "critical_wal_already_removed", + "critical_orphan_filling_disk", + "warning_inactive_and_growing", + "warning_retaining_wal_growth_unknown", + "warning_retaining_wal", + "info_inactive_growth_unknown", + "info_inactive", + "ok", + }; + + for (var i = 1; i < ladder.Length; i++) + { + Assert.True( + DarlingMcpPgSlotTools.Rank(ladder[i - 1]) > DarlingMcpPgSlotTools.Rank(ladder[i]), + $"'{ladder[i - 1]}' must outrank '{ladder[i]}'"); + } + + Assert.Equal(0, DarlingMcpPgSlotTools.Rank("ok")); + Assert.Equal(0, DarlingMcpPgSlotTools.Rank("a_label_rank_does_not_know")); + } + + /// + /// Every non-ok label Classify can emit holds a rung above ok, so a future Classify arm that skips + /// Rank cannot ship a severity that silently never headlines. + /// + [Fact] + public void EveryClassifiableSeverityHasARung() + { + var emittable = new[] + { + DarlingMcpPgSlotTools.Classify("lost", false, false), + DarlingMcpPgSlotTools.Classify("unreserved", false, false), + DarlingMcpPgSlotTools.Classify("extended", false, true), + DarlingMcpPgSlotTools.Classify("extended", false, null), + DarlingMcpPgSlotTools.Classify("extended", true, false), + DarlingMcpPgSlotTools.Classify("reserved", false, true), + DarlingMcpPgSlotTools.Classify("reserved", false, null), + DarlingMcpPgSlotTools.Classify("reserved", false, false), + }; + + Assert.All(emittable, severity => Assert.True( + DarlingMcpPgSlotTools.Rank(severity) > 0, + $"'{severity}' ranks 0 — it ties with ok and can never be picked as worst_slot")); + } + /// /// wal_status is NULL on a slot with no restart_lsn yet, and on Aurora it can be absent entirely. /// That must fall through to the activity/growth branches rather than throwing or reading as healthy @@ -161,8 +248,10 @@ public void SeveritiesAreDistinct() DarlingMcpPgSlotTools.Classify("lost", false, false), DarlingMcpPgSlotTools.Classify("unreserved", false, false), DarlingMcpPgSlotTools.Classify("extended", false, true), + DarlingMcpPgSlotTools.Classify("extended", false, null), DarlingMcpPgSlotTools.Classify("extended", true, false), DarlingMcpPgSlotTools.Classify("reserved", false, true), + DarlingMcpPgSlotTools.Classify("reserved", false, null), DarlingMcpPgSlotTools.Classify("reserved", false, false), DarlingMcpPgSlotTools.Classify("reserved", true, false), }; diff --git a/Darling/Darling.Tests/DarlingPgSlotVerdictLiveTests.cs b/Darling/Darling.Tests/DarlingPgSlotVerdictLiveTests.cs new file mode 100644 index 000000000..7a97c5208 --- /dev/null +++ b/Darling/Darling.Tests/DarlingPgSlotVerdictLiveTests.cs @@ -0,0 +1,165 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Linq; +using System.Text.Json; +using System.Threading; +using System.Threading.Tasks; +using Npgsql; +using PerformanceMonitor.Collectors; +using PerformanceMonitor.Darling.Service.Mcp; +using PerformanceMonitor.Darling.Storage; +using Xunit; + +namespace Darling.Tests; + +/// +/// get_pg_replication_slots end to end against a real store — the #3535 verdict seam. +/// +/// The test that carries this class is the severity-versus-size pick. The tool's own design +/// note says the size of the hole matters less than whether it is still being dug, and then worst_slot +/// was picked by retained bytes: an active 45 GB keeping-pace slot ("ok") headlined over an inactive +/// 2→8 GB grower ("critical_orphan_filling_disk") — the one slot the caller needed to see first. The +/// ordering and pick live inline in the tool, so only a real round-trip exercises them. +/// +/// And the growth a sentinel cannot measure. The collector's -1 not-applicable sentinel used +/// to become a measured zero — "no growth" — so an unmeasurable slot read as stable; a one-sample window +/// is the same fabrication from a single point. Both surface as null growth plus an unknown-growth +/// verdict, never as flat. +/// +[Collection("live-postgres")] +public sealed class DarlingPgSlotVerdictLiveTests +{ + private const int ServerId = -853918; + private const string ServerName = "pg-slot-verdict-e2e"; + + private const long Gb = 1024L * 1024 * 1024; + + private static string? ConnectionString => Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + + [Fact] + public async Task WorstSlotIsTheWorstClassified_AndUnmeasurableGrowthReadsUnknown() + { + var cs = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(cs), + "Set DARLING_TEST_PG to a Postgres connection string to run the live slot verdict test."); + + var ct = TestContext.Current.CancellationToken; + using var connection = new NpgsqlConnection(cs); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + await DeleteRowsAsync(connection, ct); + + await using var dataSource = NpgsqlDataSource.Create(cs!); + var bodySucceeded = false; + + try + { + await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); + + var t0 = MinutesAgo(30); + var t1 = t0.AddMinutes(10); + + /* fat_ok: the biggest hole and the healthiest slot here — active, reserved, holding 45 GB + steady. The size pick made this the headline (#3535's scenario, verbatim). */ + await SeedAsync(connection, ct, t0, "fat_ok", isActive: true, walStatus: "reserved", retainedWalBytes: 45 * Gb); + await SeedAsync(connection, ct, t1, "fat_ok", isActive: true, walStatus: "reserved", retainedWalBytes: 45 * Gb); + + /* thin_orphan: a fraction of the size, inactive, extended, and GROWING — the disk bomb. */ + await SeedAsync(connection, ct, t0, "thin_orphan", isActive: false, walStatus: "extended", retainedWalBytes: 2 * Gb); + await SeedAsync(connection, ct, t1, "thin_orphan", isActive: false, walStatus: "extended", retainedWalBytes: 8 * Gb); + + /* sentinel: retained bytes NULL in both readings (the -1 sentinel downstream) — growth is + unknowable, and "stable" would be a fabricated claim. */ + await SeedAsync(connection, ct, t0, "sentinel", isActive: false, walStatus: "extended", retainedWalBytes: null); + await SeedAsync(connection, ct, t1, "sentinel", isActive: false, walStatus: "extended", retainedWalBytes: null); + + /* single: one reading with a measured size — the size is real, the growth claim is not. */ + await SeedAsync(connection, ct, t1, "single", isActive: false, walStatus: "reserved", retainedWalBytes: 1 * Gb); + + var payload = JsonDocument.Parse( + await DarlingMcpPgSlotTools.GetPgReplicationSlots(dataSource, ServerName, 4)).RootElement; + + Assert.Equal("slots_present", payload.GetProperty("status").GetString()); + + /* THE assertion: the headline is the worst-CLASSIFIED slot, not the fattest, and the list + leads with it. */ + Assert.Equal("thin_orphan", payload.GetProperty("worst_slot").GetString()); + Assert.Equal("critical_orphan_filling_disk", payload.GetProperty("worst_severity").GetString()); + + var slots = payload.GetProperty("slots").EnumerateArray().ToArray(); + Assert.Equal("thin_orphan", slots[0].GetProperty("slot_name").GetString()); + + /* The fat slot keeps its honest verdict and its figure — it just no longer buys the + headline with it. */ + var fat = Row(slots, "fat_ok"); + Assert.Equal("ok", fat.GetProperty("severity").GetString()); + Assert.Equal(45.0, fat.GetProperty("retained_wal_gb").GetDouble(), 2); + + /* Sentinel: null size (raw AND _gb — the raw field used to serialize -1), null growth, and + the unknown-growth verdict rather than the measured-flat one. */ + var sentinel = Row(slots, "sentinel"); + Assert.Equal("warning_retaining_wal_growth_unknown", sentinel.GetProperty("severity").GetString()); + Assert.Equal(JsonValueKind.Null, sentinel.GetProperty("retained_wal_bytes").ValueKind); + Assert.Equal(JsonValueKind.Null, sentinel.GetProperty("retained_wal_gb").ValueKind); + Assert.Equal(JsonValueKind.Null, sentinel.GetProperty("retained_wal_growth_bytes").ValueKind); + Assert.Equal(JsonValueKind.Null, sentinel.GetProperty("retained_wal_growth_gb_per_hour").ValueKind); + + /* One sample: measured size, unmeasurable growth. */ + var single = Row(slots, "single"); + Assert.Equal("info_inactive_growth_unknown", single.GetProperty("severity").GetString()); + Assert.Equal(1.0, single.GetProperty("retained_wal_gb").GetDouble(), 2); + Assert.Equal(JsonValueKind.Null, single.GetProperty("retained_wal_growth_bytes").ValueKind); + + /* The total sums only measured sizes — a sentinel never subtracts phantom gigabytes. */ + Assert.Equal(54.0, payload.GetProperty("total_retained_wal_gb").GetDouble(), 1); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, async (cleanup, cleanupCt) => + await DeleteRowsAsync(cleanup, cleanupCt)); + } + } + + /// The row for one slot, asserted present with a message naming what DID come back. + private static JsonElement Row(JsonElement[] rows, string slotName) + { + var row = rows.FirstOrDefault(r => r.GetProperty("slot_name").GetString() == slotName); + + Assert.True( + row.ValueKind == JsonValueKind.Object, + $"no row for '{slotName}' — the read returned [{string.Join(", ", rows.Select(r => r.GetProperty("slot_name").GetString()))}]"); + + return row; + } + + private static DateTime MinutesAgo(int minutes) => + DarlingMcpTestData.TruncateToSeconds(DateTime.UtcNow.AddMinutes(-minutes)); + + private static async Task SeedAsync( + NpgsqlConnection connection, CancellationToken ct, DateTime collectionTimeUtc, string slotName, + bool isActive, string walStatus, long? retainedWalBytes) => + await DarlingMcpTestData.ExecAsync(connection, ct, @" +INSERT INTO collect.pg_replication_slot_stats + (collection_id, collection_time, server_id, server_name, slot_name, slot_type, plugin, + database_name, is_active, wal_status, retained_wal_bytes, conflicting) +VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11, $12)", + CollectionIdGenerator.Next(), DarlingMcpTestData.Naive(collectionTimeUtc), + ServerId, ServerName, slotName, "logical", "pgoutput", "appdb", isActive, walStatus, + retainedWalBytes, false); + + private static async Task DeleteRowsAsync(NpgsqlConnection connection, CancellationToken ct) + { + await DarlingMcpTestData.ExecAsync(connection, ct, "DELETE FROM collect.pg_replication_slot_stats WHERE server_id = $1", ServerId); + await DarlingMcpTestData.ExecAsync(connection, ct, "DELETE FROM servers WHERE server_id = $1", ServerId); + await DarlingMcpTestData.ExecAsync(connection, ct, "DELETE FROM config_monitored_servers WHERE server_id = $1", ServerId); + } +} diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgAutovacuumTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgAutovacuumTools.cs index 5a092a3c6..5236f71ee 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgAutovacuumTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgAutovacuumTools.cs @@ -29,8 +29,13 @@ public sealed class DarlingMcpPgAutovacuumTools /// count is routine on a large table and urgent on a small one, so the ratio is the only comparable /// figure — and whether the pile is growing separates autovacuum losing a race from autovacuum not /// running at all. + /// The ratio handed in is the WORSE of the dead-tuple and insert-only ratios — the same GREATEST + /// the read ranks by. Classifying from the dead ratio alone let the #1-ranked table, an append-only + /// one far past its INSERT threshold, carry severity "ok" (#3534). + /// Growth is nullable because a one-sample window cannot measure it: null never escalates, and + /// never reads as "flat" either. /// - internal static string Classify(bool autovacuumDisabled, double? thresholdRatio, bool deadTuplesGrowing) + internal static string Classify(bool autovacuumDisabled, double? thresholdRatio, bool? deadTuplesGrowing) { /* A configuration finding, and the one case where the count is beside the point: this table will never be vacuumed by autovacuum no matter how bad it gets, and it holds back the whole @@ -50,7 +55,7 @@ internal static string Classify(bool autovacuumDisabled, double? thresholdRatio, /* Ten times past the line is not a busy table, it is a stuck one — most often autovacuum being cancelled repeatedly by conflicting locks, or starved of workers. */ >= 10 => "critical_far_past_threshold", - >= 2 when deadTuplesGrowing => "warning_past_threshold_and_growing", + >= 2 when deadTuplesGrowing == true => "warning_past_threshold_and_growing", >= 2 => "warning_past_threshold", >= 1 => "info_at_threshold", _ => "ok", @@ -113,19 +118,37 @@ table we know nothing about above tables we have measured. */ double? analyzeRatio = r.AnalyzeThreshold > 0 ? Math.Round((double)r.ModsSinceAnalyze / r.AnalyzeThreshold, 2) : null; - var growing = r.DeadTuples > r.FirstDeadTuples; + /* The insert-side twin, mirroring the read's ORDER BY CASE: the -1 sentinel (a major + without autovacuum_vacuum_insert_threshold, or an unreadable inserts figure) produces + no ratio rather than a negative one. */ + double? insertRatio = r.InsertVacuumThreshold > 0 && r.InsertsSinceVacuum >= 0 + ? Math.Round((double)r.InsertsSinceVacuum / r.InsertVacuumThreshold, 2) + : null; + /* Severity comes from the axis the ranking already uses — GREATEST(dead, insert), NULLs + ignored, exactly as the read orders. Classifying from the dead ratio alone let an + append-only worst_table read "ok" (#3534). */ + double? worstRatio = ratio is null ? insertRatio + : insertRatio is null ? ratio + : Math.Max(ratio.Value, insertRatio.Value); + /* One sample cannot measure growth: first == latest is the same reading, and a false + there converts into "autovacuum blocked or not running" territory the window cannot + support. Null, not false — and the change figure goes with it. */ + bool? growing = r.FirstSeenAt == r.MeasuredAt ? null : r.DeadTuples > r.FirstDeadTuples; return new { database_name = r.DatabaseName, table_name = $"{r.SchemaName}.{r.TableName}", - severity = Classify(r.AutovacuumDisabled, ratio, growing), + severity = Classify(r.AutovacuumDisabled, worstRatio, growing), dead_tuples = r.DeadTuples, vacuum_threshold = r.VacuumThreshold >= 0 ? r.VacuumThreshold : (long?)null, /* The headline number: 1.0 means autovacuum should be triggering right now. */ threshold_ratio = ratio, dead_tuples_growing = growing, - dead_tuple_change = r.DeadTuples - r.FirstDeadTuples, + dead_tuple_change = growing is null ? null : (long?)(r.DeadTuples - r.FirstDeadTuples), + /* The window the growth claim is measured over, so a caller can see how much history + stands behind it — and that null growth means one sample, not missing data. */ + first_seen_at = r.FirstSeenAt, live_tuples = r.LiveTuples, /* The analyze half. Stale statistics produce bad row estimates and bad plans, which is a different symptom from bloat and gets missed because both come from one process. */ @@ -136,6 +159,9 @@ a different symptom from bloat and gets missed because both come from one proces rule, and a table never vacuumed is never frozen either. */ inserts_since_vacuum = r.InsertsSinceVacuum >= 0 ? r.InsertsSinceVacuum : (long?)null, insert_vacuum_threshold = r.InsertVacuumThreshold >= 0 ? r.InsertVacuumThreshold : (long?)null, + /* threshold_ratio's insert-side sibling, so the figure severity ranks on is visible + when the dead-tuple ratio is not the one that put the table here. */ + insert_threshold_ratio = insertRatio, autovacuum_disabled = r.AutovacuumDisabled, total_bytes = r.TotalBytes >= 0 ? r.TotalBytes : (long?)null, total_gb = r.TotalBytes >= 0 ? Math.Round(r.TotalBytes / 1024.0 / 1024.0 / 1024.0, 2) : (double?)null, @@ -158,9 +184,14 @@ a different symptom from bloat and gets missed because both come from one proces hours_back, status = "tables_with_pending_maintenance", table_count = tables.Count, + /* Page-scoped counts: each is computed over the rows the read's LIMIT let through, not + the server. limit_reached is the discriminator that makes that caveat actionable — when + it bit, the caller knows these are top-N figures and can raise the limit (the + get_pg_database_stats pattern). */ autovacuum_disabled_count = tables.Count(t => t.autovacuum_disabled), - past_threshold_count = tables.Count(t => t.threshold_ratio >= 1), - growing_count = tables.Count(t => t.dead_tuples_growing), + past_threshold_count = tables.Count(t => t.threshold_ratio >= 1 || t.insert_threshold_ratio >= 1), + growing_count = tables.Count(t => t.dead_tuples_growing == true), + limit_reached = tables.Count >= limit, worst_table = tables[0].table_name, worst_severity = tables[0].severity, tables, diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgSlotTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgSlotTools.cs index af0eb207a..b3499c09e 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgSlotTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgSlotTools.cs @@ -28,8 +28,12 @@ public sealed class DarlingMcpPgSlotTools /// Severity from slot state, not from the retained figure alone. The size of the hole matters far /// less than whether it is still being dug: a slot holding 45 GB steadily is a consumer keeping pace, /// while one that grew from 2 GB to 45 GB in an hour is a volume filling in front of you. + /// Growth is nullable because it is not always measurable — the collector's -1 sentinel on + /// either endpoint, or a one-sample window. Unknown never escalates to the growing verdict, and never + /// reads as the flat one either: "flat" is a measured claim, and it is the claim that separates a + /// consumer between polls from a volume filling (#3535). /// - internal static string Classify(string? walStatus, bool isActive, bool retainedWalGrowing) => + internal static string Classify(string? walStatus, bool isActive, bool? retainedWalGrowing) => walStatus switch { /* The slot is already unusable — its consumer cannot resume and needs recreating. */ @@ -37,13 +41,37 @@ internal static string Classify(string? walStatus, bool isActive, bool retainedW /* Required WAL has been removed; the consumer is about to find that out. */ "unreserved" => "critical_wal_already_removed", /* WAL is being retained BECAUSE of this slot. Inactive and still growing is the disk bomb. */ - "extended" when !isActive && retainedWalGrowing => "critical_orphan_filling_disk", + "extended" when !isActive && retainedWalGrowing == true => "critical_orphan_filling_disk", + "extended" when !isActive && retainedWalGrowing is null => "warning_retaining_wal_growth_unknown", "extended" => "warning_retaining_wal", - _ when !isActive && retainedWalGrowing => "warning_inactive_and_growing", + _ when !isActive && retainedWalGrowing == true => "warning_inactive_and_growing", + _ when !isActive && retainedWalGrowing is null => "info_inactive_growth_unknown", _ when !isActive => "info_inactive", _ => "ok", }; + /// + /// Rank by how bad the label is, so the headline slot is the worst-CLASSIFIED one rather than the + /// fattest one (the get_pg_wraparound_risk pattern). Picking worst_slot by retained size contradicted + /// this type's own design note: an active 45 GB keeping-pace slot ("ok") outranked an inactive 2→8 GB + /// grower ("critical_orphan_filling_disk") — the one slot the caller needed to see first (#3535). + /// Size still breaks ties within a label. + /// + internal static int Rank(string severity) => severity switch + { + "critical_slot_lost" => 8, + "critical_wal_already_removed" => 7, + "critical_orphan_filling_disk" => 6, + "warning_inactive_and_growing" => 5, + /* Above the measured-flat warning: this is the orphan shape with its discriminator unmeasured, + which deserves the look before a slot known to be holding steady. */ + "warning_retaining_wal_growth_unknown" => 4, + "warning_retaining_wal" => 3, + "info_inactive_growth_unknown" => 2, + "info_inactive" => 1, + _ => 0, + }; + [McpServerTool(Name = "get_pg_replication_slots"), Description("Gets PostgreSQL replication slot health, including whether any slot is retaining WAL without bound. An abandoned slot is one of the few PostgreSQL conditions that can take a server down by itself, and it does so two independent ways: it retains every WAL segment its consumer has not confirmed - unbounded by default, so it will fill the volume and stop the server - and it simultaneously pins the vacuum horizon so nothing gets reclaimed cluster-wide. Reports whether retained WAL is still GROWING across the window, which is the difference between a consumer that is merely behind and a volume filling in front of you. Common orphan sources are a removed CDC task, a finished blue/green deployment, a decommissioned Debezium consumer, or a failed major-version upgrade. Works on any PostgreSQL target.")] public static async Task GetPgReplicationSlots( NpgsqlDataSource postgres, @@ -94,11 +122,16 @@ collector has never run and never will. */ var slots = rows.Select(r => { - var growthBytes = r.RetainedWalBytes >= 0 && r.FirstRetainedWalBytes >= 0 - ? r.RetainedWalBytes - r.FirstRetainedWalBytes - : 0; + /* Growth exists only when BOTH endpoints were measured across a window that spans time. + The -1 sentinel used to become a measured 0 here — "no growth" — feeding growing=false + into Classify, so an unmeasurable slot read as stable (#3535); and a one-sample window + is the same fabrication one step milder, "flat" from a single point. Null, both. */ + long? growthBytes = + r.MeasuredAt != r.FirstSeenAt && r.RetainedWalBytes >= 0 && r.FirstRetainedWalBytes >= 0 + ? r.RetainedWalBytes - r.FirstRetainedWalBytes + : null; var hours = Math.Max((r.MeasuredAt - r.FirstSeenAt).TotalHours, 0); - var growing = growthBytes > 0; + bool? growing = growthBytes is null ? null : growthBytes > 0; return new { @@ -109,15 +142,17 @@ collector has never run and never will. */ database_name = r.DatabaseName, is_active = r.IsActive, wal_status = r.WalStatus, - retained_wal_bytes = r.RetainedWalBytes, + /* -1 is the collector's sentinel, not a size — null it here as the _gb twin below + always has, rather than serializing a figure no slot can hold. */ + retained_wal_bytes = r.RetainedWalBytes >= 0 ? r.RetainedWalBytes : (long?)null, retained_wal_gb = r.RetainedWalBytes >= 0 ? Math.Round(r.RetainedWalBytes / 1024.0 / 1024.0 / 1024.0, 2) : (double?)null, /* Growth is the actionable half. Rate is only reported when the window actually spans time, so a single-sample window cannot produce a fabricated per-hour figure. */ retained_wal_growth_bytes = growthBytes, - retained_wal_growth_gb_per_hour = hours >= 0.05 - ? Math.Round(growthBytes / 1024.0 / 1024.0 / 1024.0 / hours, 3) + retained_wal_growth_gb_per_hour = growthBytes is not null && hours >= 0.05 + ? Math.Round(growthBytes.Value / 1024.0 / 1024.0 / 1024.0 / hours, 3) : (double?)null, /* -1 is the collector's not-applicable sentinel: safe_wal_size is NULL whenever max_slot_wal_keep_size is -1, which is the default, so on a stock server there is @@ -132,7 +167,11 @@ no configured ceiling at all. */ conflicting = r.Conflicting, }; }) - .OrderByDescending(s => s.retained_wal_bytes) + /* Worst-CLASSIFIED first; size only breaks ties. Ordering by size alone put the fattest slot + in worst_slot regardless of verdict (#3535). Unknown sizes sort after measured ones within + a label. */ + .OrderByDescending(s => Rank(s.severity)) + .ThenByDescending(s => s.retained_wal_bytes ?? -1) .ToList(); var worst = slots[0]; @@ -147,7 +186,7 @@ no configured ceiling at all. */ worst_slot = worst.slot_name, worst_severity = worst.severity, total_retained_wal_gb = Math.Round( - slots.Where(s => s.retained_wal_bytes > 0).Sum(s => s.retained_wal_bytes) / 1024.0 / 1024.0 / 1024.0, 2), + slots.Where(s => s.retained_wal_bytes > 0).Sum(s => s.retained_wal_bytes ?? 0) / 1024.0 / 1024.0 / 1024.0, 2), slots, }, McpHelpers.JsonOptions); } From 0c86d7ad88c0b1b43d1e25fc74a5a48641493014 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Thu, 17 Sep 2026 23:54:29 -0400 Subject: [PATCH 04/69] Fix the honesty trio: fake granted-memory zero, dropped physical-reads column, caveat-free RCSI advice (#3550) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * Fix three honesty defects: null the fake granted-memory field, read the dropped physical-reads column, mirror the RCSI caveats #3529: get_memory_trend shipped a hardcoded total_granted_mb = 0.0 on both SKUs — an agent investigating memory-grant pressure read "granted was 0 all window" and ruled out the real culprit. The field is now an explicit null, the envelope carries a granted_note naming get_memory_grants as the grants series' source, and both tool descriptions stop promising granted memory. Pinned by payload tests (Lite real-DuckDB, Darling gated-live) and per-SKU description pins. #3530: Lite's QueryStore time-slice reader mapped TotalReads AND TotalLogicalReads to ordinal 4 (the logical aggregate — a deliberate alias, matching the QueryStats slicer) but never read ordinal 6, so the SELECT's total_physical_reads was computed and dropped. TotalPhysicalReads now maps ordinal 6, pinned by a fixture whose I/O columns carry pairwise-distinct values — equal values are exactly how the slip stayed invisible. #3531: the LCK_M_IS advice handed out the RCSI ALTER calling it "strictly better" with none of the caveats its LCK_M_S twin carried. Both twins now name the brief exclusive lock the ALTER takes, the tempdb version-store cost (phrased after FactRiskDisclosure's RCSI risk items), and the test-on-a-copy warning, pinned by a shared content test. Fixes #3529, fixes #3530, fixes #3531 Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx * Update the two doc lines that still described total_granted_mb as an always-0 parity placeholder Boundary extension approved by the wave lead: doc lines going stale because of the #3529 change belong in the same PR. Darling/README.md's tool-catalog line and DarlingTrendReader's MemoryTrendPoint doc comment now describe the shipped behavior — explicit null plus a granted_note naming get_memory_grants. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx * QueryStore slicer: physical-reads sort plots the physical series, bars and overlay both #3547, folded in per the wave lead. The grid's sorting handler mapped "AvgPhysicalReads" to the LOGICAL series under a physical-reads label — a forced workaround while the reader dropped the physical column, which the #3530 fix ended. The mapping now targets TotalPhysicalReads with the bucket.Value case to match, and the selected-row overlay follows: the timeline read carries avg_physical_io_reads through the same deduped CTE (no new dedup site), the point record gains PhysicalReads, and the overlay selector plots it — otherwise honest bars would have drawn under an elapsed-ms overlay, a new mismatch replacing the old label lie. The distinct-per-column test now covers the timeline's read path too, so a logical/physical swap on either the bars' or the overlay's source goes red. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx --------- Co-authored-by: Claude Fable 5 --- .../DarlingMcpTrendToolsTests.cs | 12 ++++++ .../Darling.Tests/DarlingTrendEmptyTests.cs | 12 +++++- .../Mcp/DarlingMcpTrendTools.cs | 12 +++--- .../Mcp/DarlingTrendReader.cs | 5 ++- Darling/README.md | 2 +- Lite.Tests/QueryStoreDedupReadTests.cs | 41 +++++++++++++++++-- Lite.Tests/TrendEmptyParityToolTests.cs | 34 +++++++++++++++ Lite/Controls/ServerTab.Grids.cs | 6 ++- Lite/Mcp/McpMemoryTools.cs | 9 +++- Lite/Services/LocalDataService.QueryStore.cs | 14 +++++-- PerformanceMonitor.Analysis/FactAdvice.cs | 4 +- .../FactAdviceCorrectnessTests.cs | 15 +++++++ 12 files changed, 146 insertions(+), 20 deletions(-) diff --git a/Darling/Darling.Tests/DarlingMcpTrendToolsTests.cs b/Darling/Darling.Tests/DarlingMcpTrendToolsTests.cs index 1a4a6c46a..6b7cdc830 100644 --- a/Darling/Darling.Tests/DarlingMcpTrendToolsTests.cs +++ b/Darling/Darling.Tests/DarlingMcpTrendToolsTests.cs @@ -101,6 +101,18 @@ public void ParamContract_ServerNameOptional_RequiredKeysAreNot() Assert.False(McpParams("get_query_trend").Single(x => x.Name == "database_name").Optional); } + /// #3529: the description promised granted memory while the payload shipped a literal 0. + /// It now points at get_memory_grants, the tool that actually serves the grants series. + [Fact] + public void MemoryTrend_Description_PointsAtTheGrantsTool_AndDoesNotPromiseGrantedMemory() + { + var method = ToolMethods().Single(m => m.GetCustomAttribute()!.Name == "get_memory_trend"); + var description = method.GetCustomAttribute()!.Description; + + Assert.DoesNotContain("and granted memory", description, StringComparison.Ordinal); + Assert.Contains("get_memory_grants", description, StringComparison.Ordinal); + } + [Fact] public void MemoryTrendSql_WindowedBothSides_CastsNumericToDouble() { diff --git a/Darling/Darling.Tests/DarlingTrendEmptyTests.cs b/Darling/Darling.Tests/DarlingTrendEmptyTests.cs index e5d913b94..9ac77e6be 100644 --- a/Darling/Darling.Tests/DarlingTrendEmptyTests.cs +++ b/Darling/Darling.Tests/DarlingTrendEmptyTests.cs @@ -83,9 +83,10 @@ public async Task AllThreeTrends_SeparateAQuietWindowFromANeverCollectedServer_A await SeedFileIoAsync(connection, ct, recent); await SeedQueryAsync(connection, ct, recent); + var memoryPayload = await DarlingMcpTrendTools.GetMemoryTrend(postgres, ServerName, 4); foreach (var payload in new[] { - await DarlingMcpTrendTools.GetMemoryTrend(postgres, ServerName, 4), + memoryPayload, await DarlingMcpTrendTools.GetFileIoTrend(postgres, ServerName, 4), await DarlingMcpTrendTools.GetQueryDurationTrend(postgres, ServerName, 4), }) @@ -95,6 +96,15 @@ await DarlingMcpTrendTools.GetQueryDurationTrend(postgres, ServerName, 4), Assert.True(root.GetProperty("trend").GetArrayLength() > 0); } + /* #3529: total_granted_mb is an explicit null with the envelope naming the real source — + never the literal 0.0 an agent read as "granted was 0 all window". */ + var memoryRoot = JsonDocument.Parse(memoryPayload).RootElement; + Assert.Contains("get_memory_grants", memoryRoot.GetProperty("granted_note").GetString(), StringComparison.Ordinal); + foreach (var point in memoryRoot.GetProperty("trend").EnumerateArray()) + { + Assert.Equal(JsonValueKind.Null, point.GetProperty("total_granted_mb").ValueKind); + } + bodySucceeded = true; } finally diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTrendTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTrendTools.cs index 86c34f9e0..2a1bf380e 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTrendTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTrendTools.cs @@ -46,7 +46,7 @@ namespace PerformanceMonitor.Darling.Service.Mcp; [McpServerToolType] public sealed class DarlingMcpTrendTools { - [McpServerTool(Name = "get_memory_trend"), Description("Gets memory usage trend over time: total server memory, target memory, buffer pool, plan cache, and granted memory. Useful for identifying memory growth patterns or pressure periods.")] + [McpServerTool(Name = "get_memory_trend"), Description("Gets memory usage trend over time: total server memory, target memory, buffer pool, and plan cache. total_granted_mb in this payload is always null — granted memory is a separate series; use get_memory_grants for it. Useful for identifying memory growth patterns or pressure periods.")] public static async Task GetMemoryTrend( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -94,16 +94,18 @@ neither. A server that collected fine and was simply quiet in THIS window wants target_server_memory_mb = p.TargetServerMemoryMb, buffer_pool_mb = p.BufferPoolMb, plan_cache_mb = p.PlanCacheMb, - /* Lite carries total_granted_mb on its MemoryTrendPoint but GetMemoryTrendAsync (a - memory_stats-only read) leaves it unset — the grant overlay is a separate chart series. - Reproduced here as the same 0 placeholder for field-for-field parity with Lite's tool. */ - total_granted_mb = 0.0 + /* The memory_stats source carries no grant data — the grant overlay is a separate series + (get_memory_grants). null, not the 0 placeholder this used to ship: a literal zero read + as "granted was 0 all window" and steered callers away from memory grants at exactly the + wrong moment (#3529). Field-for-field parity with Lite's tool, which nulls it the same way. */ + total_granted_mb = (double?)null }); return JsonSerializer.Serialize(new { server = resolved.ServerName, hours_back, + granted_note = "total_granted_mb is not sourced by this tool — the memory_stats series carries no grant data. Use get_memory_grants for the granted-memory series.", trend = result }, McpHelpers.JsonOptions); } diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingTrendReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingTrendReader.cs index 67f163a01..19a117191 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingTrendReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingTrendReader.cs @@ -41,8 +41,9 @@ internal static class DarlingTrendReader /* ─────────────────────────── result records ─────────────────────────── */ /// One memory-trend point: the four MB metrics per collection (Lite's MemoryTrendPoint - /// minus its always-0 TotalGrantedMb overlay field, which the tool carries as a Lite parity - /// placeholder — see ). + /// minus its TotalGrantedMb overlay field, which the memory_stats source never fills — the tool + /// publishes it as an explicit null with a note naming get_memory_grants (#3529) — see + /// ). public sealed record MemoryTrendPoint( DateTime CollectionTime, double TotalServerMemoryMb, double TargetServerMemoryMb, double BufferPoolMb, double PlanCacheMb); diff --git a/Darling/README.md b/Darling/README.md index fd95186b0..2824acf72 100644 --- a/Darling/README.md +++ b/Darling/README.md @@ -539,7 +539,7 @@ The embedded MCP server, over Streamable HTTP bound to `localhost` by default (s - **Trend data-read tools** — windowed time-series siblings of the core reads, each a stored read of the collected series over the window (BOTH-sides, naive-UTC): - `get_memory_trend` (total / target server memory, buffer pool, plan cache over time), `get_perfmon_trend` (a single counter's value + delta, `counter_name` required), `get_file_io_trend` (per-database read/write latency, top-10 busiest files), `get_query_trend` (one query's per-collection history by `query_hash` + `database_name`), `get_query_duration_trend` (overall elapsed-ms/sec + executions/sec). - Each mirrors the viewer's proven chart read (byte-identical Postgres SQL); the shape follows Lite where the SKUs diverge. `get_perfmon_trend` reproduces Lite's miss vocabulary (Page Life Expectancy is intentionally not collected; an unknown counter hands back the collected names). `get_memory_trend` carries a `total_granted_mb` field for field-for-field parity with Lite, where its memory_stats-only read leaves it 0 (the grant overlay is a separate chart series). + Each mirrors the viewer's proven chart read (byte-identical Postgres SQL); the shape follows Lite where the SKUs diverge. `get_perfmon_trend` reproduces Lite's miss vocabulary (Page Life Expectancy is intentionally not collected; an unknown counter hands back the collected names). `get_memory_trend` carries a `total_granted_mb` field for field-for-field parity with Lite, published as an explicit null with a `granted_note` naming `get_memory_grants` — the memory_stats source has no grant data, and the grants series is its own tool (#3529). - **System-health parse-on-read tools** — the Dashboard's `get_health_parser_*` family, over Darling's raw `system_health_events`: - `get_health_parser_system_health` (corruption + contention counters), `get_health_parser_severe_errors` (severity ≥ 19, with `database_id` resolved to a name), `get_health_parser_scheduler_issues`, `get_health_parser_memory_conditions`, `get_health_parser_memory_broker`, `get_health_parser_memory_node_oom`, `get_health_parser_cpu_tasks`, `get_health_parser_io_issues`. diff --git a/Lite.Tests/QueryStoreDedupReadTests.cs b/Lite.Tests/QueryStoreDedupReadTests.cs index 3f79276fc..e4a24c3bf 100644 --- a/Lite.Tests/QueryStoreDedupReadTests.cs +++ b/Lite.Tests/QueryStoreDedupReadTests.cs @@ -94,7 +94,9 @@ private async Task SeedAsync( long avgReads, string queryHash, long? intervalId = null, - DateTime? intervalStart = null) + DateTime? intervalStart = null, + long avgWrites = 0, + long avgPhysicalReads = 0) { using var readLock = _duckDb.AcquireReadLock(); var connection = await SeedConnectionAsync(); @@ -124,8 +126,8 @@ INSERT INTO query_store_stats cmd.Parameters.Add(new DuckDBParameter { Value = avgCpuUs }); cmd.Parameters.Add(new DuckDBParameter { Value = avgDurationUs }); cmd.Parameters.Add(new DuckDBParameter { Value = avgReads }); - cmd.Parameters.Add(new DuckDBParameter { Value = 0L }); - cmd.Parameters.Add(new DuckDBParameter { Value = 0L }); + cmd.Parameters.Add(new DuckDBParameter { Value = avgWrites }); + cmd.Parameters.Add(new DuckDBParameter { Value = avgPhysicalReads }); cmd.Parameters.Add(new DuckDBParameter { Value = $"0xPLAN{planId}" }); cmd.Parameters.Add(new DuckDBParameter { Value = false }); cmd.Parameters.Add(new DuckDBParameter { Value = 0L }); @@ -195,6 +197,39 @@ public async Task SlicerBucket_CountsARecollectedIntervalOnce_AtItsLatestValues( Assert.Equal(TrueBucketReads, bucket.TotalReads, precision: 6); } + /// + /// #3530: the slicer's reader mapped BOTH read fields to ordinal 4 and never read ordinal 6, so the + /// SELECT's total_physical_reads was computed and dropped on the floor. The seed's three I/O columns + /// carry values no other column can reproduce — equal fixture values are exactly how the slip stayed + /// invisible, so distinct-per-column is the point of this test, not a nicety. + /// + [Fact] + public async Task SlicerBucket_MapsWritesAndPhysicalReads_ToTheirOwnColumns() + { + await SeedAsync(BucketStart.AddMinutes(5), queryId: 7, planId: 77, FirstExecA, + executionCount: 10, avgCpuUs: 1_000, avgDurationUs: 2_000, avgReads: 11, queryHash: "0xIOMAP", + intervalId: 9301, intervalStart: BucketStart, avgWrites: 3, avgPhysicalReads: 5); + + var service = new LocalDataService(_duckDb); + var bucket = Assert.Single(await service.GetQueryStoreSlicerDataAsync(ServerId, hoursBack: 24)); + + /* 10 executions x the averages: logical 110, writes 30, physical 50 — all pairwise distinct. + TotalReads and TotalLogicalReads are deliberate aliases of the LOGICAL aggregate (ordinal 4), + the same shape the query-stats slicer maps; physical rides its own column at ordinal 6. */ + Assert.Equal(110.0, bucket.TotalReads, precision: 6); + Assert.Equal(110.0, bucket.TotalLogicalReads, precision: 6); + Assert.Equal(30.0, bucket.TotalWrites, precision: 6); + Assert.Equal(50.0, bucket.TotalPhysicalReads, precision: 6); + + /* #3547's data half: the slicer OVERLAY reads the same rows through the timeline, whose SELECT + carried no physical column at all — so a physical-sorted chart had nothing honest to draw. + Same distinct values, so a logical/physical swap on either side goes red here. (The overlay's + metric->field switch itself is WPF code-behind, untestable here like its sibling arms.) */ + var point = Assert.Single(await service.GetQueryStoreItemTimelineAsync(ServerId, Db, queryId: 7, planId: 77, hoursBack: 24)); + Assert.Equal(110.0, point.Reads, precision: 6); + Assert.Equal(50.0, point.PhysicalReads, precision: 6); + } + [Fact] public async Task TopQueries_ReportTheLatestCumulativeExecutionCount_NotTheSumOfSnapshots() { diff --git a/Lite.Tests/TrendEmptyParityToolTests.cs b/Lite.Tests/TrendEmptyParityToolTests.cs index 40814645b..84d0d08ac 100644 --- a/Lite.Tests/TrendEmptyParityToolTests.cs +++ b/Lite.Tests/TrendEmptyParityToolTests.cs @@ -7,7 +7,9 @@ */ using System; +using System.ComponentModel; using System.IO; +using System.Reflection; using System.Text.Json; using System.Threading.Tasks; using DuckDB.NET.Data; @@ -111,6 +113,38 @@ public async Task QueryDurationTrend_NeverCollected_AndAQuietWindow_AreDifferent AssertPayload(await McpQueryTools.GetQueryDurationTrend(service, _serverManager, ServerName, 4)); } + /// + /// #3529: the payload used to carry a hardcoded total_granted_mb of 0.0 — an agent investigating + /// RESOURCE_SEMAPHORE read "granted was 0 all window" and ruled out memory grants, the exact wrong + /// turn. The field is now an explicit null and the envelope names the real source. + /// + [Fact] + public async Task MemoryTrend_GrantedMemoryIsNullWithANoteNamingTheGrantsTool_NeverALiteralZero() + { + await SeedMemoryAsync(DateTime.UtcNow.AddMinutes(-10)); + + var payload = await McpMemoryTools.GetMemoryTrend(new LocalDataService(_duckDb), _serverManager, ServerName, 4); + var root = JsonDocument.Parse(payload).RootElement; + + Assert.True(root.GetProperty("trend").GetArrayLength() > 0); + Assert.Contains("get_memory_grants", root.GetProperty("granted_note").GetString(), StringComparison.Ordinal); + foreach (var point in root.GetProperty("trend").EnumerateArray()) + { + Assert.Equal(JsonValueKind.Null, point.GetProperty("total_granted_mb").ValueKind); + } + } + + /// #3529's description half: the tool promised granted memory it never delivered. + [Fact] + public void MemoryTrend_Description_PointsAtTheGrantsTool_AndDoesNotPromiseGrantedMemory() + { + var description = typeof(McpMemoryTools).GetMethod(nameof(McpMemoryTools.GetMemoryTrend))! + .GetCustomAttribute()!.Description; + + Assert.DoesNotContain("and granted memory", description, StringComparison.Ordinal); + Assert.Contains("get_memory_grants", description, StringComparison.Ordinal); + } + /// Nothing has ever been stored for this server: NOT an empty window, and widening it would /// never help. private static void AssertNeverCollected(string payload) diff --git a/Lite/Controls/ServerTab.Grids.cs b/Lite/Controls/ServerTab.Grids.cs index 92e158b4f..affc567af 100644 --- a/Lite/Controls/ServerTab.Grids.cs +++ b/Lite/Controls/ServerTab.Grids.cs @@ -205,6 +205,7 @@ to the RIGHT of the bar describing the same work. The overlay read places AND wi { "TotalCpu" or "AvgCpu" => p => p.CpuMs, "TotalReads" or "AvgReads" => p => p.Reads, + "TotalPhysReads" => p => p.PhysicalReads, _ => p => p.ElapsedMs, }; @@ -368,7 +369,9 @@ private void QueryStoreGrid_Sorting(object sender, DataGridSortingEventArgs e) "AvgDurationMs" => ("AvgElapsed", "Avg Duration (ms)"), "AvgLogicalReads" => ("TotalReads", "Avg Reads"), "AvgLogicalWrites" => ("TotalWrites", "Avg Writes"), - "AvgPhysicalReads" => ("TotalReads", "Avg Physical Reads"), + /* #3547: this arm mapped physical to the LOGICAL series while the reader dropped the physical + column; #3530 populates TotalPhysicalReads, so the label and the series finally agree. */ + "AvgPhysicalReads" => ("TotalPhysReads", "Total Physical Reads"), "TotalExecutions" => ("Sessions", "Executions"), _ => ("TotalCpu", "Total CPU (ms)"), }; @@ -387,6 +390,7 @@ private void QueryStoreGrid_Sorting(object sender, DataGridSortingEventArgs e) "AvgElapsed" => bucket.TotalElapsed / n, "TotalReads" => bucket.TotalReads, "TotalWrites" => bucket.TotalWrites, + "TotalPhysReads" => bucket.TotalPhysicalReads, "Sessions" => bucket.SessionCount, _ => bucket.TotalCpu, }; diff --git a/Lite/Mcp/McpMemoryTools.cs b/Lite/Mcp/McpMemoryTools.cs index 8f97b980f..d5851d16f 100644 --- a/Lite/Mcp/McpMemoryTools.cs +++ b/Lite/Mcp/McpMemoryTools.cs @@ -48,7 +48,7 @@ public static async Task GetMemoryStats( } } - [McpServerTool(Name = "get_memory_trend"), Description("Gets memory usage trend over time: total server memory, target memory, buffer pool, plan cache, and granted memory. Useful for identifying memory growth patterns or pressure periods.")] + [McpServerTool(Name = "get_memory_trend"), Description("Gets memory usage trend over time: total server memory, target memory, buffer pool, and plan cache. total_granted_mb in this payload is always null — granted memory is a separate series; use get_memory_grants for it. Useful for identifying memory growth patterns or pressure periods.")] public static async Task GetMemoryTrend( LocalDataService dataService, ServerManager serverManager, @@ -98,13 +98,18 @@ A bare empty array here told an MCP client nothing at all -- and Darling's twin target_server_memory_mb = p.TargetServerMemoryMb, buffer_pool_mb = p.BufferPoolMb, plan_cache_mb = p.PlanCacheMb, - total_granted_mb = p.TotalGrantedMb + /* GetMemoryTrendAsync is a memory_stats-only read and never assigns TotalGrantedMb — the + grant overlay is a separate series (get_memory_grants). null, not the model's default 0: + a literal zero read as "granted was 0 all window" and steered callers away from memory + grants at exactly the wrong moment (#3529). */ + total_granted_mb = (double?)null }); return JsonSerializer.Serialize(new { server = resolved.ServerName, hours_back, + granted_note = "total_granted_mb is not sourced by this tool — the memory_stats series carries no grant data. Use get_memory_grants for the granted-memory series.", trend = result }, McpHelpers.JsonOptions); } diff --git a/Lite/Services/LocalDataService.QueryStore.cs b/Lite/Services/LocalDataService.QueryStore.cs index 01a7c1af3..365cbff97 100644 --- a/Lite/Services/LocalDataService.QueryStore.cs +++ b/Lite/Services/LocalDataService.QueryStore.cs @@ -146,9 +146,14 @@ GROUP BY date_trunc('hour', COALESCE(interval_start_time_utc, collection_time)) SessionCount = reader.IsDBNull(1) ? 0 : Convert.ToInt64(reader.GetValue(1)), TotalCpu = reader.IsDBNull(2) ? 0 : ToDouble(reader.GetValue(2)), TotalElapsed = reader.IsDBNull(3) ? 0 : ToDouble(reader.GetValue(3)), + /* Ordinal 4 (total_reads) is the LOGICAL-reads aggregate — TotalReads and TotalLogicalReads + are deliberate aliases of it, same as the query-stats slicer. Physical reads ride + separately at ordinal 6; this reader shipped without that mapping, so the physical + column was computed and then dropped on the floor (#3530). */ TotalReads = reader.IsDBNull(4) ? 0 : ToDouble(reader.GetValue(4)), TotalWrites = reader.IsDBNull(5) ? 0 : ToDouble(reader.GetValue(5)), TotalLogicalReads = reader.IsDBNull(4) ? 0 : ToDouble(reader.GetValue(4)), + TotalPhysicalReads = reader.IsDBNull(6) ? 0 : ToDouble(reader.GetValue(6)), Value = reader.IsDBNull(2) ? 0 : ToDouble(reader.GetValue(2)), }); } @@ -569,7 +574,7 @@ ON c.database_name IS NOT DISTINCT FROM b.database_name /// One point on the Query Store slicer overlay: the interval's per-interval totals, placed at the hour /// the work RAN. /// - public sealed record QueryStoreItemTimelinePoint(DateTime PointTime, double CpuMs, double ElapsedMs, double Reads); + public sealed record QueryStoreItemTimelinePoint(DateTime PointTime, double CpuMs, double ElapsedMs, double Reads, double PhysicalReads); /// /// The selected Query Store row's execution timeline, for the grid→slicer overlay (#683). @@ -618,6 +623,7 @@ WITH deduped AS avg_cpu_time_us, avg_duration_us, avg_logical_io_reads, + avg_physical_io_reads, ROW_NUMBER() OVER ( PARTITION BY database_name, query_id, plan_id, runtime_stats_interval_id, first_execution_time, execution_type_desc, replica_role @@ -646,7 +652,8 @@ AND COALESCE(interval_start_time_utc, collection_time) <= $6 point_time, COALESCE(CAST(avg_cpu_time_us AS DOUBLE PRECISION) * execution_count, 0) / 1000.0 AS cpu_ms, COALESCE(CAST(avg_duration_us AS DOUBLE PRECISION) * execution_count, 0) / 1000.0 AS elapsed_ms, - COALESCE(CAST(avg_logical_io_reads AS DOUBLE PRECISION) * execution_count, 0) AS reads + COALESCE(CAST(avg_logical_io_reads AS DOUBLE PRECISION) * execution_count, 0) AS reads, + COALESCE(CAST(avg_physical_io_reads AS DOUBLE PRECISION) * execution_count, 0) AS physical_reads FROM deduped WHERE rn = 1 -- Ordered on the axis the points are PLOTTED on: a series whose x-values are not monotonic draws as a @@ -668,7 +675,8 @@ FROM deduped reader.GetDateTime(0), reader.IsDBNull(1) ? 0 : ToDouble(reader.GetValue(1)), reader.IsDBNull(2) ? 0 : ToDouble(reader.GetValue(2)), - reader.IsDBNull(3) ? 0 : ToDouble(reader.GetValue(3)))); + reader.IsDBNull(3) ? 0 : ToDouble(reader.GetValue(3)), + reader.IsDBNull(4) ? 0 : ToDouble(reader.GetValue(4)))); } return points; } diff --git a/PerformanceMonitor.Analysis/FactAdvice.cs b/PerformanceMonitor.Analysis/FactAdvice.cs index f9db172b4..a6f96b9c2 100644 --- a/PerformanceMonitor.Analysis/FactAdvice.cs +++ b/PerformanceMonitor.Analysis/FactAdvice.cs @@ -2121,7 +2121,7 @@ private static Dictionary BuildAdviceTable() Investigation: "SELECT queries are queueing behind UPDATE/INSERT/DELETE transactions on the same rows or pages. The per-mode breakdown shows LCK_M_S ranked against the other lock modes for the window; when DB_CONFIG co-fired, the databases where RCSI is off are the candidates for the fix. Open Configuration → Database Configuration in-app to see the full per-database RCSI/auto-shrink/auto-close/page-verify state. Reader/writer deadlocks are the same pattern escalated — if DEADLOCKS co-fired, the graph will show shared-lock victims.", Remediation: - "Enable READ_COMMITTED_SNAPSHOT on the affected database: `ALTER DATABASE SET READ_COMMITTED_SNAPSHOT ON;`. Readers stop taking shared locks and instead read the previous-committed row version, which eliminates the entire wait class for everything running at READ COMMITTED or below. The ALTER needs a brief exclusive lock on the database, so do it at a quiet moment. If the application uses NOLOCK / READUNCOMMITTED hints to work around this same problem today, RCSI is strictly better — it returns committed data instead of dirty reads — but test on a copy first if any code relies on dirty-read behaviour."); + "Enable READ_COMMITTED_SNAPSHOT on the affected database: `ALTER DATABASE SET READ_COMMITTED_SNAPSHOT ON;`. Readers stop taking shared locks and instead read the previous-committed row version, which eliminates the entire wait class for everything running at READ COMMITTED or below. The ALTER needs a brief exclusive lock on the database, so do it at a quiet moment, and the versioning is not free: row versions live in tempdb, where a long-running reader can grow the version store. If the application uses NOLOCK / READUNCOMMITTED hints to work around this same problem today, RCSI is strictly better — it returns committed data instead of dirty reads — but test on a copy first if any code relies on dirty-read behaviour."); t["LCK_M_IS"] = new AdviceBlock( Headline: @@ -2129,7 +2129,7 @@ private static Dictionary BuildAdviceTable() Investigation: "Intent-shared locks are the page/table-level locks SELECT takes to declare 'I will be reading rows on this page'. Waiting on IS means the page or table is held in an incompatible mode by a writer — same reader-blocked-by-writer pattern as LCK_M_S, one level up the lock hierarchy. The per-mode breakdown shows LCK_M_IS ranked against the other lock modes for the window. When DB_CONFIG co-fired, the databases with RCSI off are the candidates. Open Configuration → Database Configuration to see the full per-database state.", Remediation: - "Enable READ_COMMITTED_SNAPSHOT on the affected database: `ALTER DATABASE SET READ_COMMITTED_SNAPSHOT ON;`. This eliminates reader/writer blocking for everything at or below READ COMMITTED — the most common isolation level by far. If the application currently uses READ UNCOMMITTED with NOLOCK hints to avoid these waits, RCSI is the strictly better mechanism: it returns committed data rather than dirty reads, and you can remove the hints once it's enabled."); + "Enable READ_COMMITTED_SNAPSHOT on the affected database: `ALTER DATABASE SET READ_COMMITTED_SNAPSHOT ON;`. This eliminates reader/writer blocking for everything at or below READ COMMITTED — the most common isolation level by far. The ALTER needs a brief exclusive lock on the database, so do it at a quiet moment, and the versioning is not free: row versions live in tempdb, where a long-running reader can grow the version store. If the application currently uses READ UNCOMMITTED with NOLOCK hints to avoid these waits, RCSI is the strictly better mechanism — it returns committed data rather than dirty reads, and you can remove the hints once it's enabled — but test on a copy first if any code relies on dirty-read behaviour."); t["SCH_M"] = new AdviceBlock( Headline: diff --git a/deprecated/Dashboard.Tests/FactAdviceCorrectnessTests.cs b/deprecated/Dashboard.Tests/FactAdviceCorrectnessTests.cs index 826fb315e..3a6f065cf 100644 --- a/deprecated/Dashboard.Tests/FactAdviceCorrectnessTests.cs +++ b/deprecated/Dashboard.Tests/FactAdviceCorrectnessTests.cs @@ -78,6 +78,21 @@ public void LatencyAnomalies_HeadlineNotRightNow(string key) Assert.DoesNotContain("right now", FactAdvice.GetForFactKey(key)!.Headline); } + // #3531: the two reader/writer lock twins hand out the same RCSI ALTER, so both must name its + // counter-objectives — the brief exclusive lock the ALTER takes, the test-on-a-copy warning for + // NOLOCK-dependent code, and the tempdb version-store cost FactRiskDisclosure discloses at apply + // time. LCK_M_IS shipped the bare ALTER calling RCSI "strictly better" with none of them. + [Theory] + [InlineData("LCK_M_S")] + [InlineData("LCK_M_IS")] + public void RcsiLockTwins_NameTheCaveatsAlongsideTheAlter(string key) + { + var remediation = FactAdvice.GetForFactKey(key)!.Remediation; + Assert.Contains("brief exclusive lock", remediation); + Assert.Contains("test on a copy", remediation); + Assert.Contains("version store", remediation); + } + // Review note (§2): SOS rewrite over-swung to "never CPU pressure". The amount + a deep runnable // queue IS demand-exceeds-capacity; the advice must point at the runnable-queue discriminator. [Fact] From deb547abcab04eef19fb5c45f6d91c6913032dca Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Thu, 17 Sep 2026 23:57:21 -0400 Subject: [PATCH 05/69] The xmin horizon alert gains a horizon-persistence arm and an observation floor (#3537) (#3555) Two edges escaped the identity-keyed persistence gate in opposite directions. False fire: the identity fraction's denominator counts only holder-bearing collections (the collector emits no rows when the horizon is unheld), so the first holder after quiet hours read as 1 of 1 - 100%, "chronic", off a single sample. False quiet: a horizon continuously pinned past the age threshold by a parade of DISTINCT holders never accumulates any single holder's fraction, so the alert never fired while its own claim ("vacuum is reclaiming nothing cluster-wide") was true the whole time. The identity arm stays as designed - it is the arm that names the thing to kill - and gains a 5-observation floor, the smallest denominator whose majority test cannot be satisfied by fewer than three sightings. The new horizon arm fires when the WINNING xmin_age sat at or above the threshold in a majority of the window's real captures, holder identity ignored. Its denominator comes from collection_log (this collector's own SUCCESS runs, so quiet zero-row captures count) rather than from the holder table - the cheapest honest "collections in the window", counting the exact collector being fractioned instead of inferring cadence from a cadence-mate table. The threshold reaches the SQL as a bind from the evaluator's own constant so read and evaluation cannot drift. A horizon-arm fire without a stable identity names the rotating-holder pattern, still carries the latest holder's remedy, and subjects the stable "rotating holders" sentinel so the host's per-subject cooldown holds across rotations instead of paging once per parade member. Tests: evaluator boundaries for both arms (1/1 and sub-floor totals no longer fire, rotating holders fire under the sentinel subject, the classic chronic holder still fires the identity arm with its original wording, below-threshold never fires), plus a live-Postgres read test that pins the new SQL's counting - ERROR runs, other collectors and other servers stay out of the capture denominator; a below-threshold winner stays out of the numerator while still counting as an observation. Fixes #3537 Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx Co-authored-by: Claude Fable 5 --- .../PostgresAlertEvaluatorTests.cs | 110 +++++++++++- .../XminHorizonPersistenceReadTests.cs | 159 ++++++++++++++++++ .../DarlingPostgresAlertReadAdapter.cs | 50 +++++- .../PostgresAlertEvaluator.cs | 103 ++++++++++-- .../PostgresAlertInfo.cs | 18 +- 5 files changed, 423 insertions(+), 17 deletions(-) create mode 100644 Darling/Darling.Tests/XminHorizonPersistenceReadTests.cs diff --git a/Darling/Darling.Tests/PostgresAlertEvaluatorTests.cs b/Darling/Darling.Tests/PostgresAlertEvaluatorTests.cs index 02ff9b5f1..e0ddafed1 100644 --- a/Darling/Darling.Tests/PostgresAlertEvaluatorTests.cs +++ b/Darling/Darling.Tests/PostgresAlertEvaluatorTests.cs @@ -224,7 +224,10 @@ public void AMissingFreezeMaxAgeSettingSilencesRatherThanFiringOnEverything(long /// /// The persistence gate is what keeps this from firing on every long-running report. Same age, same - /// holder — only the persistence differs, and only the chronic one alerts. + /// holder — only the persistence differs, and only the chronic one alerts. The chronic case is the + /// classic single-pid incident, and it still fires through the IDENTITY arm with the holder as the + /// subject and the original "in N of M observations" wording — truthful now that the floor guarantees + /// M is a real sample. /// [Fact] public void XminFiresOnlyForAChronicHolderNotALongQuery() @@ -232,15 +235,118 @@ public void XminFiresOnlyForAChronicHolderNotALongQuery() var chronic = new PostgresXminHorizonAlertInfo("session", "12345", 100_000_000, 30, 40, "idle in transaction"); var transient = new PostgresXminHorizonAlertInfo("session", "12345", 100_000_000, 2, 40, "running"); - Assert.NotNull(PostgresAlertEvaluator.EvaluateXmin(chronic)); + var finding = PostgresAlertEvaluator.EvaluateXmin(chronic); + Assert.NotNull(finding); + Assert.Equal("session:12345", finding!.Subject); + Assert.Contains("in 30 of 40 observations", finding.ShortMessage, StringComparison.Ordinal); + Assert.Null(PostgresAlertEvaluator.EvaluateXmin(transient)); } + /// + /// #3537 edge 1: the identity denominator counts only holder-bearing collections, so the first holder + /// after quiet hours arrived as 1 win in 1 observation — 100%, "chronic", off a single sample. The + /// observation floor closes every denominator too small for its majority to mean anything. + /// + [Theory] + [InlineData(1, 1)] + [InlineData(2, 2)] + [InlineData(4, 4)] + [InlineData(3, 4)] + public void XminDoesNotFireBeneathTheObservationFloorHoweverTotalTheFraction(int held, int total) + { + Assert.Null(PostgresAlertEvaluator.EvaluateXmin( + new PostgresXminHorizonAlertInfo("session", "1", 900_000_000, held, total, null))); + } + + /// The floor is a floor, not a fudge: at exactly the minimum, a majority still fires. + [Fact] + public void XminIdentityArmFiresAtExactlyTheObservationFloor() + { + Assert.NotNull(PostgresAlertEvaluator.EvaluateXmin( + new PostgresXminHorizonAlertInfo( + "session", "1", 900_000_000, + PostgresAlertEvaluator.XminMinimumObservations, + PostgresAlertEvaluator.XminMinimumObservations, + null))); + } + + /// + /// #3537 edge 2: a horizon continuously pinned past the threshold by a PARADE of distinct holders never + /// accumulates any single holder's identity fraction, and the old gate never fired while the alert's own + /// claim was true the whole time. The horizon arm fires on the horizon's persistence across the window's + /// real captures, names the rotating pattern, still carries the latest holder's remedy — and subjects + /// the stable sentinel, not the latest member, so the host's per-subject cooldown holds across + /// rotations. + /// + [Fact] + public void XminRotatingHoldersFireTheHorizonArmUnderTheStableSubject() + { + var finding = PostgresAlertEvaluator.EvaluateXmin(new PostgresXminHorizonAlertInfo( + "session", "9101", 80_000_000, ObservationsHeld: 1, ObservationsTotal: 60, + "state=idle in transaction", ObservationsAboveThreshold: 70, CapturesInWindow: 120)); + + Assert.NotNull(finding); + Assert.Equal(AlertSeverityLevel.Warning, finding!.Severity); + Assert.Equal(PostgresAlertEvaluator.XminRotatingHoldersSubject, finding.Subject); + Assert.Contains("succession of different holders", finding.ShortMessage, StringComparison.Ordinal); + Assert.Contains("session:9101", finding.ShortMessage, StringComparison.Ordinal); + Assert.Contains("70 of the window's 120 collections", finding.ShortMessage, StringComparison.Ordinal); + /* The remedy names the latest holder's cause — the one actionable thing either way. */ + Assert.Contains("idle in transaction", finding.ShortMessage, StringComparison.OrdinalIgnoreCase); + } + + /// + /// The horizon arm's own boundaries: a majority of the window's real captures fires, one capture short + /// of it does not, and a window with fewer captures than the floor cannot fire at any fraction — which + /// is also what keeps the compat default (no capture data supplied) silent. + /// + [Theory] + [InlineData(60, 120, true)] // exactly the majority + [InlineData(59, 120, false)] // one capture short + [InlineData(4, 4, false)] // 100%, but beneath the capture floor + [InlineData(0, 0, false)] // no capture data supplied — the compat default + public void XminHorizonArmNeedsAMajorityOfAtLeastTheFloorsWorthOfCaptures( + int above, int captures, bool fires) + { + var finding = PostgresAlertEvaluator.EvaluateXmin(new PostgresXminHorizonAlertInfo( + "session", "1", 80_000_000, ObservationsHeld: 1, ObservationsTotal: 60, null, + ObservationsAboveThreshold: above, CapturesInWindow: captures)); + + Assert.Equal(fires, finding is not null); + } + + /// + /// The overlap case, told apart by the identity FRACTION alone: a service restarted into an incident + /// already underway sees a stable holder through a window still too young for the identity floor. The + /// horizon arm supplies the persistence evidence, but the wording must not claim a "succession" and + /// the subject stays the holder — there is exactly one, and it is the thing to kill. + /// + [Fact] + public void XminStableHolderInAYoungWindowKeepsTheHolderSubjectOnAHorizonArmFire() + { + var finding = PostgresAlertEvaluator.EvaluateXmin(new PostgresXminHorizonAlertInfo( + "session", "77", 80_000_000, ObservationsHeld: 3, ObservationsTotal: 3, null, + ObservationsAboveThreshold: 3, CapturesInWindow: 6)); + + Assert.NotNull(finding); + Assert.Equal("session:77", finding!.Subject); + Assert.DoesNotContain("succession", finding.ShortMessage, StringComparison.Ordinal); + Assert.Contains("3 of the window's 6 collections", finding.ShortMessage, StringComparison.Ordinal); + } + [Fact] public void XminBelowTheAgeThresholdNeverFiresHoweverPersistent() { Assert.Null(PostgresAlertEvaluator.EvaluateXmin( new PostgresXminHorizonAlertInfo("session", "1", 1_000_000, 40, 40, null))); + + /* Both arms saturated, current age below the bar: the age gate answers first. The latest reading + is what the alert would quote as "current", so a horizon that has already come back under the + threshold must not page off its own history. */ + Assert.Null(PostgresAlertEvaluator.EvaluateXmin( + new PostgresXminHorizonAlertInfo("session", "1", 1_000_000, 40, 40, null, + ObservationsAboveThreshold: 120, CapturesInWindow: 120))); } /// A zero denominator must not divide — it means nothing was observed, so nothing fires. diff --git a/Darling/Darling.Tests/XminHorizonPersistenceReadTests.cs b/Darling/Darling.Tests/XminHorizonPersistenceReadTests.cs new file mode 100644 index 000000000..9faa683e4 --- /dev/null +++ b/Darling/Darling.Tests/XminHorizonPersistenceReadTests.cs @@ -0,0 +1,159 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Threading; +using System.Threading.Tasks; +using Npgsql; +using PerformanceMonitor.Alerting; +using PerformanceMonitor.Collectors; +using PerformanceMonitor.Darling.Service; +using PerformanceMonitor.Darling.Storage; +using Xunit; + +namespace Darling.Tests; + +/// +/// #3537: the xmin read's two window figures, against a real Postgres — the horizon-arm numerator (winning +/// age at/above the threshold, holder ignored) and its denominator (the collector's OWN successful runs, +/// off collection_log, so quiet captures count). +/// +/// Why live rather than a source pin. XminSql has no execution coverage anywhere else: +/// a pin can assert the captures CTE exists, but only a store can prove the filters count — that an +/// ERROR run, another collector's run and another server's run all stay out of the denominator while a +/// zero-row healthy run stays in, and that a below-threshold winner stays out of the numerator while still +/// counting as a holder-bearing observation. Each seeded row here is one of those predicates. +/// +[Collection("live-postgres")] +public sealed class XminHorizonPersistenceReadTests +{ + private const int ServerId = -353701; + private const int OtherServerId = -353702; + private const string ServerName = "xmin-persistence-read"; + private const string Collector = "pg_xmin_horizon"; + + private static string? ConnectionString => Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + + /// + /// One rotating-holder window, end to end: four held collections under three distinct winning pids + /// (one below the age bar), two quiet captures, and three log rows that must not count. The read's + /// figures are asserted exactly, then handed to the evaluator to prove the fire this issue exists for: + /// the horizon arm, under the stable rotating-holders subject, where the identity fraction (1 of 4) + /// never could. + /// + [Fact] + public async Task TheXminRead_CountsCapturesAndAboveThreshold_AgainstDevPostgres() + { + var cs = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(cs), + "Set DARLING_TEST_PG to a Postgres connection string to run the live xmin-persistence test."); + + var ct = TestContext.Current.CancellationToken; + using var connection = new NpgsqlConnection(cs); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + await DeleteRowsAsync(connection, ct); + + var bodySucceeded = false; + try + { + await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); + await using var postgres = NpgsqlDataSource.Create(cs!); + var adapter = new DarlingPostgresAlertReadAdapter(postgres); + + /* The empty store first: no holder rows means no row at all, the healthy null — and the one + execution of the full statement that cannot hide behind seeded data if the SQL stops + parsing. */ + Assert.Null(await adapter.GetXminHorizonAsync(ServerId, ct)); + + /* Four collections, three distinct winning pids — the parade. The 30M winner sits below the + 50M evaluator threshold, so it is a holder-bearing observation that must NOT count as an + above-threshold one. The t-10 loser row shares its winner's collection_time, pinning that + extra rows per collection inflate nothing. */ + await SeedHolderAsync(connection, ct, MinutesAgo(10), "session", 60_000_000, "101", "state=idle in transaction", isWinner: true); + await SeedHolderAsync(connection, ct, MinutesAgo(10), "replication_slot", 10_000_000, "slot_a", null, isWinner: false); + await SeedHolderAsync(connection, ct, MinutesAgo(8), "session", 70_000_000, "102", null, isWinner: true); + await SeedHolderAsync(connection, ct, MinutesAgo(6), "session", 30_000_000, "103", null, isWinner: true); + await SeedHolderAsync(connection, ct, MinutesAgo(4), "session", 80_000_000, "104", "state=idle in transaction", isWinner: true); + + /* The capture denominator: the four holder-bearing runs, two QUIET (zero-row, healthy) runs + the holder table cannot see — the whole reason the denominator lives in collection_log — + and three rows that must stay out of it: a run that failed, another collector's run, and + another server's. */ + await SeedLogAsync(connection, ct, MinutesAgo(10), ServerId, Collector, "SUCCESS", 2); + await SeedLogAsync(connection, ct, MinutesAgo(8), ServerId, Collector, "SUCCESS", 1); + await SeedLogAsync(connection, ct, MinutesAgo(6), ServerId, Collector, "SUCCESS", 1); + await SeedLogAsync(connection, ct, MinutesAgo(4), ServerId, Collector, "SUCCESS", 1); + await SeedLogAsync(connection, ct, MinutesAgo(2), ServerId, Collector, "SUCCESS", 0); + await SeedLogAsync(connection, ct, MinutesAgo(1), ServerId, Collector, "SUCCESS", 0); + await SeedLogAsync(connection, ct, MinutesAgo(3), ServerId, Collector, "ERROR", 0); + await SeedLogAsync(connection, ct, MinutesAgo(5), ServerId, "pg_database_stats", "SUCCESS", 4); + await SeedLogAsync(connection, ct, MinutesAgo(7), OtherServerId, Collector, "SUCCESS", 1); + + var info = await adapter.GetXminHorizonAsync(ServerId, ct); + + Assert.NotNull(info); + Assert.Equal("session", info!.Source); + Assert.Equal("104", info.Identifier); + Assert.Equal(80_000_000L, info.XminAge); + Assert.Equal(1, info.ObservationsHeld); + Assert.Equal(4, info.ObservationsTotal); + Assert.Equal(3, info.ObservationsAboveThreshold); + Assert.Equal(6, info.CapturesInWindow); + + /* And the consequence the figures exist for: 3 of 6 captures is the horizon arm's majority + where the identity fraction is 1 of 4 — the rotating-holder fire, under the subject the + host's per-subject cooldown can actually hold across rotations. */ + var finding = PostgresAlertEvaluator.EvaluateXmin(info); + Assert.NotNull(finding); + Assert.Equal(PostgresAlertEvaluator.XminRotatingHoldersSubject, finding!.Subject); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, DeleteRowsAsync); + } + } + + /* ── helpers ── */ + + private static DateTime MinutesAgo(int minutes) => + DarlingMcpTestData.TruncateToSeconds(DateTime.UtcNow.AddMinutes(-minutes)); + + private static Task SeedHolderAsync( + NpgsqlConnection connection, CancellationToken ct, + DateTime collectionTimeUtc, string source, long xminAge, string holder, string? detail, bool isWinner) => + DarlingMcpTestData.ExecAsync(connection, ct, @" +INSERT INTO pg_xmin_horizon + (collection_id, collection_time, server_id, server_name, source, xmin_age, holder, detail, is_winner) +VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9)", + CollectionIdGenerator.Next(), collectionTimeUtc, ServerId, ServerName, + source, xminAge, holder, detail, isWinner); + + private static Task SeedLogAsync( + NpgsqlConnection connection, CancellationToken ct, + DateTime collectionTimeUtc, int serverId, string collectorName, string status, int rowsCollected) => + DarlingMcpTestData.ExecAsync(connection, ct, @" +INSERT INTO collection_log + (log_id, server_id, server_name, collector_name, collection_time, + duration_ms, status, error_message, rows_collected, sql_duration_ms, duckdb_duration_ms) +VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11)", + CollectionIdGenerator.Next(), serverId, ServerName, collectorName, + collectionTimeUtc, 50, status, null, rowsCollected, 40, 10); + + private static async Task DeleteRowsAsync(NpgsqlConnection connection, CancellationToken ct) + { + await DarlingMcpTestData.ExecAsync(connection, ct, + "DELETE FROM pg_xmin_horizon WHERE server_id = $1", ServerId); + await DarlingMcpTestData.ExecAsync(connection, ct, + "DELETE FROM collection_log WHERE server_id IN ($1, $2)", ServerId, OtherServerId); + await DarlingMcpTestData.ExecAsync(connection, ct, + "DELETE FROM servers WHERE server_id = $1", ServerId); + } +} diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingPostgresAlertReadAdapter.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingPostgresAlertReadAdapter.cs index 23ed7d0b7..84b22ea28 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingPostgresAlertReadAdapter.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingPostgresAlertReadAdapter.cs @@ -74,6 +74,25 @@ FROM pg_wraparound_stats /// times something held it, how often was it this one" — which is the question. Note this became reachable /// only once the collector stopped attributing its own backend: while Darling's own snapshot was always a /// session holder, every collection had a holder and the distinction was invisible. + /// #3537: two window figures the identity fraction cannot supply. + /// observations_above_threshold counts the collections whose WINNING age sat at or above the + /// evaluator's warning threshold, holder identity ignored — the alert's own claim is about the horizon, + /// and a horizon continuously pinned by a parade of DISTINCT holders never accumulates any single + /// holder's fraction. The threshold arrives as a bind from + /// so the condition counted here and the + /// one evaluated there cannot drift apart. + /// captures_in_window is the horizon arm's denominator, and it comes from + /// collection_log — this collector's own SUCCESS rows — rather than from this table, whose + /// distinct collection times count only holder-bearing collections (see above) and so read the first + /// holder after quiet hours as 1 of 1, 100%, "chronic". The log gets a row per run INCLUDING zero-row + /// (healthy, unheld) runs, sits behind its (server_id, collection_time) index, and counts the exact + /// collector whose captures are being fractioned — cheaper and more honest than inferring the cadence + /// from a cadence-mate table like pg_database_stats, which measures a different collector's + /// fate. Runs that stored nothing (ERROR / ABANDONED / PERMISSIONS / YIELDED) are excluded: a cycle + /// that could not look is not evidence the horizon was clear. The log write is failure-isolated and + /// can silently skip a row, so the count may UNDERCOUNT — which only inflates the fraction of a + /// horizon already measured above threshold, and the evaluator's minimum-captures floor keeps a + /// near-empty log from firing at all. /// internal const string XminSql = """ WITH latest AS ( @@ -92,10 +111,22 @@ window_stats AS ( WHERE is_winner AND source = (SELECT source FROM latest) AND holder IS NOT DISTINCT FROM (SELECT holder FROM latest) - ) AS observations_held + ) AS observations_held, + COUNT(DISTINCT collection_time) FILTER ( + WHERE is_winner + AND xmin_age >= $3 + ) AS observations_above_threshold FROM pg_xmin_horizon WHERE server_id = $1 AND collection_time >= $2 + ), + captures AS ( + SELECT COUNT(*) AS captures_in_window + FROM collection_log + WHERE server_id = $1 + AND collector_name = 'pg_xmin_horizon' + AND collection_time >= $2 + AND status = 'SUCCESS' ) SELECT l.source, @@ -103,9 +134,12 @@ FROM pg_xmin_horizon l.xmin_age, w.observations_held, w.observations_total, - l.detail + l.detail, + w.observations_above_threshold, + c.captures_in_window FROM latest AS l CROSS JOIN window_stats AS w + CROSS JOIN captures AS c """; /// @@ -241,6 +275,11 @@ answer consistently off one query. 0 reads as "no window data" -> FreezingIsKeep command.CommandTimeout = DarlingAlertReadAdapter.AlertPassCommandTimeoutSeconds; command.Parameters.AddWithValue(serverId); command.Parameters.AddWithValue(NaiveUtcNow() - Freshness); + /* The evaluator's own age threshold, not a local copy: the SQL counts "collections above + threshold" and the evaluator fractions that count against the SAME bar, so read and evaluation + must agree on it or the horizon arm silently means something else — the PoisonWaitSql window + discipline, applied to a level. */ + command.Parameters.AddWithValue(PostgresAlertEvaluator.XminAgeWarningThreshold); await using var reader = await command.ExecuteReaderAsync(cancellationToken); if (!await reader.ReadAsync(cancellationToken)) { @@ -253,7 +292,12 @@ answer consistently off one query. 0 reads as "no window data" -> FreezingIsKeep reader.IsDBNull(2) ? 0 : reader.GetInt64(2), reader.IsDBNull(3) ? 0 : (int)reader.GetInt64(3), reader.IsDBNull(4) ? 0 : (int)reader.GetInt64(4), - reader.IsDBNull(5) ? null : reader.GetString(5)); + reader.IsDBNull(5) ? null : reader.GetString(5), + /* ordinals 6/7: the #3537 horizon-arm figures. 0 reads as "no window data" in both — the + evaluator's floor keeps that from firing, the same conservative default the wraparound + window peaks take. */ + reader.IsDBNull(6) ? 0 : (int)reader.GetInt64(6), + reader.IsDBNull(7) ? 0 : (int)reader.GetInt64(7)); } public async Task> GetReplicationSlotRiskAsync( diff --git a/PerformanceMonitor.Alerting/PostgresAlertEvaluator.cs b/PerformanceMonitor.Alerting/PostgresAlertEvaluator.cs index c8b1d3a9c..e11ab9816 100644 --- a/PerformanceMonitor.Alerting/PostgresAlertEvaluator.cs +++ b/PerformanceMonitor.Alerting/PostgresAlertEvaluator.cs @@ -81,8 +81,38 @@ relative Warning arm now sits AT the setting (1.0x, the crossing point itself) a holder seen in a majority of the window's observations is chronic, while one seen once is a query that ran long, and only the first is worth waking anyone for. */ public const long XminAgeWarningThreshold = 50_000_000; + + /// + /// The majority standard both persistence arms apply (#3537) — one fraction, two denominators. The + /// IDENTITY arm asks it of the collections that recorded any holder ("of the times something held it, + /// how often was it this one"); the HORIZON arm asks it of the window's real captures ("of the times we + /// looked, how often was the horizon pinned past the threshold"). Majority is the line in both readings + /// because it is where "keeps happening" stops being arguable: below it every fire needs a judgment + /// call about how much less than half still counts, and there is no mechanics-derived number under 0.5 + /// to anchor one. + /// public const double XminPersistenceFraction = 0.5; + /// + /// #3537: the floor under BOTH persistence denominators. Without it the identity arm read the first + /// holder after quiet hours as 1 win in 1 observation — 100%, "chronic", off a single sample — because + /// its denominator counts only holder-bearing collections and quiet hours contribute none. 5 because it + /// is the smallest denominator whose majority test cannot be satisfied by fewer than three sightings + /// (at 1–4 observations, one or two sightings clear 50% — the exact shape of the false fire), and at + /// the collector's 1-minute cadence three sightings means the condition spanned minutes, not a moment. + /// Higher would buy little: a holder that has aged the horizon 50 million transactions has existed for + /// minutes on any workload fast enough for the extra samples to cost real delay. + /// + public const int XminMinimumObservations = 5; + + /// + /// The stable subject a horizon-arm fire carries when no single holder owns the incident (#3537). A + /// constant, like the metric names above, because the host's per-subject cooldown and history dedup + /// key on it: subjecting each parade member in turn would present every rotation as a brand-new + /// incident and page once per member for one continuously-pinned horizon. + /// + public const string XminRotatingHoldersSubject = "rotating holders"; + /* Replication slots. No byte threshold for the terminal states — `lost` and `unreserved` are failures that have already happened, at any size. For a slot merely retaining WAL, 10 GB is the point where an unbounded pile stops being noise on any volume worth monitoring; growth is what escalates it, @@ -395,30 +425,83 @@ private static (AlertSeverityLevel Severity, long Breached, bool ViaCeiling)? Gr return null; } - /* The persistence gate. Without it this fires on any long-running report, which is how an alert - earns a mute rule instead of a response. */ - var persistent = xmin.ObservationsTotal > 0 + /* The persistence gate, two arms (#3537). Without any gate this fires on any long-running report, + which is how an alert earns a mute rule instead of a response. + + The IDENTITY arm is the original: THIS holder won a majority of the collections that recorded + any holder — the chronic-holder shape, and the arm that can name the thing to kill. Its + denominator counts holder-bearing collections only (deliberately: see the read adapter), which + is why it also needs the observation floor — the first holder after quiet hours is 1 win in 1 + observation, and 100% of one sample is not "chronic" however the fraction reads. + + The HORIZON arm covers what the identity fraction structurally cannot see: a horizon pinned + past the age threshold in a majority of the window's REAL captures while the holder identity + rotates. Each parade member is individually transient, so no identity fraction ever accumulates + — but the alert's own claim ("vacuum is reclaiming nothing cluster-wide") is about the horizon, + not the holder, and it is continuously true. Same majority standard, honest denominator for + each claim: the identity claim is about the collections that had a holder, the horizon claim is + about every time the collector looked. A capture count of 0 — an adapter that supplied none, or + a log write that failed — floors the arm out rather than firing, the conservative default. */ + var identityFractionHolds = xmin.ObservationsTotal > 0 && (double)xmin.ObservationsHeld / xmin.ObservationsTotal >= XminPersistenceFraction; - if (!persistent) + var identityArm = identityFractionHolds && xmin.ObservationsTotal >= XminMinimumObservations; + + var horizonArm = xmin.CapturesInWindow >= XminMinimumObservations + && (double)xmin.ObservationsAboveThreshold / xmin.CapturesInWindow >= XminPersistenceFraction; + + if (!identityArm && !horizonArm) { return null; } - var subject = string.IsNullOrWhiteSpace(xmin.Identifier) + var holder = string.IsNullOrWhiteSpace(xmin.Identifier) ? xmin.Source : $"{xmin.Source}:{xmin.Identifier}"; + if (identityArm) + { + return new Finding( + XminHorizonMetric, + AlertSeverityLevel.Warning, + holder, + $"{xmin.XminAge:N0} transactions held by {holder}", + $"{XminAgeWarningThreshold:N0} transactions, held in at least " + + $"{XminPersistenceFraction:P0} of observations", + $"Vacuum is reclaiming nothing cluster-wide: {holder} is holding the xmin horizon " + + $"{xmin.XminAge:N0} transactions back, in {xmin.ObservationsHeld} of " + + $"{xmin.ObservationsTotal} observations. {RemedyFor(xmin.Source)}" + + (string.IsNullOrWhiteSpace(xmin.Detail) ? string.Empty : $" ({xmin.Detail})"), + xmin.XminAge, + XminAgeWarningThreshold); + } + + /* Horizon-arm fire. Two shapes reach here, told apart by the identity FRACTION alone (the floor + is what a freshly-started window cannot yet satisfy): when the fraction holds, the latest + holder has won the collections that recorded one — a chronic holder observed through a window + still too young for the identity arm, so it keeps the subject and the naming. When it does not, + the holders are rotating, and the subject must NOT be the latest member: the incident is the + horizon, and a per-member subject would sidestep the host's per-subject cooldown to page once + per parade member. The remedy still names the latest holder's cause — it is the one thing + currently actionable either way. */ + var rotating = !identityFractionHolds; + var subject = rotating ? XminRotatingHoldersSubject : holder; + var holderClause = rotating + ? $"by a succession of different holders rather than one chronic one — the latest is {holder}" + : $"by {holder}, the winner in {xmin.ObservationsHeld} of the {xmin.ObservationsTotal} " + + "collections that recorded a holder"; + return new Finding( XminHorizonMetric, AlertSeverityLevel.Warning, subject, $"{xmin.XminAge:N0} transactions held by {subject}", - $"{XminAgeWarningThreshold:N0} transactions, held in at least " - + $"{XminPersistenceFraction:P0} of observations", - $"Vacuum is reclaiming nothing cluster-wide: {subject} is holding the xmin horizon " - + $"{xmin.XminAge:N0} transactions back, in {xmin.ObservationsHeld} of " - + $"{xmin.ObservationsTotal} observations. {RemedyFor(xmin.Source)}" + $"{XminAgeWarningThreshold:N0} transactions, behind in at least " + + $"{XminPersistenceFraction:P0} of the window's captures", + $"Vacuum is reclaiming nothing cluster-wide: the xmin horizon has been at least " + + $"{XminAgeWarningThreshold:N0} transactions behind in {xmin.ObservationsAboveThreshold} of " + + $"the window's {xmin.CapturesInWindow} collections, held {holderClause}; it currently " + + $"stands {xmin.XminAge:N0} back. {RemedyFor(xmin.Source)}" + (string.IsNullOrWhiteSpace(xmin.Detail) ? string.Empty : $" ({xmin.Detail})"), xmin.XminAge, XminAgeWarningThreshold); diff --git a/PerformanceMonitor.Alerting/PostgresAlertInfo.cs b/PerformanceMonitor.Alerting/PostgresAlertInfo.cs index 2bcae3a7d..f46f33fcc 100644 --- a/PerformanceMonitor.Alerting/PostgresAlertInfo.cs +++ b/PerformanceMonitor.Alerting/PostgresAlertInfo.cs @@ -88,15 +88,29 @@ means. A peak of 0 (no window supplied) makes this false — conservative, not o /// How far behind the horizon this holder is holding, in transactions. /// How many collections in the window showed this source winning — the /// chronic-versus-transient discriminator. -/// Collections in the window, so a caller can read the ratio. +/// Collections in the window that recorded ANY holder — the identity +/// fraction's denominator. The collector emits no rows when the horizon is unheld, so this counts +/// holder-bearing collections only, which is what makes held/total mean "of the times something held it, +/// how often was it this one" — and also why it cannot serve as "collections in the window" for the +/// horizon arm (#3537). /// Free-text state the collector captured (e.g. "state=idle in transaction"). +/// #3537: collections in the window whose WINNING xmin_age sat at +/// or above the evaluator's warning threshold, holder identity ignored — the horizon arm's numerator. 0 +/// (the default) means no window data was supplied and keeps that arm quiet, the conservative +/// fail-direction already established. +/// #3537: how many times the collector actually captured in the window — +/// the horizon arm's denominator, sourced from the collection log rather than from the holder table, so +/// quiet (zero-row, healthy) captures count. 0 (the default) reads as "no capture count supplied" and +/// floors the horizon arm out rather than firing. public sealed record PostgresXminHorizonAlertInfo( string Source, string? Identifier, long XminAge, int ObservationsHeld, int ObservationsTotal, - string? Detail); + string? Detail, + int ObservationsAboveThreshold = 0, + int CapturesInWindow = 0); /// /// Accumulated pressure for one poison wait event over the alert's evaluation window (#2711). From fa5c1ecf199058d865ad445dcf0e79300c3102f3 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Thu, 17 Sep 2026 23:58:04 -0400 Subject: [PATCH 06/69] get_pg_plans pins query_id in the SQL; get_pg_io_stats reports track_io_timing (fixes #3533, fixes #3536) (#3558) #3533: the queryid filter ran client-side over a fetched top-duration page (limit * 10), so a plan ranked below the page was unfindable at any window size - and the filtered-empty branch then told the caller capture was working, no plan was captured, and it was "not the query to look at" while the plan sat in the store. The predicate now travels into DarlingPgPlanCaptureReader's SQL (($4::bigint IS NULL OR query_id = $4); null leaves the top-duration page unchanged), the over-fetch and the C# filter are gone (the Viewer's existing call keeps its overload), and the miss text says what was actually searched - the whole window, server-side - and what a miss can mean: never ran (get_pg_top_queries confirms), never crossed auto_explain.log_min_duration, capture not working when it ran (get_pg_plan_capture_readiness), or aged out of retention. #3536: track_io_timing is OFF by default in PostgreSQL, and this read rendered that as 0.000 ms latencies - an impossibly fast disk instead of "not measured". It now mirrors the trend sibling's contract: the setting is read from pg_server_config bounded by the window end (inferred from the window's data when the config was never collected, and io_timing_source says which), io_timing_tracked and timing_note are published, every time-derived field (read_time_ms, avg_read_ms, write_time_ms, extend_time_ms, the read-time shares, total_read_time_ms) is null when untracked, and busiest_basis states what the ranking used - the reader's secondary ORDER BY key (reads) is the entire ordering over a store of zeros, and is now documented as load-bearing. busiest_by_read_time keeps its key because the web tile reads it by name. Tests: SQL predicate pins; both miss-text arms; BuildIoJson wire shape for tracked, untracked, inferred, setting-overrides-observation, and the Aurora null-write interplay; and a gated live fixture seeding 20 expensive shapes plus a 21st cheap one - exactly the limit * 10 page the old path fetched - proving the pinned read finds it, the top page does not contain it, and an absent queryid gets the honest miss. Live parse analysis of every shipped PG read green on PostgreSQL 18. Fixes #3533, fixes #3536 Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx Co-authored-by: Claude Fable 5 --- .../Darling.Tests/DarlingMcpPgIoToolsTests.cs | 192 +++++++++++++ .../DarlingMcpPgPlanToolsTests.cs | 184 +++++++++++++ .../Mcp/DarlingMcpPgIoTools.cs | 252 +++++++++++------- .../Mcp/DarlingMcpPgPlanTools.cs | 57 ++-- .../DarlingPgIoReader.cs | 5 +- .../DarlingPgPlanCaptureReader.cs | 24 +- 6 files changed, 594 insertions(+), 120 deletions(-) create mode 100644 Darling/Darling.Tests/DarlingMcpPgIoToolsTests.cs diff --git a/Darling/Darling.Tests/DarlingMcpPgIoToolsTests.cs b/Darling/Darling.Tests/DarlingMcpPgIoToolsTests.cs new file mode 100644 index 000000000..79053e243 --- /dev/null +++ b/Darling/Darling.Tests/DarlingMcpPgIoToolsTests.cs @@ -0,0 +1,192 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Linq; +using System.Text.Json; +using PerformanceMonitor.Darling.Service.Mcp; +using PerformanceMonitor.Darling.Storage; +using Xunit; + +namespace Darling.Tests; + +/// +/// Wire shape for get_pg_io_stats, asserted against the real projection (#3536). +/// +/// The subject is the timing contract the trend sibling already shipped and this read lacked: +/// track_io_timing is OFF by default in PostgreSQL, so on a stock server every time counter in the +/// store is zero — and read_time_ms / reads over that is 0.000 ms, a latency that reads as an +/// impossibly fast disk rather than an unmeasured one. The zero is a fact about the configuration; printed +/// as a measurement it is the most reassuring wrong number on the surface. +/// +public class DarlingMcpPgIoToolsTests +{ + /// + /// Two combinations the way the reader hands them over: ordered by read time then by read count, so + /// over a store of zero times (timing off) the count IS the ordering. The busiest-by-reads row leads. + /// + private static List Rows(bool timed) => new() + { + new DarlingPgIoReader.PgIoRow( + BackendType: "client backend", ObjectType: "relation", Context: "normal", + Reads: 5_000, ReadTimeMs: timed ? 2_500 : 0, + Hits: 95_000, Extends: 10, ExtendTimeMs: timed ? 40 : 0, + Evictions: 5, Reuses: 0, + Writes: 200, WriteTimeMs: timed ? 90 : 0, + OpBytes: 8_192, WriteCountersTracked: true, StatsReset: null, + ReadBytes: 0, WriteBytes: 0, ExtendBytes: 0, ByteCountersTracked: false), + new DarlingPgIoReader.PgIoRow( + BackendType: "autovacuum worker", ObjectType: "relation", Context: "vacuum", + Reads: 1_000, ReadTimeMs: timed ? 700 : 0, + Hits: 3_000, Extends: 0, ExtendTimeMs: 0, + Evictions: 0, Reuses: 40, + Writes: 50, WriteTimeMs: timed ? 25 : 0, + OpBytes: 8_192, WriteCountersTracked: true, StatsReset: null, + ReadBytes: 0, WriteBytes: 0, ExtendBytes: 0, ByteCountersTracked: false), + }; + + private static JsonElement Parse(string json) + { + using var doc = JsonDocument.Parse(json); + return doc.RootElement.Clone(); + } + + /// + /// The point of #3536. With track_io_timing off — PostgreSQL's DEFAULT — every time field + /// is null rather than the 0.0 the arithmetic produces, top to bottom: the per-row times, the per-read + /// latency, the read-time shares, and the window total. The counters beside them survive untouched, + /// because the operation counts are real measurements whatever the timing setting is. + /// + [Fact] + public void TimingUntracked_NullsEveryTimeField_AndKeepsTheCounts() + { + var root = Parse(DarlingMcpPgIoTools.BuildIoJson("srv", 24, Rows(timed: false), timingSetting: false)); + + Assert.False(root.GetProperty("io_timing_tracked").GetBoolean()); + Assert.Equal(JsonValueKind.Null, root.GetProperty("total_read_time_ms").ValueKind); + + var row = root.GetProperty("combinations")[0]; + Assert.Equal(JsonValueKind.Null, row.GetProperty("read_time_ms").ValueKind); + Assert.Equal(JsonValueKind.Null, row.GetProperty("avg_read_ms").ValueKind); + Assert.Equal(JsonValueKind.Null, row.GetProperty("write_time_ms").ValueKind); + Assert.Equal(JsonValueKind.Null, row.GetProperty("extend_time_ms").ValueKind); + Assert.Equal(JsonValueKind.Null, row.GetProperty("pct_of_total_read_time").ValueKind); + + /* The counts are measured regardless of the timing setting and must not be dragged down with it. */ + Assert.Equal(5_000, row.GetProperty("reads").GetInt64()); + Assert.Equal(95_000, row.GetProperty("hits").GetInt64()); + Assert.Equal(95.0, row.GetProperty("hit_pct").GetDouble()); + Assert.Equal(200, row.GetProperty("writes").GetInt64()); + + var note = root.GetProperty("timing_note").GetString()!; + Assert.Contains("off by DEFAULT", note, StringComparison.Ordinal); + Assert.Contains("does not measure I/O time", note, StringComparison.Ordinal); + } + + /// + /// The busiest field keeps its key — the web tile reads it by name — and busiest_basis beside it + /// says what the ranking actually used: read time when the server measures it, the read COUNT when it + /// does not. The reader's ORDER BY carries the count as its second key, so over a store of zeros the + /// count is the entire ordering rather than a tiebreak, and the payload has to say so. + /// + [Fact] + public void TheBusiestBasis_IsReadsWhenUntracked_AndReadTimeWhenTracked() + { + var untracked = Parse(DarlingMcpPgIoTools.BuildIoJson("srv", 24, Rows(timed: false), timingSetting: false)); + Assert.Equal("client backend/relation/normal", untracked.GetProperty("busiest_by_read_time").GetString()); + Assert.Contains("read count", untracked.GetProperty("busiest_basis").GetString(), StringComparison.Ordinal); + Assert.Contains("does not measure I/O time", untracked.GetProperty("busiest_basis").GetString(), StringComparison.Ordinal); + + var tracked = Parse(DarlingMcpPgIoTools.BuildIoJson("srv", 24, Rows(timed: true), timingSetting: true)); + Assert.Equal("client backend/relation/normal", tracked.GetProperty("busiest_by_read_time").GetString()); + Assert.Contains("read time", tracked.GetProperty("busiest_basis").GetString(), StringComparison.Ordinal); + } + + /// + /// With timing tracked the measured figures flow through unchanged — the untracked arm must not cost + /// the measuring server anything: per-read latency is time over reads, the shares add up, and the + /// window total is the sum of the rows. + /// + [Fact] + public void TimingTracked_KeepsTheMeasuredFigures() + { + var root = Parse(DarlingMcpPgIoTools.BuildIoJson("srv", 24, Rows(timed: true), timingSetting: true)); + + Assert.True(root.GetProperty("io_timing_tracked").GetBoolean()); + Assert.Equal(3_200.0, root.GetProperty("total_read_time_ms").GetDouble()); + Assert.Contains("pg_server_config", root.GetProperty("io_timing_source").GetString(), StringComparison.Ordinal); + Assert.Contains("track_io_timing on", root.GetProperty("timing_note").GetString(), StringComparison.Ordinal); + + var row = root.GetProperty("combinations")[0]; + Assert.Equal(2_500.0, row.GetProperty("read_time_ms").GetDouble()); + Assert.Equal(0.5, row.GetProperty("avg_read_ms").GetDouble()); + Assert.Equal(78.1, row.GetProperty("pct_of_total_read_time").GetDouble()); + Assert.Equal(90.0, row.GetProperty("write_time_ms").GetDouble()); + } + + /// + /// A store without the server's configuration answers from the only evidence left — whether any + /// non-zero time appears in the window — and SAYS it is inferring, exactly as the trend sibling does. + /// "We do not know" and "the server does not measure it" license different readings of a zero. + /// + [Fact] + public void AnUncollectedSetting_IsInferredFromTheData_AndSaysSo() + { + var quiet = Parse(DarlingMcpPgIoTools.BuildIoJson("srv", 24, Rows(timed: false), timingSetting: null)); + Assert.False(quiet.GetProperty("io_timing_tracked").GetBoolean()); + Assert.Contains("inferred from the data", quiet.GetProperty("io_timing_source").GetString(), StringComparison.Ordinal); + + var timed = Parse(DarlingMcpPgIoTools.BuildIoJson("srv", 24, Rows(timed: true), timingSetting: null)); + Assert.True(timed.GetProperty("io_timing_tracked").GetBoolean()); + Assert.Contains("inferred from the data", timed.GetProperty("io_timing_source").GetString(), StringComparison.Ordinal); + } + + /// + /// The setting wins over the observation in BOTH directions, mirroring the trend sibling: collected + /// configuration is the authority, and inference is only for a store that never collected it. + /// + [Fact] + public void TheCollectedSetting_OverridesTheObservation() + { + /* Setting says on, window happens to be all zeros: tracked, with honest zeros, not nulls. */ + var on = Parse(DarlingMcpPgIoTools.BuildIoJson("srv", 24, Rows(timed: false), timingSetting: true)); + Assert.True(on.GetProperty("io_timing_tracked").GetBoolean()); + Assert.Equal(0.0, on.GetProperty("combinations")[0].GetProperty("read_time_ms").GetDouble()); + + /* Setting says off, stale non-zero times in the window: untracked wins and the times are null. */ + var off = Parse(DarlingMcpPgIoTools.BuildIoJson("srv", 24, Rows(timed: true), timingSetting: false)); + Assert.False(off.GetProperty("io_timing_tracked").GetBoolean()); + Assert.Equal(JsonValueKind.Null, off.GetProperty("combinations")[0].GetProperty("read_time_ms").ValueKind); + } + + /// + /// The Aurora contract is untouched by the timing one: write COUNTERS tracked/untracked is a different + /// axis from time measured/unmeasured, and a timing-on Aurora still reports null writes. The two flags + /// travel separately because their remedies are different — one is a platform fact, one is a setting. + /// + [Fact] + public void AuroraNullWrites_SurviveTheTimingGate() + { + var aurora = Rows(timed: true) + .Select(r => r with { Writes = 0, WriteTimeMs = 0, WriteCountersTracked = false }) + .ToList(); + + var root = Parse(DarlingMcpPgIoTools.BuildIoJson("srv", 24, aurora, timingSetting: true)); + + Assert.True(root.GetProperty("io_timing_tracked").GetBoolean()); + Assert.False(root.GetProperty("write_counters_tracked_anywhere").GetBoolean()); + Assert.Contains("Aurora", root.GetProperty("note").GetString(), StringComparison.Ordinal); + + var row = root.GetProperty("combinations")[0]; + Assert.Equal(JsonValueKind.Null, row.GetProperty("writes").ValueKind); + Assert.Equal(JsonValueKind.Null, row.GetProperty("write_time_ms").ValueKind); + /* The read side still reports: timing is on and reads are tracked everywhere. */ + Assert.Equal(2_500.0, row.GetProperty("read_time_ms").GetDouble()); + } +} diff --git a/Darling/Darling.Tests/DarlingMcpPgPlanToolsTests.cs b/Darling/Darling.Tests/DarlingMcpPgPlanToolsTests.cs index 1dbf82747..f8a46e06e 100644 --- a/Darling/Darling.Tests/DarlingMcpPgPlanToolsTests.cs +++ b/Darling/Darling.Tests/DarlingMcpPgPlanToolsTests.cs @@ -8,8 +8,14 @@ using System; using System.Collections.Generic; +using System.Globalization; using System.Linq; using System.Text.Json; +using System.Threading; +using System.Threading.Tasks; +using Npgsql; +using PerformanceMonitor.Collectors; +using PerformanceMonitor.Common; using PerformanceMonitor.Darling.Service.Mcp; using PerformanceMonitor.Darling.Storage; using Xunit; @@ -128,6 +134,62 @@ public void AnUnparseablePlan_IsShownRatherThanDropped() Assert.Equal("{not valid json", plan.GetString()); } + /* ── the queryid filter and the empty answers (#3533) ── */ + + /// + /// The queryid filter runs IN the SQL, over every capture in the window. It used to be applied in C# + /// over a fetched top-duration page, which made any plan ranked below the page unfindable — the ranking + /// is by total duration, so a cheap-but-asked-about statement sits arbitrarily far down and no page + /// size reaches it. NULL must leave the read as the top page, which is what the OR arm is. + /// + [Fact] + public void TheQueryIdFilter_RunsInTheStore_NotOverAFetchedPage() + { + var sql = DarlingPgPlanCaptureReader.PgPlanCaptureSql; + + Assert.Contains("($4::bigint IS NULL OR query_id = $4)", sql, StringComparison.Ordinal); + Assert.Contains("LIMIT $5", sql, StringComparison.Ordinal); + } + + /// + /// The queryid miss says what was actually searched — the whole window, not a page — and names what a + /// miss can mean: the statement never ran (get_pg_top_queries confirms), it never crossed the capture + /// threshold, or capture was not working when it ran (get_pg_plan_capture_readiness has the facets). + /// The text this replaced declared the query "not the query to look at" over a fetched page the plan + /// could legitimately sit below, which is the confident wrong verdict #3533 exists to remove. + /// + [Fact] + public void TheQueryIdMiss_SaysTheWholeWindowWasSearched_AndWhatAMissCanMean() + { + var text = DarlingMcpPgPlanTools.NoPlanCapturedMessage(BigQueryId, 24); + + Assert.Contains("not a top-N page", text, StringComparison.Ordinal); + Assert.Contains("the last 24 hour(s)", text, StringComparison.Ordinal); + Assert.Contains("get_pg_top_queries", text, StringComparison.Ordinal); + Assert.Contains("auto_explain.log_min_duration", text, StringComparison.Ordinal); + Assert.Contains("get_pg_plan_capture_readiness", text, StringComparison.Ordinal); + Assert.Contains("plan_content_retention_days", text, StringComparison.Ordinal); + + Assert.DoesNotContain("not the query to look at", text, StringComparison.Ordinal); + } + + /// + /// The unfiltered miss is a statement about EVERY statement — the read has no filter, so zero rows + /// means the window is genuinely empty, and the message must not borrow the per-query verdict. + /// + [Fact] + public void TheUnfilteredMiss_IsAboutEveryStatement_AndNamesBothCauses() + { + var text = DarlingMcpPgPlanTools.NoPlanCapturedMessage(null, 48); + + Assert.Contains("every statement", text, StringComparison.Ordinal); + Assert.Contains("the last 48 hour(s)", text, StringComparison.Ordinal); + Assert.Contains("auto_explain.log_min_duration", text, StringComparison.Ordinal); + Assert.Contains("plan_content_retention_days", text, StringComparison.Ordinal); + + Assert.DoesNotContain("not the query to look at", text, StringComparison.Ordinal); + } + /* ── get_pg_plan_capture_readiness (#3070) ── */ /// @@ -264,3 +326,125 @@ public void ACappedResult_WithholdsTheUnsatisfiedSummary_RatherThanDescribingThe Assert.Contains("TRUNCATED", root.GetProperty("note").GetString(), StringComparison.Ordinal); } } + +/// +/// Gated (DARLING_TEST_PG) proof of #3533's mechanism, with the fixture the bug requires: a plan ranked +/// BELOW the page the old path fetched. The old code pulled the top limit * 10 shapes by total +/// duration and filtered them in C#, so with limit 2 a plan ranked 21st was unreachable at any +/// window size — and the miss was then reported as "capture is working, this plan was never captured". +/// The predicate now runs in the store, so the same call must return the plan; and a queryid that was +/// genuinely never captured must get the honest whole-window miss rather than the old verdict. +/// +[Collection("live-postgres")] +public sealed class DarlingMcpPgPlanQueryIdLiveTests +{ + private const string ServerName = "darling-pg-plans-queryid-e2e"; + private static readonly int ServerId = ServerIdHelper.GetDeterministicHashCode(ServerName); + + /// Past 2^53 and negative, the way real pg_stat_statements ids look (#2548). + private const long WantedQueryId = -8126435036642491494; + + /// Never seeded, so the filtered read over it must miss honestly. + private const long AbsentQueryId = -7000000000000000001; + + private static string? ConnectionString => Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + + [Fact] + public async Task AQueryIdRankedBelowTheTopPage_IsFound_AndAGenuineMissIsReportedHonestly() + { + var cs = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(cs), + "Set DARLING_TEST_PG to a Postgres connection string to run the live get_pg_plans queryid test."); + + var ct = TestContext.Current.CancellationToken; + using var connection = new NpgsqlConnection(cs); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + await DeleteRowsAsync(connection, ct); + + await using var postgres = NpgsqlDataSource.Create(cs!); + var bodySucceeded = false; + + try + { + await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); + await DarlingMcpTestData.ExecAsync(connection, ct, + "UPDATE servers SET engine_kind = $2 WHERE server_id = $1", + ServerId, MonitoredEngineKind.Postgres); + + var seen = DarlingMcpTestData.TruncateToSeconds(DateTime.UtcNow).AddMinutes(-30); + + /* 20 expensive shapes — exactly the limit * 10 page the OLD path fetched for limit 2 — and the + wanted plan 21st, cheaper than all of them. Ranked by total duration it sits one row below + everything the old path could ever see. */ + for (var i = 0; i < 20; i++) + { + await SeedCaptureAsync(connection, ct, seen, queryId: 9_100_000_000_000_000_001 + i, + planHash: $"EXPENSIVE{i:D2}", durationMs: 10_000 - (i * 100), + planJson: """{"Plan":{"Node Type":"Hash Join"}}"""); + } + + await SeedCaptureAsync(connection, ct, seen, WantedQueryId, + planHash: "WANTED", durationMs: 1.5, + planJson: """{"Plan":{"Node Type":"Index Scan","Relation Name":"orders"}}"""); + + /* The premise, demonstrated rather than assumed: the top page at this limit does not contain + the wanted plan. If this ever fails the fixture has stopped modeling the bug. */ + var topPage = JsonDocument.Parse( + await DarlingMcpPgPlanTools.GetPgPlans(postgres, ServerName, 24, 2)).RootElement; + var pageIds = topPage.GetProperty("plans").EnumerateArray() + .Select(p => p.GetProperty("queryid").GetString()) + .ToArray(); + Assert.Equal(2, pageIds.Length); + Assert.DoesNotContain(WantedQueryId.ToString(CultureInfo.InvariantCulture), pageIds); + + /* The fix: the same limit, pinned to the queryid, finds the plan the old path could not. */ + var found = JsonDocument.Parse(await DarlingMcpPgPlanTools.GetPgPlans( + postgres, ServerName, 24, 2, WantedQueryId.ToString(CultureInfo.InvariantCulture))).RootElement; + + var plan = Assert.Single(found.GetProperty("plans").EnumerateArray().ToArray()); + Assert.Equal(WantedQueryId.ToString(CultureInfo.InvariantCulture), + plan.GetProperty("queryid").GetString()); + Assert.Equal("WANTED", plan.GetProperty("plan_hash").GetString()); + Assert.Equal("Index Scan", + plan.GetProperty("plan").GetProperty("Plan").GetProperty("Node Type").GetString()); + + /* A queryid that was never captured now misses HONESTLY: the whole window was searched, and + the answer says what a miss can mean instead of declaring the query healthy. */ + var miss = JsonDocument.Parse(await DarlingMcpPgPlanTools.GetPgPlans( + postgres, ServerName, 24, 2, AbsentQueryId.ToString(CultureInfo.InvariantCulture))).RootElement; + + Assert.Equal("empty", miss.GetProperty("status").GetString()); + var message = miss.GetProperty("message").GetString()!; + Assert.Contains("not a top-N page", message, StringComparison.Ordinal); + Assert.Contains("get_pg_plan_capture_readiness", message, StringComparison.Ordinal); + Assert.DoesNotContain("not the query to look at", message, StringComparison.Ordinal); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, async (cleanup, cleanupCt) => + await DeleteRowsAsync(cleanup, cleanupCt)); + } + } + + private static async Task SeedCaptureAsync( + NpgsqlConnection connection, CancellationToken ct, DateTime collectionTimeUtc, + long queryId, string planHash, double durationMs, string planJson) => + await DarlingMcpTestData.ExecAsync(connection, ct, @" +INSERT INTO pg_plan_capture + (collection_id, collection_time, server_id, server_name, query_id, plan_hash, duration_ms, + node_count, top_node_type, plan_json) +VALUES ($1, $2, $3, $4, $5, $6, $7, 3, 'Seeded', $8)", + CollectionIdGenerator.Next(), DarlingMcpTestData.Naive(collectionTimeUtc), ServerId, ServerName, + queryId, planHash, durationMs, planJson); + + private static async Task DeleteRowsAsync(NpgsqlConnection connection, CancellationToken ct) + { + await DarlingMcpTestData.ExecAsync(connection, ct, "DELETE FROM pg_plan_capture WHERE server_id = $1", ServerId); + await DarlingMcpTestData.ExecAsync(connection, ct, "DELETE FROM pg_plan_capture_readiness WHERE server_id = $1", ServerId); + await DarlingMcpTestData.ExecAsync(connection, ct, "DELETE FROM servers WHERE server_id = $1", ServerId); + await DarlingMcpTestData.ExecAsync(connection, ct, "DELETE FROM config_monitored_servers WHERE server_id = $1", ServerId); + } +} diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgIoTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgIoTools.cs index 66e8dd3d2..055148757 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgIoTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgIoTools.cs @@ -7,6 +7,7 @@ */ using System; +using System.Collections.Generic; using System.ComponentModel; using System.Linq; using System.Text.Json; @@ -33,7 +34,7 @@ public sealed class DarlingMcpPgIoTools /// internal static string ContextMeaning(string? context) => DarlingPgIoReader.ContextMeaning(context); - [McpServerTool(Name = "get_pg_io_stats"), Description("Gets PostgreSQL I/O attributed to WHO did it, to WHAT, and WHY - the (backend_type, object, context) breakdown from pg_stat_io, differenced across the requested window. Richer than SQL Server's file-level dm_io_virtual_file_stats: instead of 'this file is busy' you get 'autovacuum workers are reading relations in the vacuum context', which names the cause. The context dimension is the one with no SQL Server equivalent and the one that changes the remedy - it separates ordinary buffer-pool misses (where more shared_buffers or a better index helps) from sequential scans that deliberately bypass the pool via a ring buffer (where it will not help at all), from vacuum's ring buffer, from a standby applying WAL. Reports whether write counters are TRACKED at all, because on Amazon Aurora they are always null - backends there do not write data files, the storage layer does - and a zero would otherwise read as 'no writes happened'. Requires PostgreSQL 16 or later; valid on a standby.")] + [McpServerTool(Name = "get_pg_io_stats"), Description("Gets PostgreSQL I/O attributed to WHO did it, to WHAT, and WHY - the (backend_type, object, context) breakdown from pg_stat_io, differenced across the requested window. Richer than SQL Server's file-level dm_io_virtual_file_stats: instead of 'this file is busy' you get 'autovacuum workers are reading relations in the vacuum context', which names the cause. The context dimension is the one with no SQL Server equivalent and the one that changes the remedy - it separates ordinary buffer-pool misses (where more shared_buffers or a better index helps) from sequential scans that deliberately bypass the pool via a ring buffer (where it will not help at all), from vacuum's ring buffer, from a standby applying WAL. Reports whether the server tracks I/O TIMING at all: track_io_timing is OFF by default in PostgreSQL, and its zero read_time would otherwise divide out to a latency of 0.000 ms that reads as an impossibly fast disk rather than an unmeasured one - the time fields are null when untracked, and busiest_basis says what the ranking actually used. Also reports whether write counters are TRACKED at all, because on Amazon Aurora they are always null - backends there do not write data files, the storage layer does - and a zero would otherwise read as 'no writes happened'. Requires PostgreSQL 16 or later; valid on a standby.")] public static async Task GetPgIoStats( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -79,108 +80,163 @@ public static async Task GetPgIoStats( }, McpHelpers.JsonOptions); } - var totalReads = rows.Sum(r => r.Reads); - var totalReadTime = rows.Sum(r => r.ReadTimeMs); + /* Asked of the server's OWN configuration rather than inferred from the zeros, exactly as the + trend sibling asks it (#3536): the two readings a zero latency permits — "the disk is + instant" and "nobody is timing it" — are not distinguishable in the counters, and + track_io_timing is OFF by default, so the zeros are the ordinary case rather than a fault. */ + var timingSetting = await DarlingPgTrendReader.GetIoTimingTrackedAsync( + postgres, resolved.ServerId, windowEnd); - var combinations = rows.Select(r => - { - var accesses = r.Reads + r.Hits; - return new - { - backend_type = r.BackendType, - object_type = r.ObjectType, - context = r.Context, - context_meaning = ContextMeaning(r.Context), - reads = r.Reads, - read_time_ms = Math.Round(r.ReadTimeMs, 1), - /* Per-read latency is the figure that separates "a lot of I/O" from "slow I/O", and - they have completely different remedies. */ - avg_read_ms = r.Reads > 0 ? Math.Round(r.ReadTimeMs / r.Reads, 3) : (double?)null, - hits = r.Hits, - /* A hit ratio scoped to this combination, which is the only scope where it means - anything: a server-wide ratio averages bulkread's deliberate misses together with - normal-context misses and understates both. */ - hit_pct = accesses > 0 ? Math.Round((double)r.Hits / accesses * 100, 1) : (double?)null, - pct_of_total_reads = totalReads > 0 ? Math.Round((double)r.Reads / totalReads * 100, 1) : 0, - pct_of_total_read_time = totalReadTime > 0 ? Math.Round(r.ReadTimeMs / totalReadTime * 100, 1) : 0, - extends = r.Extends, - extend_time_ms = Math.Round(r.ExtendTimeMs, 1), - evictions = r.Evictions, - /* Ring-buffer reuse, NOT eviction pressure. Conflating the two is the standard - misreading of this view: reuses are a bulk operation recycling its OWN buffers. */ - reuses = r.Reuses, - writes = r.WriteCountersTracked ? r.Writes : (long?)null, - write_time_ms = r.WriteCountersTracked ? Math.Round(r.WriteTimeMs, 1) : (double?)null, - write_counters_tracked = r.WriteCountersTracked, - /* The block size an operation moves. Gone from 18, where a read is no longer one - block, so it is null there and read_bytes below is measured instead of derived. */ - block_bytes = r.OpBytes > 0 ? r.OpBytes : (long?)null, - /* One name for the volume answer, and bytes_source says how it was arrived at. From 18 - these are measured totals; below 18 they are reads x block size. Never both, and - never silently swapped: the two are different quantities, and on 18 the old estimate - would UNDERCOUNT because a vectored read covers several blocks. */ - read_bytes = r.ByteCountersTracked - ? r.ReadBytes - : (r.OpBytes > 0 ? r.Reads * r.OpBytes : (decimal?)null), - write_bytes = r.ByteCountersTracked - ? r.WriteBytes - : (r.OpBytes > 0 && r.WriteCountersTracked ? r.Writes * r.OpBytes : (decimal?)null), - extend_bytes = r.ByteCountersTracked ? r.ExtendBytes : (decimal?)null, - bytes_source = r.ByteCountersTracked - ? "measured" - : (r.OpBytes > 0 ? "estimated_from_block_size" : "unavailable"), - stats_reset = r.StatsReset, - }; - }) - .ToList(); - - var anyWritesTracked = rows.Any(r => r.WriteCountersTracked); - /* #2655: PostgreSQL 18 replaced op_bytes with measured byte totals. Said once at the top for - the same reason the write flag is: a caller has to know which quantity it is reading before - it compares two servers, and the two are not comparable. */ - var bytesMeasured = rows.Any(r => r.ByteCountersTracked); - var bytesEstimated = !bytesMeasured && rows.Any(r => r.OpBytes > 0); - - return JsonSerializer.Serialize(new - { - server = resolved.ServerName, - hours_back, - status = "io_activity", - combination_count = combinations.Count, - total_reads = totalReads, - total_read_time_ms = Math.Round(totalReadTime, 1), - busiest_by_read_time = $"{rows[0].BackendType}/{rows[0].ObjectType}/{rows[0].Context}", - /* Said once at the top rather than repeated per row: on Aurora this is false everywhere, - and a caller needs to know the write side is unmeasured before it concludes anything - from the absence of writes. */ - write_counters_tracked_anywhere = anyWritesTracked, - bytes_source = bytesMeasured - ? "measured" - : (bytesEstimated ? "estimated_from_block_size" : "unavailable"), - note = anyWritesTracked - ? "All counters are windowed differences, clamped per interval so a stats reset cannot " - + "produce a negative figure." - : "All counters are windowed differences. This server tracks NO write counters — the " - + "signature of Amazon Aurora, where backends do not write data files and the storage " - + "layer does. Absent writes here mean unmeasured, not zero.", - bytes_note = bytesMeasured - ? "Byte totals are MEASURED, from PostgreSQL 18's read_bytes/write_bytes/extend_bytes. " - + "They are not comparable with the figures a pre-18 server reports, which are " - + "reads x block size - 18 reads several blocks per operation, so the older estimate " - + "undercounts." - : (bytesEstimated - ? "Byte totals are ESTIMATED as count x block_bytes, which is exact below " - + "PostgreSQL 18 because one operation moves one block. PostgreSQL 18 measures " - + "them directly instead." - : "This server reports no byte figures at all: op_bytes is absent and the measured " - + "columns PostgreSQL 18 replaced it with are not being collected. The counts and " - + "times above are unaffected."), - combinations, - }, McpHelpers.JsonOptions); + return BuildIoJson(resolved.ServerName, hours_back, rows, timingSetting); } catch (Exception ex) { return McpHelpers.Status("error", $"Reading PostgreSQL I/O stats failed: {ex.Message}"); } } + + /// + /// The response body, split out so the WIRE SHAPE can be asserted without a live store — the same + /// reason the plan tools' BuildPlansJson is separate. + /// + /// timingSetting is the target's own track_io_timing from pg_server_config, + /// or null when the configuration has not been collected — in which case whether any non-zero time + /// appears in the window is the only evidence available and is used, stated as inference. + /// + internal static string BuildIoJson( + string serverName, + int hoursBack, + IReadOnlyList rows, + bool? timingSetting) + { + var timingObserved = rows.Any(r => r.ReadTimeMs > 0 || r.WriteTimeMs > 0); + var timingTracked = timingSetting ?? timingObserved; + + var totalReads = rows.Sum(r => r.Reads); + var totalReadTime = rows.Sum(r => r.ReadTimeMs); + + var combinations = rows.Select(r => + { + var accesses = r.Reads + r.Hits; + return new + { + backend_type = r.BackendType, + object_type = r.ObjectType, + context = r.Context, + context_meaning = ContextMeaning(r.Context), + reads = r.Reads, + /* Every time figure is null when the server does not measure I/O time (#3536), rather + than the 0.0 the arithmetic produces: that zero is a fact about the configuration, and + printed as a time it is the most reassuring wrong number here. */ + read_time_ms = timingTracked ? Math.Round(r.ReadTimeMs, 1) : (double?)null, + /* Per-read latency is the figure that separates "a lot of I/O" from "slow I/O", and + they have completely different remedies. */ + avg_read_ms = timingTracked && r.Reads > 0 ? Math.Round(r.ReadTimeMs / r.Reads, 3) : (double?)null, + hits = r.Hits, + /* A hit ratio scoped to this combination, which is the only scope where it means + anything: a server-wide ratio averages bulkread's deliberate misses together with + normal-context misses and understates both. */ + hit_pct = accesses > 0 ? Math.Round((double)r.Hits / accesses * 100, 1) : (double?)null, + pct_of_total_reads = totalReads > 0 ? Math.Round((double)r.Reads / totalReads * 100, 1) : 0, + pct_of_total_read_time = timingTracked + ? (totalReadTime > 0 ? Math.Round(r.ReadTimeMs / totalReadTime * 100, 1) : 0) + : (double?)null, + extends = r.Extends, + extend_time_ms = timingTracked ? Math.Round(r.ExtendTimeMs, 1) : (double?)null, + evictions = r.Evictions, + /* Ring-buffer reuse, NOT eviction pressure. Conflating the two is the standard + misreading of this view: reuses are a bulk operation recycling its OWN buffers. */ + reuses = r.Reuses, + writes = r.WriteCountersTracked ? r.Writes : (long?)null, + write_time_ms = timingTracked && r.WriteCountersTracked ? Math.Round(r.WriteTimeMs, 1) : (double?)null, + write_counters_tracked = r.WriteCountersTracked, + /* The block size an operation moves. Gone from 18, where a read is no longer one + block, so it is null there and read_bytes below is measured instead of derived. */ + block_bytes = r.OpBytes > 0 ? r.OpBytes : (long?)null, + /* One name for the volume answer, and bytes_source says how it was arrived at. From 18 + these are measured totals; below 18 they are reads x block size. Never both, and + never silently swapped: the two are different quantities, and on 18 the old estimate + would UNDERCOUNT because a vectored read covers several blocks. */ + read_bytes = r.ByteCountersTracked + ? r.ReadBytes + : (r.OpBytes > 0 ? r.Reads * r.OpBytes : (decimal?)null), + write_bytes = r.ByteCountersTracked + ? r.WriteBytes + : (r.OpBytes > 0 && r.WriteCountersTracked ? r.Writes * r.OpBytes : (decimal?)null), + extend_bytes = r.ByteCountersTracked ? r.ExtendBytes : (decimal?)null, + bytes_source = r.ByteCountersTracked + ? "measured" + : (r.OpBytes > 0 ? "estimated_from_block_size" : "unavailable"), + stats_reset = r.StatsReset, + }; + }) + .ToList(); + + var anyWritesTracked = rows.Any(r => r.WriteCountersTracked); + /* #2655: PostgreSQL 18 replaced op_bytes with measured byte totals. Said once at the top for + the same reason the write flag is: a caller has to know which quantity it is reading before + it compares two servers, and the two are not comparable. */ + var bytesMeasured = rows.Any(r => r.ByteCountersTracked); + var bytesEstimated = !bytesMeasured && rows.Any(r => r.OpBytes > 0); + + return JsonSerializer.Serialize(new + { + server = serverName, + hours_back = hoursBack, + status = "io_activity", + combination_count = combinations.Count, + total_reads = totalReads, + total_read_time_ms = timingTracked ? Math.Round(totalReadTime, 1) : (double?)null, + /* The key survives untracked timing for the web tile's sake; busiest_basis beside it says + what the ranking actually used. The reader orders by read time and then by read count, so + over a store of zero times the count IS the ordering rather than a tiebreak. */ + busiest_by_read_time = $"{rows[0].BackendType}/{rows[0].ObjectType}/{rows[0].Context}", + busiest_basis = timingTracked + ? "total read time (read_time_ms), then read count" + : "read count (reads) — this server does not measure I/O time, so every read-time figure " + + "is zero in the store and cannot rank anything; the ordering falls back to the counter " + + "that exists", + /* Said once at the top rather than repeated per row: on Aurora this is false everywhere, + and a caller needs to know the write side is unmeasured before it concludes anything + from the absence of writes. */ + write_counters_tracked_anywhere = anyWritesTracked, + io_timing_tracked = timingTracked, + io_timing_source = timingSetting is null + ? "inferred from the data - this server's configuration has not been collected, so " + + "track_io_timing is unknown and the answer here is simply whether any non-zero I/O " + + "time appears in the window" + : "the target's own track_io_timing, as collected into pg_server_config", + bytes_source = bytesMeasured + ? "measured" + : (bytesEstimated ? "estimated_from_block_size" : "unavailable"), + note = anyWritesTracked + ? "All counters are windowed differences, clamped per interval so a stats reset cannot " + + "produce a negative figure." + : "All counters are windowed differences. This server tracks NO write counters — the " + + "signature of Amazon Aurora, where backends do not write data files and the storage " + + "layer does. Absent writes here mean unmeasured, not zero.", + timing_note = timingTracked + ? "read_time_ms, avg_read_ms, write_time_ms and extend_time_ms are measured I/O times: " + + "this server has track_io_timing on." + : "read_time_ms, avg_read_ms, write_time_ms, extend_time_ms and the read-time shares are " + + "NULL throughout because this server does not measure I/O time. track_io_timing is off " + + "by DEFAULT in PostgreSQL, so this is the ordinary configuration rather than a fault - " + + "but it means the operation counts are the only I/O evidence here, and nothing in this " + + "store can say whether the storage is slow. Turning it on costs a clock read per " + + "operation; measure that on the platform before enabling it fleet-wide.", + bytes_note = bytesMeasured + ? "Byte totals are MEASURED, from PostgreSQL 18's read_bytes/write_bytes/extend_bytes. " + + "They are not comparable with the figures a pre-18 server reports, which are " + + "reads x block size - 18 reads several blocks per operation, so the older estimate " + + "undercounts." + : (bytesEstimated + ? "Byte totals are ESTIMATED as count x block_bytes, which is exact below " + + "PostgreSQL 18 because one operation moves one block. PostgreSQL 18 measures " + + "them directly instead." + : "This server reports no byte figures at all: op_bytes is absent and the measured " + + "columns PostgreSQL 18 replaced it with are not being collected. The counts and " + + "times above are unaffected."), + combinations, + }, McpHelpers.JsonOptions); + } } diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgPlanTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgPlanTools.cs index f34a708b4..1e17300c7 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgPlanTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgPlanTools.cs @@ -53,7 +53,7 @@ public static async Task GetPgPlans( [Description("Server name or display name.")] string? server_name = null, [Description("Hours of history to analyze. Default 24.")] int hours_back = 24, [Description("Maximum plan shapes to return. Default 10.")] int limit = 10, - [Description("Only return plans for this queryid, as a string. Optional.")] string? query_id = null, + [Description("Only return plans for this queryid, as a string. Optional. The filter is applied in the store over EVERY capture in the window, not over the top-duration page, so a statement ranked far below the busiest shapes is still found - and an empty answer with this set genuinely means no plan for it was captured in the window.")] string? query_id = null, [Description(McpHelpers.AsOfDescription)] string? as_of = null) { var (resolved, error) = await DarlingServerResolver.ResolveOrErrorAsync(postgres, server_name); @@ -87,17 +87,16 @@ public static async Task GetPgPlans( var now = windowEnd; var start = now.AddHours(-hours_back); + /* The queryid pin travels INTO the SQL (#3533). It used to be applied here, over a fetched + top-duration page, which made every plan ranked below the page unfindable and then reported + the miss as "not captured" — the predicate has to run where the rows are. */ var rows = await DarlingPgPlanCaptureReader.GetPgPlanCaptureAsync( - postgres, resolved.ServerId, start, now, wantedQueryId is null ? limit : limit * 10); - - if (wantedQueryId is not null) - { - rows = rows.Where(r => r.QueryId == wantedQueryId.Value).Take(limit).ToList(); - } + postgres, resolved.ServerId, start, now, limit, wantedQueryId); if (rows.Count == 0) { - return await NoPlansStatusAsync(postgres, resolved.ServerId, resolved.ServerName, wantedQueryId); + return await NoPlansStatusAsync( + postgres, resolved.ServerId, resolved.ServerName, wantedQueryId, hours_back); } return BuildPlansJson(resolved.ServerName, hours_back, rows, limit); @@ -259,7 +258,7 @@ to avoid. The unsatisfied facets are NAMED instead. */ /// the threshold" become a true statement rather than a guess. /// private static async Task NoPlansStatusAsync( - NpgsqlDataSource postgres, int serverId, string serverName, long? wantedQueryId) + NpgsqlDataSource postgres, int serverId, string serverName, long? wantedQueryId, int hoursBack) { var gated = await DarlingEngineCapability.NotCollectedStatusAsync( postgres, serverId, serverName, "pg_plan_capture"); @@ -291,18 +290,38 @@ private static async Task NoPlansStatusAsync( + "satisfied or not, in the order they have to be fixed in."); } - var subject = wantedQueryId is null ? "any statement" : "this statement"; - - return McpHelpers.Status( - "empty", - $"Capture is configured on this server and no plan was captured for {subject} in this window. " - + "Two things produce that and they are different: the statement never ran longer than " - + "auto_explain.log_min_duration, which is the healthy answer and means it is not the query to " - + "look at; or a plan was captured earlier and has aged out, which plan_content_retention_days " - + "governs — widen hours_back to tell those apart, because a plan that exists further back will " - + "reappear and one that never existed will not."); + return McpHelpers.Status("empty", NoPlanCapturedMessage(wantedQueryId, hoursBack)); } + /// + /// The empty answer once capture is known to be configured, split by whether a queryid was asked for. + /// + /// The queryid arm changed with #3533 and its honesty depends on the reader: the filter now runs + /// in the SQL over every capture in the window, so "no plan for this statement was captured" is a fact + /// rather than a statement about a fetched page. The text this replaced said the query was "not the + /// query to look at" — over a top-N page that was an invented verdict, and even over the whole window + /// it collapses three different causes into the most reassuring one. + /// + internal static string NoPlanCapturedMessage(long? wantedQueryId, int hoursBack) => wantedQueryId is null + ? $"Capture is configured on this server and nothing was captured in the last {hoursBack} hour(s) — " + + "this read has no filter, so that is a fact about every statement, not about a page of them. " + + "Two things produce it and they are different: no statement ran longer than " + + "auto_explain.log_min_duration, which is the healthy answer on a server whose statements are " + + "all fast; or plans were captured earlier and have aged out, which plan_content_retention_days " + + "governs — widen hours_back to tell those apart, because a plan that exists further back will " + + "reappear and one that never existed will not." + : $"Every capture in the last {hoursBack} hour(s) was searched for this query_id — the whole " + + "window, in the store, not a top-N page — and none matches, so no plan for this statement was " + + "captured in this window. That has three different causes with three different remedies: the " + + "statement did not run in this window at all, which get_pg_top_queries can confirm from its " + + "call counts; it ran but never crossed auto_explain.log_min_duration, so auto_explain never " + + "wrote a plan — the healthy reading for a statement that is fast HERE, not a verdict about it " + + "at other times; or capture was not working when it ran — get_pg_plan_capture_readiness " + + "reports every precondition with the remedy beside it, and its facets are the latest reading " + + "rather than the window's history, so a server that is ready now can still have missed an " + + "earlier run. A plan captured before this window has aged out under " + + "plan_content_retention_days; widen hours_back to reach further back."; + /// /// The unsatisfied readiness facets, newest reading per facet. Read directly rather than through the /// readiness tool so this stays a fact lookup rather than one MCP tool narrating another's prose. diff --git a/Darling/PerformanceMonitor.Darling.Storage/DarlingPgIoReader.cs b/Darling/PerformanceMonitor.Darling.Storage/DarlingPgIoReader.cs index 3695b4618..244a3f636 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/DarlingPgIoReader.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/DarlingPgIoReader.cs @@ -120,7 +120,10 @@ ORDER BY collection_time FROM differenced GROUP BY backend_type, object_type, context /* Anything that moved, ordered by the work that actually costs time. A combination with no - activity in the window is not a finding and would crowd out the ones that are. */ + activity in the window is not a finding and would crowd out the ones that are. The read-count + key is NOT a cosmetic tiebreak: track_io_timing is off by DEFAULT, so on a stock server every + time sum here is zero and the count is the entire ordering — the tool reports which key decided + (#3536), and dropping the second key would make the untracked case effectively unordered. */ HAVING coalesce(SUM(d_reads), 0) + coalesce(SUM(d_writes), 0) + coalesce(SUM(d_extends), 0) + coalesce(SUM(d_hits), 0) > 0 ORDER BY coalesce(SUM(d_read_time_ms), 0) DESC, coalesce(SUM(d_reads), 0) DESC diff --git a/Darling/PerformanceMonitor.Darling.Storage/DarlingPgPlanCaptureReader.cs b/Darling/PerformanceMonitor.Darling.Storage/DarlingPgPlanCaptureReader.cs index 9fee5b34c..e93ffaa06 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/DarlingPgPlanCaptureReader.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/DarlingPgPlanCaptureReader.cs @@ -11,6 +11,7 @@ using System.Threading; using System.Threading.Tasks; using Npgsql; +using NpgsqlTypes; namespace PerformanceMonitor.Darling.Storage; @@ -67,14 +68,26 @@ FROM pg_plan_capture WHERE server_id = $1 AND collection_time >= $2 AND collection_time <= $3 + /* The optional queryid pin, IN the SQL rather than filtered client-side over a page (#3533): the + ranking is by total duration, so a cheap-but-wanted query sits arbitrarily far below the top and + no page size makes it reachable — filtering a fetched page turned "ranked low" into "was never + captured". NULL leaves the read as the top-duration page. */ + AND ($4::bigint IS NULL OR query_id = $4) GROUP BY query_id, plan_hash ORDER BY sum(duration_ms) DESC - LIMIT $4 + LIMIT $5 """; + public static Task> GetPgPlanCaptureAsync( + NpgsqlDataSource postgres, int serverId, DateTime startUtc, DateTime endUtc, int limit, + CancellationToken cancellationToken = default) => + GetPgPlanCaptureAsync(postgres, serverId, startUtc, endUtc, limit, queryId: null, cancellationToken); + + /// Pins the read to one statement's plans, server-side, over the whole window. + /// Null returns the top page by total duration instead. public static async Task> GetPgPlanCaptureAsync( NpgsqlDataSource postgres, int serverId, DateTime startUtc, DateTime endUtc, int limit, - CancellationToken cancellationToken = default) + long? queryId, CancellationToken cancellationToken = default) { ArgumentNullException.ThrowIfNull(postgres); @@ -85,6 +98,13 @@ public static async Task> GetPgPlanCaptureAsync( /* SpecifyKind(Unspecified) at the BIND, the convention every PostgreSQL read here follows. */ command.Parameters.AddWithValue(DateTime.SpecifyKind(startUtc, DateTimeKind.Unspecified)); command.Parameters.AddWithValue(DateTime.SpecifyKind(endUtc, DateTimeKind.Unspecified)); + /* Typed explicitly rather than through AddWithValue: DBNull carries no type for Npgsql to infer, + so an untyped null fails at bind time — the same reason the trend reader's TextOrNull exists. */ + command.Parameters.Add(new NpgsqlParameter + { + NpgsqlDbType = NpgsqlDbType.Bigint, + Value = (object?)queryId ?? DBNull.Value, + }); command.Parameters.AddWithValue(limit); await using var reader = await command.ExecuteReaderAsync(cancellationToken); From 69a248b4af9359732f71d5222438b99573402cda Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 00:03:28 -0400 Subject: [PATCH 07/69] Perfmon analysis reads divide the per-interval delta by its measured interval (#3527) (#3560) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The PERFMON_*_SEC facts, the batch-request anomaly window, and both SKUs' batch-request baselines consumed delta_cntr_value — the per-COLLECTION-INTERVAL delta — as if it were already per-second: 60x truth at a 60s cadence, 300x at 5min. All three consuming reads now divide by the interval consistently, in both SKUs, so the window statistic, the baseline population, and the requests/sec floors (BatchRequestFloor 500, BatchRequestFallback 5000) finally meet in one unit: - Fact collectors (Pg + DuckDb): select the row's measured sample_interval_seconds (#2234) alongside the delta, emit delta / interval as the fact value, and keep cntr_value/delta_cntr_value raw in the metadata with the divisor added. Rows with interval <= 0 (no delta was knowable: first sighting, reset, gap past the delta policy) are filtered so rn = 1 lands on the newest USABLE row — an unknowable rate is never emitted as 0 or as the raw delta. - Anomaly windows (PgAnomalyDetector + Lite AnomalyDetector, Lite-verbatim SQL): AVG/MAX over delta * 1.0 / NULLIF(sample_interval_seconds, 0) with interval <= 0 rows excluded from the sample count. - Baselines: Lite divides by the stored measured interval on the raw rows; Darling's arm reads the perfmon_baseline continuous aggregate, which materializes no interval column and cannot grow one without forfeiting the 31 of 35 days of history the 4-day raw tier can't refill — so it derives interval_sec from LAG(collection_time) over the collapsed series, the established WaitMsPerSec idiom, with the io-arm's DOUBLE PRECISION cast. The restart-exclusion signature stays on the RAW delta in both SKUs (its > 1000 bar predates the division). No stored baseline state needs migration: both providers compute baselines on read (1-hour in-process cache only, cleared by the upgrade's restart), and the CAGG stores per-collection raw deltas — unit-neutral facts — not statistics, so all 35 days of materialized history serve the new unit immediately. Tests, both SKUs: the interval-60/delta-6000 fixture produces a fact of 100; interval-0 rows are skipped at every read (never rates of 0); the anomaly floors demonstrably compare per-second values (a raw delta past the floor whose rate is under it stays quiet); live-Postgres end-to-end proofs for the fact collector and the detector+baseline pair, including the LAG-derived baseline mean and a window that skips an interval-0 row. Fixes #3527 Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx Co-authored-by: Claude Fable 5 --- .../DarlingAnomalyBaselineTests.cs | 164 ++++++++++++++++++ Darling/Darling.Tests/PgFactCollectorTests.cs | 120 +++++++++++++ .../PgAnomalyDetector.cs | 12 +- .../PgBaselineProvider.cs | 36 ++-- .../PgFactCollector.Resources.cs | 23 ++- Lite.Tests/AnomalyDetectorTests.cs | 60 ++++++- Lite.Tests/BaselineProviderTests.cs | 38 +++- Lite.Tests/FactCollectorTests.cs | 58 +++++++ Lite.Tests/TestDataSeeder.cs | 40 ++++- Lite/Analysis/AnomalyDetector.cs | 12 +- Lite/Analysis/BaselineProvider.cs | 8 +- .../Analysis/DuckDbFactCollector.Resources.cs | 23 ++- 12 files changed, 548 insertions(+), 46 deletions(-) diff --git a/Darling/Darling.Tests/DarlingAnomalyBaselineTests.cs b/Darling/Darling.Tests/DarlingAnomalyBaselineTests.cs index d4878b17f..988687e28 100644 --- a/Darling/Darling.Tests/DarlingAnomalyBaselineTests.cs +++ b/Darling/Darling.Tests/DarlingAnomalyBaselineTests.cs @@ -242,6 +242,170 @@ QUALIFY NOT (delta_cntr_value = 0 AND COALESCE(LAG(delta_cntr_value) OVER /* The pre-window row filter is unchanged from Lite. */ Assert.Contains("counter_name = 'Batch Requests/sec'", TimescaleSupport.CreatePerfmonBaselineSql, StringComparison.Ordinal); Assert.Contains("delta_cntr_value >= 0", TimescaleSupport.CreatePerfmonBaselineSql, StringComparison.Ordinal); + + /* #3527: v is the PER-SECOND rate. The perfmon_baseline supply materializes only + (collection_time, delta_cntr_value) — no stored interval, and a continuous aggregate cannot + grow a column without forfeiting the history the 4-day raw tier can't refill — so the divisor + is derived from LAG(collection_time) over the collapsed series (the WaitMsPerSec idiom), + computed in the SAME windowed CTE (window-before-filter holds for it too), with the + interval-less first row filtered alongside the restart exclusion. The DOUBLE PRECISION cast + is the io-arm rule: STDDEV_SAMP over numeric can overflow System.Decimal. */ + var intervalAt = sql.IndexOf("extract(epoch FROM (date_trunc('second', collection_time) - date_trunc('second', LAG(collection_time) OVER (ORDER BY collection_time)))) AS interval_sec", StringComparison.Ordinal); + Assert.True(intervalAt >= 0 && intervalAt < fromCteAt, "interval_sec must be derived inside the windowed CTE"); + Assert.Contains("delta_cntr_value::DOUBLE PRECISION / interval_sec AS v", sql, StringComparison.Ordinal); + var intervalFilterAt = sql.IndexOf("interval_sec > 0", StringComparison.Ordinal); + Assert.True(intervalFilterAt > fromCteAt, "the interval filter must sit OUTSIDE the windowed CTE, with the exclusion"); + } + + /// + /// #3527: the batch-request WINDOW statistic must be requests/sec — the per-interval + /// delta_cntr_value divided by the row's measured sample_interval_seconds — or the + /// BatchRequestFloor/Fallback thresholds (defined in requests/sec) admit 60-300x-inflated + /// deltas and the comparison against the per-second baseline is cross-unit. Interval <= 0 + /// rows carry NO knowable delta and must be filtered, never read as a rate of 0 or as the + /// raw delta. + /// + [Fact] + public void BatchRequestWindow_DividesByMeasuredInterval_AndSkipsUnknowableRows() + { + var sql = PgAnomalyDetector.BatchRequestWindowSql; + + Assert.Contains("AVG(delta_cntr_value * 1.0 / NULLIF(sample_interval_seconds, 0))", sql, StringComparison.Ordinal); + Assert.Contains("MAX(delta_cntr_value * 1.0 / NULLIF(sample_interval_seconds, 0))", sql, StringComparison.Ordinal); + Assert.Contains("sample_interval_seconds > 0", sql, StringComparison.Ordinal); + + /* A raw AVG/MAX of the delta is exactly the #3527 defect — pin its absence. */ + Assert.DoesNotContain("AVG(delta_cntr_value)", sql, StringComparison.Ordinal); + Assert.DoesNotContain("MAX(delta_cntr_value)", sql, StringComparison.Ordinal); + } + + /// + /// #3527 proven live, both halves in one place: the BatchRequests BASELINE arm derives its + /// per-second unit from LAG(collection_time) over the perfmon_baseline supply, and the DETECTOR's + /// window read divides by the stored measured interval — so the two sides meet in the same + /// requests/sec unit and the absolute bars judge honest rates. + /// + /// History (one Monday-10:00 bucket, 12 collections at 300s spacing, delta 30000 each — + /// 100 req/sec): c1 has no prior (interval NULL → dropped), c6 is a restart zero (prior 30000 > + /// 1000 → excluded), c7 is a genuine idle zero (prior 0 → kept at 0/sec). 10 samples, mean + /// (9x100 + 0)/10 = 90 — in requests/sec, where the raw-delta unit would read 27000. + /// + /// Window (the following Monday): three rows at stored interval 60, delta 600000 — 10000 + /// req/sec — plus one interval-0 row with a wild delta that must be SKIPPED, not rated. The one + /// planted history day leaves the bucket untrustworthy (Full tier needs 3 distinct days), so the + /// detector fires on the absolute BatchRequestFallback bar (5000 req/sec): peak 10000 clears it + /// honestly. Pre-#3527 the raw deltas cleared every bar by orders of magnitude regardless of + /// workload; post-fix the emitted Value, peak/avg metadata, and baseline_mean are all per-second. + /// + [Fact] + public async Task EndToEnd_BatchRequestArm_PerSecondBaselineAndWindow_AgainstDevPostgres() + { + var connectionString = Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + Assert.SkipWhen(string.IsNullOrEmpty(connectionString), + "Set DARLING_TEST_PG to a Postgres connection string to run the live batch-request test."); + + var ct = TestContext.Current.CancellationToken; + const int batchServerId = TestServerId + 2; // own id — this test cleans its own rows + const string batchServerName = "batch-per-second-e2e"; + + using var connection = new NpgsqlConnection(connectionString); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + + await using (var cleanup = new NpgsqlCommand( + $"DELETE FROM perfmon_stats WHERE server_id = {batchServerId}; " + + $"DELETE FROM wait_stats WHERE server_id = {batchServerId};", connection)) + { + await cleanup.ExecuteNonQueryAsync(ct); + } + + await using var postgres = NpgsqlDataSource.Create(connectionString!); + var bodySucceeded = false; + try + { + var day = DateTime.UtcNow.Date.AddDays(-8); + while (day.DayOfWeek != DayOfWeek.Monday) day = day.AddDays(-1); + var historyStart = DateTime.SpecifyKind(day.AddHours(10), DateTimeKind.Unspecified); + + for (var i = 0; i < 12; i++) + { + var delta = (i == 5 || i == 6) ? 0L : 30000L; + await InsertAsync(connection, + "INSERT INTO perfmon_stats (collection_id, collection_time, server_id, server_name, object_name, counter_name, instance_name, cntr_value, delta_cntr_value, sample_interval_seconds) VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10)", + (long)(200 + i), historyStart.AddMinutes(5 * i), batchServerId, batchServerName, + "SQLServer:SQL Statistics", "Batch Requests/sec", "", delta * 2, delta, 300); + } + + /* The baseline supply must exist (see the wait test's note) — plain fallback views. */ + await TimescaleSupport.EnsureBaselineFallbackViewsAsync(connection, null, ct); + + var provider = new PgBaselineProvider(postgres); + var analysisTime = historyStart.AddDays(7); + + var baseline = await provider.GetBaselineAsync(batchServerId, MetricNames.BatchRequests, analysisTime); + Assert.Equal(10L, baseline.SampleCount); + Assert.Equal(90.0, baseline.Mean, 0.001); + Assert.Equal(BaselineTier.Full, baseline.Tier); + Assert.Equal(10, baseline.HourOfDay); + Assert.Equal((int)DayOfWeek.Monday, baseline.DayOfWeek); + + /* Canary for the HasBaselineData gate — OUTSIDE the analysis window so the wait + detector's own window read stays empty and it emits nothing. */ + await InsertAsync(connection, + "INSERT INTO wait_stats (collection_id, collection_time, server_id, server_name, wait_type, delta_waiting_tasks, delta_wait_time_ms) VALUES ($1, $2, $3, $4, $5, $6, $7)", + 300L, historyStart, batchServerId, batchServerName, TestWaitType, 1L, 100L); + + /* The anomalous current window: 10000 req/sec (delta 600000 over a measured 60s), + plus one interval-0 row whose wild delta must never be rated. */ + for (var i = 0; i < 3; i++) + { + await InsertAsync(connection, + "INSERT INTO perfmon_stats (collection_id, collection_time, server_id, server_name, object_name, counter_name, instance_name, cntr_value, delta_cntr_value, sample_interval_seconds) VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10)", + (long)(400 + i), analysisTime.AddMinutes(5 * (i + 1)), batchServerId, batchServerName, + "SQLServer:SQL Statistics", "Batch Requests/sec", "", 1200000L, 600000L, 60); + } + await InsertAsync(connection, + "INSERT INTO perfmon_stats (collection_id, collection_time, server_id, server_name, object_name, counter_name, instance_name, cntr_value, delta_cntr_value, sample_interval_seconds) VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10)", + 403L, analysisTime.AddMinutes(20), batchServerId, batchServerName, + "SQLServer:SQL Statistics", "Batch Requests/sec", "", 0L, 999999999L, 0); + + var detector = new PgAnomalyDetector(postgres, provider); + var context = new AnalysisContext + { + ServerId = batchServerId, + ServerName = batchServerName, + TimeRangeStart = analysisTime, + TimeRangeEnd = analysisTime.AddMinutes(30), + ServerUtcOffset = TimeSpan.Zero + }; + + var anomalies = await detector.DetectAnomaliesAsync(context); + + var fact = Assert.Single(anomalies); + Assert.Equal("ANOMALY_BATCH_REQUESTS", fact.Key); + Assert.Equal(10000.0, fact.Value, 0.001); // per-second, not 600000 + Assert.Equal(10000.0, fact.Metadata["peak_batch_requests"], 0.001); + Assert.Equal(10000.0, fact.Metadata["avg_batch_requests"], 0.001); + Assert.Equal(3.0, fact.Metadata["window_samples"]); // the interval-0 row is NOT a sample + Assert.Equal(90.0, fact.Metadata["baseline_mean"], 0.001); // same unit as the window + Assert.Equal(1.0, fact.Metadata["baseline_low_quality"]); // one distinct day → absolute bar + Assert.Equal(2.0, fact.Metadata["fallback_exceedance"], 0.001); // 10000 / the 5000 req/sec bar + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(connectionString!, bodySucceeded, async (cleanup, cleanupCt) => + { + using (var command = new NpgsqlCommand( + $"DELETE FROM perfmon_stats WHERE server_id = {batchServerId}; " + + $"DELETE FROM wait_stats WHERE server_id = {batchServerId};", cleanup)) + { + await command.ExecuteNonQueryAsync(cleanupCt); + } + await DropBaselineFallbackViewsAsync(cleanup, cleanupCt); + }); + } } [Fact] diff --git a/Darling/Darling.Tests/PgFactCollectorTests.cs b/Darling/Darling.Tests/PgFactCollectorTests.cs index 5b5b60c09..f36394448 100644 --- a/Darling/Darling.Tests/PgFactCollectorTests.cs +++ b/Darling/Darling.Tests/PgFactCollectorTests.cs @@ -341,6 +341,126 @@ await LiveStoreCleanup.RunAsync(connectionString!, bodySucceeded, async (cleanup } } + /* ---------------- #3527: perfmon facts are per-second rates ---------------- */ + + /// + /// #3527: delta_cntr_value spans one COLLECTION INTERVAL, not one second — read raw, the + /// PERFMON_*_SEC facts overstate by the cadence (60x at 60s, 300x at 5min). The query must + /// select the row's measured sample_interval_seconds (#2234) for the division and filter + /// interval <= 0 rows (no delta was knowable: first sighting, reset, gap) so rn = 1 lands on + /// the newest row a rate can honestly be derived from. + /// + [Fact] + public void PerfmonSql_SelectsTheMeasuredInterval_AndFiltersUnknowableRows() + { + var sql = PgFactCollector.PerfmonSql; + + Assert.Contains("delta_cntr_value, sample_interval_seconds", sql, StringComparison.Ordinal); + Assert.Contains("sample_interval_seconds > 0", sql, StringComparison.Ordinal); + } + + /// + /// #3527 live fixture: a 'Batch Requests/sec' row with delta 6000 over a measured 60s interval + /// must emit PERFMON_BATCH_REQ_SEC = 100 (not 6000), with the raw delta and the divisor in the + /// metadata. A NEWER interval-0 row (unknowable delta) must be skipped — the fact still comes + /// from the older usable row — and a counter with ONLY interval-0 rows emits no fact at all, + /// never a fact of 0. + /// + [Fact] + public async Task EndToEnd_PerfmonFacts_DivideDeltaByMeasuredInterval_AgainstDevPostgres() + { + var connectionString = Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + Assert.SkipWhen(string.IsNullOrEmpty(connectionString), + "Set DARLING_TEST_PG to a Postgres connection string to run the live perfmon fact test."); + + var ct = TestContext.Current.CancellationToken; + const int perfmonServerId = TestServerId - 2; // own id — this test cleans its own rows + + using var connection = new NpgsqlConnection(connectionString); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + + await using (var cleanup = new NpgsqlCommand( + $"DELETE FROM perfmon_stats WHERE server_id = {perfmonServerId};", connection)) + { + await cleanup.ExecuteNonQueryAsync(ct); + } + + await using var postgres = NpgsqlDataSource.Create(connectionString!); + var collector = new PgFactCollector(postgres); + + var bodySucceeded = false; + try + { + var windowEnd = TruncateToSeconds(DateTime.UtcNow); + var windowStart = windowEnd.AddHours(-1); + + async Task PlantAsync(long id, DateTime time, string counter, long delta, int intervalSeconds) + { + using var plant = new NpgsqlCommand(@" +INSERT INTO perfmon_stats + (collection_id, collection_time, server_id, server_name, + object_name, counter_name, instance_name, cntr_value, delta_cntr_value, sample_interval_seconds) +VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10)", connection); + plant.Parameters.AddWithValue(id); + plant.Parameters.AddWithValue(time); + plant.Parameters.AddWithValue(perfmonServerId); + plant.Parameters.AddWithValue("perfmon-per-second-e2e"); + plant.Parameters.AddWithValue("SQLServer:SQL Statistics"); + plant.Parameters.AddWithValue(counter); + plant.Parameters.AddWithValue(""); + plant.Parameters.AddWithValue(delta * 2); + plant.Parameters.AddWithValue(delta); + plant.Parameters.AddWithValue(intervalSeconds); + await plant.ExecuteNonQueryAsync(ct); + } + + /* Batch requests: an older USABLE row (delta 6000 / 60s = 100/sec), then a NEWER + interval-0 row that must not become the fact. */ + await PlantAsync(1, windowStart.AddMinutes(20), "Batch Requests/sec", 6000, 60); + await PlantAsync(2, windowStart.AddMinutes(25), "Batch Requests/sec", 0, 0); + + /* Compilations: one usable row, 300 / 60s = 5/sec. */ + await PlantAsync(3, windowStart.AddMinutes(20), "SQL Compilations/sec", 300, 60); + + /* Re-compilations: ONLY an interval-0 row — no rate is knowable, so no fact. */ + await PlantAsync(4, windowStart.AddMinutes(20), "SQL Re-Compilations/sec", 0, 0); + + var context = new AnalysisContext + { + ServerId = perfmonServerId, + ServerName = "perfmon-per-second-e2e", + TimeRangeStart = windowStart, + TimeRangeEnd = windowEnd, + ServerUtcOffset = TimeSpan.Zero + }; + + var facts = await collector.CollectFactsAsync(context); + + var batch = Assert.Single(facts, f => f.Key == "PERFMON_BATCH_REQ_SEC"); + Assert.Equal(100.0, batch.Value, precision: 10); + Assert.Equal(6000, batch.Metadata["delta_cntr_value"]); + Assert.Equal(60, batch.Metadata["sample_interval_seconds"]); + + var compilations = Assert.Single(facts, f => f.Key == "PERFMON_COMPILATIONS_SEC"); + Assert.Equal(5.0, compilations.Value, precision: 10); + + Assert.DoesNotContain(facts, f => f.Key == "PERFMON_RECOMPILATIONS_SEC"); + Assert.Equal(2, facts.Count); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(connectionString!, bodySucceeded, async (cleanup, cleanupCt) => + { + using var command = new NpgsqlCommand( + $"DELETE FROM perfmon_stats WHERE server_id = {perfmonServerId};", cleanup); + await command.ExecuteNonQueryAsync(cleanupCt); + }); + } + } + private static DateTime TruncateToSeconds(DateTime value) => DateTime.SpecifyKind(new DateTime(value.Ticks - (value.Ticks % TimeSpan.TicksPerSecond)), DateTimeKind.Unspecified); diff --git a/Darling/PerformanceMonitor.Darling.Analysis/PgAnomalyDetector.cs b/Darling/PerformanceMonitor.Darling.Analysis/PgAnomalyDetector.cs index 776d66e4c..4a4c0d85c 100644 --- a/Darling/PerformanceMonitor.Darling.Analysis/PgAnomalyDetector.cs +++ b/Darling/PerformanceMonitor.Darling.Analysis/PgAnomalyDetector.cs @@ -207,14 +207,20 @@ FROM v_file_io_stats WHERE server_id = $1 AND collection_time >= $2 AND collection_time <= $3 AND (delta_reads > 0 OR delta_writes > 0)"; + /* Batch-request window: per-second rate per sample (#3527) — delta_cntr_value spans one collection + interval, so divide by the row's MEASURED sample_interval_seconds (#2234). Interval <= 0 marks an + unknowable delta (first sighting/reset/gap) and the row is skipped, never read as 0. Keeps the + window statistic in the same requests/sec unit as the baseline and the BatchRequestFloor/Fallback + thresholds. */ public const string BatchRequestWindowSql = @" -SELECT AVG(delta_cntr_value) AS avg_batch, - MAX(delta_cntr_value) AS peak_batch, +SELECT AVG(delta_cntr_value * 1.0 / NULLIF(sample_interval_seconds, 0)) AS avg_batch, + MAX(delta_cntr_value * 1.0 / NULLIF(sample_interval_seconds, 0)) AS peak_batch, COUNT(*) AS sample_count FROM v_perfmon_stats WHERE server_id = $1 AND collection_time >= $2 AND collection_time <= $3 AND counter_name = 'Batch Requests/sec' -AND delta_cntr_value >= 0"; +AND delta_cntr_value >= 0 +AND sample_interval_seconds > 0"; public const string SessionWindowSql = @" WITH per_collection AS ( diff --git a/Darling/PerformanceMonitor.Darling.Analysis/PgBaselineProvider.cs b/Darling/PerformanceMonitor.Darling.Analysis/PgBaselineProvider.cs index cc8e10ab5..e987c868b 100644 --- a/Darling/PerformanceMonitor.Darling.Analysis/PgBaselineProvider.cs +++ b/Darling/PerformanceMonitor.Darling.Analysis/PgBaselineProvider.cs @@ -366,23 +366,19 @@ FROM cpu_utilization_stats /* QUALIFY rewrite 1 of 4 — cumulative counter, restart exclusion. Excludes samples where the delta drops to 0 when the prior sample was > 1000 - (restart signature for cumulative counters). Lite's DuckDB original: - - SELECT EXTRACT(HOUR FROM collection_time)::INT AS hour_of_day, - EXTRACT(DOW FROM collection_time)::INT AS day_of_week, - AVG(delta_cntr_value) AS mean_val, - STDDEV_SAMP(delta_cntr_value) AS stddev_val, - COUNT(*) AS sample_count - FROM ( - SELECT collection_time, delta_cntr_value + (restart signature for cumulative counters). Lite's DuckDB original (#3527 unit: + v is delta / the row's measured sample_interval_seconds — a per-second rate): + + WITH clean AS ( + SELECT collection_time, delta_cntr_value * 1.0 / NULLIF(sample_interval_seconds, 0) AS v FROM v_perfmon_stats WHERE server_id = $1 AND collection_time >= $2 AND collection_time < $3 AND counter_name = 'Batch Requests/sec' AND delta_cntr_value >= 0 + AND sample_interval_seconds > 0 QUALIFY NOT (delta_cntr_value = 0 AND COALESCE(LAG(delta_cntr_value) OVER (ORDER BY collection_time), 0) > 1000) ) - GROUP BY hour_of_day, day_of_week QUALIFY evaluates AFTER window computation: LAG runs over every WHERE-surviving row (including rows QUALIFY itself is about to drop), THEN the predicate prunes. @@ -390,18 +386,32 @@ The rewrite computes the SAME LAG over the SAME WHERE-filtered rowset inside a CTE and applies the IDENTICAL predicate in the outer WHERE — window-before-filter is preserved, so only the FIRST zero after a >1000 sample is dropped, and a zero following another zero keeps LAG = 0 and SURVIVES (genuine idle, not a restart). - Row selection is exactly the original's. */ + Row selection is exactly the original's. + + #3527 divisor: this arm reads the perfmon_baseline supply (the #1757 CAGG / fallback + view), which materializes only (collection_time, delta_cntr_value) — it does NOT carry + sample_interval_seconds, and a continuous aggregate cannot grow a column without a + drop-and-rebuild that would forfeit the 31 days of baseline history the 4-day raw tier + can no longer refill. So the interval is DERIVED from the gap between consecutive + collections — LAG(collection_time) over the collapsed series, byte-for-byte the + WaitMsPerSec arm's idiom (see CreateWaitStatsBaselineSql's note: the provider computes + interval_sec, nothing extra is stored). Same requests/sec unit as Lite and as the + detector's window read; the first row of the window has no prior (interval NULL) and is + skipped, exactly like WaitMsPerSec. The ::DOUBLE PRECISION cast is the io-arm rule: + STDDEV_SAMP over numeric can overflow System.Decimal at materialization. */ MetricNames.BatchRequests => @" WITH windowed AS ( SELECT collection_time, delta_cntr_value, - COALESCE(LAG(delta_cntr_value) OVER (ORDER BY collection_time), 0) AS prior_delta + COALESCE(LAG(delta_cntr_value) OVER (ORDER BY collection_time), 0) AS prior_delta, + extract(epoch FROM (date_trunc('second', collection_time) - date_trunc('second', LAG(collection_time) OVER (ORDER BY collection_time)))) AS interval_sec FROM perfmon_baseline WHERE server_id = $1 AND collection_time >= $2 AND collection_time < $3 ), clean AS ( - SELECT collection_time, delta_cntr_value AS v + SELECT collection_time, delta_cntr_value::DOUBLE PRECISION / interval_sec AS v FROM windowed WHERE NOT (delta_cntr_value = 0 AND prior_delta > 1000) + AND interval_sec > 0 )," + RobustTierScaffold, /* QUALIFY rewrite 2 of 4 — cumulative counter, multiple rows per collection (per diff --git a/Darling/PerformanceMonitor.Darling.Analysis/PgFactCollector.Resources.cs b/Darling/PerformanceMonitor.Darling.Analysis/PgFactCollector.Resources.cs index 8b60fdd25..8c126431b 100644 --- a/Darling/PerformanceMonitor.Darling.Analysis/PgFactCollector.Resources.cs +++ b/Darling/PerformanceMonitor.Darling.Analysis/PgFactCollector.Resources.cs @@ -331,22 +331,30 @@ private async Task CollectCpuUtilizationFactsAsync(AnalysisContext context, List } } + /* #3527: delta_cntr_value spans one COLLECTION INTERVAL, not one second — at a 60s cadence the raw + delta is 60x the true rate. The honest rate divides by the row's MEASURED sample_interval_seconds + (#2234); interval <= 0 marks an unknowable delta (first sighting, counter reset, gap past the + policy), so those rows are filtered rather than emitted as 0 — rn = 1 lands on the newest row a + rate can honestly be derived from. */ public const string PerfmonSql = @" WITH latest AS ( - SELECT counter_name, cntr_value, delta_cntr_value, + SELECT counter_name, cntr_value, delta_cntr_value, sample_interval_seconds, ROW_NUMBER() OVER (PARTITION BY counter_name ORDER BY collection_time DESC) AS rn FROM perfmon_stats WHERE server_id = $1 AND collection_time >= $2 AND collection_time <= $3 AND counter_name IN ('Batch Requests/sec', 'SQL Compilations/sec', 'SQL Re-Compilations/sec') + AND sample_interval_seconds > 0 ) -SELECT counter_name, cntr_value, delta_cntr_value +SELECT counter_name, cntr_value, delta_cntr_value, sample_interval_seconds FROM latest WHERE rn = 1"; /// /// Collects key perfmon throughput counters: Batch Requests/sec, compilations, recompilations. /// Unscored context that distinguishes a busy server from a sick one (used by the AI surfaces). + /// Fact values are per-second rates: the per-interval delta divided by the row's measured + /// sample_interval_seconds (#3527); the raw delta and the divisor ride the metadata. /// private async Task CollectPerfmonFactsAsync(AnalysisContext context, List facts) { @@ -365,6 +373,7 @@ private async Task CollectPerfmonFactsAsync(AnalysisContext context, List var counterName = reader.GetString(0); var cntrValue = reader.IsDBNull(1) ? 0L : ToInt64(reader.GetValue(1)); var deltaValue = reader.IsDBNull(2) ? 0L : ToInt64(reader.GetValue(2)); + var intervalSeconds = reader.IsDBNull(3) ? 0L : ToInt64(reader.GetValue(3)); var (factKey, source) = counterName switch { @@ -376,8 +385,11 @@ private async Task CollectPerfmonFactsAsync(AnalysisContext context, List if (factKey == null) continue; - // All remaining counters are per-second rates — use the delta. - var value = (double)deltaValue; + /* The delta spans one collection interval — divide by the measured interval for the + per-second rate (#3527). The SQL already filters interval <= 0 (unknowable delta); + this guard keeps a raw or zero value from ever escaping if that filter regresses. */ + if (intervalSeconds <= 0) continue; + var value = deltaValue / (double)intervalSeconds; facts.Add(new Fact { @@ -388,7 +400,8 @@ private async Task CollectPerfmonFactsAsync(AnalysisContext context, List Metadata = new Dictionary { ["cntr_value"] = cntrValue, - ["delta_cntr_value"] = deltaValue + ["delta_cntr_value"] = deltaValue, + ["sample_interval_seconds"] = intervalSeconds } }); } diff --git a/Lite.Tests/AnomalyDetectorTests.cs b/Lite.Tests/AnomalyDetectorTests.cs index 073cd6f80..198b5d14b 100644 --- a/Lite.Tests/AnomalyDetectorTests.cs +++ b/Lite.Tests/AnomalyDetectorTests.cs @@ -100,10 +100,10 @@ private async Task ExecuteSeedAsync(string sql) [Fact] public async Task DetectBatchRequestAnomalies_Spike_DetectsAnomaly() { - // Baseline: normal batch requests (~5000) + // Baseline: normal batch requests (delta ~5000 per 10s interval = ~500/sec) await SeedBaselinePerfmon("Batch Requests/sec", 5000, variance: 200); - // Analysis window: spike to 15000 + // Analysis window: spike to delta 15000 per 10s interval = 1500/sec for (int i = 0; i < 16; i++) await SeedPerfmonAsync(_analysisStart.AddMinutes(i * 15), "Batch Requests/sec", 15000); @@ -139,11 +139,51 @@ public async Task DetectBatchRequestAnomalies_Normal_NoAnomaly() [Fact] public async Task DetectBatchRequestAnomalies_LowVolumeSpike_NoAnomaly() { - // Low-throughput server: a 6x relative spike but the peak stays at 300/sec - // — below the 500/sec BatchRequestFloor — must NOT be flagged. - await SeedBaselinePerfmon("Batch Requests/sec", 50, variance: 5); + // Low-throughput server: a 6x relative spike but the peak stays at 300/sec (delta 3000 over + // a 10s interval) — below the 500/sec BatchRequestFloor — must NOT be flagged. The RAW + // per-interval delta (3000) is past the floor: only the per-second division (#3527) keeps + // this quiet, so this test also pins that the floor compares requests/sec, not the delta. + await SeedBaselinePerfmon("Batch Requests/sec", 500, variance: 50); for (int i = 0; i < 16; i++) - await SeedPerfmonAsync(_analysisStart.AddMinutes(i * 15), "Batch Requests/sec", 300); + await SeedPerfmonAsync(_analysisStart.AddMinutes(i * 15), "Batch Requests/sec", 3000); + + await SeedBaselineCpu(10, variance: 2); + + var anomalies = await _detector.DetectAnomaliesAsync(CreateContext()); + + Assert.DoesNotContain(anomalies, f => f.Key == "ANOMALY_BATCH_REQUESTS"); + } + + [Fact] + public async Task DetectBatchRequestAnomalies_WindowValuesArePerSecond() + { + // #3527 fixture through the detector: window delta 15000 over a 10s interval = 1500/sec. + // The emitted Value and peak/avg metadata must be the per-second numbers, in the same unit + // as the baseline (delta ~5000 / 10s = ~500/sec) — not the raw per-interval deltas. + await SeedBaselinePerfmon("Batch Requests/sec", 5000, variance: 200); + for (int i = 0; i < 16; i++) + await SeedPerfmonAsync(_analysisStart.AddMinutes(i * 15), "Batch Requests/sec", 15000); + + await SeedBaselineCpu(10, variance: 2); + + var anomalies = await _detector.DetectAnomaliesAsync(CreateContext()); + + var fact = anomalies.First(f => f.Key == "ANOMALY_BATCH_REQUESTS"); + Assert.Equal(1500.0, fact.Value); + Assert.Equal(1500.0, fact.Metadata["peak_batch_requests"]); + Assert.Equal(1500.0, fact.Metadata["avg_batch_requests"]); + Assert.InRange(fact.Metadata["baseline_mean"], 450, 550); + } + + [Fact] + public async Task DetectBatchRequestAnomalies_IntervalZeroRowsAreSkipped_NotReadAsRates() + { + // #3527: interval-0 rows carry NO knowable delta. A window holding only interval-0 rows — + // however wild their deltas — has zero usable samples, so the detector stays silent instead + // of reading the raw deltas as a spike (or the rows as rate 0). + await SeedBaselinePerfmon("Batch Requests/sec", 5000, variance: 200); + for (int i = 0; i < 16; i++) + await SeedPerfmonAsync(_analysisStart.AddMinutes(i * 15), "Batch Requests/sec", 999_999, intervalSeconds: 0); await SeedBaselineCpu(10, variance: 2); @@ -600,7 +640,10 @@ private async Task SeedCpuAsync(DateTime time, int cpuValue) await cmd.ExecuteNonQueryAsync(); } - private async Task SeedPerfmonAsync(DateTime time, string counterName, long deltaValue) + /// Seeds one perfmon row. deltaValue is the PER-INTERVAL delta; the detector divides it by + /// intervalSeconds (#3527), so at the default 10s interval the per-second rate is deltaValue / 10. + /// intervalSeconds = 0 plants the unknowable-delta marker the reads must skip. + private async Task SeedPerfmonAsync(DateTime time, string counterName, long deltaValue, int intervalSeconds = 10) { using var readLock = _duckDb.AcquireReadLock(); var conn = await SeedConnectionAsync(); @@ -608,12 +651,13 @@ private async Task SeedPerfmonAsync(DateTime time, string counterName, long delt cmd.CommandText = @"INSERT INTO perfmon_stats (collection_id, collection_time, server_id, server_name, object_name, counter_name, instance_name, cntr_value, delta_cntr_value, sample_interval_seconds) - VALUES ($1, $2, $3, 'TestServer', 'SQLServer:SQL Statistics', $4, '', $5, $5, 10)"; + VALUES ($1, $2, $3, 'TestServer', 'SQLServer:SQL Statistics', $4, '', $5, $5, $6)"; cmd.Parameters.Add(new DuckDBParameter { Value = _nextId-- }); cmd.Parameters.Add(new DuckDBParameter { Value = time }); cmd.Parameters.Add(new DuckDBParameter { Value = ServerId }); cmd.Parameters.Add(new DuckDBParameter { Value = counterName }); cmd.Parameters.Add(new DuckDBParameter { Value = deltaValue }); + cmd.Parameters.Add(new DuckDBParameter { Value = intervalSeconds }); await cmd.ExecuteNonQueryAsync(); } diff --git a/Lite.Tests/BaselineProviderTests.cs b/Lite.Tests/BaselineProviderTests.cs index 1ce9f111e..ad04f8bd9 100644 --- a/Lite.Tests/BaselineProviderTests.cs +++ b/Lite.Tests/BaselineProviderTests.cs @@ -286,8 +286,34 @@ public async Task GetBaseline_BatchRequests_ExcludesRestartDrop() _provider.ClearCache(); var baseline = await _provider.GetBaselineAsync(ServerId, MetricNames.BatchRequests, AnalysisTime); - // The restart drop (0) should be excluded, so mean should be near 5000, not pulled toward 0 - Assert.True(baseline.Mean > 4000, $"Mean {baseline.Mean} should not be poisoned by restart drop"); + // #3527: the baseline is per-second now — deltas of ~5000 over 10s intervals are ~500/sec. + // The restart drop (0) should be excluded, so the mean should sit near 500, not pulled toward 0. + Assert.True(baseline.Mean > 400, $"Mean {baseline.Mean} should not be poisoned by restart drop"); + Assert.InRange(baseline.Mean, 400, 600); + } + + [Fact] + public async Task GetBaseline_BatchRequests_IsPerSecond_AndSkipsIntervalZeroRows() + { + // #3527 fixture: deltas of 6000 over measured 60s intervals are 100 requests/sec — the + // baseline population must be that division, in the same unit as the detector's window read. + // Interspersed interval-0 rows (unknowable delta) must not enter the population at all: + // read as raw deltas they would inflate the mean; read as 0 they would drag it down. + for (int d = 1; d <= 4; d++) + { + var day = AnalysisTime.AddDays(-7 * d); + for (int i = 0; i < 5; i++) + await SeedPerfmonAsync(day.AddMinutes(i * 10), "Batch Requests/sec", 6000, intervalSeconds: 60); + + // One unknowable-delta row per day, wedged between the usable samples. + await SeedPerfmonAsync(day.AddMinutes(55), "Batch Requests/sec", 0, intervalSeconds: 0); + } + + _provider.ClearCache(); + var baseline = await _provider.GetBaselineAsync(ServerId, MetricNames.BatchRequests, AnalysisTime); + + Assert.Equal(20, baseline.SampleCount); // 5 usable rows x 4 days; interval-0 rows contribute nothing + Assert.Equal(100.0, baseline.Mean, 3); } // ── Wait stats: per-collection aggregation ── @@ -442,7 +468,10 @@ private async Task SeedCpuAsync(DateTime time, int cpuValue, int serverId = Serv await cmd.ExecuteNonQueryAsync(); } - private async Task SeedPerfmonAsync(DateTime time, string counterName, long deltaValue) + /// Seeds one perfmon row. deltaValue is the PER-INTERVAL delta; the baseline divides it by + /// intervalSeconds (#3527), so at the default 10s interval the per-second value is deltaValue / 10. + /// intervalSeconds = 0 plants the unknowable-delta marker the baseline must skip. + private async Task SeedPerfmonAsync(DateTime time, string counterName, long deltaValue, int intervalSeconds = 10) { using var readLock = _duckDb.AcquireReadLock(); var conn = await SeedConnectionAsync(); @@ -450,12 +479,13 @@ private async Task SeedPerfmonAsync(DateTime time, string counterName, long delt cmd.CommandText = @"INSERT INTO perfmon_stats (collection_id, collection_time, server_id, server_name, object_name, counter_name, instance_name, cntr_value, delta_cntr_value, sample_interval_seconds) - VALUES ($1, $2, $3, 'TestServer', 'SQLServer:SQL Statistics', $4, '', $5, $5, 10)"; + VALUES ($1, $2, $3, 'TestServer', 'SQLServer:SQL Statistics', $4, '', $5, $5, $6)"; cmd.Parameters.Add(new DuckDBParameter { Value = _nextId-- }); cmd.Parameters.Add(new DuckDBParameter { Value = time }); cmd.Parameters.Add(new DuckDBParameter { Value = ServerId }); cmd.Parameters.Add(new DuckDBParameter { Value = counterName }); cmd.Parameters.Add(new DuckDBParameter { Value = deltaValue }); + cmd.Parameters.Add(new DuckDBParameter { Value = intervalSeconds }); await cmd.ExecuteNonQueryAsync(); } diff --git a/Lite.Tests/FactCollectorTests.cs b/Lite.Tests/FactCollectorTests.cs index b226f1f72..915c63579 100644 --- a/Lite.Tests/FactCollectorTests.cs +++ b/Lite.Tests/FactCollectorTests.cs @@ -213,6 +213,64 @@ public async Task CollectFacts_Perfmon_ReturnsRateCounters() var facts = await SeedAndCollectAsync(s => s.SeedEverythingOnFireServerAsync()); Assert.True(facts.ContainsKey("PERFMON_BATCH_REQ_SEC"), "PERFMON_BATCH_REQ_SEC should be collected"); + + /* #3527: the seeder plants delta 30000 over a measured 60s interval — the fact must be the + per-second rate 500, not the raw per-interval delta. */ + var batch = facts["PERFMON_BATCH_REQ_SEC"]; + Assert.Equal(500.0, batch.Value); + Assert.Equal(30000, batch.Metadata["delta_cntr_value"]); + Assert.Equal(60, batch.Metadata["sample_interval_seconds"]); + } + + /// + /// #3527 fixture: delta 6000 over a measured 60s interval is 100 requests/sec — the fact value is + /// the division, with the raw delta and the divisor preserved in metadata. Before the fix the fact + /// carried the raw 6000 (60x truth at this cadence). + /// + [Fact] + public async Task CollectFacts_Perfmon_DividesDeltaByMeasuredInterval() + { + var facts = await SeedAndCollectAsync(async s => + { + await s.ClearTestDataAsync(); + await s.SeedTestServerAsync(); + await s.SeedPerfmonRawAsync("Batch Requests/sec", deltaValue: 6000, sampleIntervalSeconds: 60); + }); + + var batch = facts["PERFMON_BATCH_REQ_SEC"]; + Assert.Equal(100.0, batch.Value); + Assert.Equal(6000, batch.Metadata["delta_cntr_value"]); + Assert.Equal(60, batch.Metadata["sample_interval_seconds"]); + } + + /// + /// #3527: sample_interval_seconds = 0 means NO delta was knowable (first sighting, counter reset, + /// gap past the delta policy) — such a row must never become a fact of 0 or of the raw delta. The + /// newest usable row wins instead, and a counter with ONLY unusable rows emits no fact at all. + /// + [Fact] + public async Task CollectFacts_Perfmon_SkipsIntervalZeroRows() + { + var facts = await SeedAndCollectAsync(async s => + { + await s.ClearTestDataAsync(); + await s.SeedTestServerAsync(); + + /* Older usable row, then a NEWER interval-0 row: the fact must come from the usable row. */ + await s.SeedPerfmonRawAsync("Batch Requests/sec", deltaValue: 6000, sampleIntervalSeconds: 60, + collectionTime: TestDataSeeder.TestPeriodEnd.AddMinutes(-5)); + await s.SeedPerfmonRawAsync("Batch Requests/sec", deltaValue: 0, sampleIntervalSeconds: 0); + + /* A counter whose only row in the window is interval-0: no fact, not a fact of 0. */ + await s.SeedPerfmonRawAsync("SQL Re-Compilations/sec", deltaValue: 0, sampleIntervalSeconds: 0); + }); + + var batch = facts["PERFMON_BATCH_REQ_SEC"]; + Assert.Equal(100.0, batch.Value); + Assert.Equal(60, batch.Metadata["sample_interval_seconds"]); + + Assert.False(facts.ContainsKey("PERFMON_RECOMPILATIONS_SEC"), + "a counter with only interval-0 rows must emit no fact — 0 is not a knowable rate"); } [Fact] diff --git a/Lite.Tests/TestDataSeeder.cs b/Lite.Tests/TestDataSeeder.cs index cca1af50a..00db2a424 100644 --- a/Lite.Tests/TestDataSeeder.cs +++ b/Lite.Tests/TestDataSeeder.cs @@ -1518,7 +1518,9 @@ INSERT INTO query_stats /// /// Seeds perfmon_stats with the collected rate counters (batch requests, compilations, - /// recompilations); all use delta_cntr_value. + /// recompilations). Parameters are PER-SECOND rates; the rows carry the per-interval delta + /// (rate x the 60s interval) plus sample_interval_seconds = 60, so the fact collector's + /// delta / interval division (#3527) reproduces the parameter exactly. /// internal async Task SeedPerfmonAsync(long batchReqSec = 500, long compilationsSec = 50, long recompilationsSec = 5) @@ -1529,9 +1531,9 @@ internal async Task SeedPerfmonAsync(long batchReqSec = 500, var counters = new (string name, long cntrValue, long deltaValue)[] { - ("Batch Requests/sec", batchReqSec * 60, batchReqSec), // cntr = cumulative, delta = rate - ("SQL Compilations/sec", compilationsSec * 60, compilationsSec), - ("SQL Re-Compilations/sec", recompilationsSec * 60, recompilationsSec) + ("Batch Requests/sec", batchReqSec * 120, batchReqSec * 60), // cntr = cumulative, delta = rate x 60s interval + ("SQL Compilations/sec", compilationsSec * 120, compilationsSec * 60), + ("SQL Re-Compilations/sec", recompilationsSec * 120, recompilationsSec * 60) }; foreach (var (name, cntr, delta) in counters) @@ -1556,6 +1558,36 @@ INSERT INTO perfmon_stats } } + /// + /// Seeds one raw perfmon_stats row with explicit delta and interval — for pinning the #3527 + /// delta / sample_interval_seconds division and the interval-0 (unknowable delta) skip. + /// + internal async Task SeedPerfmonRawAsync(string counterName, long deltaValue, int sampleIntervalSeconds, + DateTime? collectionTime = null) + { + using var readLock = _duckDb.AcquireReadLock(); + var connection = await SeedConnectionAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +INSERT INTO perfmon_stats + (collection_id, collection_time, server_id, server_name, + object_name, counter_name, cntr_value, delta_cntr_value, sample_interval_seconds) +VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9)"; + + cmd.Parameters.Add(new DuckDBParameter { Value = _nextId-- }); + cmd.Parameters.Add(new DuckDBParameter { Value = collectionTime ?? TestPeriodEnd }); + cmd.Parameters.Add(new DuckDBParameter { Value = TestServerId }); + cmd.Parameters.Add(new DuckDBParameter { Value = TestServerName }); + cmd.Parameters.Add(new DuckDBParameter { Value = "SQLServer:SQL Statistics" }); + cmd.Parameters.Add(new DuckDBParameter { Value = counterName }); + cmd.Parameters.Add(new DuckDBParameter { Value = deltaValue * 2 }); + cmd.Parameters.Add(new DuckDBParameter { Value = deltaValue }); + cmd.Parameters.Add(new DuckDBParameter { Value = sampleIntervalSeconds }); + + await cmd.ExecuteNonQueryAsync(); + } + /// /// Seeds memory_clerks with clerk type → MB mappings. /// diff --git a/Lite/Analysis/AnomalyDetector.cs b/Lite/Analysis/AnomalyDetector.cs index 30e0ce0a1..c06096799 100644 --- a/Lite/Analysis/AnomalyDetector.cs +++ b/Lite/Analysis/AnomalyDetector.cs @@ -716,14 +716,20 @@ private async Task DetectBatchRequestAnomalies(AnalysisContext context, List= $2 AND collection_time <= $3 AND counter_name = 'Batch Requests/sec' -AND delta_cntr_value >= 0"; +AND delta_cntr_value >= 0 +AND sample_interval_seconds > 0"; cmd.Parameters.Add(new DuckDBParameter { Value = context.ServerId }); cmd.Parameters.Add(new DuckDBParameter { Value = context.TimeRangeStart }); diff --git a/Lite/Analysis/BaselineProvider.cs b/Lite/Analysis/BaselineProvider.cs index a517765c8..c558e9836 100644 --- a/Lite/Analysis/BaselineProvider.cs +++ b/Lite/Analysis/BaselineProvider.cs @@ -294,13 +294,19 @@ FROM v_cpu_utilization_stats // Cumulative counter — restart exclusion via subquery with QUALIFY. // Excludes samples where delta drops to 0 when prior sample was > 1000 // (restart signature for cumulative counters). + // #3527: v is the per-second rate — the per-interval delta divided by the row's measured + // sample_interval_seconds — so the baseline population is in the same requests/sec unit as + // the detector's window statistic. Interval <= 0 rows (unknowable delta) are skipped, never + // read as 0. The restart signature stays on the RAW delta: its > 1000 bar predates the + // division and marks a counter reset regardless of cadence. MetricNames.BatchRequests => @" WITH clean AS ( - SELECT collection_time, delta_cntr_value AS v + SELECT collection_time, delta_cntr_value * 1.0 / NULLIF(sample_interval_seconds, 0) AS v FROM v_perfmon_stats WHERE server_id = $1 AND collection_time >= $2 AND collection_time < $3 AND counter_name = 'Batch Requests/sec' AND delta_cntr_value >= 0 + AND sample_interval_seconds > 0 QUALIFY NOT (delta_cntr_value = 0 AND COALESCE(LAG(delta_cntr_value) OVER (ORDER BY collection_time), 0) > 1000) )," + RobustTierScaffold, diff --git a/Lite/Analysis/DuckDbFactCollector.Resources.cs b/Lite/Analysis/DuckDbFactCollector.Resources.cs index fbdf2975e..32d272e43 100644 --- a/Lite/Analysis/DuckDbFactCollector.Resources.cs +++ b/Lite/Analysis/DuckDbFactCollector.Resources.cs @@ -339,6 +339,8 @@ FROM v_cpu_utilization_stats /// /// Collects key perfmon throughput counters: Batch Requests/sec, compilations, recompilations. /// Unscored context that distinguishes a busy server from a sick one (used by the AI surfaces). + /// Fact values are per-second rates: the per-interval delta divided by the row's measured + /// sample_interval_seconds (#3527); the raw delta and the divisor ride the metadata. /// private async Task CollectPerfmonFactsAsync(AnalysisContext context, List facts) { @@ -349,17 +351,23 @@ private async Task CollectPerfmonFactsAsync(AnalysisContext context, List await connection.OpenAsync(context.CancellationToken); using var cmd = connection.CreateCommand(); + /* #3527: delta_cntr_value spans one COLLECTION INTERVAL, not one second — at a 60s cadence the + raw delta is 60x the true rate. The honest rate divides by the row's MEASURED + sample_interval_seconds (#2234); interval <= 0 marks an unknowable delta (first sighting, + counter reset, gap past the policy), so those rows are filtered rather than emitted as 0 — + rn = 1 lands on the newest row a rate can honestly be derived from. */ cmd.CommandText = @" WITH latest AS ( - SELECT counter_name, cntr_value, delta_cntr_value, + SELECT counter_name, cntr_value, delta_cntr_value, sample_interval_seconds, ROW_NUMBER() OVER (PARTITION BY counter_name ORDER BY collection_time DESC) AS rn FROM v_perfmon_stats WHERE server_id = $1 AND collection_time >= $2 AND collection_time <= $3 AND counter_name IN ('Batch Requests/sec', 'SQL Compilations/sec', 'SQL Re-Compilations/sec') + AND sample_interval_seconds > 0 ) -SELECT counter_name, cntr_value, delta_cntr_value +SELECT counter_name, cntr_value, delta_cntr_value, sample_interval_seconds FROM latest WHERE rn = 1"; cmd.Parameters.Add(new DuckDBParameter { Value = context.ServerId }); @@ -372,6 +380,7 @@ AND counter_name IN ('Batch Requests/sec', 'SQL Compilations/sec', 'SQL Re-Com var counterName = reader.GetString(0); var cntrValue = reader.IsDBNull(1) ? 0L : ToInt64(reader.GetValue(1)); var deltaValue = reader.IsDBNull(2) ? 0L : ToInt64(reader.GetValue(2)); + var intervalSeconds = reader.IsDBNull(3) ? 0L : ToInt64(reader.GetValue(3)); var (factKey, source) = counterName switch { @@ -383,8 +392,11 @@ AND counter_name IN ('Batch Requests/sec', 'SQL Compilations/sec', 'SQL Re-Com if (factKey == null) continue; - // All remaining counters are per-second rates — use the delta. - var value = (double)deltaValue; + /* The delta spans one collection interval — divide by the measured interval for the + per-second rate (#3527). The SQL already filters interval <= 0 (unknowable delta); + this guard keeps a raw or zero value from ever escaping if that filter regresses. */ + if (intervalSeconds <= 0) continue; + var value = deltaValue / (double)intervalSeconds; facts.Add(new Fact { @@ -395,7 +407,8 @@ AND counter_name IN ('Batch Requests/sec', 'SQL Compilations/sec', 'SQL Re-Com Metadata = new Dictionary { ["cntr_value"] = cntrValue, - ["delta_cntr_value"] = deltaValue + ["delta_cntr_value"] = deltaValue, + ["sample_interval_seconds"] = intervalSeconds } }); } From 537cbbd9275691a8f9ef5db33d4080e479ebe70f Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 00:06:11 -0400 Subject: [PATCH 08/69] Floor the SQL count thresholds, give Store Disk Pressure a GB floor, and count measured metrics in the fleet Healthy label (#3528) (#3562) Three alert-semantics honesty fixes from the 2026-09 brains-review campaign: 1. DeadlockCountThreshold and BlockingCountThreshold now floor at PostgresAlertEvaluator.CountThresholdFloor on read, mirroring their V122 PostgreSQL twins, so a store row hand-edited to 0 can no longer make count >= threshold fire on a quiet server. The MCP write bounds name the same constant instead of a literal 1, and the PG twins' doc comments stop calling the gap pre-existing. 2. Store Disk Pressure gains a GB floor (V126: self_disk_free_warn_gb, default 50): the percent trigger additionally requires free space below the floor before firing, the pvs_floor_gb AND-composition, so 400 GB free on a 4 TB store volume stops paging CRITICAL; 0 removes the floor. Full knob plumbing: darling.json field, clamped settings property, seed/read, get_alert_settings/update_alert_settings key parity, and the viewer schema-probe arm. The Settings window box is deliberately deferred to the viewer pass and pinned as an abstinence. 3. Fleet cards carry measured_metric_count/metric_count beside the band (the worst-of fold skips Unknown, so an online server with five of six metrics structurally Unknown still bands Healthy); the web fleet page renders the qualifier ("1 of 6 measured") on partially-measured cards. Rank-neutrality of Unknown is unchanged. Fixes #3528 Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx Co-authored-by: Claude Fable 5 --- Darling/Darling.Tests/AlertEngineTests.cs | 1 + .../Darling.Tests/AlertStoredValueTests.cs | 1 + .../CollectorDatabaseScopeRungTests.cs | 50 ++-- .../DarlingAlertTuningKnobsTests.cs | 7 +- .../Darling.Tests/DarlingFleetReaderTests.cs | 11 + .../DarlingMcpAlertToolsTests.cs | 6 +- .../Darling.Tests/DarlingSelfAlertTests.cs | 63 ++++- .../PgAlertCountKnobRungTests.cs | 27 +- .../SelfDiskWarnGbFloorRungTests.cs | 248 ++++++++++++++++++ .../ServerHealthClassifierTests.cs | 63 +++++ .../Darling.Tests/StoreSelfMetricsTests.cs | 1 + .../DarlingAlertSettings.cs | 41 ++- .../DarlingConfig.cs | 10 +- .../DarlingSelfAlertEvaluator.cs | 58 ++-- .../Mcp/DarlingAlertReader.cs | 11 +- .../Mcp/DarlingFleetReader.cs | 18 ++ .../Mcp/DarlingMcpAlertTools.cs | 18 +- .../StoreConfigProvider.cs | 18 +- .../wwwroot/js/pages/fleet.js | 10 +- .../PgMigrations.cs | 28 ++ .../StorageVersion.cs | 2 +- .../ViewerDataService.cs | 34 ++- Lite/Services/AppAlertEngineSettings.cs | 1 + .../IAlertEngineSettings.cs | 10 + .../ServerHealthBands.cs | 24 ++ 25 files changed, 669 insertions(+), 92 deletions(-) create mode 100644 Darling/Darling.Tests/SelfDiskWarnGbFloorRungTests.cs diff --git a/Darling/Darling.Tests/AlertEngineTests.cs b/Darling/Darling.Tests/AlertEngineTests.cs index ac94fffaf..9d505a2b5 100644 --- a/Darling/Darling.Tests/AlertEngineTests.cs +++ b/Darling/Darling.Tests/AlertEngineTests.cs @@ -69,6 +69,7 @@ test switches on exactly the check it pins (a disabled check must not even fetch public int DiskCriticalFreePercent { get; set; } = 3; public int DiskCriticalFreeGb { get; set; } = 2; public int SelfDiskFreeWarnPercent { get; set; } = 10; + public int SelfDiskFreeWarnGb { get; set; } = 50; public int CollectionStaleMinutes { get; set; } = 30; public int CollectionFailureThreshold { get; set; } = 10; /* #1984: DarlingConfig defaults (40% / 1 GB); enable stays the class's opt-in OFF. */ diff --git a/Darling/Darling.Tests/AlertStoredValueTests.cs b/Darling/Darling.Tests/AlertStoredValueTests.cs index d43b0754e..1df8ae3ad 100644 --- a/Darling/Darling.Tests/AlertStoredValueTests.cs +++ b/Darling/Darling.Tests/AlertStoredValueTests.cs @@ -91,6 +91,7 @@ private sealed class Settings : IAlertEngineSettings public int DiskCriticalFreePercent { get; set; } = 3; public int DiskCriticalFreeGb { get; set; } = 2; public int SelfDiskFreeWarnPercent { get; set; } = 10; + public int SelfDiskFreeWarnGb { get; set; } = 50; public int CollectionStaleMinutes { get; set; } = 30; public int CollectionFailureThreshold { get; set; } = 10; public int PvsThresholdPercent { get; set; } = 40; diff --git a/Darling/Darling.Tests/CollectorDatabaseScopeRungTests.cs b/Darling/Darling.Tests/CollectorDatabaseScopeRungTests.cs index 9099ad3f3..4f54ab322 100644 --- a/Darling/Darling.Tests/CollectorDatabaseScopeRungTests.cs +++ b/Darling/Darling.Tests/CollectorDatabaseScopeRungTests.cs @@ -8,7 +8,6 @@ using System; using System.Collections.Generic; -using System.Globalization; using System.Linq; using System.Reflection; using Npgsql; @@ -36,18 +35,20 @@ namespace Darling.Tests; /// because the composed predicate is scoped-in AND NOT excluded. New databases stay OUT of a /// non-empty scope until named — the allow-list-not-deny-list argument the issue was won on. /// -/// This file carries the "I am the top rung" claims that moved off -/// (V124) when this rung landed, the same handoff that -/// file received from (V123) — a fully-migrated store must map -/// to EXACTLY this version, or the viewer's connect-time gate refuses a store that is actually -/// current. +/// The "I am the top rung" claims have moved ON to +/// (V126), the same handoff this file received from +/// (V124) and that file received from (V123). What stays here +/// is everything true of this rung wherever it sits in the ladder; what left is every claim that was +/// really about being NEWEST — keeping a copy of those would assert this rung is still the top, +/// which is how the NEXT rung's build goes red. /// public sealed class CollectorDatabaseScopeRungTests { private const int RungVersion = 125; private const int PreviousVersion = 124; - /// This rung's sentinel ordinal in the viewer probe — the newest, so the last argument. + /// This rung's sentinel ordinal in the viewer probe. No longer the last argument — V126 + /// appended its own — so this is a position within the signature rather than its end. private const int ProbeOrdinal = 100; private const string ScopeColumn = "databases"; @@ -55,7 +56,7 @@ public sealed class CollectorDatabaseScopeRungTests /* ---- the rung ------------------------------------------------------------------------------------ */ [Fact] - public void TheRungIsRegisteredAtTheTopOfADenseLadder() + public void TheRungIsRegisteredInADenseLadder() { var versions = PgMigrations.Scripts.Select(s => s.Version).ToList(); @@ -65,7 +66,11 @@ public void TheRungIsRegisteredAtTheTopOfADenseLadder() Assert.Equal(StorageVersion.SchemaVersion, PgMigrations.Scripts[^1].Version); Assert.Equal(StorageVersion.SchemaVersion, versions.Max()); - Assert.Equal(RungVersion, StorageVersion.SchemaVersion); + + /* Not `RungVersion == SchemaVersion` any more: that asserted this rung is the newest, which + stopped being true when V126 landed. The invariant that outlives the handoff is that the + LADDER's top and the declared version agree, which the two lines above already say. */ + Assert.True(RungVersion < StorageVersion.SchemaVersion); Assert.Equal(versions.Distinct().OrderBy(v => v), versions); } @@ -113,15 +118,15 @@ this rung out of MigrationDataMovingRungCensusPins' register. */ /* ---- the probe (three sites, top arm) ------------------------------------------------------------- */ /// - /// The viewer probe's three sites carry this rung's sentinel, and the map treats it as the TOP arm. + /// The viewer probe's three sites carry this rung's sentinel, and its arm still answers. /// /// The probe asks the question, the caller reads the answer, the map has the parameter — three /// sites, and a sentinel present at only some of them shifts every LATER ordinal onto the wrong column. - /// Miss all three and a fully-migrated store probes one rung short, so the connect-time gate refuses a - /// store that is in fact current — permanently, because no later upgrade changes the answer. + /// The top-arm claims (last argument, textual newest-first ordering) moved to + /// with the V126 handoff. /// [Fact] - public void TheProbeMapsAFullyMigratedStoreToThisTopRung() + public void TheProbeCarriesThisRungsSentinel_AndAFullyMigratedStoreMapsToTheLaddersTop() { Assert.Contains($"column_name = '{ScopeColumn}'", ViewerDataService.StoreSchemaProbeSql, StringComparison.Ordinal); Assert.Contains("table_name = 'config_collector_schedules'", ViewerDataService.StoreSchemaProbeSql, StringComparison.Ordinal); @@ -136,8 +141,10 @@ public void TheProbeMapsAFullyMigratedStoreToThisTopRung() .GetMethod("MapProbedSchemaVersion", BindingFlags.NonPublic | BindingFlags.Static)!; var arity = method.GetParameters().Length; - /* The top rung's sentinel IS the last argument. */ - Assert.Equal(ProbeOrdinal, arity - 1); + /* The ordinal has to be a position that exists, and one that is no longer the last: `arity - 1` + asserted this rung is the NEWEST sentinel, which stopped being true the moment V126 appended + its own. Strictly-less is the form every other non-top rung's test here uses. */ + Assert.True(ProbeOrdinal < arity - 1); /* Every sentinel true = a fully-migrated store, which must map to exactly this version. Built by reflection so the arity tracks the signature. */ @@ -154,19 +161,6 @@ test of that rung. */ var behind = (object[])atThisRung.Clone(); behind[ProbeOrdinal] = false; Assert.Equal(PreviousVersion, (int)method.Invoke(null, behind)!); - - /* And in the source, the arm sits ABOVE V124's — newest-first is the whole contract of that method — - and returns this build's version rather than a literal that could drift from it. This is the - textual half of the top-arm claim, inherited from FleetSweepCadenceKnobRungTests the way that - file inherited it from FleetSweepStateRungTests. */ - var v125 = viewer.IndexOf("if (hasCollectorScheduleDatabases)", StringComparison.Ordinal); - var v124 = viewer.IndexOf("if (hasFleetSweepCadenceKnobs)", StringComparison.Ordinal); - Assert.True(v125 >= 0, "the viewer has no V125 sentinel arm — a fully-migrated store would map to 124"); - Assert.True(v124 >= 0, "the V124 arm is gone, so this pin is comparing against nothing"); - Assert.True(v125 < v124, "the V125 arm sits below V124's, so a current store maps one rung low"); - Assert.Contains( - "return " + StorageVersion.SchemaVersion.ToString(CultureInfo.InvariantCulture) + ";", - viewer[v125..], StringComparison.Ordinal); } /* ---- every schedule-row surface handles the column ------------------------------------------------ */ diff --git a/Darling/Darling.Tests/DarlingAlertTuningKnobsTests.cs b/Darling/Darling.Tests/DarlingAlertTuningKnobsTests.cs index c6a275606..2afc9f4f2 100644 --- a/Darling/Darling.Tests/DarlingAlertTuningKnobsTests.cs +++ b/Darling/Darling.Tests/DarlingAlertTuningKnobsTests.cs @@ -38,8 +38,10 @@ public void SelfAlertKnobs_DefaultsAreTheConstantsTheyReplaced_AndReadsClampLike var config = new DarlingConfig(); var settings = new DarlingAlertSettings(config); - /* Defaults mirror the V55 DDL — the compile-time constants these knobs replaced. */ + /* Defaults mirror the V55 DDL — the compile-time constants these knobs replaced — and the V126 + GB floor mirrors its own shipped constant (#3528). */ Assert.Equal(10, settings.SelfDiskFreeWarnPercent); + Assert.Equal(50, settings.SelfDiskFreeWarnGb); Assert.Equal(30, settings.CollectionStaleMinutes); Assert.Equal(10, settings.CollectionFailureThreshold); Assert.Equal(3, settings.DiskCriticalFreePercent); @@ -51,6 +53,7 @@ public void SelfAlertKnobs_DefaultsAreTheConstantsTheyReplaced_AndReadsClampLike 0 failure threshold on the fast path would fire on any single failure, and the analysis cooldown keeps the shared engine's documented [30, 10080]. */ config.Alerts.SelfDiskFreeWarnPercent = 150; + config.Alerts.SelfDiskFreeWarnGb = -1; config.Alerts.CollectionStaleMinutes = 0; config.Alerts.CollectionFailureThreshold = 0; config.Alerts.DiskCriticalFreePercent = -5; @@ -58,6 +61,8 @@ public void SelfAlertKnobs_DefaultsAreTheConstantsTheyReplaced_AndReadsClampLike config.Alerts.AnalysisNotifyCooldownMinutes = 99999; Assert.Equal(100, settings.SelfDiskFreeWarnPercent); + /* #3528: floored at 0 like the sibling GB knobs — 0 is meaningful (it removes the floor). */ + Assert.Equal(0, settings.SelfDiskFreeWarnGb); Assert.Equal(5, settings.CollectionStaleMinutes); Assert.Equal(1, settings.CollectionFailureThreshold); Assert.Equal(0, settings.DiskCriticalFreePercent); diff --git a/Darling/Darling.Tests/DarlingFleetReaderTests.cs b/Darling/Darling.Tests/DarlingFleetReaderTests.cs index 54d419822..e1d92c7bf 100644 --- a/Darling/Darling.Tests/DarlingFleetReaderTests.cs +++ b/Darling/Darling.Tests/DarlingFleetReaderTests.cs @@ -238,6 +238,10 @@ public void FleetServerCard_SerializesSnakeCase_WithStringBands() FailedCollectorCount = 0, CollectorSeverity = HealthSeverity.Healthy, OverallMetricSeverity = HealthSeverity.Critical, + /* #3528: deliberately measured < total, so the value pins below cannot pass off a card that + serialized one count under both keys. */ + MeasuredMetricCount = 1, + MetricCount = 6, }; var json = JsonSerializer.Serialize(card, DarlingFleetReader.JsonOptions); @@ -251,11 +255,18 @@ public void FleetServerCard_SerializesSnakeCase_WithStringBands() "\"deadlock_count\"", "\"deadlock_last_seen\"", "\"deadlock_rate_per_hour\"", "\"deadlock_severity\"", "\"threads_severity\"", "\"failed_collector_count\"", "\"collector_severity\"", "\"overall_metric_severity\"", + "\"measured_metric_count\"", "\"metric_count\"", }) { Assert.Contains(field, json, StringComparison.Ordinal); } + /* #3528: the coverage counts ride every card so a consumer can qualify the band label + ("Healthy — 1 of 6 measured") — values pinned, not just keys, so the two cannot be swapped or + collapsed into one. */ + JsonAssert.Contains("\"measured_metric_count\": 1", json); + JsonAssert.Contains("\"metric_count\": 6", json); + /* Bands / severities serialize as strings, not ordinals — the frontend maps a name to a color. */ JsonAssert.Contains("\"band\": \"Critical\"", json); JsonAssert.Contains("\"cpu_severity\": \"Critical\"", json); diff --git a/Darling/Darling.Tests/DarlingMcpAlertToolsTests.cs b/Darling/Darling.Tests/DarlingMcpAlertToolsTests.cs index b02884432..dd8e7ffec 100644 --- a/Darling/Darling.Tests/DarlingMcpAlertToolsTests.cs +++ b/Darling/Darling.Tests/DarlingMcpAlertToolsTests.cs @@ -748,7 +748,11 @@ SQL Server numbers above (3 and 5) — equal pairs would let a payload that emit /* #3466 (V124): inside the write bound [15, 1440], deliberately NOT the shipped 60 — a sample equal to the default would let a payload that dropped the column and fell back to the default still match — and enabled deliberately FALSE against the shipped TRUE for the same reason. */ - FleetSweepEnabled: false, FleetSweepIntervalMinutes: 240); + FleetSweepEnabled: false, FleetSweepIntervalMinutes: 240, + /* #3528 (V126): inside the write bound (>= 0), deliberately NOT the shipped 50 — a sample equal + to the default would let a payload that dropped the column and fell back to the default still + match. */ + SelfDiskFreeWarnGb: 75); [Fact] public void AlertSettingsSql_ReadsSingleGlobalRow() diff --git a/Darling/Darling.Tests/DarlingSelfAlertTests.cs b/Darling/Darling.Tests/DarlingSelfAlertTests.cs index 96cc96170..2bc3193c9 100644 --- a/Darling/Darling.Tests/DarlingSelfAlertTests.cs +++ b/Darling/Darling.Tests/DarlingSelfAlertTests.cs @@ -78,6 +78,7 @@ private sealed class FakeSettings : IAlertEngineSettings public int DiskCriticalFreePercent { get; set; } = 3; public int DiskCriticalFreeGb { get; set; } = 2; public int SelfDiskFreeWarnPercent { get; set; } = 10; + public int SelfDiskFreeWarnGb { get; set; } = 50; public int CollectionStaleMinutes { get; set; } = 30; public int CollectionFailureThreshold { get; set; } = 10; public int PvsThresholdPercent { get; set; } = 40; @@ -824,11 +825,39 @@ this must not collapse to 0 on the not-pressure path. */ [Fact] public void IsDiskPressure_JustBelowThreshold_Pressure() { - /* 9.9% free trips it — the threshold is a real edge, not a wide band. */ - Assert.True(DarlingSelfAlertEvaluator.IsDiskPressure(99 * Gib, 1000 * Gib, out _, out var percentFree)); + /* 9.9% free trips it — the threshold is a real edge, not a wide band. The volume is small enough + (100 GiB) that the #3528 GB floor is far above the free space, so the percent is what decides. */ + Assert.True(DarlingSelfAlertEvaluator.IsDiskPressure( + (long)(9.9 * Gib), 100 * Gib, out _, out var percentFree)); Assert.Equal(9.9, percentFree, precision: 6); } + [Fact] + public void IsDiskPressure_BigVolumeAtLowPercent_IsNotPressure_TheGbFloorQualifies() + { + /* #3528's own example, scaled: 99 GiB free on a 1000 GiB store volume is 9.9% — below the percent + threshold — but 99 GiB of runway is nothing to page CRITICAL about, and it is ABOVE the shipped + 50 GB floor, so the composed shipped default stays quiet. Before #3528 this exact call fired. */ + Assert.False(DarlingSelfAlertEvaluator.IsDiskPressure(99 * Gib, 1000 * Gib, out _, out var percentFree)); + + /* Measured whenever the total is usable, firing or not (#1881). */ + Assert.Equal(9.9, percentFree, precision: 6); + + /* The floor only QUALIFIES: once free space is genuinely below it too, the same volume fires. */ + Assert.True(DarlingSelfAlertEvaluator.IsDiskPressure(45 * Gib, 1000 * Gib, out _, out _)); + } + + [Fact] + public void IsDiskPressure_FloorOfZero_RestoresThePercentOnlyCondition() + { + /* 0 removes the floor (the pvs_floor_gb reading), so the pre-#3528 percent-only behaviour is one + setting away — and the percent-only overload is that same condition, pinned equal here. */ + Assert.True(DarlingSelfAlertEvaluator.IsDiskPressure( + 99 * Gib, 1000 * Gib, DarlingSelfAlertEvaluator.DiskFreeWarnPercent, 0.0, out _, out _)); + Assert.True(DarlingSelfAlertEvaluator.IsDiskPressure( + 99 * Gib, 1000 * Gib, DarlingSelfAlertEvaluator.DiskFreeWarnPercent, out _, out _)); + } + [Fact] public void IsDiskPressure_PlentyFree_NotPressure() { @@ -892,6 +921,36 @@ public async Task DiskPressure_FiresOnce_ThenStaysQuietAtUnchangedLevel_ReFiresO Assert.Equal(2, h.Deliverer.Outcomes.Count); } + [Fact] + public async Task DiskPressure_ReadsTheGbFloorThroughTheSettingsSeam() + { + var h = new Harness(); + var e = h.Build(); + + /* 9% free on a 1 TiB store volume: the percent is breached, but ~92 GiB of runway sits above the + fake's 50 GB floor — the sweep stays quiet. This drives the SEAM (the store-backed knob the + sweep reads), not the constant the pure tests pin, and the fired threshold text below is what + proves which condition judged. */ + await e.ApplyDiskPressureAsync(92 * Gib, 1024 * Gib, null, Ct); + Assert.Empty(h.Deliverer.Outcomes); + + /* An operator setting the floor to 0 restores the percent-only condition on the next sweep — + read live through the by-reference seam, no rebuild. */ + h.Settings.SelfDiskFreeWarnGb = 0; + await e.ApplyDiskPressureAsync(92 * Gib, 1024 * Gib, null, Ct); + var fired = Assert.Single(h.Deliverer.Outcomes); + Assert.Equal("Store Disk Pressure", fired.MetricName); + /* With the floor off the threshold string names the percent alone... */ + Assert.Equal("10% free", fired.ThresholdValue); + + /* ...and with it on, a breach BELOW both gates fires and the string names both. */ + var h2 = new Harness(); + var e2 = h2.Build(); + await e2.ApplyDiskPressureAsync(40 * Gib, 1024 * Gib, null, Ct); + var both = Assert.Single(h2.Deliverer.Outcomes); + Assert.Equal("10% free and under 50 GB", both.ThresholdValue); + } + /* ---------------- custom-alert-rule health edge (#3304) ---------------- */ private static CustomAlertHealthReport HealthReport(int broken, int neverFiring) diff --git a/Darling/Darling.Tests/PgAlertCountKnobRungTests.cs b/Darling/Darling.Tests/PgAlertCountKnobRungTests.cs index 2eba2622c..180f52ebb 100644 --- a/Darling/Darling.Tests/PgAlertCountKnobRungTests.cs +++ b/Darling/Darling.Tests/PgAlertCountKnobRungTests.cs @@ -230,17 +230,20 @@ public void TheMcpWriteBound_TheSettingsClamp_AndTheViewerGate_AreTheSameConstan var window = RepoFile.ReadRepoFile( "Darling", "PerformanceMonitor.Darling.Viewer", "SettingsWindow.xaml.cs"); - /* The floor appears TWICE per surface — once per knob. ONE is what a half-migration looks like: - the deadlock knob bounded by the constant and the blocking knob by a literal beside it. The - surface name rides in the failure message so a count mismatch does not send the reader to - three files. */ - foreach (var (what, text) in new[] + /* The floor appears once per knob per surface — FEWER is what a half-migration looks like: one + knob bounded by the constant and its sibling by a literal beside it. #3528 floored the two SQL + Server twins with the same constant, so the engine clamp and the MCP write bound now reach it + four times (pg + SQL, blocking + deadlocks) while the Settings window still gates its SQL boxes + with the numerically-identical `> 0` — the viewer pass is deliberately deferred (backend-first), + and the window's two constant references remain the PG boxes'. The surface name rides in the + failure message so a count mismatch does not send the reader to three files. */ + foreach (var (what, text, perSurface) in new[] { - ("engine clamp", settings), ("mcp write bound", tools), ("settings window gate", window), + ("engine clamp", settings, 4), ("mcp write bound", tools, 4), ("settings window gate", window, 2), }) { Assert.True( - CountOf(text, "PostgresAlertEvaluator.CountThresholdFloor") == 2, + CountOf(text, "PostgresAlertEvaluator.CountThresholdFloor") == perSurface, $"the {what} must reach the shared floor constant once per knob, not a literal"); } @@ -367,8 +370,8 @@ public void TheShippedDefaults_ReproduceTheFireAtOneBehaviour() /// /// The clamp floors a hand-edited store row, so the gate cannot fire on a count of zero — the failure /// the floor exists for, stated on 's own doc. - /// The SQL Server twins take the same write bound and have no read-side floor; that asymmetry is named - /// on DarlingAlertSettings rather than copied here. + /// Since #3528 the SQL Server twins floor at the same constant, so all four count gates are asserted + /// together — the asymmetry this doc used to record is gone. /// [Theory] [InlineData(0)] @@ -378,10 +381,16 @@ public void AHandEditedRowBelowTheFloor_ReadsAsTheFloor_SoNothingFiresOnNothing( var config = new DarlingConfig(); config.Alerts.PgDeadlockCountThreshold = stored; config.Alerts.PgBlockingCountThreshold = stored; + config.Alerts.DeadlockCountThreshold = stored; + config.Alerts.BlockingCountThreshold = stored; var settings = new DarlingAlertSettings(config); Assert.Equal(PostgresAlertEvaluator.CountThresholdFloor, settings.PgDeadlockCountThreshold); Assert.Equal(PostgresAlertEvaluator.CountThresholdFloor, settings.PgBlockingCountThreshold); + /* #3528: the SQL Server twins, floored at the same constant — a store row hand-edited to 0 no + longer makes count >= threshold true on a deadlock-free (or blocking-free) server. */ + Assert.Equal(PostgresAlertEvaluator.CountThresholdFloor, settings.DeadlockCountThreshold); + Assert.Equal(PostgresAlertEvaluator.CountThresholdFloor, settings.BlockingCountThreshold); Assert.False(RollingCountAlertGate.Evaluate( 0, settings.PgDeadlockCountThreshold, watermark: 0, cooldownElapsed: true, suppressed: false).Fire); diff --git a/Darling/Darling.Tests/SelfDiskWarnGbFloorRungTests.cs b/Darling/Darling.Tests/SelfDiskWarnGbFloorRungTests.cs new file mode 100644 index 000000000..094165dcb --- /dev/null +++ b/Darling/Darling.Tests/SelfDiskWarnGbFloorRungTests.cs @@ -0,0 +1,248 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Globalization; +using System.Linq; +using System.Reflection; +using PerformanceMonitor.Darling.Service; +using PerformanceMonitor.Darling.Service.Mcp; +using PerformanceMonitor.Darling.Storage; +using PerformanceMonitor.Darling.Viewer; +using Xunit; + +namespace Darling.Tests; + +/// +/// V126 / #3528: the Store Disk Pressure warning's GB floor moves onto config.config_alert_settings. +/// +/// The self-store warn condition was percent-only, its own comment calling a GB floor "a trivial +/// follow-up if an operator ever wants one" — and #3528's example is the want: 400 GB free on a 4 TB store +/// volume fired a CRITICAL "act now". The floor is an AND qualifier (the pvs_floor_gb composition, +/// deliberately not the target-volume pair's OR, whose GB dimension ADDS fires), so a large volume at a +/// low percent stays quiet until absolute free space is genuinely short; 0 removes the floor. +/// +/// This file carries the "I am the top rung" claims that moved off +/// (V125) when this rung landed, the same handoff that file +/// received from (V124) — a fully-migrated store must map to +/// EXACTLY this version, or the viewer's connect-time gate refuses a store that is actually current. +/// +public sealed class SelfDiskWarnGbFloorRungTests +{ + private const int RungVersion = 126; + private const int PreviousVersion = 125; + + /// This rung's sentinel ordinal in the viewer probe — the newest, so the last argument. + private const int ProbeOrdinal = 101; + + private const string FloorColumn = "self_disk_free_warn_gb"; + + /* ---- the rung ------------------------------------------------------------------------------------ */ + + [Fact] + public void TheRungIsRegisteredAtTheTopOfADenseLadder() + { + var versions = PgMigrations.Scripts.Select(s => s.Version).ToList(); + + Assert.Equal( + "self-disk-warn-gb-floor", + PgMigrations.Scripts.Single(s => s.Version == RungVersion).Name); + + Assert.Equal(StorageVersion.SchemaVersion, PgMigrations.Scripts[^1].Version); + Assert.Equal(StorageVersion.SchemaVersion, versions.Max()); + Assert.Equal(RungVersion, StorageVersion.SchemaVersion); + + Assert.Equal(versions.Distinct().OrderBy(v => v), versions); + } + + /// + /// The rung adds ONE column to the singleton settings row, schema-qualified, with the shipped constant + /// as its default. + /// + /// The DEFAULT is compared against DarlingSelfAlertEvaluator.DiskFreeWarnFloorGb rather + /// than a literal (restated in the rung only because a rung is a SQL string), so a moved shipped + /// default cannot leave upgraded stores qualifying at a floor no surface reports. The default is + /// NON-ZERO on upgrade deliberately, unlike V122's knobs: their acceptance was "an untouched store + /// fires exactly where it did", while #3528's is that the untouched firing IS the defect — the issue's + /// own example is a default-configured store paging with 400 GB of runway. Any store volume at or + /// under 500 GB (floor ÷ warn percent) keeps the exact pre-#3528 percent behaviour. + /// + [Fact] + public void TheRungAddsTheColumn_SchemaQualified_WithTheShippedConstantAsDefault() + { + var rung = PgMigrations.Scripts.Single(s => s.Version == RungVersion).Sql; + + /* Schema-qualified for the reason every config rung is: the migrate session's search_path puts + collect first, so a bare name would resolve to the wrong schema (and the wrong ACL). */ + Assert.Equal(1, CountOf(rung, "ALTER TABLE config.config_alert_settings")); + Assert.DoesNotContain("ALTER TABLE config_alert_settings", rung, StringComparison.Ordinal); + + /* IF NOT EXISTS so re-running the ladder over a store that already has it is a no-op rather than + a 42701 that aborts the whole migration. integer, matching self_disk_free_warn_percent and the + low-disk GB columns on this same table — a whole-GB knob has no meaningful fractional part. */ + Assert.Equal(1, CountOf(rung, "ADD COLUMN IF NOT EXISTS")); + Assert.DoesNotContain("double precision", rung, StringComparison.Ordinal); + Assert.Contains( + $"ADD COLUMN IF NOT EXISTS {FloorColumn} integer NOT NULL DEFAULT " + + ((int)DarlingSelfAlertEvaluator.DiskFreeWarnFloorGb).ToString(CultureInfo.InvariantCulture) + ";", + rung, StringComparison.Ordinal); + + /* And the C# seed names the same figure, so a fresh file-plane config and an upgraded store row + agree without either citing the other. The constant is whole-valued by construction — the cast + in the assertion above must not be hiding a fractional shipped default. */ + Assert.Equal(DarlingSelfAlertEvaluator.DiskFreeWarnFloorGb, (int)DarlingSelfAlertEvaluator.DiskFreeWarnFloorGb); + Assert.Equal((int)DarlingSelfAlertEvaluator.DiskFreeWarnFloorGb, new AlertsConfig().SelfDiskFreeWarnGb); + + /* No reload beacon of its own: config_alert_settings already carries V17's statement-level + trg_bump_alert_settings, so a second trigger here would be a duplicate bump per write. */ + Assert.DoesNotContain("config_bump_version", rung, StringComparison.Ordinal); + + /* And no per-table GRANT: this table carries table-level grants with no column carve, which is + what every earlier knob rung on it says. */ + Assert.DoesNotContain("GRANT", rung, StringComparison.Ordinal); + } + + /* ---- the probe (three sites, top arm) ------------------------------------------------------------- */ + + /// + /// The viewer probe's three sites carry this rung's sentinel, and the map treats it as the TOP arm. + /// + /// The probe asks the question, the caller reads the answer, the map has the parameter — three + /// sites, and a sentinel present at only some of them shifts every LATER ordinal onto the wrong column. + /// Miss all three and a fully-migrated store probes one rung short, so the connect-time gate refuses a + /// store that is in fact current — permanently, because no later upgrade changes the answer. + /// + [Fact] + public void TheProbeMapsAFullyMigratedStoreToThisTopRung() + { + Assert.Contains($"column_name = '{FloorColumn}'", ViewerDataService.StoreSchemaProbeSql, StringComparison.Ordinal); + + var viewer = RepoFile.ReadRepoFile("Darling", "PerformanceMonitor.Darling.Viewer", "ViewerDataService.cs"); + Assert.Contains($"reader.GetBoolean({ProbeOrdinal})", viewer, StringComparison.Ordinal); + Assert.Contains("hasSelfDiskWarnGbFloor", viewer, StringComparison.Ordinal); + + Assert.Equal(StorageVersion.SchemaVersion, ViewerDataService.RequiredStoreSchemaVersion); + + var method = typeof(ViewerDataService) + .GetMethod("MapProbedSchemaVersion", BindingFlags.NonPublic | BindingFlags.Static)!; + var arity = method.GetParameters().Length; + + /* The top rung's sentinel IS the last argument. */ + Assert.Equal(ProbeOrdinal, arity - 1); + + /* Every sentinel true = a fully-migrated store, which must map to exactly this version. Built by + reflection so the arity tracks the signature. */ + var all = Enumerable.Repeat((object)true, arity).ToArray(); + Assert.Equal(StorageVersion.SchemaVersion, (int)method.Invoke(null, all)!); + + /* This rung's own arm answers for a store that stopped here. Expressed as "false above" rather than + as one named ordinal, so a rung landing on top of this one does not quietly turn this case into a + test of that rung. */ + var atThisRung = Enumerable.Range(0, arity).Select(i => (object)(i <= ProbeOrdinal)).ToArray(); + Assert.Equal(RungVersion, (int)method.Invoke(null, atThisRung)!); + + /* One rung behind: the same store WITHOUT this rung's sentinel reports the previous rung. */ + var behind = (object[])atThisRung.Clone(); + behind[ProbeOrdinal] = false; + Assert.Equal(PreviousVersion, (int)method.Invoke(null, behind)!); + + /* And in the source, the arm sits ABOVE V125's — newest-first is the whole contract of that method — + and returns this build's version rather than a literal that could drift from it. This is the + textual half of the top-arm claim, inherited from CollectorDatabaseScopeRungTests the way that + file inherited it from FleetSweepCadenceKnobRungTests. */ + var v126 = viewer.IndexOf("if (hasSelfDiskWarnGbFloor)", StringComparison.Ordinal); + var v125 = viewer.IndexOf("if (hasCollectorScheduleDatabases)", StringComparison.Ordinal); + Assert.True(v126 >= 0, "the viewer has no V126 sentinel arm — a fully-migrated store would map to 125"); + Assert.True(v125 >= 0, "the V125 arm is gone, so this pin is comparing against nothing"); + Assert.True(v126 < v125, "the V126 arm sits below V125's, so a current store maps one rung low"); + Assert.Contains( + "return " + StorageVersion.SchemaVersion.ToString(CultureInfo.InvariantCulture) + ";", + viewer[v126..], StringComparison.Ordinal); + } + + /* ---- every settings-row surface handles the column ------------------------------------------------ */ + + /// + /// EVERY wired surface that reads or writes the settings row names the column — and the ONE surface + /// deliberately not wired yet is pinned to its abstinence. The wired lists drive ordinals or parameter + /// positions, so a column added to one and not the others re-maps reads and writes at once. + /// + /// The viewer's select/upsert is the pinned abstinence, unlike every earlier knob rung: + /// this knob lands backend-first (store plane + the two MCP tools), and the Settings window's box + /// follows in the viewer pass. The viewer's explicit column lists mean its select and save are + /// untouched by the new column — nothing throws, and a viewer Save cannot null the floor out. When the + /// viewer pass wires the box, this assertion is where that decision flips. + /// + [Fact] + public void EveryWiredSettingsRowSurfaceNamesTheColumn_AndTheViewerAbstains() + { + var service = RepoFile.ReadRepoFile( + "Darling", "PerformanceMonitor.Darling.Service", "StoreConfigProvider.cs"); + var tools = RepoFile.ReadRepoFile( + "Darling", "PerformanceMonitor.Darling.Service", "Mcp", "DarlingMcpAlertTools.cs"); + + /* The service must READ it, not merely select it — ApplyToConfig replaces config.Alerts wholesale, + so a selected-but-unread column resets the floor to the shipped default on every worker start. + Both halves, because one of the two being present is what an off-by-one produces. */ + Assert.Contains(FloorColumn, service, StringComparison.Ordinal); + Assert.Contains("SelfDiskFreeWarnGb = reader.GetInt32(", service, StringComparison.Ordinal); + + /* The MCP read names the column and the report/accept pair carries the wire key — readable AND + writable, because a read-only knob leaves the UPDATE in someone's runbook. */ + Assert.Contains(FloorColumn, DarlingAlertReader.AlertSettingsSelectSql, StringComparison.Ordinal); + Assert.Contains("disk_free_warn_gb = s.SelfDiskFreeWarnGb", tools, StringComparison.Ordinal); + Assert.Contains( + "case \"disk_free_warn_gb\": AddInt(\"self_disk_free_warn_gb\", n, \"self_alerts.disk_free_warn_gb\", 0, int.MaxValue); break;", + tools, StringComparison.Ordinal); + + /* The deliberate abstinence: the viewer's settings surface does not name the column yet. */ + var viewerSettings = RepoFile.ReadRepoFile( + "Darling", "PerformanceMonitor.Darling.Viewer", "ViewerDataService.AlertSettings.cs"); + Assert.DoesNotContain(FloorColumn, viewerSettings, StringComparison.Ordinal); + } + + /* ---- the seam reaches the gate -------------------------------------------------------------------- */ + + /// + /// The settings adapter defaults to the shipped constant and clamps a hand-edited store value at the + /// 0 floor — the raw-in/clamped-out split every knob on this table uses, with 0 IN range because it + /// removes the floor (the pvs_floor_gb reading) rather than being nonsense. The write bound in + /// is the same + /// [0, int.MaxValue], so no accepted value is one this clamp rewrites. + /// + [Fact] + public void TheSettingsSeamDefaultsToTheConstant_AndClampsAtZero() + { + var config = new DarlingConfig(); + var settings = new DarlingAlertSettings(config); + + Assert.Equal((int)DarlingSelfAlertEvaluator.DiskFreeWarnFloorGb, settings.SelfDiskFreeWarnGb); + + config.Alerts.SelfDiskFreeWarnGb = -5; + Assert.Equal(0, settings.SelfDiskFreeWarnGb); + + config.Alerts.SelfDiskFreeWarnGb = 0; + Assert.Equal(0, settings.SelfDiskFreeWarnGb); + + config.Alerts.SelfDiskFreeWarnGb = 400; + Assert.Equal(400, settings.SelfDiskFreeWarnGb); + } + + private static int CountOf(string haystack, string needle) + { + var count = 0; + for (var at = haystack.IndexOf(needle, StringComparison.Ordinal); + at >= 0; + at = haystack.IndexOf(needle, at + needle.Length, StringComparison.Ordinal)) + { + count++; + } + + return count; + } +} diff --git a/Darling/Darling.Tests/ServerHealthClassifierTests.cs b/Darling/Darling.Tests/ServerHealthClassifierTests.cs index 237c786b8..5819c80a3 100644 --- a/Darling/Darling.Tests/ServerHealthClassifierTests.cs +++ b/Darling/Darling.Tests/ServerHealthClassifierTests.cs @@ -198,6 +198,69 @@ public void OverallMetricSeverity_AllCalm_IsHealthy_UnknownNeverEscalates() Assert.Equal(HealthSeverity.Healthy, ServerHealthClassifier.OverallMetricSeverity(m)); } + /* ── measured-metric coverage (#3528) ── */ + + [Fact] + public void MeasuredMetricCounts_FullyMeasuredBundle_CountsAllSix() + { + var m = new ServerHealthMetrics + { + CpuPercentForAlert = 50, + TotalThreads = 512, + AvailableThreads = 400, + HasMemoryPressure = false, + BlockingCount = 0, + DeadlockCount = 0, + DeadlockWindow = TimeSpan.FromHours(1), + }; + + Assert.Equal((6, 6), ServerHealthClassifier.MeasuredMetricCounts(m)); + } + + [Fact] + public void MeasuredMetricCounts_UnknownHeavyBundle_SaysSo_WhileTheFoldStillReadsHealthy() + { + /* The PostgreSQL-card shape #3528 was filed about: five of the six metrics structurally Unknown + (no CPU/threads snapshot, DMV-sourced memory/blocking/deadlocks nulled), only the collector row + measured. The fold deliberately skips Unknown, so the band label is still Healthy — and the + counts are what let a consumer render that label as "Healthy — 1 of 6 measured" instead of an + unqualified green. */ + var m = new ServerHealthMetrics(); + + Assert.Equal((1, 6), ServerHealthClassifier.MeasuredMetricCounts(m)); + + var overall = ServerHealthClassifier.OverallMetricSeverity(m); + Assert.Equal(HealthSeverity.Healthy, overall); + Assert.Equal(FleetHealthBand.Healthy, + ServerHealthClassifier.ClassifyBand(isOnline: true, awaitingFirstCollection: false, collectionStale: false, overall)); + } + + [Fact] + public void MeasuredMetricCounts_AreRankNeutral() + { + /* The counts describe, they never rank: two bundles differing only in how many metrics are + measured score identically, which is the Unknown rank-neutrality + UnmeasuredMetricsAreNotHealthyTests pins, restated against the new fields' own inputs. */ + var measured = new ServerHealthMetrics + { + CpuPercentForAlert = 50, + TotalThreads = 512, + AvailableThreads = 400, + HasMemoryPressure = false, + BlockingCount = 0, + DeadlockCount = 0, + DeadlockWindow = TimeSpan.FromHours(1), + }; + var unmeasured = new ServerHealthMetrics(); + + Assert.NotEqual( + ServerHealthClassifier.MeasuredMetricCounts(measured), + ServerHealthClassifier.MeasuredMetricCounts(unmeasured)); + Assert.Equal( + ServerHealthClassifier.FleetHealthScore(FleetHealthBand.Healthy, measured), + ServerHealthClassifier.FleetHealthScore(FleetHealthBand.Healthy, unmeasured)); + } + /* ── fleet band (collapse) ── */ [Fact] diff --git a/Darling/Darling.Tests/StoreSelfMetricsTests.cs b/Darling/Darling.Tests/StoreSelfMetricsTests.cs index 90683d6b5..885290f00 100644 --- a/Darling/Darling.Tests/StoreSelfMetricsTests.cs +++ b/Darling/Darling.Tests/StoreSelfMetricsTests.cs @@ -751,6 +751,7 @@ private sealed class CadenceFakeSettings : IAlertEngineSettings public int DiskCriticalFreePercent { get; set; } = 3; public int DiskCriticalFreeGb { get; set; } = 2; public int SelfDiskFreeWarnPercent { get; set; } = 10; + public int SelfDiskFreeWarnGb { get; set; } = 50; public int CollectionStaleMinutes { get; set; } = 30; public int CollectionFailureThreshold { get; set; } = 10; public int PvsThresholdPercent { get; set; } = 40; diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingAlertSettings.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingAlertSettings.cs index a422342a5..df1462435 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingAlertSettings.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingAlertSettings.cs @@ -56,12 +56,26 @@ public DarlingAlertSettings(DarlingConfig config) public bool ForcePlanFailureEnabled => true; public int CpuThresholdPercent => _config.Alerts.CpuThresholdPercent; - public int BlockingCountThreshold => _config.Alerts.BlockingCountThreshold; + + /// #3528: floored at the same named constant as its PostgreSQL twin below, for the identical + /// reason — a store row hand-edited to 0 makes the gate's count >= threshold test true for a + /// count of zero and fires on a server with no blocking. The floor is also the + /// update_alert_settings lower write bound and the Settings window's save gate, so no value any + /// writer accepts is a value this clamp then rewrites. + public int BlockingCountThreshold => Math.Max( + PostgresAlertEvaluator.CountThresholdFloor, + _config.Alerts.BlockingCountThreshold); /* #1839: floored at 0 (= off) so a negative in darling.json or the store can't make the "is it above threshold" test true for every snapshot. */ public int BlockingWaitSecondsThreshold => Math.Max(0, _config.Alerts.BlockingWaitSecondsThreshold); - public int DeadlockCountThreshold => _config.Alerts.DeadlockCountThreshold; + + /// #3528: floored like above and + /// below — the constant lives on + /// only by birthplace; the failure it closes is engine-neutral. + public int DeadlockCountThreshold => Math.Max( + PostgresAlertEvaluator.CountThresholdFloor, + _config.Alerts.DeadlockCountThreshold); public int PoisonWaitThresholdMs => _config.Alerts.PoisonWaitThresholdMs; public int LongRunningQueryThresholdMinutes => _config.Alerts.LongRunningQueryThresholdMinutes; public int TempDbSpaceThresholdPercent => _config.Alerts.TempDbSpaceThresholdPercent; @@ -75,6 +89,12 @@ public DarlingAlertSettings(DarlingConfig config) public int DiskCriticalFreePercent => Math.Clamp(_config.Alerts.DiskCriticalFreePercent, 0, 100); public int DiskCriticalFreeGb => Math.Max(0, _config.Alerts.DiskCriticalFreeGb); public int SelfDiskFreeWarnPercent => Math.Clamp(_config.Alerts.SelfDiskFreeWarnPercent, 0, 100); + + /// #3528: the store warning's GB floor — Store Disk Pressure fires only when the percent above + /// is breached AND free space is below this many GB, so a large volume at a low percent (400 GB free on + /// a 4 TB store) stops paging CRITICAL. Zero removes the floor (the percent-only pre-#3528 condition); + /// the 0-floor GB shape is 's, whose AND-qualifier composition this mirrors. + public int SelfDiskFreeWarnGb => Math.Max(0, _config.Alerts.SelfDiskFreeWarnGb); public int CollectionStaleMinutes => Math.Clamp(_config.Alerts.CollectionStaleMinutes, 5, 1440); /// #2136: the Store Job Over Cadence warning percent. Clamped [5, 100]. @@ -122,16 +142,13 @@ public DarlingAlertSettings(DarlingConfig config) TimescaleSupport.RetentionHoldRatioFloor, TimescaleSupport.RetentionHoldRatioCeiling); - /// #3444 (V122): the PostgreSQL Deadlocks alert's rolling-window count threshold. - /// - /// Floored, where its SQL Server twin is not. passes - /// _config.Alerts through raw, so a store row hand-edited to 0 makes the gate's - /// count >= threshold test true for a count of zero and fires on a server with no deadlocks. - /// That is a pre-existing gap on the twin rather than a shape to copy: the floor here matches - /// two screens up, which floors for the identical reason. - /// The floor is also the update_alert_settings lower write bound, so no value the write path - /// accepts is a value this clamp then rewrites — the "setting did not stick" failure the write-bound - /// parity pins exist for. + /// #3444 (V122): the PostgreSQL Deadlocks alert's rolling-window count threshold, floored so a + /// store row hand-edited to 0 cannot make the gate's count >= threshold test true for a count + /// of zero and fire on a server with no deadlocks. Its SQL Server twin + /// () floors at the same constant since #3528, so the two engines' + /// gates are now the one shape. The floor is also the update_alert_settings lower write bound, + /// so no value the write path accepts is a value this clamp then rewrites — the "setting did not stick" + /// failure the write-bound parity pins exist for. public int PgDeadlockCountThreshold => Math.Max( PostgresAlertEvaluator.CountThresholdFloor, _config.Alerts.PgDeadlockCountThreshold); diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingConfig.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingConfig.cs index 60613fcd3..8f938ba17 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingConfig.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingConfig.cs @@ -615,10 +615,18 @@ public sealed class AlertsConfig public int FileGrowthLookbackMinutes { get; set; } = 60; /// #2107: the store volume's self-alert warning percent (was a compile-time 10.0; - /// 0 disables the check — percent is its only trigger). + /// 0 disables the check). [JsonPropertyName("selfDiskFreeWarnPercent")] public int SelfDiskFreeWarnPercent { get; set; } = 10; + /// #3528: the store warning's GB floor — the percent above additionally requires free space + /// below this many GB, an AND qualifier so a large volume at a low percent never pages (0 removes the + /// floor). The PVS-floor composition, not the target pair's OR. 50 puts the crossover at a 500 GB + /// store volume: below that the percent governs exactly as before; above it, 50 GB free is the line — + /// which is what stops 400 GB free on a 4 TB volume reading as "act now". + [JsonPropertyName("selfDiskFreeWarnGb")] + public int SelfDiskFreeWarnGb { get; set; } = 50; + /// #2107: how long collection may go quiet before Collection Stopped / Agent Not /// Running fire (was a compile-time 30 minutes). Defaults to the shared constant behind the /// display's Offline band, so the two definitions of "collection stopped" agree out of the diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingSelfAlertEvaluator.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingSelfAlertEvaluator.cs index 62efe5959..bf5e4c3c9 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingSelfAlertEvaluator.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingSelfAlertEvaluator.cs @@ -94,15 +94,22 @@ faster than the staleness backstop when a connected server's collectors are erro cycle. 10 spans a couple of minutes of total failure across the frequently-scheduled collectors. */ internal const int ConsecutiveFailureThreshold = 10; - /* Store Disk Pressure fires when the store volume drops below this percent free — a percentage (not an - absolute floor) so it scales from a small managed box to a large fleet disk; 10% is the universal DBA - "act now" threshold for a database volume, and mirrors the shared engine's target-server low-disk - percent (LowDiskThresholdPercent). Percent-only by design (defaults over speculative config — a GB - floor is a trivial follow-up if an operator ever wants one). The condition no-ops when free space is - undeterminable (a remote BYO store), so it never false-alarms; the managed store's own volume is the - case it exists to protect. */ + /* Store Disk Pressure fires when the store volume drops below this percent free — a percentage so it + scales from a small managed box to a large fleet disk; 10% is the universal DBA "act now" threshold + for a database volume, and mirrors the shared engine's target-server low-disk percent + (LowDiskThresholdPercent). The condition no-ops when free space is undeterminable (a remote BYO + store), so it never false-alarms; the managed store's own volume is the case it exists to protect. */ internal const double DiskFreeWarnPercent = 10.0; + /* #3528: the percent's GB floor — pressure additionally requires free space BELOW this many GB, an AND + qualifier so a big volume at a low percent (400 GB free on a 4 TB store) never pages CRITICAL. This + was "percent-only by design (a GB floor is a trivial follow-up if an operator ever wants one)"; #3528 + is that want. The composition is PvsFloorGb's (percent triggers, the floor keeps it honest, 0 removes + the floor), deliberately NOT the target-volume pair's OR — there the GB dimension ADDS fires, which + would make this alert noisier, the opposite of the complaint. 50 puts the crossover at a 500 GB + volume: below that the percent governs exactly as before; above it, 50 GB free is the line. */ + internal const double DiskFreeWarnFloorGb = 50.0; + private readonly IAlertEngineSettings _settings; private readonly IAlertDeliverer _deliverer; private readonly IAlertHistoryStore _historyStore; @@ -2697,10 +2704,11 @@ private static string DescribeAgDatabaseKey(string key) /* ---------------- store disk pressure (fleet-level, polled) ---------------- */ /// - /// Pure disk-pressure decision: the store volume is under pressure when its FREE space is below - /// of the volume total. No I/O, so it pins directly. A non-positive - /// total is treated as "can't tell" (false — the caller also guards this). The percentage scales across - /// disk sizes; see the constant for why it is percent-only. + /// Pure disk-pressure decision at the SHIPPED defaults: the store volume is under pressure when its + /// FREE space is below of the volume total AND below + /// absolute (#3528 — see the floor constant for the composition). + /// No I/O, so it pins directly. A non-positive total is treated as "can't tell" (false — the caller + /// also guards this). /// is the measurement the alert is ABOUT, handed back so the fire /// site can store it as a real numeric instead of leaving the history store to find it again by /// scanning for digits (#1881). It is computed whenever the total is usable, @@ -2709,12 +2717,16 @@ private static string DescribeAgDatabaseKey(string key) /// the one dangerous ambiguity this metric must never have back into the signature. /// internal static bool IsDiskPressure(long freeBytes, long totalBytes, out string reason, out double percentFree) - => IsDiskPressure(freeBytes, totalBytes, DiskFreeWarnPercent, out reason, out percentFree); + => IsDiskPressure(freeBytes, totalBytes, DiskFreeWarnPercent, DiskFreeWarnFloorGb, out reason, out percentFree); - /// #2107: the configurable form — the sweep passes the store-backed - /// SelfDiskFreeWarnPercent; the constant-threshold overload keeps the shipped default - /// for the tests pinning it. + /// #2107: the percent-only form — floor disabled, kept for the tests that pin the percent + /// edge on its own. The sweep calls the two-knob overload below. internal static bool IsDiskPressure(long freeBytes, long totalBytes, double warnPercent, out string reason, out double percentFree) + => IsDiskPressure(freeBytes, totalBytes, warnPercent, 0.0, out reason, out percentFree); + + /// #3528: the configurable form — the sweep passes the store-backed + /// SelfDiskFreeWarnPercent AND SelfDiskFreeWarnGb (0 = no floor). + internal static bool IsDiskPressure(long freeBytes, long totalBytes, double warnPercent, double floorGb, out string reason, out double percentFree) { if (totalBytes <= 0) { @@ -2724,7 +2736,8 @@ internal static bool IsDiskPressure(long freeBytes, long totalBytes, double warn } percentFree = (double)freeBytes / totalBytes * 100.0; - if (percentFree < warnPercent) + double freeGb = freeBytes / (1024.0 * 1024.0 * 1024.0); + if (percentFree < warnPercent && (floorGb <= 0 || freeGb < floorGb)) { reason = $"The monitor store's disk volume has only {percentFree.ToString("0.#", CultureInfo.InvariantCulture)}% free ({FormatGb(freeBytes)} of {FormatGb(totalBytes)})."; return true; @@ -2791,10 +2804,11 @@ internal async Task ApplyDiskPressureAsync( } var now = _utcNow(); - /* #2107: store-backed threshold (clamped on read); the constant remains only as the - shipped default. */ + /* #2107/#3528: store-backed thresholds (clamped on read); the constants remain only as the + shipped defaults. */ double warnPercent = _settings.SelfDiskFreeWarnPercent; - bool pressure = IsDiskPressure(free, total, warnPercent, out var reason, out var percentFree); + double floorGb = _settings.SelfDiskFreeWarnGb; + bool pressure = IsDiskPressure(free, total, warnPercent, floorGb, out var reason, out var percentFree); if (pressure) { @@ -2827,7 +2841,11 @@ shipped default. */ : ""; await FireAsync( StoreKey(DiskKey), _storeLabel, DiskPressureMetric, reason, - $"{warnPercent.ToString("0.#", CultureInfo.InvariantCulture)}% free", + /* #3528: the threshold string names BOTH gates when the floor is active, so the history + row's threshold column states the condition that actually fired. */ + floorGb > 0 + ? $"{warnPercent.ToString("0.#", CultureInfo.InvariantCulture)}% free and under {floorGb.ToString("0.#", CultureInfo.InvariantCulture)} GB" + : $"{warnPercent.ToString("0.#", CultureInfo.InvariantCulture)}% free", detail: reason + storeText + " When the store volume fills, collection and every write stop " + "for the WHOLE fleet, and a headless service has no dashboard to warn you. Free space on the " + "store volume, shorten retention (config_collector_schedules), enable TimescaleDB compression, " + diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingAlertReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingAlertReader.cs index ca09dc0f4..166fa2998 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingAlertReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingAlertReader.cs @@ -178,7 +178,9 @@ and the two engines carry separate calibrations on purpose (see the V122 rung). /* #3466 (V124): the fleet sweep's cadence knobs. APPENDED, same reason. Darling-only: Lite has no fleet to sweep, so McpAlertSettingsKeyTests records the omitted group as a decision. */ bool FleetSweepEnabled, - int FleetSweepIntervalMinutes); + int FleetSweepIntervalMinutes, + /* #3528 (V126): the Store Disk Pressure warning's GB floor. APPENDED, same reason. */ + int SelfDiskFreeWarnGb); /// The single global alert-settings row (id=1) — the viewer's AlertSettingsSelectSql. The /// columns are read in the SAME order the service reads them (StoreConfigProvider), and @@ -209,7 +211,8 @@ and the two engines carry separate calibrations on purpose (see the V122 rung). retention_hold_warn_ratio, retention_hold_critical_ratio, deadlock_warn_per_hour, deadlock_critical_per_hour, pg_deadlock_count_threshold, pg_blocking_count_threshold, - fleet_sweep_enabled, fleet_sweep_interval_minutes + fleet_sweep_enabled, fleet_sweep_interval_minutes, + self_disk_free_warn_gb FROM config_alert_settings WHERE id = 1"; @@ -260,7 +263,9 @@ FROM config_alert_settings /* #3444: V122 PostgreSQL Deadlocks/Blocking count thresholds at 62–63. */ reader.GetInt32(62), reader.GetInt32(63), /* #3466: V124 fleet-sweep cadence knobs at 64–65. */ - reader.GetBoolean(64), reader.GetInt32(65)); + reader.GetBoolean(64), reader.GetInt32(65), + /* #3528: V126 store-disk-warn GB floor at 66. */ + reader.GetInt32(66)); } /* ─────────────────────── delivery cooldown (a SECOND config table) ─────────────────────── */ diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingFleetReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingFleetReader.cs index ae7a07932..d455be0fd 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingFleetReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingFleetReader.cs @@ -547,6 +547,10 @@ because cpuForAlert there is percent of an allocation that moves. The source is var overall = ServerHealthClassifier.OverallMetricSeverity(metrics); var band = ServerHealthClassifier.ClassifyBand(isOnline, awaitingFirstCollection, collectionStale, overall); + /* #3528: the fold above skips Unknown, so an online server with five of six metrics structurally + Unknown still bands Healthy. The counts ride the card so every consumer of the band label can + qualify it ("Healthy — 1 of 6 measured") instead of rendering an unqualified green. */ + var (measuredMetrics, totalMetrics) = ServerHealthClassifier.MeasuredMetricCounts(metrics); /* Per-server platform (design D4): the reliable signal the composer's measure auto-greying matches a measure's appliesTo against — see ClassifyPlatform for the edition mapping and why AWS RDS / msdb are @@ -618,6 +622,8 @@ the same card correctly reads Unknown. The card chip and the viewer detail line FailedCollectorCount = collectors.Failing, CollectorSeverity = ServerHealthClassifier.CollectorSeverity(collectors.Failing), OverallMetricSeverity = overall, + MeasuredMetricCount = measuredMetrics, + MetricCount = totalMetrics, }; } @@ -1389,6 +1395,18 @@ public sealed class FleetServerCard [JsonPropertyName("overall_metric_severity")] public HealthSeverity OverallMetricSeverity { get; init; } + /// How many of the card's per-metric severities carried a real reading when it banded (#3528) + /// — the band's fold skips Unknown, so a card can read Healthy off one measured metric of six. When + /// this is below , the band label deserves the qualifier ("Healthy — 1 of 6 + /// measured"); the web fleet page renders exactly that. Purely descriptive: it feeds neither the band + /// nor the worst-first score, so rank-neutrality of Unknown is unchanged. + [JsonPropertyName("measured_metric_count")] public int MeasuredMetricCount { get; init; } + + /// The denominator for — how many per-metric severities the + /// card carries at all. Published rather than assumed at six so a consumer never hardcodes a figure the + /// next metric row changes. + [JsonPropertyName("metric_count")] public int MetricCount { get; init; } + /// The card's raw per-metric inputs, for re-scoring in the rollup (not serialized). [JsonIgnore] public ServerHealthMetrics ToHealthMetricsValue => ToHealthMetrics(); diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpAlertTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpAlertTools.cs index f63b07c28..0f610bcfd 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpAlertTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpAlertTools.cs @@ -257,6 +257,10 @@ invite tuning one and expecting the other to move. */ self_alerts = new { disk_free_warn_percent = s.SelfDiskFreeWarnPercent, + /* #3528 (V126): the percent's GB floor — pressure requires BOTH the percent above breached + AND free space below this many GB (0 removes the floor), so a large store volume at a low + percent stops paging CRITICAL. The pvs.floor_gb composition, not low_disk's OR pair. */ + disk_free_warn_gb = s.SelfDiskFreeWarnGb, collection_stale_minutes = s.CollectionStaleMinutes, collection_failure_threshold = s.CollectionFailureThreshold, /* #2136: the Store Job Over Cadence warning percent (Critical is fixed at 100). */ @@ -1317,7 +1321,10 @@ letting one silently win. */ switch (k) { case "enabled": AddBool("blocking_enabled", n, "blocking.enabled"); break; - case "count_threshold": AddInt("blocking_count_threshold", n, "blocking.count_threshold", 1, int.MaxValue); break; + /* #3528: the floor is the named constant rather than the literal 1 it always + was, because the read side now clamps to it — the same structural parity the + pg twin below has held since V122. */ + case "count_threshold": AddInt("blocking_count_threshold", n, "blocking.count_threshold", PostgresAlertEvaluator.CountThresholdFloor, int.MaxValue); break; /* #2417: get_alert_settings has emitted this key since #1839 and the writer never took it, so handing a whole read payload back -- the round trip this tool's own description tells the caller to perform -- was rejected with @@ -1328,8 +1335,7 @@ keeps disabling the second gate rather than becoming invalid. */ /* #3444 (V122): the PostgreSQL count gate. The floor is the SAME named constant DarlingAlertSettings clamps to, not a retyped 1, so this writer cannot ACCEPT a value the read-side clamp then rewrites. The twin above - takes the identical bound and has no read-side clamp; that gap is named on - DarlingAlertSettings rather than reproduced here. */ + holds the identical bound-and-clamp pair since #3528. */ case "pg_count_threshold": AddInt("pg_blocking_count_threshold", n, "blocking.pg_count_threshold", PostgresAlertEvaluator.CountThresholdFloor, int.MaxValue); break; default: error = $"Unknown field 'blocking.{k}'."; break; } @@ -1342,7 +1348,8 @@ DarlingAlertSettings rather than reproduced here. */ switch (k) { case "enabled": AddBool("deadlock_enabled", n, "deadlocks.enabled"); break; - case "count_threshold": AddInt("deadlock_count_threshold", n, "deadlocks.count_threshold", 1, int.MaxValue); break; + /* #3528: the named constant for its blocking sibling's reason. */ + case "count_threshold": AddInt("deadlock_count_threshold", n, "deadlocks.count_threshold", PostgresAlertEvaluator.CountThresholdFloor, int.MaxValue); break; /* #3444 (V122): the PostgreSQL count gate — same bound sourcing as its blocking sibling. */ case "pg_count_threshold": AddInt("pg_deadlock_count_threshold", n, "deadlocks.pg_count_threshold", PostgresAlertEvaluator.CountThresholdFloor, int.MaxValue); break; @@ -1443,6 +1450,9 @@ the value the sweep uses. */ switch (k) { case "disk_free_warn_percent": AddInt("self_disk_free_warn_percent", n, "self_alerts.disk_free_warn_percent", 0, 100); break; + /* #3528: bound mirrors DarlingAlertSettings' Math.Max(0, ...) — 0 is IN range + because it removes the floor (the pvs.floor_gb reading), not nonsense. */ + case "disk_free_warn_gb": AddInt("self_disk_free_warn_gb", n, "self_alerts.disk_free_warn_gb", 0, int.MaxValue); break; case "collection_stale_minutes": AddInt("collection_stale_minutes", n, "self_alerts.collection_stale_minutes", 5, 1440); break; case "collection_failure_threshold": AddInt("collection_failure_threshold", n, "self_alerts.collection_failure_threshold", 1, 1000); break; case "store_job_cadence_warn_percent": AddInt("store_job_cadence_warn_percent", n, "self_alerts.store_job_cadence_warn_percent", 5, 100); break; diff --git a/Darling/PerformanceMonitor.Darling.Service/StoreConfigProvider.cs b/Darling/PerformanceMonitor.Darling.Service/StoreConfigProvider.cs index 67e03f56b..8059e75d3 100644 --- a/Darling/PerformanceMonitor.Darling.Service/StoreConfigProvider.cs +++ b/Darling/PerformanceMonitor.Darling.Service/StoreConfigProvider.cs @@ -992,11 +992,12 @@ INSERT INTO config_alert_settings ( retention_hold_warn_ratio, retention_hold_critical_ratio, deadlock_warn_per_hour, deadlock_critical_per_hour, pg_deadlock_count_threshold, pg_blocking_count_threshold, - fleet_sweep_enabled, fleet_sweep_interval_minutes) + fleet_sweep_enabled, fleet_sweep_interval_minutes, + self_disk_free_warn_gb) VALUES (1, $1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11, $12, $13, $14, $15, $16, $17, $18, $19, $20, $21, $22, $23, $24, $25, $26, $27, $28, $29, $30, $31, $32, $33, $34, $35, $36, $37, $38, $39, $40, $41, $42, $43, $44, $45, $46, $47, $48, $49, $50, $51, $52, $53, $54, $55, $56, $57, $58, $59, $60, $61, $62, $63, - $64, $65, $66, $67) + $64, $65, $66, $67, $68) ON CONFLICT (id) DO NOTHING", connection) { CommandTimeout = ServiceCommandDeadlines.BootstrapSeconds }; command.Parameters.AddWithValue(a.Enabled); command.Parameters.AddWithValue(a.CpuEnabled); @@ -1088,6 +1089,10 @@ reports back. */ holds is what get_alert_settings reports back. */ command.Parameters.AddWithValue(a.FleetSweepEnabled); command.Parameters.AddWithValue(a.FleetSweepIntervalMinutes); + /* #3528, bound in the same order the V126 column was appended. Seeded RAW like every sibling: + the floor-at-0 lives on DarlingAlertSettings, so what the store holds is what + get_alert_settings reports back. */ + command.Parameters.AddWithValue(a.SelfDiskFreeWarnGb); await command.ExecuteNonQueryAsync(ct); } @@ -1361,7 +1366,8 @@ internal static string NormalizePlanXmlCompression(string? value) => retention_hold_warn_ratio, retention_hold_critical_ratio, deadlock_warn_per_hour, deadlock_critical_per_hour, pg_deadlock_count_threshold, pg_blocking_count_threshold, - fleet_sweep_enabled, fleet_sweep_interval_minutes + fleet_sweep_enabled, fleet_sweep_interval_minutes, + self_disk_free_warn_gb FROM config_alert_settings WHERE id = 1", connection) { CommandTimeout = ServiceCommandDeadlines.SerialLoopSeconds }; using var reader = await command.ExecuteReaderAsync(ct); if (!await reader.ReadAsync(ct)) @@ -1486,6 +1492,12 @@ but not read here -- or read but not selected -- would silently reset the knob t default on every worker start. */ FleetSweepEnabled = reader.GetBoolean(64), FleetSweepIntervalMinutes = reader.GetInt32(65), + + /* #3528 store-disk-warn GB floor appended (V126) at ordinal 66. Same reachability rule as + every appended knob: ApplyToConfig replaces config.Alerts wholesale, so a column selected + but not read here -- or read but not selected -- would silently reset the floor to the + shipped default on every worker start. */ + SelfDiskFreeWarnGb = reader.GetInt32(66), }; var analysis = new AnalysisConfig { diff --git a/Darling/PerformanceMonitor.Darling.Service/wwwroot/js/pages/fleet.js b/Darling/PerformanceMonitor.Darling.Service/wwwroot/js/pages/fleet.js index 44a3e1ba2..81e24dd02 100644 --- a/Darling/PerformanceMonitor.Darling.Service/wwwroot/js/pages/fleet.js +++ b/Darling/PerformanceMonitor.Darling.Service/wwwroot/js/pages/fleet.js @@ -545,9 +545,17 @@ function rollup(d) { function serverCard(c) { const cls = bandClass(c.band); + /* #3528: the band's fold skips Unknown, so a card can read Healthy off one measured metric of six — + say so instead of rendering an unqualified green. The counts are the server's own (R1: read the + pre-computed field, never re-derive); an awaiting card keeps its plain status, which already says + nothing has been measured yet. */ + const coverage = + c.metric_count > 0 && c.measured_metric_count < c.metric_count + ? " · " + c.measured_metric_count + " of " + c.metric_count + " measured" + : ""; const statusLine = c.awaiting_first_collection ? el("div", { class: "status-line awaiting", text: c.status }) - : el("div", { class: "status-line", text: c.status + " · last collect " + localClock(c.last_collection) }); + : el("div", { class: "status-line", text: c.status + " · last collect " + localClock(c.last_collection) + coverage }); return el( "div", diff --git a/Darling/PerformanceMonitor.Darling.Storage/PgMigrations.cs b/Darling/PerformanceMonitor.Darling.Storage/PgMigrations.cs index 1cb523dc8..176e7781d 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/PgMigrations.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/PgMigrations.cs @@ -198,6 +198,7 @@ Generated from the collector definition rather than by threading a specific late new Migration(123, "fleet-sweep-state", V123Sql), new Migration(124, "fleet-sweep-cadence-knobs", V124Sql), new Migration(125, "collector-database-scope", V125Sql), + new Migration(126, "self-disk-warn-gb-floor", V126Sql), }; /// @@ -651,6 +652,33 @@ ALTER TABLE config.config_alert_settings ALTER TABLE config.config_collector_schedules ADD COLUMN IF NOT EXISTS databases text[];"; + /// + /// V126 — the Store Disk Pressure warning's GB floor on the singleton config_alert_settings + /// row (#3528): the self-alert's percent trigger additionally requires free space below this many + /// GB before it fires, an AND qualifier so a large volume at a low percent (400 GB free on a 4 TB + /// store) stops paging CRITICAL. 0 removes the floor and restores the percent-only condition — + /// the pvs_floor_gb composition, deliberately not the target-volume pair's OR, whose GB + /// dimension ADDS fires. + /// + /// The column default IS the shipped constant + /// (DarlingSelfAlertEvaluator.DiskFreeWarnFloorGb — restated as a literal here only because + /// a rung is a SQL string, and pinned equal by SelfDiskWarnGbFloorRungTests). Non-zero on + /// upgrade DELIBERATELY, unlike the V122 knobs: their acceptance was "an untouched store fires + /// exactly where it did", while #3528's is that the untouched firing IS the defect — the issue's + /// own example is a default-configured store paging "act now" with 400 GB of runway. 50 puts the + /// crossover at a 500 GB volume, so any store volume at or under that keeps the exact pre-#3528 + /// percent behaviour. + /// + /// No CHECK enforcing the bound, matching V119/V120/V122/V124: the floor-at-0 is enforced as + /// the update_alert_settings write bound and DarlingAlertSettings' read-side clamp — + /// the raw-in/clamped-out split every knob on this table uses. No reload beacon of its own: V17's + /// statement-level trg_bump_alert_settings already bumps config_service.config_version + /// on any write here. No GRANT: this table carries table-level grants with no column carve. + /// + private const string V126Sql = @" +ALTER TABLE config.config_alert_settings + ADD COLUMN IF NOT EXISTS self_disk_free_warn_gb integer NOT NULL DEFAULT 50;"; + /// /// V2 — the service's observability store: the servers registry (upserted on every /// successful connect) and the per-run collection_log. Column names deliberately mirror diff --git a/Darling/PerformanceMonitor.Darling.Storage/StorageVersion.cs b/Darling/PerformanceMonitor.Darling.Storage/StorageVersion.cs index 91cd2df86..9d5f85a11 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/StorageVersion.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/StorageVersion.cs @@ -16,5 +16,5 @@ namespace PerformanceMonitor.Darling.Storage; /// public static class StorageVersion { - public const int SchemaVersion = 125; + public const int SchemaVersion = 126; } diff --git a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.cs b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.cs index 98dfc3149..de225b5e4 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.cs @@ -806,7 +806,12 @@ the parent every read of this feature starts from — a store with the children table existence cannot separate the rungs, and the rung adds exactly one column — there is no sibling to prefer or to explain not preferring. */ EXISTS (SELECT 1 FROM information_schema.columns WHERE table_name = 'config_collector_schedules' - AND column_name = 'databases')"; + AND column_name = 'databases'), + /* V126 probes a COLUMN for V57's reason: config_alert_settings has existed since V17, so table + existence cannot separate the rungs, and the rung adds exactly one column — there is no + sibling to prefer or to explain not preferring. */ + EXISTS (SELECT 1 FROM information_schema.columns WHERE table_name = 'config_alert_settings' + AND column_name = 'self_disk_free_warn_gb')"; /// The store schema version this viewer build requires — the highest migration it knows /// (). The connect-time gate blocks a store below this. @@ -828,7 +833,7 @@ sibling to prefer or to explain not preferring. */ await using var reader = await command.ExecuteReaderAsync(cancellationToken); if (await reader.ReadAsync(cancellationToken)) { - return MapProbedSchemaVersion(reader.GetBoolean(0), reader.GetBoolean(1), reader.GetBoolean(2), reader.GetBoolean(3), reader.GetBoolean(4), reader.GetBoolean(5), reader.GetBoolean(6), reader.GetBoolean(7), reader.GetBoolean(8), reader.GetBoolean(9), reader.GetBoolean(10), reader.GetBoolean(11), reader.GetBoolean(12), reader.GetBoolean(13), reader.GetBoolean(14), reader.GetBoolean(15), reader.GetBoolean(16), reader.GetBoolean(17), reader.GetBoolean(18), reader.GetBoolean(19), reader.GetBoolean(20), reader.GetBoolean(21), reader.GetBoolean(22), reader.GetBoolean(23), reader.GetBoolean(24), reader.GetBoolean(25), reader.GetBoolean(26), reader.GetBoolean(27), reader.GetBoolean(28), reader.GetBoolean(29), reader.GetBoolean(30), reader.GetBoolean(31), reader.GetBoolean(32), reader.GetBoolean(33), reader.GetBoolean(34), reader.GetBoolean(35), reader.GetBoolean(36), reader.GetBoolean(37), reader.GetBoolean(38), reader.GetBoolean(39), reader.GetBoolean(40), reader.GetBoolean(41), reader.GetBoolean(42), reader.GetBoolean(43), reader.GetBoolean(44), reader.GetBoolean(45), reader.GetBoolean(46), reader.GetBoolean(47), reader.GetBoolean(48), reader.GetBoolean(49), reader.GetBoolean(50), reader.GetBoolean(51), reader.GetBoolean(52), reader.GetBoolean(53), reader.GetBoolean(54), reader.GetBoolean(55), reader.GetBoolean(56), reader.GetBoolean(57), reader.GetBoolean(58), reader.GetBoolean(59), reader.GetBoolean(60), reader.GetBoolean(61), reader.GetBoolean(62), reader.GetBoolean(63), reader.GetBoolean(64), reader.GetBoolean(65), reader.GetBoolean(66), reader.GetBoolean(67), reader.GetBoolean(68), reader.GetBoolean(69), reader.GetBoolean(70), reader.GetBoolean(71), reader.GetBoolean(72), reader.GetBoolean(73), reader.GetBoolean(74), reader.GetBoolean(75), reader.GetBoolean(76), reader.GetBoolean(77), reader.GetBoolean(78), reader.GetBoolean(79), reader.GetBoolean(80), reader.GetBoolean(81), reader.GetBoolean(82), reader.GetBoolean(83), reader.GetBoolean(84), reader.GetBoolean(85), reader.GetBoolean(86), reader.GetBoolean(87), reader.GetBoolean(88), reader.GetBoolean(89), reader.GetBoolean(90), reader.GetBoolean(91), reader.GetBoolean(92), reader.GetBoolean(93), reader.GetBoolean(94), reader.GetBoolean(95), reader.GetBoolean(96), reader.GetBoolean(97), reader.GetBoolean(98), reader.GetBoolean(99), reader.GetBoolean(100)); + return MapProbedSchemaVersion(reader.GetBoolean(0), reader.GetBoolean(1), reader.GetBoolean(2), reader.GetBoolean(3), reader.GetBoolean(4), reader.GetBoolean(5), reader.GetBoolean(6), reader.GetBoolean(7), reader.GetBoolean(8), reader.GetBoolean(9), reader.GetBoolean(10), reader.GetBoolean(11), reader.GetBoolean(12), reader.GetBoolean(13), reader.GetBoolean(14), reader.GetBoolean(15), reader.GetBoolean(16), reader.GetBoolean(17), reader.GetBoolean(18), reader.GetBoolean(19), reader.GetBoolean(20), reader.GetBoolean(21), reader.GetBoolean(22), reader.GetBoolean(23), reader.GetBoolean(24), reader.GetBoolean(25), reader.GetBoolean(26), reader.GetBoolean(27), reader.GetBoolean(28), reader.GetBoolean(29), reader.GetBoolean(30), reader.GetBoolean(31), reader.GetBoolean(32), reader.GetBoolean(33), reader.GetBoolean(34), reader.GetBoolean(35), reader.GetBoolean(36), reader.GetBoolean(37), reader.GetBoolean(38), reader.GetBoolean(39), reader.GetBoolean(40), reader.GetBoolean(41), reader.GetBoolean(42), reader.GetBoolean(43), reader.GetBoolean(44), reader.GetBoolean(45), reader.GetBoolean(46), reader.GetBoolean(47), reader.GetBoolean(48), reader.GetBoolean(49), reader.GetBoolean(50), reader.GetBoolean(51), reader.GetBoolean(52), reader.GetBoolean(53), reader.GetBoolean(54), reader.GetBoolean(55), reader.GetBoolean(56), reader.GetBoolean(57), reader.GetBoolean(58), reader.GetBoolean(59), reader.GetBoolean(60), reader.GetBoolean(61), reader.GetBoolean(62), reader.GetBoolean(63), reader.GetBoolean(64), reader.GetBoolean(65), reader.GetBoolean(66), reader.GetBoolean(67), reader.GetBoolean(68), reader.GetBoolean(69), reader.GetBoolean(70), reader.GetBoolean(71), reader.GetBoolean(72), reader.GetBoolean(73), reader.GetBoolean(74), reader.GetBoolean(75), reader.GetBoolean(76), reader.GetBoolean(77), reader.GetBoolean(78), reader.GetBoolean(79), reader.GetBoolean(80), reader.GetBoolean(81), reader.GetBoolean(82), reader.GetBoolean(83), reader.GetBoolean(84), reader.GetBoolean(85), reader.GetBoolean(86), reader.GetBoolean(87), reader.GetBoolean(88), reader.GetBoolean(89), reader.GetBoolean(90), reader.GetBoolean(91), reader.GetBoolean(92), reader.GetBoolean(93), reader.GetBoolean(94), reader.GetBoolean(95), reader.GetBoolean(96), reader.GetBoolean(97), reader.GetBoolean(98), reader.GetBoolean(99), reader.GetBoolean(100), reader.GetBoolean(101)); } return null; @@ -853,7 +858,7 @@ sibling to prefer or to explain not preferring. */ /// is unit-tested without a live store; any schema bump past the newest arm trips the pinning test that keeps /// this in step with . /// - internal static int MapProbedSchemaVersion(bool hasConfigControlPlane, bool hasAlertDeliveryOverride, bool hasAnalysisState, bool hasAlertTuningKnobs, bool hasDefaultTraceEvents, bool hasIndexObjectStatsLatestIndex, bool hasCollectionLogHypertableOrPlainPg, bool hasJobHistory, bool hasAgentStatus, bool hasGenericWebhook, bool hasDeadlocksDatabaseName, bool hasQueryStoreReplicaRole, bool hasLongQueryCompletions, bool hasWebDashboardConfig, bool hasCustomViews, bool hasServerTags, bool hasConnectionRefireKnobs = false, bool hasAgCollectors = false, bool hasAgAlertKnobs = false, bool hasAgLatencyColumns = false, bool hasAgDisconnectRefire = false, bool hasPayloadDimensions = false, bool hasDimFloorIndexes = false, bool hasBlockingWaitThreshold = false, bool hasQueryStoreIntervalIdentity = false, bool hasPagerDutyWebhook = false, bool hasPagerDutyProxy = false, bool hasCollectorState = false, bool hasPlanCorrection = false, bool hasPvsStats = false, bool hasPvsPressureKnobs = false, bool hasDatabaseStateAlert = false, bool hasServerTagColour = false, bool hasQueryStatsHostObject = false, bool hasFindingDrillDown = false, bool hasStoreMetrics = false, bool hasPlanDimGzip = false, bool hasSelfAlertKnobs = false, bool hasJobMetricsColumns = false, bool hasJobCadenceKnob = false, bool hasBackfillSwitch = false, bool hasCollectorMemoryKnobs = false, bool hasDatabaseStateEdgeMemory = false, bool hasIncidentOccurrences = false, bool hasPlanXmlCompressionKnob = false, bool hasMonitoredServerEngine = false, bool hasPgBlockingEdges = false, bool hasQueryStorePlanMap = false, bool hasPgStatementText = false, bool hasQueryStoreText = false, bool hasPlanContentRetentionKnob = false, bool hasQueryStoreHealth = false, bool hasQueryStoreTextHash = false, bool hasComposeTimeoutKnob = false, bool hasFileGrowthAlert = false, bool hasCollectionLogFanoutRollup = false, bool hasTempDbMaxSize = false, bool hasServerEngineKind = false, bool hasPgDatabaseStats = false, bool hasPgIndexUsageStats = false, bool hasPgTableBloatStats = false, bool hasPgSessionStates = false, bool hasPgPlanCaptureReadiness = false, bool hasPgWriteStats = false, bool hasPgExtensionAvailability = false, bool hasPgLockStats = false, bool hasPgColumnStats = false, bool hasPgReplicationStats = false, bool hasPgBufferUsage = false, bool hasPgIndexBloat = false, bool hasPgPerDatabaseAttribution = false, bool hasPgWaitSampling = false, bool hasPgKernelStats = false, bool hasPgPredicateStats = false, bool hasPgPlanCapture = false, bool hasPgMajorVersion = false, bool hasPg18IoBytes = false, bool hasPgServerConfig = false, bool hasPgDeadlocks = false, bool hasPgDeadlockIdentity = false, bool hasCollectorCost = false, bool hasPgCpuUtilization = false, bool hasPlanForceActions = false, bool hasCollectionLogPhaseSplit = false, bool hasCollectionLogDrainForensics = false, bool hasCollectionLogFetchPhaseSums = false, bool hasStoreLogSelfMonitoring = false, bool hasCollectorStallProbes = false, bool hasRemediationCredentialAndActor = false, bool hasPgIndexBloatEstimate = false, bool hasPgCpuCapacityHeadroom = false, bool hasCustomAlertCore = false, bool hasMuteRuleReloadBeacon = false, bool hasBuiltinAlertPersistence = false, bool hasRetentionHoldRatioKnobs = false, bool hasDeadlockRateBandKnobs = false, bool hasOversizedPlanBacklog = false, bool hasPgAlertCountKnobs = false, bool hasFleetSweepState = false, bool hasFleetSweepCadenceKnobs = false, bool hasCollectorScheduleDatabases = false) + internal static int MapProbedSchemaVersion(bool hasConfigControlPlane, bool hasAlertDeliveryOverride, bool hasAnalysisState, bool hasAlertTuningKnobs, bool hasDefaultTraceEvents, bool hasIndexObjectStatsLatestIndex, bool hasCollectionLogHypertableOrPlainPg, bool hasJobHistory, bool hasAgentStatus, bool hasGenericWebhook, bool hasDeadlocksDatabaseName, bool hasQueryStoreReplicaRole, bool hasLongQueryCompletions, bool hasWebDashboardConfig, bool hasCustomViews, bool hasServerTags, bool hasConnectionRefireKnobs = false, bool hasAgCollectors = false, bool hasAgAlertKnobs = false, bool hasAgLatencyColumns = false, bool hasAgDisconnectRefire = false, bool hasPayloadDimensions = false, bool hasDimFloorIndexes = false, bool hasBlockingWaitThreshold = false, bool hasQueryStoreIntervalIdentity = false, bool hasPagerDutyWebhook = false, bool hasPagerDutyProxy = false, bool hasCollectorState = false, bool hasPlanCorrection = false, bool hasPvsStats = false, bool hasPvsPressureKnobs = false, bool hasDatabaseStateAlert = false, bool hasServerTagColour = false, bool hasQueryStatsHostObject = false, bool hasFindingDrillDown = false, bool hasStoreMetrics = false, bool hasPlanDimGzip = false, bool hasSelfAlertKnobs = false, bool hasJobMetricsColumns = false, bool hasJobCadenceKnob = false, bool hasBackfillSwitch = false, bool hasCollectorMemoryKnobs = false, bool hasDatabaseStateEdgeMemory = false, bool hasIncidentOccurrences = false, bool hasPlanXmlCompressionKnob = false, bool hasMonitoredServerEngine = false, bool hasPgBlockingEdges = false, bool hasQueryStorePlanMap = false, bool hasPgStatementText = false, bool hasQueryStoreText = false, bool hasPlanContentRetentionKnob = false, bool hasQueryStoreHealth = false, bool hasQueryStoreTextHash = false, bool hasComposeTimeoutKnob = false, bool hasFileGrowthAlert = false, bool hasCollectionLogFanoutRollup = false, bool hasTempDbMaxSize = false, bool hasServerEngineKind = false, bool hasPgDatabaseStats = false, bool hasPgIndexUsageStats = false, bool hasPgTableBloatStats = false, bool hasPgSessionStates = false, bool hasPgPlanCaptureReadiness = false, bool hasPgWriteStats = false, bool hasPgExtensionAvailability = false, bool hasPgLockStats = false, bool hasPgColumnStats = false, bool hasPgReplicationStats = false, bool hasPgBufferUsage = false, bool hasPgIndexBloat = false, bool hasPgPerDatabaseAttribution = false, bool hasPgWaitSampling = false, bool hasPgKernelStats = false, bool hasPgPredicateStats = false, bool hasPgPlanCapture = false, bool hasPgMajorVersion = false, bool hasPg18IoBytes = false, bool hasPgServerConfig = false, bool hasPgDeadlocks = false, bool hasPgDeadlockIdentity = false, bool hasCollectorCost = false, bool hasPgCpuUtilization = false, bool hasPlanForceActions = false, bool hasCollectionLogPhaseSplit = false, bool hasCollectionLogDrainForensics = false, bool hasCollectionLogFetchPhaseSums = false, bool hasStoreLogSelfMonitoring = false, bool hasCollectorStallProbes = false, bool hasRemediationCredentialAndActor = false, bool hasPgIndexBloatEstimate = false, bool hasPgCpuCapacityHeadroom = false, bool hasCustomAlertCore = false, bool hasMuteRuleReloadBeacon = false, bool hasBuiltinAlertPersistence = false, bool hasRetentionHoldRatioKnobs = false, bool hasDeadlockRateBandKnobs = false, bool hasOversizedPlanBacklog = false, bool hasPgAlertCountKnobs = false, bool hasFleetSweepState = false, bool hasFleetSweepCadenceKnobs = false, bool hasCollectorScheduleDatabases = false, bool hasSelfDiskWarnGbFloor = false) { /* V71 (the PostgreSQL blocking-edges rung): a table-existence sentinel and now the newest-first arm. A collector table would ordinarily get no arm at all — see the V63-V69 note below — but the TOP @@ -1002,13 +1007,30 @@ information_schema lines but cannot strip a comment. */ StorageVersion.SchemaVersion (116) rather than falling through to 115 and showing a spurious upgrade banner on a store that is current. The table is named only in the probe line, not this prose, per the V71 finding (the coverage ratchet strips information_schema lines but cannot strip a comment). */ + /* V126 (#3528): the Store Disk Pressure warning's GB floor on config_alert_settings — the AND + qualifier that stops a large store volume at a low percent paging CRITICAL. COLUMN-existence + sentinel (the table has existed since V17, so table existence cannot separate the rungs), + newest-first, and the TOP rung, so a fully-migrated store maps to EXACTLY + StorageVersion.SchemaVersion rather than falling through to 125 and showing a spurious + upgrade banner on a store that is current. + + The reason to gate is that standing invariant rather than a viewer read that would throw: + nothing in the viewer reads this column yet — the knob is backend-first (store plane + the + two MCP tools), and the Settings window's box follows in the viewer pass. The column is + named only in the probe line, not this prose, per the V71 finding: the coverage ratchet + strips information_schema lines but cannot strip a comment. */ + if (hasSelfDiskWarnGbFloor) + { + return 126; + } + /* V125 (#3477): the per-collector database scope on config_collector_schedules — the allow-list that lets an expensive per-database collector run against a representative sample instead of being turned off for the whole server. COLUMN-existence sentinel (the table has existed since V17, so table existence cannot separate the rungs), newest-first, - and the TOP rung, so a fully-migrated store maps to EXACTLY StorageVersion.SchemaVersion - rather than falling through to 124 and showing a spurious upgrade banner on a store that - is current. + one rung behind the top — a store carrying this and not V126 above maps to 125, which is + the honest answer for it and also what makes the upgrade banner correct in both + directions. The gate earns its place beyond that standing invariant: the schedule editor's select now names the column, so a viewer pointed below this rung would throw a raw 42703 on OPENING diff --git a/Lite/Services/AppAlertEngineSettings.cs b/Lite/Services/AppAlertEngineSettings.cs index 31f35363c..6dbb91274 100644 --- a/Lite/Services/AppAlertEngineSettings.cs +++ b/Lite/Services/AppAlertEngineSettings.cs @@ -80,6 +80,7 @@ would be silently reset on the first config_version bump. If this needs to be co public int DiskCriticalFreePercent => App.AlertDiskCriticalFreePercent; public int DiskCriticalFreeGb => App.AlertDiskCriticalFreeGb; public int SelfDiskFreeWarnPercent => 10; + public int SelfDiskFreeWarnGb => 50; public int CollectionStaleMinutes => ServerHealthThresholds.CollectionStoppedMinutesDefault; public int CollectionFailureThreshold => 10; public int PvsThresholdPercent => App.AlertPvsThresholdPercent; diff --git a/PerformanceMonitor.Alerting/IAlertEngineSettings.cs b/PerformanceMonitor.Alerting/IAlertEngineSettings.cs index 19523b240..2b341b928 100644 --- a/PerformanceMonitor.Alerting/IAlertEngineSettings.cs +++ b/PerformanceMonitor.Alerting/IAlertEngineSettings.cs @@ -169,6 +169,16 @@ public interface IAlertEngineSettings /// int SelfDiskFreeWarnPercent { get; } + /// + /// The store warning's GB floor (#3528): the percent above additionally requires free space + /// below this many GB before Store Disk Pressure fires — an AND qualifier so a large volume at + /// a low percent (400 GB free on a 4 TB store) never pages CRITICAL; 0 removes the floor and + /// restores the percent-only condition. The PVS floor's composition, deliberately NOT the + /// target-volume pair's OR: there the GB dimension ADDS fires, here it keeps them honest. On + /// the engine surface for the reason its percent sibling is. + /// + int SelfDiskFreeWarnGb { get; } + /// How long collection may go quiet before Collection Stopped / Agent Not Running /// fire (#2107; was a compile-time 30 minutes). Lite returns the default. int CollectionStaleMinutes { get; } diff --git a/PerformanceMonitor.Common/ServerHealthBands.cs b/PerformanceMonitor.Common/ServerHealthBands.cs index 83495a987..ae2502bf9 100644 --- a/PerformanceMonitor.Common/ServerHealthBands.cs +++ b/PerformanceMonitor.Common/ServerHealthBands.cs @@ -849,6 +849,30 @@ public static HealthSeverity OverallMetricSeverity(in ServerHealthMetrics m) return worst; } + /// + /// How many of the six card metrics carry a real reading (#3528): measured = severities that are + /// not , out of the per-metric total. The worst-of fold above + /// SKIPS Unknown — deliberately, so an unmeasured metric can never escalate — which means an online + /// server with five of six metrics structurally Unknown still folds to Healthy. These counts let a + /// consumer say so ("Healthy — 1 of 6 measured") instead of rendering that fold as an unqualified + /// green. Rank-neutrality is untouched: nothing here feeds . + /// + public static (int Measured, int Total) MeasuredMetricCounts(in ServerHealthMetrics m) + { + var measured = 0; + var total = 0; + foreach (var s in MetricSeverities(m)) + { + total++; + if (s != HealthSeverity.Unknown) + { + measured++; + } + } + + return (measured, total); + } + /// /// Collapses a server's health to one fleet band, mirroring the card border: offline collection -> Offline; /// a never-collected (queued-during-bootstrap) server -> Warning (attention-worthy but not the red overlay); From b74a22c0be1369745139cbd0db98f19364460665 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 00:17:06 -0400 Subject: [PATCH 09/69] Schedule surfaces refuse delta-family cadences past the shared gap policy (#3532) (#3559) A delta-family collector scheduled above the CollectorDeltaCalculator gap policy (DefaultMaxGapSeconds = 3600) takes the reset branch every cycle: every delta is (0, 0), facts vanish, baselines fill with zeros, and the product reads green precisely because it stopped measuring. The bound lives once, in PerformanceMonitor.Collectors: the delta family set (the calculator's exact caller census - 8 SQL Server + 2 PostgreSQL collectors, pinned by a source-scan test) and MaxDeltaFrequencyMinutes, half the policy so even an entirely missed cycle still yields a real delta (a cadence AT the policy already fails every cycle, since the gap is the cadence plus scheduling latency and the reset comparison is strict). Enforcement, per surface: - Lite ScheduleManager: UpdateSchedule / SetScheduleForServer refuse (before mutating), and the load path clamps a hand-edited or pre-fix collection_schedule.json with a warning, since there is no user to bounce it back to. - Lite schedule editor: validates before save (mirroring the Darling editor, which Lite previously had no counterpart to), commits pending cell edits first, and shows the cap in a footer line built from the shared constants. - Darling viewer editor: ValidateSchedule moved to the pure CollectorScheduleOverlay (now testable) and extended with the delta bound; footer hint appended from the same constants. - Darling service: StoreConfigProvider.ResolveSchedule's ValidFrequency now knows the collector, so a poisoned store row (hand-written or pre-fix) degrades to "no override" and falls through, exactly like a negative frequency. The web/MCP command path only writes enabled flags, so it needed nothing. Snapshot collectors keep long cadences - the bound is per-collector-kind. Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx Co-authored-by: Claude Fable 5 --- .../Darling.Tests/StoreConfigProviderTests.cs | 38 +++ .../ViewerControlPlaneStage3bTests.cs | 43 ++++ .../StoreConfigProvider.cs | 23 +- .../CollectorScheduleEditorWindow.xaml | 4 +- .../CollectorScheduleEditorWindow.xaml.cs | 35 +-- .../CollectorScheduleOverlay.cs | 38 +++ Lite.Tests/DeltaFamilyScheduleBoundTests.cs | 221 ++++++++++++++++++ Lite/Services/ScheduleManager.cs | 76 ++++++ .../CollectorScheduleEditorWindow.xaml | 10 +- .../CollectorScheduleEditorWindow.xaml.cs | 51 ++++ .../CollectorDeltaCalculator.cs | 62 +++++ 11 files changed, 566 insertions(+), 35 deletions(-) create mode 100644 Lite.Tests/DeltaFamilyScheduleBoundTests.cs diff --git a/Darling/Darling.Tests/StoreConfigProviderTests.cs b/Darling/Darling.Tests/StoreConfigProviderTests.cs index c90fab686..a1b3f5ff5 100644 --- a/Darling/Darling.Tests/StoreConfigProviderTests.cs +++ b/Darling/Darling.Tests/StoreConfigProviderTests.cs @@ -197,6 +197,44 @@ public void Resolve_RejectsDestructiveRetention_AndNegativeFrequency_FallingBack Assert.Equal(1, eff.RetentionDays); } + /// + /// #3532: a delta-family cadence past + /// would exceed the shared delta gap policy every cycle — the collector re-baselines each run and + /// stores (0, 0) forever, fabricating permanent quiet. The viewer's editor refuses to write such a + /// row, but a hand-written or pre-fix row can still exist, so the resolver treats it as "no override" + /// and falls through to the next level, exactly like a negative frequency. + /// + [Fact] + public void Resolve_RejectsADeltaFamilyCadencePastTheGapPolicyCap_FallingThrough() + { + var def = CollectorScheduleDefaults.All["wait_stats"]; + var cap = CollectorDeltaCalculator.MaxDeltaFrequencyMinutes; + + /* A poisoned fleet row falls all the way through to the code default. */ + var fleetBad = new[] { new ScheduleOverride(null, "wait_stats", cap + 60, null, true) }; + Assert.Equal(def.FrequencyMinutes, StoreConfigProvider.ResolveSchedule("wait_stats", 1, fleetBad).FrequencyMinutes); + + /* A poisoned per-server row falls through to a VALID fleet row, per-column. */ + var layered = new[] + { + new ScheduleOverride(null, "wait_stats", 15, null, true), + new ScheduleOverride(1, "wait_stats", cap + 1, null, true), + }; + Assert.Equal(15, StoreConfigProvider.ResolveSchedule("wait_stats", 1, layered).FrequencyMinutes); + + /* The cap itself is honored, the PostgreSQL delta family is covered, and a snapshot collector + keeps its long cadence — the bound is per-collector-kind, not blanket. */ + var atCap = new[] { new ScheduleOverride(null, "wait_stats", cap, null, true) }; + Assert.Equal(cap, StoreConfigProvider.ResolveSchedule("wait_stats", 1, atCap).FrequencyMinutes); + + var pgBad = new[] { new ScheduleOverride(null, "pg_wait_stats", cap + 60, null, true) }; + Assert.Equal(CollectorScheduleDefaults.All["pg_wait_stats"].FrequencyMinutes, + StoreConfigProvider.ResolveSchedule("pg_wait_stats", 1, pgBad).FrequencyMinutes); + + var snapshot = new[] { new ScheduleOverride(null, "database_size_stats", cap + 60, null, true) }; + Assert.Equal(cap + 60, StoreConfigProvider.ResolveSchedule("database_size_stats", 1, snapshot).FrequencyMinutes); + } + /* ---------------- live (DARLING_TEST_PG): the V17 bump trigger, rolled back ---------------- */ [Fact] diff --git a/Darling/Darling.Tests/ViewerControlPlaneStage3bTests.cs b/Darling/Darling.Tests/ViewerControlPlaneStage3bTests.cs index 70093fb9f..f870fa25f 100644 --- a/Darling/Darling.Tests/ViewerControlPlaneStage3bTests.cs +++ b/Darling/Darling.Tests/ViewerControlPlaneStage3bTests.cs @@ -437,6 +437,49 @@ public void ServerHasOverride_ReflectsTheServersRows() Assert.True(CollectorScheduleOverlay.ServerHasOverride(overrides, 7)); Assert.False(CollectorScheduleOverlay.ServerHasOverride(overrides, 8)); } + + /// + /// #3532: the editor refuses a delta-family cadence past the shared gap-policy cap before the write — + /// past it every cycle exceeds , re-baselines, + /// and stores zeros forever. The refusal names the cap and the policy; snapshot collectors stay exempt. + /// + [Fact] + public void ValidateSchedule_RefusesADeltaCadencePastTheCap_NamingThePolicy() + { + var edited = CollectorSchedulePresets.BuildDefaultSchedule(); + edited.First(s => s.Name == "wait_stats").FrequencyMinutes = 90; + + Assert.False(CollectorScheduleOverlay.ValidateSchedule(edited, out var error)); + Assert.Contains("wait_stats", error); + Assert.Contains(CollectorDeltaCalculator.MaxDeltaFrequencyMinutes.ToString(), error); + Assert.Contains($"{CollectorDeltaCalculator.DefaultMaxGapSeconds / 60}-minute delta gap policy", error); + } + + [Fact] + public void ValidateSchedule_AllowsTheCapSnapshotLongCadences_AndTheShippedDefaults() + { + /* The shipped defaults must validate as-is (index_object_stats ships at 1440 — snapshot, exempt). */ + var edited = CollectorSchedulePresets.BuildDefaultSchedule(); + Assert.True(CollectorScheduleOverlay.ValidateSchedule(edited, out _)); + + edited.First(s => s.Name == "wait_stats").FrequencyMinutes = CollectorDeltaCalculator.MaxDeltaFrequencyMinutes; + edited.First(s => s.Name == "database_size_stats").FrequencyMinutes = 90; + Assert.True(CollectorScheduleOverlay.ValidateSchedule(edited, out _)); + } + + [Fact] + public void ValidateSchedule_StillRefusesNegativeFrequency_AndSubDayRetention() + { + var edited = CollectorSchedulePresets.BuildDefaultSchedule(); + edited.First(s => s.Name == "wait_stats").FrequencyMinutes = -1; + Assert.False(CollectorScheduleOverlay.ValidateSchedule(edited, out var negativeError)); + Assert.Contains("can't be negative", negativeError); + + edited.First(s => s.Name == "wait_stats").FrequencyMinutes = 1; + edited.First(s => s.Name == "wait_stats").RetentionDays = 0; + Assert.False(CollectorScheduleOverlay.ValidateSchedule(edited, out var retentionError)); + Assert.Contains("at least 1", retentionError); + } } /// The viewer's control commands agree with the service executor's dispatch (the two ends must use the diff --git a/Darling/PerformanceMonitor.Darling.Service/StoreConfigProvider.cs b/Darling/PerformanceMonitor.Darling.Service/StoreConfigProvider.cs index 8059e75d3..10fa21a68 100644 --- a/Darling/PerformanceMonitor.Darling.Service/StoreConfigProvider.cs +++ b/Darling/PerformanceMonitor.Darling.Service/StoreConfigProvider.cs @@ -1745,10 +1745,12 @@ public static EffectiveSchedule ResolveSchedule(string collectorName, int server } /* Sanitize operator-supplied overrides before they drive scheduling / a destructive purge: a - negative frequency or a retention < 1 (0 would invert the purge cutoff and wipe the table) - is treated as "no override" and falls through to the next level. Defense in depth with the - V17 CHECK constraints and the DarlingRetention sink clamp. */ - var frequency = ValidFrequency(perServer?.FrequencyMinutes) ?? ValidFrequency(fleet?.FrequencyMinutes) ?? def.FrequencyMinutes; + negative frequency, a retention < 1 (0 would invert the purge cutoff and wipe the table), or a + delta-family cadence past the gap-policy cap (#3532 — every cycle would exceed + CollectorDeltaCalculator.DefaultMaxGapSeconds, re-baseline, and store a zero delta forever) is + treated as "no override" and falls through to the next level. Defense in depth with the + V17 CHECK constraints, the viewer editor's ValidateSchedule, and the DarlingRetention sink clamp. */ + var frequency = ValidFrequency(collectorName, perServer?.FrequencyMinutes) ?? ValidFrequency(collectorName, fleet?.FrequencyMinutes) ?? def.FrequencyMinutes; var retention = ValidRetention(perServer?.RetentionDays) ?? ValidRetention(fleet?.RetentionDays) ?? def.RetentionDays; /* No override row falls back to the collector's shared default enabled state — true for nearly every collector, but false for an opt-in one like long_query_completions (#1496). Falling back @@ -1843,9 +1845,16 @@ collecting rather than silently collecting nothing on a row full of whitespace. /// cutoff and delete everything, so it degrades to "no override" (fall through to the default). private static int? ValidRetention(int? days) => days is int v && v >= 1 ? v : null; - /// A frequency override is honored only when >= 0 (0 = on-load-only); negative degrades to - /// "no override" so a bad value can't make a collector run every sweep. - private static int? ValidFrequency(int? minutes) => minutes is int v && v >= 0 ? v : null; + /// A frequency override is honored only when >= 0 (0 = on-load-only) and, for a + /// delta-family collector, no slower than + /// (#3532 — a cadence past the delta gap policy fabricates permanent zeros); a bad value degrades to + /// "no override" so it can't make a collector run every sweep or stop measuring. + private static int? ValidFrequency(string collectorName, int? minutes) => + minutes is int v + && v >= 0 + && !(CollectorDeltaCalculator.IsDeltaFamily(collectorName) && v > CollectorDeltaCalculator.MaxDeltaFrequencyMinutes) + ? v + : null; /* ---------------- helpers ---------------- */ diff --git a/Darling/PerformanceMonitor.Darling.Viewer/CollectorScheduleEditorWindow.xaml b/Darling/PerformanceMonitor.Darling.Viewer/CollectorScheduleEditorWindow.xaml index 216d685a2..811b73337 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/CollectorScheduleEditorWindow.xaml +++ b/Darling/PerformanceMonitor.Darling.Viewer/CollectorScheduleEditorWindow.xaml @@ -91,7 +91,9 @@ - + diff --git a/Darling/PerformanceMonitor.Darling.Viewer/CollectorScheduleEditorWindow.xaml.cs b/Darling/PerformanceMonitor.Darling.Viewer/CollectorScheduleEditorWindow.xaml.cs index 7b87db9d2..6bcc5ede6 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/CollectorScheduleEditorWindow.xaml.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/CollectorScheduleEditorWindow.xaml.cs @@ -11,6 +11,7 @@ using System.Linq; using System.Windows; using System.Windows.Controls; +using PerformanceMonitor.Collectors; namespace PerformanceMonitor.Darling.Viewer; @@ -62,6 +63,15 @@ private async void OnLoaded(object sender, RoutedEventArgs e) PopulateScopeCombos(); + /* The delta cadence cap (#3532), from the shared constants so the shown numbers and collector + list can never drift from what ValidateSchedule enforces. */ + var deltaNames = CollectorSchedulePresets.BuildDefaultSchedule() + .Where(s => CollectorDeltaCalculator.IsDeltaFamily(s.Name)) + .Select(s => s.Name); + FooterHintText.Text += + $" Delta collectors ({string.Join(", ", deltaNames)}) accept at most {CollectorDeltaCalculator.MaxDeltaFrequencyMinutes} minutes: " + + $"past the {CollectorDeltaCalculator.DefaultMaxGapSeconds / 60}-minute delta gap policy every reading would be discarded as stale and recorded as zero."; + try { _allOverrides = await _dataService.GetCollectorSchedulesAsync(); @@ -290,7 +300,7 @@ private async void SaveButton_Click(object sender, RoutedEventArgs e) var usesDefault = _scopeServerId is not null && UseDefaultCheckBox.IsChecked == true; - if (!usesDefault && !ValidateSchedule(out var error)) + if (!usesDefault && !CollectorScheduleOverlay.ValidateSchedule(_editing, out var error)) { MessageBox.Show(error, "Collector Schedules", MessageBoxButton.OK, MessageBoxImage.Warning); return; @@ -332,29 +342,6 @@ private async void SaveButton_Click(object sender, RoutedEventArgs e) } } - /// Enforces the V17 CHECK constraints before the write (frequency >= 0, retention >= 1) so a - /// bad value surfaces as a friendly message rather than a raw Postgres error. - private bool ValidateSchedule(out string error) - { - foreach (var item in _editing) - { - if (item.FrequencyMinutes < 0) - { - error = $"'{item.Name}': frequency (minutes) can't be negative. Use 0 to collect once on server load."; - return false; - } - - if (item.RetentionDays < 1) - { - error = $"'{item.Name}': retention (days) must be at least 1."; - return false; - } - } - - error = ""; - return true; - } - /// /// "Apply Default to All Servers" — the fleet-scale bulk reset: removes EVERY server's per-server schedule /// override in one write () so they all fall back diff --git a/Darling/PerformanceMonitor.Darling.Viewer/CollectorScheduleOverlay.cs b/Darling/PerformanceMonitor.Darling.Viewer/CollectorScheduleOverlay.cs index 033ad4630..94e49c1dd 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/CollectorScheduleOverlay.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/CollectorScheduleOverlay.cs @@ -90,6 +90,44 @@ public static List ParseDatabases(string? text) => .Where(s => s.Length > 0) .ToList(); + /// + /// Enforces what the store and the collection pipeline can honor before the write, so a bad value + /// surfaces as a friendly message rather than a raw Postgres error or silent bad data: the V17 CHECK + /// constraints (frequency >= 0, retention >= 1) plus the delta gap-policy cadence cap (#3532) — + /// a delta-family collector past would + /// exceed every cycle and record permanent + /// zeros (the service's StoreConfigProvider.ResolveSchedule refuses such a row too, by falling + /// through to the default). Pure, so Darling.Tests exercise it without a Window. + /// + public static bool ValidateSchedule(IReadOnlyList edited, out string error) + { + ArgumentNullException.ThrowIfNull(edited); + + foreach (var item in edited) + { + if (item.FrequencyMinutes < 0) + { + error = $"'{item.Name}': frequency (minutes) can't be negative. Use 0 to collect once on server load."; + return false; + } + + if (CollectorDeltaCalculator.DeltaFrequencyError(item.Name, item.FrequencyMinutes) is string frequencyError) + { + error = frequencyError; + return false; + } + + if (item.RetentionDays < 1) + { + error = $"'{item.Name}': retention (days) must be at least 1."; + return false; + } + } + + error = ""; + return true; + } + /// True when the store holds any override row for this server (the editor's "custom vs. use /// default" initial state). public static bool ServerHasOverride(IReadOnlyList allOverrides, int serverId) diff --git a/Lite.Tests/DeltaFamilyScheduleBoundTests.cs b/Lite.Tests/DeltaFamilyScheduleBoundTests.cs new file mode 100644 index 000000000..7fa24cf5e --- /dev/null +++ b/Lite.Tests/DeltaFamilyScheduleBoundTests.cs @@ -0,0 +1,221 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor Lite. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.IO; +using System.Linq; +using System.Text.RegularExpressions; +using PerformanceMonitor.Collectors; +using PerformanceMonitorLite.Services; +using Xunit; + +namespace Lite.Tests; + +/// +/// #3532: a delta-family collector scheduled past the shared gap policy takes the reset branch every +/// cycle and fabricates permanent quiet — all (0, 0) deltas, zero facts, a green product that stopped +/// measuring. These tests pin the cadence cap: the census (the delta family equals the calculator's +/// caller set, so a new delta call site can't dodge the bound), the bound's derivation from +/// , and the three Lite enforcement points +/// (the write APIs refuse, the load path clamps, snapshot collectors stay exempt). +/// +public sealed class DeltaFamilyScheduleBoundTests : IDisposable +{ + private readonly string _configDir = Directory.CreateTempSubdirectory("pm-lite-schedule-tests-").FullName; + + public void Dispose() + { + try { Directory.Delete(_configDir, recursive: true); } catch { /* best effort */ } + } + + /* ---------------- the census ---------------- */ + + /// + /// must equal the set of collectors that + /// actually call the delta calculator (context.Deltas.CalculateDelta*), found by scanning the + /// shared collector sources. A collector that grows a delta call without joining the family would + /// accept the poisoned cadence; a listed collector that stopped calling would refuse a cadence that + /// is now harmless. Both directions fail here. + /// + [Fact] + public void DeltaFamily_EqualsTheCalculatorCallerSet() + { + var collectorsDir = FindRepoDirectory("PerformanceMonitor.Collectors"); + var callSite = new Regex(@"Deltas\s*\.\s*CalculateDelta", RegexOptions.Compiled); + var namePin = new Regex("override string Name\\s*=>\\s*\"([^\"]+)\"", RegexOptions.Compiled); + + var callers = Directory.EnumerateFiles(collectorsDir, "*.cs", SearchOption.AllDirectories) + .Select(File.ReadAllText) + .Where(source => callSite.IsMatch(source)) + .Select(source => + { + var match = namePin.Match(source); + Assert.True(match.Success, "a file calling Deltas.CalculateDelta has no Name => \"...\" pin — the census can't classify it"); + return match.Groups[1].Value; + }) + .ToHashSet(StringComparer.OrdinalIgnoreCase); + + Assert.True(callers.Count > 0, $"no Deltas.CalculateDelta call sites found under {collectorsDir} — the census regex is broken"); + Assert.Equal( + CollectorDeltaCalculator.DeltaFamilyCollectors.OrderBy(n => n, StringComparer.OrdinalIgnoreCase), + callers.OrderBy(n => n, StringComparer.OrdinalIgnoreCase)); + } + + /* ---------------- the bound itself ---------------- */ + + [Fact] + public void Cap_IsHalfTheGapPolicy_SoAMissedCycleStillYieldsARealDelta() + { + /* Derived, not coincidental: at the cap, even a gap of two full cadences (one entirely missed + cycle) is exactly the policy — still inside the strict '>' comparison the reset branch uses. */ + Assert.Equal(CollectorDeltaCalculator.DefaultMaxGapSeconds / 60 / 2, CollectorDeltaCalculator.MaxDeltaFrequencyMinutes); + Assert.True(CollectorDeltaCalculator.MaxDeltaFrequencyMinutes * 60 * 2 <= CollectorDeltaCalculator.DefaultMaxGapSeconds); + } + + [Fact] + public void DeltaFrequencyError_NamesTheCapAndThePolicy() + { + var error = CollectorDeltaCalculator.DeltaFrequencyError("wait_stats", 90); + + Assert.NotNull(error); + Assert.Contains("wait_stats", error); + Assert.Contains(CollectorDeltaCalculator.MaxDeltaFrequencyMinutes.ToString(), error); + Assert.Contains($"{CollectorDeltaCalculator.DefaultMaxGapSeconds / 60}-minute delta gap policy", error); + } + + [Fact] + public void DeltaFrequencyError_AllowsTheCapOnLoadOnlyAndEveryNonDeltaCadence() + { + Assert.Null(CollectorDeltaCalculator.DeltaFrequencyError("wait_stats", CollectorDeltaCalculator.MaxDeltaFrequencyMinutes)); + Assert.Null(CollectorDeltaCalculator.DeltaFrequencyError("wait_stats", 0)); + /* Snapshot collectors keep long cadences — index_object_stats ships at 1440 by default. */ + Assert.Null(CollectorDeltaCalculator.DeltaFrequencyError("index_object_stats", 1440)); + Assert.Null(CollectorDeltaCalculator.DeltaFrequencyError("database_size_stats", 90)); + } + + /* ---------------- Lite's write APIs refuse ---------------- */ + + [Fact] + public void UpdateSchedule_RefusesADeltaCadencePastTheCap_AndNamesThePolicy() + { + var manager = new ScheduleManager(_configDir); + + var ex = Assert.Throws(() => manager.UpdateSchedule("wait_stats", frequencyMinutes: 90)); + Assert.Contains("delta gap policy", ex.Message); + Assert.Contains(CollectorDeltaCalculator.MaxDeltaFrequencyMinutes.ToString(), ex.Message); + + /* Refused means unchanged — and unsaved. */ + Assert.Equal(1, manager.GetDefaultSchedule().First(s => s.Name == "wait_stats").FrequencyMinutes); + } + + [Fact] + public void UpdateSchedule_RefusesNegative_AllowsTheCapAndSnapshotLongCadences() + { + var manager = new ScheduleManager(_configDir); + + Assert.Throws(() => manager.UpdateSchedule("wait_stats", frequencyMinutes: -1)); + + manager.UpdateSchedule("wait_stats", frequencyMinutes: CollectorDeltaCalculator.MaxDeltaFrequencyMinutes); + Assert.Equal(CollectorDeltaCalculator.MaxDeltaFrequencyMinutes, + manager.GetDefaultSchedule().First(s => s.Name == "wait_stats").FrequencyMinutes); + + /* A snapshot collector is exempt: 90 minutes on database sizes loses nothing. */ + manager.UpdateSchedule("database_size_stats", frequencyMinutes: 90); + Assert.Equal(90, manager.GetDefaultSchedule().First(s => s.Name == "database_size_stats").FrequencyMinutes); + } + + [Fact] + public void SetScheduleForServer_RefusesADeltaCadencePastTheCap() + { + var manager = new ScheduleManager(_configDir); + + var schedules = ScheduleManager.GetDefaultSchedules(); + schedules.First(s => s.Name == "latch_stats").FrequencyMinutes = 90; + + var ex = Assert.Throws(() => manager.SetScheduleForServer("srv1", schedules)); + Assert.Contains("latch_stats", ex.Message); + Assert.False(manager.HasServerOverride("srv1")); + + /* The same list with the cadence inside the cap saves fine. */ + schedules.First(s => s.Name == "latch_stats").FrequencyMinutes = CollectorDeltaCalculator.MaxDeltaFrequencyMinutes; + manager.SetScheduleForServer("srv1", schedules); + Assert.True(manager.HasServerOverride("srv1")); + } + + /* ---------------- the load path clamps (no user to bounce the value back to) ---------------- */ + + [Fact] + public void LoadSchedules_ClampsAPersistedDeltaCadence_LeavesSnapshotsAlone() + { + /* A hand-edited (or pre-fix) collection_schedule.json carrying the poisoned cadence — in the + default schedule AND in a per-server override. */ + var json = """ + { + "version": 2, + "default_schedule": [ + { "name": "wait_stats", "enabled": true, "frequency_minutes": 90, "retention_days": 30 }, + { "name": "database_size_stats", "enabled": true, "frequency_minutes": 90, "retention_days": 90 } + ], + "server_overrides": { + "srv1": { + "collectors": [ + { "name": "latch_stats", "enabled": true, "frequency_minutes": 240, "retention_days": 30 } + ] + } + } + } + """; + File.WriteAllText(Path.Combine(_configDir, "collection_schedule.json"), json); + + var manager = new ScheduleManager(_configDir); + + Assert.Equal(CollectorDeltaCalculator.MaxDeltaFrequencyMinutes, + manager.GetDefaultSchedule().First(s => s.Name == "wait_stats").FrequencyMinutes); + Assert.Equal(CollectorDeltaCalculator.MaxDeltaFrequencyMinutes, + manager.GetSchedulesForServer("srv1").First(s => s.Name == "latch_stats").FrequencyMinutes); + + /* The snapshot cadence survives untouched, and the clamp was persisted — a fresh load sees it. */ + Assert.Equal(90, manager.GetDefaultSchedule().First(s => s.Name == "database_size_stats").FrequencyMinutes); + var reloaded = new ScheduleManager(_configDir); + Assert.Equal(CollectorDeltaCalculator.MaxDeltaFrequencyMinutes, + reloaded.GetDefaultSchedule().First(s => s.Name == "wait_stats").FrequencyMinutes); + } + + /* ---------------- presets stay inside the cap ---------------- */ + + [Fact] + public void EveryPreset_KeepsEveryDeltaCollectorInsideTheCap() + { + foreach (var (presetName, intervals) in ScheduleManager.s_presets) + { + foreach (var (collector, frequency) in intervals) + { + Assert.Null(CollectorDeltaCalculator.DeltaFrequencyError(collector, frequency)); + } + + Assert.NotNull(presetName); + } + } + + /* ---------------- helpers ---------------- */ + + private static string FindRepoDirectory(string relativePath) + { + var dir = AppContext.BaseDirectory; + for (int i = 0; i < 8 && dir is not null; i++) + { + var candidate = Path.Combine(dir, relativePath); + if (Directory.Exists(candidate)) + { + return candidate; + } + dir = Path.GetDirectoryName(dir); + } + throw new DirectoryNotFoundException($"Could not locate {relativePath} walking up from {AppContext.BaseDirectory}"); + } +} diff --git a/Lite/Services/ScheduleManager.cs b/Lite/Services/ScheduleManager.cs index b8e953456..6a45b07ac 100644 --- a/Lite/Services/ScheduleManager.cs +++ b/Lite/Services/ScheduleManager.cs @@ -13,6 +13,7 @@ using System.Text.Json; using System.Text.Json.Serialization; using Microsoft.Extensions.Logging; +using PerformanceMonitor.Collectors; using PerformanceMonitorLite.Models; namespace PerformanceMonitorLite.Services; @@ -136,6 +137,13 @@ public void UpdateSchedule(string collectorName, bool? enabled = null, int? freq throw new InvalidOperationException($"Collector '{collectorName}' not found"); } + /* Refuse before mutating anything, so a bad frequency can't leave a half-applied update. */ + if (frequencyMinutes.HasValue + && FrequencyError(collectorName, frequencyMinutes.Value) is string frequencyError) + { + throw new InvalidOperationException(frequencyError); + } + if (enabled.HasValue) { schedule.Enabled = enabled.Value; @@ -295,6 +303,14 @@ public void SetScheduleForServer(string serverId, List schedu { lock (_lock) { + foreach (var schedule in schedules) + { + if (FrequencyError(schedule.Name, schedule.FrequencyMinutes) is string frequencyError) + { + throw new InvalidOperationException(frequencyError); + } + } + _serverOverrides[serverId] = new ServerScheduleOverride { Collectors = schedules }; SaveSchedules(); @@ -421,6 +437,7 @@ private void LoadSchedules() } MergeNewDefaults(); + SanitizeDeltaFrequencies(); } catch (Exception ex) { @@ -436,6 +453,7 @@ private void LoadSchedules() if (TryLoadV2(bakJson)) { _logger?.LogInformation("Restored schedules from backup file"); + SanitizeDeltaFrequencies(); return; } @@ -443,6 +461,7 @@ private void LoadSchedules() _defaultSchedule = bakConfig?.Collectors ?? GetDefaultSchedules(); _serverOverrides = new Dictionary(); _logger?.LogInformation("Restored v1 schedules from backup file"); + SanitizeDeltaFrequencies(); return; } catch { /* backup also corrupt, fall through to defaults */ } @@ -532,6 +551,63 @@ private bool MergeIntoList(List list, List // Helpers // ────────────────────────────────────────────────────────────────── + /// + /// Why a frequency can't be honored for this collector, or null when it can (#3532). Negative is + /// nonsense on any collector (0 = on-load only), and a delta-family collector past + /// would exceed the shared delta gap + /// policy every cycle and record permanent zeros. The editor shows this message before saving; the + /// write APIs throw it as a backstop. + /// + internal static string? FrequencyError(string collectorName, int frequencyMinutes) + { + if (frequencyMinutes < 0) + { + return $"'{collectorName}': frequency (minutes) can't be negative. Use 0 to collect once on server load."; + } + + return CollectorDeltaCalculator.DeltaFrequencyError(collectorName, frequencyMinutes); + } + + /// + /// Clamps any loaded delta-family frequency above the gap-policy cap back to the cap (#3532) — the + /// write APIs refuse such a cadence, but a hand-edited or pre-fix collection_schedule.json can still + /// carry one, and honoring it would fabricate permanent quiet (every cycle past the gap policy + /// re-baselines and stores a zero delta). Load-time has no user to bounce the value back to, so it + /// clamps and logs instead of refusing. Saves when anything changed. + /// + private void SanitizeDeltaFrequencies() + { + var changed = false; + + var lists = new List> { _defaultSchedule }; + foreach (var over in _serverOverrides.Values) + { + lists.Add(over.Collectors); + } + + foreach (var list in lists) + { + foreach (var schedule in list) + { + if (schedule.FrequencyMinutes > CollectorDeltaCalculator.MaxDeltaFrequencyMinutes + && CollectorDeltaCalculator.IsDeltaFamily(schedule.Name)) + { + _logger?.LogWarning( + "Collector '{Name}' was scheduled every {Bad}m, above the {Max}m cap for delta collectors — past the {Policy}s delta gap policy every reading would be discarded and recorded as zero. Clamped to {Max}m.", + schedule.Name, schedule.FrequencyMinutes, CollectorDeltaCalculator.MaxDeltaFrequencyMinutes, + CollectorDeltaCalculator.DefaultMaxGapSeconds, CollectorDeltaCalculator.MaxDeltaFrequencyMinutes); + schedule.FrequencyMinutes = CollectorDeltaCalculator.MaxDeltaFrequencyMinutes; + changed = true; + } + } + } + + if (changed) + { + SaveSchedules(); + } + } + /// /// Detects which preset matches a list of collector schedules, or "Custom". This is the single /// source of preset logic; and the collector-schedule editor window both diff --git a/Lite/Windows/CollectorScheduleEditorWindow.xaml b/Lite/Windows/CollectorScheduleEditorWindow.xaml index 3a5124983..ff9bbb84a 100644 --- a/Lite/Windows/CollectorScheduleEditorWindow.xaml +++ b/Lite/Windows/CollectorScheduleEditorWindow.xaml @@ -74,9 +74,13 @@ - - + + + + + diff --git a/Lite/Windows/CollectorScheduleEditorWindow.xaml.cs b/Lite/Windows/CollectorScheduleEditorWindow.xaml.cs index edd7b34d5..c360a8db2 100644 --- a/Lite/Windows/CollectorScheduleEditorWindow.xaml.cs +++ b/Lite/Windows/CollectorScheduleEditorWindow.xaml.cs @@ -11,6 +11,7 @@ using System.Linq; using System.Windows; using System.Windows.Controls; +using PerformanceMonitor.Collectors; using PerformanceMonitorLite.Models; using PerformanceMonitorLite.Services; @@ -53,6 +54,7 @@ public CollectorScheduleEditorWindow( SetupCopyFromServerCombo(); LoadServerSchedule(); + SetDeltaBoundHint(); } /// @@ -78,6 +80,7 @@ public CollectorScheduleEditorWindow( _editingSchedules = CloneScheduleList(_scheduleManager.GetDefaultSchedule()); ScheduleGrid.ItemsSource = _editingSchedules; DetectActivePreset(); + SetDeltaBoundHint(); } private void SetupCopyFromServerCombo() @@ -221,6 +224,16 @@ private void CopyFromServer_Click(object sender, RoutedEventArgs e) private void SaveButton_Click(object sender, RoutedEventArgs e) { + /* Flush any in-progress cell edit into the bound items before we read them. */ + ScheduleGrid.CommitEdit(DataGridEditingUnit.Row, true); + + bool revertingToDefault = !_isEditingDefault && UseDefaultCheckBox.IsChecked == true; + if (!revertingToDefault && !ValidateSchedule(out var error)) + { + MessageBox.Show(error, "Collector Schedules", MessageBoxButton.OK, MessageBoxImage.Warning); + return; + } + if (_isEditingDefault) { /* Save to default schedule */ @@ -256,6 +269,44 @@ private void CancelButton_Click(object sender, RoutedEventArgs e) // Helpers // ────────────────────────────────────────────────────────────────── + /// Refuses a schedule the pipeline can't honor before it is saved (mirrors the Darling + /// viewer's editor): negative frequency, retention under a day, or a delta-family cadence past the + /// gap-policy cap (#3532) — the message names the policy so the refusal isn't mysterious. + private bool ValidateSchedule(out string error) + { + foreach (var item in _editingSchedules) + { + if (ScheduleManager.FrequencyError(item.Name, item.FrequencyMinutes) is string frequencyError) + { + error = frequencyError; + return false; + } + + if (item.RetentionDays < 1) + { + error = $"'{item.Name}': retention (days) must be at least 1."; + return false; + } + } + + error = ""; + return true; + } + + /// The always-visible cadence-cap note under the grid, built from the shared constants so + /// the shown numbers and collector list can never drift from what the validation enforces. + private void SetDeltaBoundHint() + { + var deltaNames = ScheduleManager.GetDefaultSchedules() + .Where(s => CollectorDeltaCalculator.IsDeltaFamily(s.Name)) + .Select(s => s.Name); + + DeltaBoundText.Text = + $"Delta collectors ({string.Join(", ", deltaNames)}) accept at most {CollectorDeltaCalculator.MaxDeltaFrequencyMinutes} minutes: " + + $"they store the change in cumulative counters between runs, and past the {CollectorDeltaCalculator.DefaultMaxGapSeconds / 60}-minute " + + "delta gap policy every reading would be discarded as stale and recorded as zero."; + } + private static List CloneScheduleList(IReadOnlyList source) { return source.Select(s => new CollectorSchedule diff --git a/PerformanceMonitor.Collectors/CollectorDeltaCalculator.cs b/PerformanceMonitor.Collectors/CollectorDeltaCalculator.cs index e5630f234..a8aac4ffa 100644 --- a/PerformanceMonitor.Collectors/CollectorDeltaCalculator.cs +++ b/PerformanceMonitor.Collectors/CollectorDeltaCalculator.cs @@ -8,6 +8,7 @@ using System; using System.Collections.Concurrent; +using System.Collections.Generic; namespace PerformanceMonitor.Collectors; @@ -41,6 +42,67 @@ public class CollectorDeltaCalculator : ICollectorDeltaCalculator /// public const int DefaultMaxGapSeconds = 3600; + /// + /// The slowest cadence (minutes) a schedule may give a delta-family collector — half of + /// (#3532). + /// + /// The gap between consecutive collections is never less than the cadence, and a gap past the + /// policy makes discard the baseline and return (0, 0) — so a cadence AT + /// the policy (60 minutes) fabricates permanent quiet: every cycle's gap is the cadence plus scheduling + /// latency, always past 3600s, so every cycle re-baselines, every delta is zero, the charts flatline, + /// and the product reads green precisely because it stopped measuring. A cadence between half the + /// policy and the policy is wrong less deterministically: one sweep overrun longer than the leftover + /// headroom zeros that interval, and the fleet measurement above saw ~15 minutes of overrun at p99.9. + /// Half the policy is the cadence at which even an entirely missed cycle (a gap of two cadences) + /// still yields a real delta. + /// + public const int MaxDeltaFrequencyMinutes = DefaultMaxGapSeconds / 60 / 2; + + /// + /// The collectors whose stored values are deltas of cumulative counters — every schedule surface caps + /// their cadence at (#3532). Membership means "calls + /// under "; the + /// DeltaFamilyScheduleBoundTests census asserts this set equals the calculator's caller set, so a new + /// delta call site that is not listed here (or a listed collector that stopped calling) fails tests. + /// Snapshot collectors are deliberately absent — a long cadence loses them nothing. + /// + public static readonly IReadOnlySet DeltaFamilyCollectors = new HashSet(StringComparer.OrdinalIgnoreCase) + { + "wait_stats", + "latch_stats", + "spinlock_stats", + "query_stats", + "procedure_stats", + "file_io_stats", + "memory_grant_stats", + "perfmon_stats", + "pg_wait_stats", + "pg_statement_stats", + }; + + /// True when the named collector stores deltas of cumulative counters and so must stay + /// inside . + public static bool IsDeltaFamily(string collectorName) => + collectorName is not null && DeltaFamilyCollectors.Contains(collectorName); + + /// + /// The refusal for a delta-family collector scheduled past , or + /// null when the cadence is fine (any cadence on a non-delta collector is). Shared by both SKUs' + /// schedule editors and Lite's ScheduleManager so the bound and its explanation live exactly once. + /// + public static string? DeltaFrequencyError(string collectorName, int frequencyMinutes) + { + if (frequencyMinutes <= MaxDeltaFrequencyMinutes || !IsDeltaFamily(collectorName)) + { + return null; + } + + return $"'{collectorName}': frequency (minutes) can't exceed {MaxDeltaFrequencyMinutes} for this collector. " + + $"It reads cumulative counters and stores the change between consecutive runs; past the " + + $"{DefaultMaxGapSeconds / 60}-minute delta gap policy every reading is discarded as too stale to subtract from, " + + "so the collector would record zero activity forever. The cap is half the policy so a slow or missed cycle still lands inside it."; + } + /// /// How far back a restart re-seed reads when restoring baselines from a host's own store. /// From ca10188a4125c49995722a0ecfc3e5fee163e386 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 00:45:28 -0400 Subject: [PATCH 10/69] Viewer Recommendations tabs render a window-empty notice instead of the false all-clear (#3551) (#3565) The tabs only branched on InsufficientDataMessage, so an analysis window that collected zero facts (dead collector, unreachable target) rendered "All clear - no current recommendations." Both viewers now consume the WindowEmptyMessage state #3553 added, one rung over from insufficient-data: - Lite's Generate now reads AnalysisService.WindowEmptyMessage directly and renders a distinct notice pointing at the Collection Health tab. - The Darling viewer never calls the engine, so the worker persists the window-empty determination into the V19 analysis_state marker as insufficient_data = false + the engine's message (a shape nothing else writes, so no schema change), and AnalysisStateMarker.WindowEmpty derives it back for the Recommendations tab on both the read and Generate now doors. The genuine all-clear (facts measured, zero findings) is unchanged, and the marker self-heals on the next facts-bearing pass - pinned through the real viewer read against live Postgres. Fixes #3551 Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx Co-authored-by: Claude Fable 5 --- .../DarlingObservabilityTests.cs | 50 +++++++++++ .../ViewerRecommendationsTests.cs | 76 +++++++++++++++++ .../DarlingObservability.cs | 17 ++-- .../DarlingWorker.cs | 23 +++-- .../MainWindow.xaml | 8 ++ .../MainWindow.xaml.cs | 30 +++++-- .../RecommendationsViewModel.cs | 85 +++++++++++++++---- .../ViewerDataService.Findings.cs | 15 +++- .../LiteRecommendationsViewModelTests.cs | 31 +++++++ .../LiteRecommendationsViewModel.cs | 44 +++++++++- Lite/Controls/RecommendationsTab.xaml | 5 ++ Lite/Controls/RecommendationsTab.xaml.cs | 27 +++++- 12 files changed, 374 insertions(+), 37 deletions(-) diff --git a/Darling/Darling.Tests/DarlingObservabilityTests.cs b/Darling/Darling.Tests/DarlingObservabilityTests.cs index aa7bedc76..26a640511 100644 --- a/Darling/Darling.Tests/DarlingObservabilityTests.cs +++ b/Darling/Darling.Tests/DarlingObservabilityTests.cs @@ -14,6 +14,7 @@ using PerformanceMonitor.Darling.Service; using PerformanceMonitor.Darling.Service.Mcp; using PerformanceMonitor.Darling.Storage; +using PerformanceMonitor.Darling.Viewer; using Xunit; namespace Darling.Tests; @@ -1000,6 +1001,55 @@ await DarlingObservability.WriteAnalysisStateAsync( await DeleteTestRowsAsync(connection); } + /// + /// #3551: the marker's window-empty encoding round-trips through the REAL viewer read and then + /// self-heals. A window-empty pass persists (false, message) — the shape nothing else writes — + /// and must read it as window-empty, not + /// insufficient; the next facts-bearing pass writes (false, null) over the same row, and the + /// viewer read must no longer say window-empty, so a recovered server sheds the notice on its + /// next pass without any extra clearing mechanism. + /// + [Fact] + public async Task WriteAnalysisState_WindowEmptyThenFactsBearingPass_SelfHealsThroughTheViewerRead_AgainstDevPostgres() + { + var connectionString = Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + Assert.SkipWhen(string.IsNullOrEmpty(connectionString), + "Set DARLING_TEST_PG to a Postgres connection string to run the window-empty marker test."); + + using var connection = new NpgsqlConnection(connectionString); + await connection.OpenAsync(TestContext.Current.CancellationToken); + await PgMigrations.MigrateAsync(connection, TestContext.Current.CancellationToken); + await DeleteTestRowsAsync(connection); + + await using var postgres = NpgsqlDataSource.Create(connectionString!); + await using var viewer = new ViewerDataService(connectionString!); + + /* A window-empty pass (the worker's #3551 arm): insufficient_data = false WITH the engine's + message. */ + await DarlingObservability.WriteAnalysisStateAsync( + postgres, TestServerId, insufficientData: false, "No facts were collected in the analysis window.", + null, TestContext.Current.CancellationToken); + + var windowEmptyMarker = await viewer.GetAnalysisStateAsync(TestServerId); + Assert.NotNull(windowEmptyMarker); + Assert.True(windowEmptyMarker!.WindowEmpty); + Assert.False(windowEmptyMarker.InsufficientData); + Assert.Equal("No facts were collected in the analysis window.", windowEmptyMarker.Message); + + /* The next facts-bearing pass (false, null) upserts the SAME row — the window-empty marker + self-heals rather than sticking to a recovered server. */ + await DarlingObservability.WriteAnalysisStateAsync( + postgres, TestServerId, insufficientData: false, null, null, TestContext.Current.CancellationToken); + + var healedMarker = await viewer.GetAnalysisStateAsync(TestServerId); + Assert.NotNull(healedMarker); + Assert.False(healedMarker!.WindowEmpty); + Assert.False(healedMarker.InsufficientData); + Assert.Null(healedMarker.Message); + + await DeleteTestRowsAsync(connection); + } + private static async Task<(bool Insufficient, string? Message)> ReadAnalysisStateAsync(NpgsqlConnection connection) { using var read = new NpgsqlCommand("SELECT insufficient_data, message FROM analysis_state WHERE server_id = $1", connection); diff --git a/Darling/Darling.Tests/ViewerRecommendationsTests.cs b/Darling/Darling.Tests/ViewerRecommendationsTests.cs index b9ca551f3..c495edc31 100644 --- a/Darling/Darling.Tests/ViewerRecommendationsTests.cs +++ b/Darling/Darling.Tests/ViewerRecommendationsTests.cs @@ -638,6 +638,82 @@ public void InsufficientData_UsesEngineMessageWhenPresent_ElseTheDefault() Assert.Equal(RecommendationsState.InsufficientData, RecommendationsViewModel.InsufficientData(null).State); } + // ── Window-empty vs all-clear state selection (#3524/#3551, the marker's false-with-a-message shape) ── + + [Fact] + public void FromFindings_ZeroFindings_WindowEmptyMarker_ShowsCollectionBroken_NotAllClear() + { + // window-empty + zero findings -> WindowEmpty ("collection appears broken"), NOT a false all-clear. + var vm = RecommendationsViewModel.FromFindings( + Array.Empty(), "SQL2022", utcOffsetMinutes: 0, + windowEmpty: true, + windowEmptyMessage: "No facts were collected in the analysis window."); + + Assert.Equal(RecommendationsState.WindowEmpty, vm.State); + Assert.Empty(vm.Sections); + Assert.StartsWith("No facts were collected in the analysis window.", vm.WindowEmptyMessage); + Assert.EndsWith(RecommendationsViewModel.WindowEmptyCollectionHealthPointer, vm.WindowEmptyMessage); + Assert.Equal(string.Empty, vm.InsufficientDataMessage); + } + + [Fact] + public void FromFindings_WindowEmptyMarker_ButFindingsPresent_FindingsWin_Loaded() + { + // Same rule as the insufficient marker: it only decides the zero-finding case. + var rows = new List { Row(1.6, "CPU is on fire", incidentId: "a") }; + + var vm = RecommendationsViewModel.FromFindings( + rows, "SQL2022", utcOffsetMinutes: 0, windowEmpty: true, windowEmptyMessage: "window empty"); + + Assert.Equal(RecommendationsState.Loaded, vm.State); + Assert.Single(vm.Sections); + Assert.Equal(string.Empty, vm.WindowEmptyMessage); + } + + [Fact] + public void WindowEmpty_UsesMarkerMessageWhenPresent_ElseTheDefault_AlwaysWithThePointer() + { + Assert.StartsWith( + RecommendationsViewModel.DefaultWindowEmptyMessage, + RecommendationsViewModel.WindowEmpty(null).WindowEmptyMessage); + Assert.StartsWith( + RecommendationsViewModel.DefaultWindowEmptyMessage, + RecommendationsViewModel.WindowEmpty(" ").WindowEmptyMessage); + Assert.StartsWith( + "engine says the window held nothing", + RecommendationsViewModel.WindowEmpty("engine says the window held nothing").WindowEmptyMessage); + Assert.EndsWith( + RecommendationsViewModel.WindowEmptyCollectionHealthPointer, + RecommendationsViewModel.WindowEmpty(null).WindowEmptyMessage); + Assert.Equal(RecommendationsState.WindowEmpty, RecommendationsViewModel.WindowEmpty(null).State); + } + + [Fact] + public void AnalysisStateMarker_WindowEmpty_IsExactlyTheFalseWithMessageShape() + { + // The marker encoding contract (#3551): false + message = window-empty, the shape only the + // worker's window-empty arm writes. Every other persisted shape must NOT read as window-empty — + // true + message is the span-gate miss, false + null/empty is a clean pass. + var at = new DateTime(2026, 9, 1, 12, 0, 0, DateTimeKind.Utc); + Assert.True(new AnalysisStateMarker(false, "window empty", at).WindowEmpty); + Assert.False(new AnalysisStateMarker(false, null, at).WindowEmpty); + Assert.False(new AnalysisStateMarker(false, "", at).WindowEmpty); + Assert.False(new AnalysisStateMarker(true, "still collecting", at).WindowEmpty); + } + + [Fact] + public void FromFindings_BothMarkers_InsufficientWinsDefensively() + { + // The writer never sets both (the engine nulls both and sets at most one), but if a skewed + // store ever did, the span-gate miss is the more fundamental answer. + var vm = RecommendationsViewModel.FromFindings( + Array.Empty(), "SQL2022", utcOffsetMinutes: 0, + insufficientData: true, insufficientDataMessage: "collecting", + windowEmpty: true, windowEmptyMessage: "window empty"); + + Assert.Equal(RecommendationsState.InsufficientData, vm.State); + } + [Fact] public void FromFindings_GroupsByIncident_HeaderNamesPrimaryPlusCount() { diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingObservability.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingObservability.cs index 86d961399..9b20be337 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingObservability.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingObservability.cs @@ -596,13 +596,16 @@ internal static (int Elapsed, int TargetMs, int StorageMs) SplitSweepPhases(long /// /// Upserts the per-server analysis-state marker (V19) after an analysis pass: insufficient_data = - /// true plus the engine's message when the pass hit the 24h data-span gate, or false + null - /// when a real pass completed on enough data. The engine ALREADY makes this determination - /// (DarlingAnalysisService.InsufficientDataMessage); this persists it so the Viewer's - /// Recommendations tab — which never calls the engine — shows "still collecting" instead of a false - /// all-clear on a young deployment's zero-finding read. One row per server, upserted on - /// server_id. Failure-isolated (Debug + no-op) like the other observability writes — an - /// analysis-state write must never break the collection loop. + /// true plus the engine's message when the pass hit the 24h data-span gate; false plus the + /// engine's message when the pass cleared the gate but the analysis window itself collected zero facts + /// (#3524/#3551 — false-with-a-message is written ONLY for that window-empty shape, which is how the + /// viewer tells it from a clean pass without a schema change); or false + null when a real pass + /// completed on measured facts. The engine ALREADY makes these determinations + /// (DarlingAnalysisService.InsufficientDataMessage / WindowEmptyMessage); this persists + /// them so the Viewer's Recommendations tab — which never calls the engine — shows "still collecting" + /// or "collection appears broken" instead of a false all-clear on a zero-finding read. One row per + /// server, upserted on server_id. Failure-isolated (Debug + no-op) like the other observability + /// writes — an analysis-state write must never break the collection loop. /// public static async Task WriteAnalysisStateAsync( NpgsqlDataSource postgres, diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs index 8ae40f4fb..95446c8ef 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs @@ -5767,12 +5767,16 @@ this wrap keeps any residue from reclassifying a perfectly good analysis pass. * } } - /* Persist the pass's insufficient-data determination (V19 marker) so the Viewer's - Recommendations tab shows "still collecting" instead of a false all-clear on a young - deployment: true + the engine's message when the pass hit the 24h data-span gate, cleared - (false) when a real pass completed on enough data. Failure-isolated like the other - observability writes. Only the two REAL terminal states write it — a Skipped/TimedOut/Error - pass (handled above / in the catch) leaves the last known marker untouched. */ + /* Persist the pass's data-state determination (V19 marker) so the Viewer's Recommendations + tab shows a reason instead of a false all-clear on a zero-finding read: true + the + engine's message when the pass hit the 24h data-span gate ("still collecting"); false + + the engine's message when the pass cleared the gate but the window itself collected zero + facts (#3524/#3551 — a dead-collector shape, "collection appears broken"; false-with-a- + message is a shape only this arm writes, so the viewer distinguishes it without a schema + change); cleared (false + null) when a real pass completed on measured facts, which is + how both miss markers self-heal. Failure-isolated like the other observability writes. + Only the REAL terminal states write it — a Skipped/TimedOut/Error pass (handled above / + in the catch) leaves the last known marker untouched. */ if (analysisService.InsufficientDataMessage is string insufficient) { await DarlingObservability.WriteAnalysisStateAsync( @@ -5780,6 +5784,13 @@ await DarlingObservability.WriteAnalysisStateAsync( return new AnalysisPassResult(AnalysisPassStatus.InsufficientData, 0, insufficient); } + if (analysisService.WindowEmptyMessage is string windowEmpty) + { + await DarlingObservability.WriteAnalysisStateAsync( + _postgres!, serverId, insufficientData: false, windowEmpty, _logger, stoppingToken); + return new AnalysisPassResult(AnalysisPassStatus.Ran, 0, windowEmpty); + } + await DarlingObservability.WriteAnalysisStateAsync( _postgres!, serverId, insufficientData: false, null, _logger, stoppingToken); return new AnalysisPassResult(AnalysisPassStatus.Ran, findings.Count, null); diff --git a/Darling/PerformanceMonitor.Darling.Viewer/MainWindow.xaml b/Darling/PerformanceMonitor.Darling.Viewer/MainWindow.xaml index 037e8f222..cdb91b9a5 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/MainWindow.xaml +++ b/Darling/PerformanceMonitor.Darling.Viewer/MainWindow.xaml @@ -1329,6 +1329,14 @@ MaxWidth="520" TextWrapping="Wrap" TextAlignment="Center" Visibility="Collapsed"/> + + + diff --git a/Darling/PerformanceMonitor.Darling.Viewer/MainWindow.xaml.cs b/Darling/PerformanceMonitor.Darling.Viewer/MainWindow.xaml.cs index 774d781ea..04a133179 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/MainWindow.xaml.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/MainWindow.xaml.cs @@ -1656,8 +1656,10 @@ private void FleetWorstServer_Click(object sender, MouseButtonEventArgs e) /// Recommendations tab's OWN selected server, and renders Lite's advise-only, incident-grouped cards. /// Advise-only: no Apply, and no mute (mute lives on the Alert History surface per the re-skin). The /// marker lets a zero-finding read on a young deployment show "still collecting" (the engine skipped - /// for its 24h data-span gate) instead of a false all-clear; a server with findings shows them, and a - /// server with enough data and zero findings shows the genuine all-clear. The tab's status line + /// for its 24h data-span gate) instead of a false all-clear, and a window-empty pass (#3524/#3551 — + /// the span gate passed but the window itself collected nothing, a dead-collector shape) show + /// "collection appears broken"; a server with findings shows them, and a server with enough data, + /// measured facts, and zero findings shows the genuine all-clear. The tab's status line /// surfaces the last analysis time; "Generate now" forces an immediate pass via the analyze_now command. /// private async Task LoadRecommendationsAsync() @@ -1681,15 +1683,19 @@ private async Task LoadRecommendationsAsync() /* Read the per-server analysis-state marker (V19) the Darling service persists after each pass, so a zero-finding read on a young deployment renders "still collecting" (the engine skipped for its - 24h data-span gate) instead of a false all-clear. The marker only decides the ZERO-finding case; - a server with findings shows them regardless. Null (no pass yet / pre-V19 store) = not insufficient. */ + 24h data-span gate) instead of a false all-clear, and a window-empty pass (#3524/#3551 — the + marker's false-with-a-message shape) renders "collection appears broken". The marker only decides + the ZERO-finding case; a server with findings shows them regardless. Null (no pass yet / pre-V19 + store) = neither. */ var analysisState = await _dataService.GetAnalysisStateAsync(server.ServerId); ApplyRecommendationsViewModel( RecommendationsViewModel.FromFindings( rows, server.DisplayName, LocalUtcOffsetMinutes(), insufficientData: analysisState?.InsufficientData == true, - insufficientDataMessage: analysisState?.Message)); + insufficientDataMessage: analysisState?.Message, + windowEmpty: analysisState?.WindowEmpty == true, + windowEmptyMessage: analysisState?.Message)); RecommendationsStatusText.Text = rows.Count > 0 ? $"Last analyzed {rows[0].AnalysisTimeLocal:yyyy-MM-dd HH:mm:ss} (local)" @@ -1707,6 +1713,7 @@ private void ApplyRecommendationsViewModel(RecommendationsViewModel vm) RecommendationsScroll.Visibility = Visibility.Collapsed; RecommendationsEmptyText.Visibility = Visibility.Collapsed; RecommendationsInsufficientText.Visibility = Visibility.Collapsed; + RecommendationsWindowEmptyText.Visibility = Visibility.Collapsed; break; case RecommendationsState.InsufficientData: @@ -1714,15 +1721,27 @@ private void ApplyRecommendationsViewModel(RecommendationsViewModel vm) RecommendationsSectionsList.ItemsSource = null; RecommendationsScroll.Visibility = Visibility.Collapsed; RecommendationsEmptyText.Visibility = Visibility.Collapsed; + RecommendationsWindowEmptyText.Visibility = Visibility.Collapsed; RecommendationsInsufficientText.Text = vm.InsufficientDataMessage; RecommendationsInsufficientText.Visibility = Visibility.Visible; break; + case RecommendationsState.WindowEmpty: + RecommendationsLoadingText.Visibility = Visibility.Collapsed; + RecommendationsSectionsList.ItemsSource = null; + RecommendationsScroll.Visibility = Visibility.Collapsed; + RecommendationsEmptyText.Visibility = Visibility.Collapsed; + RecommendationsInsufficientText.Visibility = Visibility.Collapsed; + RecommendationsWindowEmptyText.Text = vm.WindowEmptyMessage; + RecommendationsWindowEmptyText.Visibility = Visibility.Visible; + break; + case RecommendationsState.Empty: RecommendationsLoadingText.Visibility = Visibility.Collapsed; RecommendationsSectionsList.ItemsSource = null; RecommendationsScroll.Visibility = Visibility.Collapsed; RecommendationsInsufficientText.Visibility = Visibility.Collapsed; + RecommendationsWindowEmptyText.Visibility = Visibility.Collapsed; RecommendationsEmptyText.Visibility = Visibility.Visible; break; @@ -1731,6 +1750,7 @@ private void ApplyRecommendationsViewModel(RecommendationsViewModel vm) RecommendationsLoadingText.Visibility = Visibility.Collapsed; RecommendationsEmptyText.Visibility = Visibility.Collapsed; RecommendationsInsufficientText.Visibility = Visibility.Collapsed; + RecommendationsWindowEmptyText.Visibility = Visibility.Collapsed; RecommendationsSectionsList.ItemsSource = vm.Sections; RecommendationsScroll.Visibility = Visibility.Visible; break; diff --git a/Darling/PerformanceMonitor.Darling.Viewer/RecommendationsViewModel.cs b/Darling/PerformanceMonitor.Darling.Viewer/RecommendationsViewModel.cs index 5a194d85f..dcf86de1e 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/RecommendationsViewModel.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/RecommendationsViewModel.cs @@ -52,6 +52,14 @@ public enum RecommendationsState /// InsufficientData, + /// + /// The read produced zero recommendations AND the persisted marker records a window-empty pass + /// (#3524/#3551): the span gate passed on lifetime history but the analysis window itself collected + /// zero facts — a dead collector or an unreachable target, not a healthy server, so a distinct + /// "collection appears broken" notice is shown, never the all-clear. + /// + WindowEmpty, + /// The read completed and produced zero recommendations — the all-clear. Empty, @@ -378,15 +386,24 @@ public sealed class RecommendationsViewModel /// public string InsufficientDataMessage { get; } + /// + /// The message shown in the state — the engine's own + /// persisted message (or when it supplied none), always + /// suffixed with the Collection Health pointer. Empty in every other state. + /// + public string WindowEmptyMessage { get; } + /// Total card count across all sections. public int TotalCount => Sections.Sum(s => s.Count); private RecommendationsViewModel( - IReadOnlyList sections, RecommendationsState state, string insufficientDataMessage) + IReadOnlyList sections, RecommendationsState state, string insufficientDataMessage, + string windowEmptyMessage = "") { Sections = sections; State = state; InsufficientDataMessage = insufficientDataMessage; + WindowEmptyMessage = windowEmptyMessage; } /// The default insufficient-data prose when the engine supplied no message (mirrors Lite's). @@ -410,29 +427,58 @@ public static RecommendationsViewModel InsufficientData(string? message) => RecommendationsState.InsufficientData, string.IsNullOrWhiteSpace(message) ? DefaultInsufficientDataMessage : message!); + /// The default window-empty prose when the marker carried no message (mirrors Lite's). + public const string DefaultWindowEmptyMessage = + "Nothing was collected in this analysis window, so nothing was measured — this is not an all-clear."; + /// - /// Builds a loaded/empty/insufficient-data view-model from the persisted finding rows and the - /// per-server analysis-state marker. Maps each row to an advise-only item, appends the co-fired + /// Appended to every window-empty message so the operator lands on the surface that diagnoses a + /// dead collector — the viewer's rendering of the same pointer the MCP analyze_server tool + /// appends (get_collection_health there, the in-app tab here). Mirrors Lite's. + /// + public const string WindowEmptyCollectionHealthPointer = + "Check the Collection Health tab to see when collectors last succeeded."; + + /// + /// Builds the window-empty-state view-model (#3524/#3551) from the persisted marker's message (or + /// the default when it is null/blank), suffixed with the Collection Health pointer — the viewer's + /// mirror of Lite's LiteRecommendationsViewModel.WindowEmpty, sourced from the V19 marker's + /// window-empty shape () rather than a live engine call. + /// + public static RecommendationsViewModel WindowEmpty(string? message) => + new( + Array.Empty(), + RecommendationsState.WindowEmpty, + string.Empty, + (string.IsNullOrWhiteSpace(message) ? DefaultWindowEmptyMessage : message!) + + " " + WindowEmptyCollectionHealthPointer); + + /// + /// Builds a loaded/empty/insufficient-data/window-empty view-model from the persisted finding rows + /// and the per-server analysis-state marker. Maps each row to an advise-only item, appends the co-fired /// cross-reference, and groups by incident. State selection: /// /// one or more findings -> (findings always win); /// zero findings AND (the persisted marker says the engine /// has not cleared its 24h data-span gate) -> /// ("still collecting"); - /// zero findings and no insufficient-data marker -> - /// (the genuine all-clear — enough data, nothing to report). + /// zero findings AND (the marker records a window-empty pass, + /// #3524/#3551) -> ("collection appears broken"); + /// zero findings and neither marker -> + /// (the genuine all-clear — enough data, facts measured, nothing to report). /// /// is carried onto each card for the Ask-AI prompt's window. The /// rows arrive pre-sorted (severity band desc, raw desc, database, title) from the read, and grouping - /// preserves that order. defaults false so the callers that carry - /// no marker keep the prior loaded/empty behavior. + /// preserves that order. and default + /// false so the callers that carry no marker keep the prior loaded/empty behavior. /// public static RecommendationsViewModel FromFindings( IReadOnlyList rows, string serverName, int utcOffsetMinutes = 0, - bool insufficientData = false, string? insufficientDataMessage = null) + bool insufficientData = false, string? insufficientDataMessage = null, + bool windowEmpty = false, string? windowEmptyMessage = null) { if (rows is null || rows.Count == 0) - return ZeroFindingState(insufficientData, insufficientDataMessage); + return ZeroFindingState(insufficientData, insufficientDataMessage, windowEmpty, windowEmptyMessage); var items = new List(rows.Count); foreach (var row in rows) @@ -443,7 +489,7 @@ public static RecommendationsViewModel FromFindings( } if (items.Count == 0) - return ZeroFindingState(insufficientData, insufficientDataMessage); + return ZeroFindingState(insufficientData, insufficientDataMessage, windowEmpty, windowEmptyMessage); AppendCoFired(items); return new(GroupByIncident(items, utcOffsetMinutes), RecommendationsState.Loaded, string.Empty); @@ -452,13 +498,20 @@ public static RecommendationsViewModel FromFindings( /// /// Picks the state for a zero-finding read: when /// the persisted marker says the analysis pass has not cleared the 24h data-span gate (so the tab - /// shows "still collecting" rather than a false all-clear), else - /// (a genuine all-clear). + /// shows "still collecting" rather than a false all-clear), + /// when it records a window-empty pass instead (#3524/#3551 — "collection appears broken", also never + /// the all-clear; the two marker shapes are mutually exclusive at the writer, and insufficient-data is + /// checked first defensively), else (a genuine all-clear). /// - private static RecommendationsViewModel ZeroFindingState(bool insufficientData, string? message) => - insufficientData - ? InsufficientData(message) - : new(Array.Empty(), RecommendationsState.Empty, string.Empty); + private static RecommendationsViewModel ZeroFindingState( + bool insufficientData, string? insufficientMessage, bool windowEmpty, string? windowEmptyMessage) + { + if (insufficientData) + return InsufficientData(insufficientMessage); + if (windowEmpty) + return WindowEmpty(windowEmptyMessage); + return new(Array.Empty(), RecommendationsState.Empty, string.Empty); + } /// /// Maps one persisted finding row to an advise-only . Reuses the diff --git a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Findings.cs b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Findings.cs index f68d3c6ad..b74011957 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Findings.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Findings.cs @@ -65,7 +65,20 @@ public sealed class ViewerFindingRow /// null when no pass has run for the server yet (no row), which the tab treats as "not /// insufficient" — an explicit marker is required to show the collecting state. /// -public sealed record AnalysisStateMarker(bool InsufficientData, string? Message, DateTime AnalysisTimeUtc); +public sealed record AnalysisStateMarker(bool InsufficientData, string? Message, DateTime AnalysisTimeUtc) +{ + /// + /// True when the marker records a WINDOW-EMPTY pass (#3524/#3551): the pass cleared the 24h + /// data-span gate ( false) but the analysis window itself collected + /// zero facts — a dead collector or an unreachable target, not a healthy server. Encoded as + /// false-with-a-message, a shape only the worker's window-empty arm writes (a completed pass on + /// enough data writes false + null; the two gate misses write true + message), so no schema change + /// was needed and a pre-#3551 marker can never satisfy it. The Recommendations tab renders the + /// "collection appears broken" notice instead of the all-clear for a zero-finding read while this + /// is set. + /// + public bool WindowEmpty => !InsufficientData && !string.IsNullOrEmpty(Message); +} public sealed partial class ViewerDataService { diff --git a/Lite.Tests/LiteRecommendationsViewModelTests.cs b/Lite.Tests/LiteRecommendationsViewModelTests.cs index 487652d27..463c65914 100644 --- a/Lite.Tests/LiteRecommendationsViewModelTests.cs +++ b/Lite.Tests/LiteRecommendationsViewModelTests.cs @@ -58,6 +58,37 @@ public void InsufficientData_BlankMessage_FallsBackToDefault() Assert.Equal(LiteRecommendationsViewModel.DefaultInsufficientDataMessage, vm.InsufficientDataMessage); } + [Fact] + public void WindowEmpty_UsesEngineMessage_AndPointsAtCollectionHealth() + { + // #3524/#3551: zero facts in the window is a dead-collector shape — a distinct state carrying + // the engine's message plus the in-app pointer, never the all-clear. + var vm = LiteRecommendationsViewModel.WindowEmpty("No facts were collected in the analysis window."); + Assert.Equal(LiteRecommendationsState.WindowEmpty, vm.State); + Assert.Empty(vm.Sections); + Assert.StartsWith("No facts were collected in the analysis window.", vm.WindowEmptyMessage); + Assert.EndsWith(LiteRecommendationsViewModel.WindowEmptyCollectionHealthPointer, vm.WindowEmptyMessage); + Assert.Equal(string.Empty, vm.InsufficientDataMessage); + } + + [Fact] + public void WindowEmpty_BlankMessage_FallsBackToDefault_StillWithThePointer() + { + var vm = LiteRecommendationsViewModel.WindowEmpty(" "); + Assert.StartsWith(LiteRecommendationsViewModel.DefaultWindowEmptyMessage, vm.WindowEmptyMessage); + Assert.EndsWith(LiteRecommendationsViewModel.WindowEmptyCollectionHealthPointer, vm.WindowEmptyMessage); + } + + [Fact] + public void WindowEmptyMessage_IsEmptyOutsideTheWindowEmptyState() + { + // The genuine all-clear (facts measured, zero findings) keeps its state and carries no + // window-empty prose — the distinction #3551 exists for. + Assert.Equal(string.Empty, LiteRecommendationsViewModel.FromItems(Array.Empty()).WindowEmptyMessage); + Assert.Equal(string.Empty, LiteRecommendationsViewModel.InsufficientData("x").WindowEmptyMessage); + Assert.Equal(string.Empty, LiteRecommendationsViewModel.Loading().WindowEmptyMessage); + } + [Fact] public void FromItems_Empty_IsEmptyState() { diff --git a/Lite/Analysis/Recommendations/LiteRecommendationsViewModel.cs b/Lite/Analysis/Recommendations/LiteRecommendationsViewModel.cs index bc39bdfbe..10f35e156 100644 --- a/Lite/Analysis/Recommendations/LiteRecommendationsViewModel.cs +++ b/Lite/Analysis/Recommendations/LiteRecommendationsViewModel.cs @@ -29,6 +29,13 @@ public enum LiteRecommendationsState /// InsufficientData, + /// + /// The analysis window itself collected zero facts even though the engine's lifetime data-span + /// gate passed (AnalysisService.WindowEmptyMessage, #3524/#3551) — a dead collector or an + /// unreachable target, not a healthy server, so a distinct notice is shown, never the all-clear. + /// + WindowEmpty, + /// The read completed and produced zero recommendations — the all-clear. Empty, @@ -217,17 +224,26 @@ public sealed class LiteRecommendationsViewModel /// public string InsufficientDataMessage { get; } + /// + /// The window-empty message to show in the + /// state (the engine's own message, or a default, always suffixed with the Collection Health + /// pointer). Empty in every other state. + /// + public string WindowEmptyMessage { get; } + /// Total card count across all sections. public int TotalCount => Sections.Sum(s => s.Count); private LiteRecommendationsViewModel( IReadOnlyList sections, LiteRecommendationsState state, - string insufficientDataMessage) + string insufficientDataMessage, + string windowEmptyMessage = "") { Sections = sections; State = state; InsufficientDataMessage = insufficientDataMessage; + WindowEmptyMessage = windowEmptyMessage; } /// The default insufficient-data prose when the engine supplied none. @@ -249,6 +265,32 @@ public static LiteRecommendationsViewModel InsufficientData(string? engineMessag LiteRecommendationsState.InsufficientData, string.IsNullOrWhiteSpace(engineMessage) ? DefaultInsufficientDataMessage : engineMessage!); + /// The default window-empty prose when the engine supplied no message. + public const string DefaultWindowEmptyMessage = + "Nothing was collected in this analysis window, so nothing was measured — this is not an all-clear."; + + /// + /// Appended to every window-empty message so the operator lands on the surface that diagnoses a + /// dead collector — the viewer's rendering of the same pointer the MCP analyze_server tool + /// appends (get_collection_health there, the in-app tab here). + /// + public const string WindowEmptyCollectionHealthPointer = + "Check the Collection Health tab to see when collectors last succeeded."; + + /// + /// Builds the window-empty-state view-model (#3524/#3551) from the engine's + /// AnalysisService.WindowEmptyMessage (or the default when it is null/blank), suffixed with + /// the Collection Health pointer. Rendered instead of the all-clear when the analysis window + /// collected zero facts on a server whose lifetime span gate passed — a dead-collector shape. + /// + public static LiteRecommendationsViewModel WindowEmpty(string? engineMessage) => + new( + Array.Empty(), + LiteRecommendationsState.WindowEmpty, + string.Empty, + (string.IsNullOrWhiteSpace(engineMessage) ? DefaultWindowEmptyMessage : engineMessage!) + + " " + WindowEmptyCollectionHealthPointer); + /// /// Builds a loaded/empty view-model from the reader's flat, already-sorted list. Groups by /// into Critical / Warning / Info sections (empty diff --git a/Lite/Controls/RecommendationsTab.xaml b/Lite/Controls/RecommendationsTab.xaml index 6cfe59c65..5df48eb47 100644 --- a/Lite/Controls/RecommendationsTab.xaml +++ b/Lite/Controls/RecommendationsTab.xaml @@ -159,6 +159,11 @@ + + + diff --git a/Lite/Controls/RecommendationsTab.xaml.cs b/Lite/Controls/RecommendationsTab.xaml.cs index 184068a4a..276124b4d 100644 --- a/Lite/Controls/RecommendationsTab.xaml.cs +++ b/Lite/Controls/RecommendationsTab.xaml.cs @@ -193,7 +193,10 @@ server selected NOW rather than the one selected when this call started. */ /// Runs an on-demand analysis for the selected server (same construction path the background /// collector uses), then renders the freshly-enriched in-memory findings directly — which, unlike /// the stored-finding read path, carry drill-down detail, so copy-paste SQL is populated. If the - /// engine reports insufficient collected history, surfaces the insufficient-data state. + /// engine reports insufficient collected history, surfaces the insufficient-data state; if it + /// reports an empty analysis window (#3524/#3551 — the span gate passed on lifetime history but + /// the window itself collected nothing, a dead-collector shape), surfaces the window-empty + /// notice instead of the false all-clear an empty findings list would render as. /// private async void GenerateNowButton_Click(object sender, RoutedEventArgs e) { @@ -243,6 +246,15 @@ _scheduleManager is null return; } + /* #3524/#3551: zero facts in the window is a dead-collector shape, not a clean bill of + health — mapping the empty findings list below would render the all-clear. */ + if (analysisService.WindowEmptyMessage is { Length: > 0 } windowEmptyMessage) + { + StatusText.Text = string.Empty; + ApplyViewModel(LiteRecommendationsViewModel.WindowEmpty(windowEmptyMessage)); + return; + } + StatusText.Text = string.Empty; var items = LiteRecommendationsReader.MapFindings(findings, serverName); ApplyViewModel(LiteRecommendationsViewModel.FromItems(items, ServerTimeHelper.UtcOffsetMinutes)); @@ -324,21 +336,33 @@ private void ApplyViewModel(LiteRecommendationsViewModel vm) SectionsScroll.Visibility = Visibility.Collapsed; EmptyMessage.Visibility = Visibility.Collapsed; InsufficientDataMessage.Visibility = Visibility.Collapsed; + WindowEmptyMessage.Visibility = Visibility.Collapsed; break; case LiteRecommendationsState.InsufficientData: LoadingOverlay.IsLoading = false; SectionsScroll.Visibility = Visibility.Collapsed; EmptyMessage.Visibility = Visibility.Collapsed; + WindowEmptyMessage.Visibility = Visibility.Collapsed; InsufficientDataMessage.Text = vm.InsufficientDataMessage; InsufficientDataMessage.Visibility = Visibility.Visible; break; + case LiteRecommendationsState.WindowEmpty: + LoadingOverlay.IsLoading = false; + SectionsScroll.Visibility = Visibility.Collapsed; + EmptyMessage.Visibility = Visibility.Collapsed; + InsufficientDataMessage.Visibility = Visibility.Collapsed; + WindowEmptyMessage.Text = vm.WindowEmptyMessage; + WindowEmptyMessage.Visibility = Visibility.Visible; + break; + case LiteRecommendationsState.Empty: LoadingOverlay.IsLoading = false; SectionsList.ItemsSource = null; SectionsScroll.Visibility = Visibility.Collapsed; InsufficientDataMessage.Visibility = Visibility.Collapsed; + WindowEmptyMessage.Visibility = Visibility.Collapsed; EmptyMessage.Visibility = Visibility.Visible; break; @@ -347,6 +371,7 @@ private void ApplyViewModel(LiteRecommendationsViewModel vm) LoadingOverlay.IsLoading = false; EmptyMessage.Visibility = Visibility.Collapsed; InsufficientDataMessage.Visibility = Visibility.Collapsed; + WindowEmptyMessage.Visibility = Visibility.Collapsed; SectionsList.ItemsSource = vm.Sections; SectionsScroll.Visibility = Visibility.Visible; break; From ec195bc5fb19e3e4c97bb44a0e3c53ab3e3e4555 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 01:03:34 -0400 Subject: [PATCH 11/69] get_memory_trend joins the grants series so total_granted_mb carries real data (#3566) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * get_memory_trend joins the grants series: total_granted_mb carries real data Fixes #3548. #3529 shipped the honest placeholder — an explicit null with a note naming get_memory_grants — because the memory_stats source has no grant data. The complete fix joins the memory-grant series both stores already collect (the same v_memory_grant_stats sum the viewers' Memory Overview overlays plot) into the trend payload on both SKUs. The join is nearest-match within 30 seconds, because each collector stamps its own DateTime.UtcNow per run: same-cycle rows sit seconds apart, so an equality join returns nothing, while a wider match would smear a slower grants cadence across points it never measured. Zero-vs-unknown discipline holds on both sides of the match: a snapshot measuring nothing granted is a genuine 0.0, an uncovered point is null, and the granted_note survives only in payloads that have a null to explain — a fully covered window is not captioned with an apology. Darling's read is a new DarlingTrendReader const pinned byte-identical to the viewer's proven MemoryGrantTrendSql; Lite reuses its overlay read with an asOfUtc pass-through so the grants window matches the memory window's. Twin tests pin the pool-summed match, the genuine zero, the uncovered null, note presence/absence, and the 30s boundary (in at 30, out at 31); the same claims run live against Postgres, plus the byte-identical SQL pin. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx * Reword a doc comment the as_of census reads as code AsOfWindowAnchorTests string-scans anchored tool bodies for the literal DateTime.UtcNow, and the tolerance constant's doc comment named it while describing the collectors' per-run stamps. Both tools are genuinely anchored (the joined grants read takes the same windowEnd as the memory read); the comment now says "capture clock" and the census passes. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx --------- Co-authored-by: Claude Fable 5 --- .../DarlingMcpTrendToolsTests.cs | 30 ++- .../DarlingMemoryTrendGrantJoinTests.cs | 135 ++++++++++++ .../Darling.Tests/DarlingTrendEmptyTests.cs | 6 +- .../Mcp/DarlingMcpTrendTools.cs | 90 ++++++-- .../Mcp/DarlingTrendReader.cs | 46 +++- Darling/README.md | 2 +- Lite.Tests/MemoryTrendGrantJoinToolTests.cs | 198 ++++++++++++++++++ Lite.Tests/TrendEmptyParityToolTests.cs | 15 +- Lite/Mcp/McpMemoryTools.cs | 90 ++++++-- .../Services/LocalDataService.MemoryGrants.cs | 4 +- 10 files changed, 574 insertions(+), 42 deletions(-) create mode 100644 Darling/Darling.Tests/DarlingMemoryTrendGrantJoinTests.cs create mode 100644 Lite.Tests/MemoryTrendGrantJoinToolTests.cs diff --git a/Darling/Darling.Tests/DarlingMcpTrendToolsTests.cs b/Darling/Darling.Tests/DarlingMcpTrendToolsTests.cs index 6b7cdc830..3b7d88e27 100644 --- a/Darling/Darling.Tests/DarlingMcpTrendToolsTests.cs +++ b/Darling/Darling.Tests/DarlingMcpTrendToolsTests.cs @@ -20,6 +20,7 @@ using PerformanceMonitor.Common; using PerformanceMonitor.Darling.Service.Mcp; using PerformanceMonitor.Darling.Storage; +using PerformanceMonitor.Darling.Viewer; using Xunit; namespace Darling.Tests; @@ -101,15 +102,18 @@ public void ParamContract_ServerNameOptional_RequiredKeysAreNot() Assert.False(McpParams("get_query_trend").Single(x => x.Name == "database_name").Optional); } - /// #3529: the description promised granted memory while the payload shipped a literal 0. - /// It now points at get_memory_grants, the tool that actually serves the grants series. + /// #3529's description half, superseded by the #3548 join: the tool now DELIVERS granted + /// memory (joined per point from the grants series), so the description may promise it again — but it + /// must name the null gap rather than promising an always-filled field, and still point at + /// get_memory_grants as the series' own tool. [Fact] - public void MemoryTrend_Description_PointsAtTheGrantsTool_AndDoesNotPromiseGrantedMemory() + public void MemoryTrend_Description_PromisesTheJoinedGrantSeries_AndNamesTheNullGap() { var method = ToolMethods().Single(m => m.GetCustomAttribute()!.Name == "get_memory_trend"); var description = method.GetCustomAttribute()!.Description; - Assert.DoesNotContain("and granted memory", description, StringComparison.Ordinal); + Assert.Contains("granted memory joined per point", description, StringComparison.Ordinal); + Assert.Contains("total_granted_mb is null", description, StringComparison.Ordinal); Assert.Contains("get_memory_grants", description, StringComparison.Ordinal); } @@ -125,6 +129,24 @@ public void MemoryTrendSql_WindowedBothSides_CastsNumericToDouble() Assert.Contains("collection_time <= $3", sql, StringComparison.Ordinal); } + /// + /// #3548: the grants-series read the get_memory_trend join rides on — byte-identical to the viewer's + /// proven overlay read (the reader's doctrine), so the MCP payload and the Memory Overview overlay can + /// never disagree about what the grants series says. + /// + [Fact] + public void MemoryGrantTrendSql_IsTheViewersOverlayRead_ByteForByte() + { + Assert.Equal(ViewerDataService.MemoryGrantTrendSql, DarlingTrendReader.MemoryGrantTrendSql); + + var sql = DarlingTrendReader.MemoryGrantTrendSql; + Assert.Contains("FROM v_memory_grant_stats", sql, StringComparison.Ordinal); + Assert.Contains("CAST(SUM(granted_memory_mb) AS double precision)", sql, StringComparison.Ordinal); + Assert.Contains("GROUP BY collection_time", sql, StringComparison.Ordinal); + Assert.Contains("collection_time >= $2", sql, StringComparison.Ordinal); + Assert.Contains("collection_time <= $3", sql, StringComparison.Ordinal); + } + [Fact] public void PerfmonTrendSql_SingleCounter_SumsInstances_CastsBigint() { diff --git a/Darling/Darling.Tests/DarlingMemoryTrendGrantJoinTests.cs b/Darling/Darling.Tests/DarlingMemoryTrendGrantJoinTests.cs new file mode 100644 index 000000000..94903ba30 --- /dev/null +++ b/Darling/Darling.Tests/DarlingMemoryTrendGrantJoinTests.cs @@ -0,0 +1,135 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Text.Json; +using System.Threading; +using System.Threading.Tasks; +using Npgsql; +using PerformanceMonitor.Collectors; +using PerformanceMonitor.Darling.Service.Mcp; +using PerformanceMonitor.Darling.Storage; +using Xunit; + +namespace Darling.Tests; + +/// +/// #3548: get_memory_trend joins the memory-grant series so total_granted_mb carries real data — the +/// complete fix #3529's null was the honest placeholder for. Lite's twin coverage is +/// MemoryTrendGrantJoinToolTests; the two pin the same three claims so the SKUs cannot drift: a +/// matched point carries the pool-summed measurement, a matched point measuring NOTHING granted is a +/// genuine 0.0 (a snapshot existed — zero is a measurement, not a fabrication), and an unmatched point is +/// null with the envelope's granted_note explaining the gap — a note that vanishes entirely when every +/// point matched. The join is nearest-match within 30 seconds because each collector stamps its own +/// DateTime.UtcNow per run: same-cycle rows sit seconds apart, so equality returns nothing, while a wider +/// match would smear a slower grants cadence across points it never measured. +/// +/// Gated on DARLING_TEST_PG like every other live class. +/// +[Collection("live-postgres")] +public sealed class DarlingMemoryTrendGrantJoinTests +{ + private const int ServerId = -949583; + private const string ServerName = "grant-join"; + + private static string? ConnectionString => Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + + [Fact] + public async Task JoinedPoints_CarryThePoolSum_AGenuineZero_ANullForTheUncovered_AndTheNoteOnlyWithAGap() + { + var cs = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(cs), + "Set DARLING_TEST_PG to a Postgres connection string to run the live grant-join test."); + + var ct = TestContext.Current.CancellationToken; + using var connection = new NpgsqlConnection(cs); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + await DeleteRowsAsync(connection, ct); + + await using var postgres = NpgsqlDataSource.Create(cs!); + var bodySucceeded = false; + + try + { + await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); + + /* ── fully covered: one memory point, one grants snapshot 3s later — value, and NO note ── */ + var t0 = MinutesAgo(30); + await SeedMemoryAsync(connection, ct, t0); + await SeedGrantAsync(connection, ct, t0.AddSeconds(3), poolId: 2, grantedMb: 50m); + + var covered = JsonDocument.Parse(await DarlingMcpTrendTools.GetMemoryTrend(postgres, ServerName, 4)).RootElement; + Assert.Equal(50.0, covered.GetProperty("trend")[0].GetProperty("total_granted_mb").GetDouble(), precision: 6); + Assert.False(covered.TryGetProperty("granted_note", out _), + "a window the grants series fully covers must not be captioned with a gap note"); + + /* ── the gap shapes: a two-pool sum, a genuine zero, and an uncovered point ── */ + var t1 = t0.AddMinutes(1); + var t2 = t0.AddMinutes(2); + var t3 = t0.AddMinutes(3); + await SeedMemoryAsync(connection, ct, t1); + await SeedMemoryAsync(connection, ct, t2); + await SeedMemoryAsync(connection, ct, t3); + + var snap1 = t1.AddSeconds(4); + await SeedGrantAsync(connection, ct, snap1, poolId: 1, grantedMb: 25m); + await SeedGrantAsync(connection, ct, snap1, poolId: 2, grantedMb: 100m); + await SeedGrantAsync(connection, ct, t2.AddSeconds(6), poolId: 2, grantedMb: 0m); + /* nothing anywhere near t3 */ + + var root = JsonDocument.Parse(await DarlingMcpTrendTools.GetMemoryTrend(postgres, ServerName, 4)).RootElement; + var trend = root.GetProperty("trend"); + Assert.Equal(4, trend.GetArrayLength()); + Assert.Equal(125.0, trend[1].GetProperty("total_granted_mb").GetDouble(), precision: 6); + Assert.Equal(JsonValueKind.Number, trend[2].GetProperty("total_granted_mb").ValueKind); + Assert.Equal(0.0, trend[2].GetProperty("total_granted_mb").GetDouble(), precision: 6); + Assert.Equal(JsonValueKind.Null, trend[3].GetProperty("total_granted_mb").ValueKind); + + var note = root.GetProperty("granted_note").GetString()!; + Assert.Contains("get_memory_grants", note, StringComparison.Ordinal); + Assert.Contains("30 seconds", note, StringComparison.Ordinal); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, async (cleanup, cleanupCt) => + await DeleteRowsAsync(cleanup, cleanupCt)); + } + } + + private static DateTime MinutesAgo(int minutes) => + DarlingMcpTestData.TruncateToSeconds(DateTime.UtcNow.AddMinutes(-minutes)); + + private static async Task SeedMemoryAsync(NpgsqlConnection connection, CancellationToken ct, DateTime t) => + await DarlingMcpTestData.ExecAsync(connection, ct, @" +INSERT INTO memory_stats + (collection_id, collection_time, server_id, server_name, + total_server_memory_mb, target_server_memory_mb, buffer_pool_mb, plan_cache_mb) +VALUES ($1, $2, $3, $4, $5, $6, $7, $8)", + CollectionIdGenerator.Next(), DarlingMcpTestData.Naive(t), ServerId, ServerName, + 40000m, 49152m, 35000m, 5000m); + + private static async Task SeedGrantAsync(NpgsqlConnection connection, CancellationToken ct, DateTime t, int poolId, decimal grantedMb) => + await DarlingMcpTestData.ExecAsync(connection, ct, @" +INSERT INTO memory_grant_stats + (collection_id, collection_time, server_id, server_name, + resource_semaphore_id, pool_id, granted_memory_mb) +VALUES ($1, $2, $3, $4, $5, $6, $7)", + CollectionIdGenerator.Next(), DarlingMcpTestData.Naive(t), ServerId, ServerName, + (short)0, poolId, grantedMb); + + private static async Task DeleteRowsAsync(NpgsqlConnection connection, CancellationToken ct) + { + await DarlingMcpTestData.ExecAsync(connection, ct, "DELETE FROM memory_grant_stats WHERE server_id = $1", ServerId); + await DarlingMcpTestData.ExecAsync(connection, ct, "DELETE FROM memory_stats WHERE server_id = $1", ServerId); + await DarlingMcpTestData.ExecAsync(connection, ct, "DELETE FROM servers WHERE server_id = $1", ServerId); + await DarlingMcpTestData.ExecAsync(connection, ct, "DELETE FROM config_monitored_servers WHERE server_id = $1", ServerId); + } +} diff --git a/Darling/Darling.Tests/DarlingTrendEmptyTests.cs b/Darling/Darling.Tests/DarlingTrendEmptyTests.cs index 9ac77e6be..8e60ff4fe 100644 --- a/Darling/Darling.Tests/DarlingTrendEmptyTests.cs +++ b/Darling/Darling.Tests/DarlingTrendEmptyTests.cs @@ -96,8 +96,10 @@ await DarlingMcpTrendTools.GetQueryDurationTrend(postgres, ServerName, 4), Assert.True(root.GetProperty("trend").GetArrayLength() > 0); } - /* #3529: total_granted_mb is an explicit null with the envelope naming the real source — - never the literal 0.0 an agent read as "granted was 0 all window". */ + /* #3529, now the #3548 join's UNCOVERED arm (no memory_grant_stats rows seeded near these + points): total_granted_mb stays an explicit null with the envelope naming the real source — + never the literal 0.0 an agent read as "granted was 0 all window". The covered arms live in + DarlingMemoryTrendGrantJoinTests. */ var memoryRoot = JsonDocument.Parse(memoryPayload).RootElement; Assert.Contains("get_memory_grants", memoryRoot.GetProperty("granted_note").GetString(), StringComparison.Ordinal); foreach (var point in memoryRoot.GetProperty("trend").EnumerateArray()) diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTrendTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTrendTools.cs index 2a1bf380e..c9029311f 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTrendTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTrendTools.cs @@ -46,7 +46,7 @@ namespace PerformanceMonitor.Darling.Service.Mcp; [McpServerToolType] public sealed class DarlingMcpTrendTools { - [McpServerTool(Name = "get_memory_trend"), Description("Gets memory usage trend over time: total server memory, target memory, buffer pool, and plan cache. total_granted_mb in this payload is always null — granted memory is a separate series; use get_memory_grants for it. Useful for identifying memory growth patterns or pressure periods.")] + [McpServerTool(Name = "get_memory_trend"), Description("Gets memory usage trend over time: total server memory, target memory, buffer pool, plan cache, and granted memory joined per point from the memory-grant series. total_granted_mb is null on points the grants series does not cover — a granted_note explains any gap; use get_memory_grants for grant detail. Useful for identifying memory growth patterns or pressure periods.")] public static async Task GetMemoryTrend( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -87,27 +87,44 @@ neither. A server that collected fine and was simply quiet in THIS window wants $"No memory stats have EVER been recorded for {resolved.ServerName}. This is not an empty window — the memory_stats collector has stored nothing at all for this server. Check that collection is running and that the server is enabled; get_memory_stats will be equally empty until it does."); } - var result = points.Select(p => new + var grants = await DarlingTrendReader.GetMemoryGrantTrendAsync(postgres, resolved.ServerId, now.AddHours(-hours_back), now); + var granted = AlignGrantSeries( + points.Select(p => p.CollectionTime).ToArray(), + grants.Select(g => (g.CollectionTime, g.TotalGrantedMb)).ToArray()); + + var result = points.Select((p, i) => new { time = p.CollectionTime.ToString("o"), total_server_memory_mb = p.TotalServerMemoryMb, target_server_memory_mb = p.TargetServerMemoryMb, buffer_pool_mb = p.BufferPoolMb, plan_cache_mb = p.PlanCacheMb, - /* The memory_stats source carries no grant data — the grant overlay is a separate series - (get_memory_grants). null, not the 0 placeholder this used to ship: a literal zero read - as "granted was 0 all window" and steered callers away from memory grants at exactly the - wrong moment (#3529). Field-for-field parity with Lite's tool, which nulls it the same way. */ - total_granted_mb = (double?)null + /* Joined from the memory-grant series (#3548): the nearest memory_grant_stats snapshot + within 30 seconds of this memory sample, SUM(granted_memory_mb) across pools — the same + series the viewer's Memory Overview overlay charts. null when no snapshot aligns, never + a fabricated 0: a literal zero read as "granted was 0 all window" and steered callers + away from memory grants at exactly the wrong moment (#3529). A genuine 0.0 still appears + when a snapshot exists with nothing granted. Field-for-field parity with Lite's tool, + which joins it the same way. */ + total_granted_mb = granted[i] }); - return JsonSerializer.Serialize(new - { - server = resolved.ServerName, - hours_back, - granted_note = "total_granted_mb is not sourced by this tool — the memory_stats series carries no grant data. Use get_memory_grants for the granted-memory series.", - trend = result - }, McpHelpers.JsonOptions); + /* The note exists to explain null points; a fully covered window gets no note at all rather + than a null-valued key (JsonOptions writes nulls). */ + return granted.Any(v => v is null) + ? JsonSerializer.Serialize(new + { + server = resolved.ServerName, + hours_back, + granted_note = GrantGapNote, + trend = result + }, McpHelpers.JsonOptions) + : JsonSerializer.Serialize(new + { + server = resolved.ServerName, + hours_back, + trend = result + }, McpHelpers.JsonOptions); } catch (Exception ex) { @@ -115,6 +132,51 @@ wrong moment (#3529). Field-for-field parity with Lite's tool, which nulls it th } } + /// + /// Half the 1-minute cadence floor both collectors share (CollectorScheduleDefaults). The two + /// series each stamp their own capture clock per collector run, so same-cycle rows sit seconds + /// apart and can never be equality-joined — while a grants series on a slower cadence must NOT smear + /// onto every memory point. Within half the finest cadence, at most one snapshot can claim a point. + /// + private static readonly TimeSpan GrantJoinTolerance = TimeSpan.FromSeconds(30); + + /// Why a point is null, stated once per payload — and only when a null point exists. + private const string GrantGapNote = + "total_granted_mb is null where no memory-grant snapshot lies within 30 seconds of the memory sample — the memory_grant_stats series is collected on its own schedule, so a gap means no grant measurement at that moment, not zero granted. Use get_memory_grants for the full grant picture."; + + /// + /// Nearest-match join of the memory-grant series onto the memory-trend points (#3548): for each trend + /// point, the closest grants snapshot within , else null — no grant + /// measurement at that moment, which is not the same claim as a genuine 0.0 from a snapshot with + /// nothing granted. Both inputs are time-ascending (both reads ORDER BY collection_time), so one + /// forward pointer finds every nearest neighbor. Twin of Lite's + /// McpMemoryTools.AlignGrantSeries — the two must stay in step so both SKUs join the same way. + /// + private static double?[] AlignGrantSeries( + DateTime[] trendTimes, + (DateTime Time, double TotalGrantedMb)[] grants) + { + var aligned = new double?[trendTimes.Length]; + if (grants.Length == 0) return aligned; + + var g = 0; + for (var t = 0; t < trendTimes.Length; t++) + { + var target = trendTimes[t]; + while (g + 1 < grants.Length && (grants[g + 1].Time - target).Duration() <= (grants[g].Time - target).Duration()) + { + g++; + } + + if ((grants[g].Time - target).Duration() <= GrantJoinTolerance) + { + aligned[t] = grants[g].TotalGrantedMb; + } + } + + return aligned; + } + [McpServerTool(Name = "get_perfmon_trend"), Description("Gets a time-series trend for a specific performance counter. Use get_perfmon_stats first to see available counter names.")] public static async Task GetPerfmonTrend( NpgsqlDataSource postgres, diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingTrendReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingTrendReader.cs index 19a117191..1711a97bd 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingTrendReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingTrendReader.cs @@ -42,12 +42,20 @@ internal static class DarlingTrendReader /// One memory-trend point: the four MB metrics per collection (Lite's MemoryTrendPoint /// minus its TotalGrantedMb overlay field, which the memory_stats source never fills — the tool - /// publishes it as an explicit null with a note naming get_memory_grants (#3529) — see + /// joins it per point from the grants series (), null with a note + /// naming get_memory_grants where no snapshot aligns (#3529, #3548) — see /// ). public sealed record MemoryTrendPoint( DateTime CollectionTime, double TotalServerMemoryMb, double TargetServerMemoryMb, double BufferPoolMb, double PlanCacheMb); + /// One memory-grant-trend point: total granted workspace memory summed across every resource + /// pool at one grants collection (#3548) — the series + /// joins onto the memory trend, and the same series the viewer's Memory Overview overlay plots. Its + /// collection_times are the grants collector's OWN stamps, seconds apart from the memory series' even + /// in the same cycle, which is why the join is nearest-match rather than equality. + public sealed record MemoryGrantTrendPoint(DateTime CollectionTime, double TotalGrantedMb); + /// One perfmon-trend point for a single counter: the counter value, the per-interval delta, /// and the wall-clock seconds that delta covers, all summed across the counter's instances at that /// collection (Lite's PerfmonTrendPoint, plus the interval Lite does not carry). @@ -177,6 +185,42 @@ public static async Task> GetMemoryTrendAsync( return items; } + /// + /// The memory-grant trend — the viewer's MemoryGrantTrendSql (Lite's + /// GetMemoryGrantTrendAsync): total granted MB across all pools per grants collection over the + /// window, for the join get_memory_trend makes onto the memory series (#3548). $1 server_id, $2/$3 + /// window (naive UTC). + /// + public const string MemoryGrantTrendSql = """ + SELECT + collection_time, + CAST(SUM(granted_memory_mb) AS double precision) AS total_granted_mb + FROM v_memory_grant_stats + WHERE server_id = $1 + AND collection_time >= $2 + AND collection_time <= $3 + GROUP BY collection_time + ORDER BY collection_time + """; + + public static async Task> GetMemoryGrantTrendAsync( + NpgsqlDataSource postgres, int serverId, DateTime startUtc, DateTime endUtc, CancellationToken cancellationToken = default) + { + var items = new List(); + await using var command = postgres.CreateCommand(MemoryGrantTrendSql); + command.CommandTimeout = McpCommandDeadlines.ReadSeconds; + DarlingMcpReadParameters.AddWindow(command, serverId, startUtc, endUtc); + await using var reader = await command.ExecuteReaderAsync(cancellationToken); + while (await reader.ReadAsync(cancellationToken)) + { + items.Add(new MemoryGrantTrendPoint( + reader.GetDateTime(0), + reader.IsDBNull(1) ? 0 : reader.GetDouble(1))); + } + + return items; + } + /* ─────────────────────────── perfmon trend ─────────────────────────── */ /// diff --git a/Darling/README.md b/Darling/README.md index 2824acf72..b1a4b573c 100644 --- a/Darling/README.md +++ b/Darling/README.md @@ -539,7 +539,7 @@ The embedded MCP server, over Streamable HTTP bound to `localhost` by default (s - **Trend data-read tools** — windowed time-series siblings of the core reads, each a stored read of the collected series over the window (BOTH-sides, naive-UTC): - `get_memory_trend` (total / target server memory, buffer pool, plan cache over time), `get_perfmon_trend` (a single counter's value + delta, `counter_name` required), `get_file_io_trend` (per-database read/write latency, top-10 busiest files), `get_query_trend` (one query's per-collection history by `query_hash` + `database_name`), `get_query_duration_trend` (overall elapsed-ms/sec + executions/sec). - Each mirrors the viewer's proven chart read (byte-identical Postgres SQL); the shape follows Lite where the SKUs diverge. `get_perfmon_trend` reproduces Lite's miss vocabulary (Page Life Expectancy is intentionally not collected; an unknown counter hands back the collected names). `get_memory_trend` carries a `total_granted_mb` field for field-for-field parity with Lite, published as an explicit null with a `granted_note` naming `get_memory_grants` — the memory_stats source has no grant data, and the grants series is its own tool (#3529). + Each mirrors the viewer's proven chart read (byte-identical Postgres SQL); the shape follows Lite where the SKUs diverge. `get_perfmon_trend` reproduces Lite's miss vocabulary (Page Life Expectancy is intentionally not collected; an unknown counter hands back the collected names). `get_memory_trend` carries a `total_granted_mb` field for field-for-field parity with Lite, joined per point from the memory-grant series (the same `v_memory_grant_stats` sum the viewer's overlay charts plot) — a point no grants snapshot covers within 30 seconds stays an explicit null, with a `granted_note` naming `get_memory_grants`, never a fabricated 0 (#3529, #3548). - **System-health parse-on-read tools** — the Dashboard's `get_health_parser_*` family, over Darling's raw `system_health_events`: - `get_health_parser_system_health` (corruption + contention counters), `get_health_parser_severe_errors` (severity ≥ 19, with `database_id` resolved to a name), `get_health_parser_scheduler_issues`, `get_health_parser_memory_conditions`, `get_health_parser_memory_broker`, `get_health_parser_memory_node_oom`, `get_health_parser_cpu_tasks`, `get_health_parser_io_issues`. diff --git a/Lite.Tests/MemoryTrendGrantJoinToolTests.cs b/Lite.Tests/MemoryTrendGrantJoinToolTests.cs new file mode 100644 index 000000000..60cb2e381 --- /dev/null +++ b/Lite.Tests/MemoryTrendGrantJoinToolTests.cs @@ -0,0 +1,198 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.IO; +using System.Text.Json; +using System.Threading.Tasks; +using DuckDB.NET.Data; +using PerformanceMonitorLite.Database; +using PerformanceMonitorLite.Mcp; +using PerformanceMonitorLite.Models; +using PerformanceMonitorLite.Services; +using Xunit; + +namespace PerformanceMonitorLite.Tests; + +/// +/// #3548: get_memory_trend joins the memory-grant series so total_granted_mb carries real data — the +/// complete fix #3529's null was the honest placeholder for. The join is nearest-match within 30 seconds +/// because the two collectors each stamp their own DateTime.UtcNow per run: same-cycle rows sit seconds +/// apart, so an equality join returns nothing, while a wider match would smear a slower grants cadence +/// across points it never measured. +/// +/// The three claims worth pinning are the three an agent acts on: a matched point carries the +/// pool-summed measurement, a matched point measuring NOTHING granted is a genuine 0.0 (a snapshot +/// existed — zero is a measurement here, not a fabrication), and an unmatched point is null with the +/// envelope's granted_note explaining the gap — which vanishes entirely when every point matched, so a +/// clean window is not captioned with an apology. +/// +public sealed class MemoryTrendGrantJoinToolTests : IClassFixture, IDisposable +{ + private const string ServerName = "GrantJoinSrv"; + + private readonly DuckDbInitializer _duckDb; + private readonly string _configDir; + private readonly ServerManager _serverManager; + private readonly int _serverId; + private DuckDBConnection? _seedConn; + private long _nextId = 910000; + + public MemoryTrendGrantJoinToolTests(SharedDuckDbFixture fixture) + { + fixture.ResetData(); + _duckDb = fixture.DuckDb; + + _configDir = Path.Combine(Path.GetTempPath(), "pmlite-grantjoin-" + Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(_configDir); + _serverManager = new ServerManager(_configDir); + + var server = new ServerConnection + { + Id = Guid.NewGuid().ToString(), + ServerName = ServerName, + IsEnabled = true, + }; + _serverManager.AddServer(server); + + /* Derived, not stored -- seeding under a hardcoded id would write rows the tool looks past. */ + _serverId = RemoteCollectorService.GetDeterministicHashCode( + RemoteCollectorService.GetServerNameForStorage(server)); + } + + public void Dispose() + { + _seedConn?.Dispose(); + try { Directory.Delete(_configDir, recursive: true); } catch (IOException) { /* temp dir */ } + } + + [Fact] + public async Task JoinedPoints_CarryThePoolSum_AGenuineZero_AndANullForTheUncoveredPoint() + { + var t0 = DateTime.UtcNow.AddMinutes(-30); + var t1 = t0.AddMinutes(1); + var t2 = t0.AddMinutes(2); + + await SeedMemoryAsync(t0); + await SeedMemoryAsync(t1); + await SeedMemoryAsync(t2); + + /* One grants snapshot 4s after t0 with TWO pools (the SUM is the contract, not a row pick), one + 6s after t1 measuring nothing granted, and nothing anywhere near t2. The offsets are the real + shape: each collector stamps its own UtcNow, so same-cycle rows land seconds apart. */ + var snap0 = t0.AddSeconds(4); + await SeedGrantAsync(snap0, poolId: 1, grantedMb: 25.0); + await SeedGrantAsync(snap0, poolId: 2, grantedMb: 100.0); + await SeedGrantAsync(t1.AddSeconds(6), poolId: 2, grantedMb: 0.0); + + var payload = await McpMemoryTools.GetMemoryTrend(new LocalDataService(_duckDb), _serverManager, ServerName, 4); + var root = JsonDocument.Parse(payload).RootElement; + + var trend = root.GetProperty("trend"); + Assert.Equal(3, trend.GetArrayLength()); + Assert.Equal(125.0, trend[0].GetProperty("total_granted_mb").GetDouble(), precision: 6); + + /* The genuine zero: a snapshot existed and measured nothing granted. Distinguishable from the + uncovered point below only because the join keeps zero-vs-unknown apart. */ + Assert.Equal(JsonValueKind.Number, trend[1].GetProperty("total_granted_mb").ValueKind); + Assert.Equal(0.0, trend[1].GetProperty("total_granted_mb").GetDouble(), precision: 6); + + Assert.Equal(JsonValueKind.Null, trend[2].GetProperty("total_granted_mb").ValueKind); + + var note = root.GetProperty("granted_note").GetString()!; + Assert.Contains("get_memory_grants", note, StringComparison.Ordinal); + Assert.Contains("30 seconds", note, StringComparison.Ordinal); + } + + [Fact] + public async Task AFullyCoveredWindow_CarriesNoGrantedNoteAtAll() + { + var t0 = DateTime.UtcNow.AddMinutes(-20); + await SeedMemoryAsync(t0); + await SeedGrantAsync(t0.AddSeconds(3), poolId: 2, grantedMb: 50.0); + + var payload = await McpMemoryTools.GetMemoryTrend(new LocalDataService(_duckDb), _serverManager, ServerName, 4); + var root = JsonDocument.Parse(payload).RootElement; + + Assert.Equal(50.0, root.GetProperty("trend")[0].GetProperty("total_granted_mb").GetDouble(), precision: 6); + Assert.False(root.TryGetProperty("granted_note", out _), + "a window the grants series fully covers must not be captioned with a gap note"); + } + + [Fact] + public async Task TheToleranceBoundary_Is30Seconds_InAt30_OutAt31() + { + var tIn = DateTime.UtcNow.AddMinutes(-40); + var tOut = tIn.AddMinutes(5); + + await SeedMemoryAsync(tIn); + await SeedMemoryAsync(tOut); + await SeedGrantAsync(tIn.AddSeconds(30), poolId: 2, grantedMb: 75.0); + await SeedGrantAsync(tOut.AddSeconds(31), poolId: 2, grantedMb: 999.0); + + var payload = await McpMemoryTools.GetMemoryTrend(new LocalDataService(_duckDb), _serverManager, ServerName, 4); + var trend = JsonDocument.Parse(payload).RootElement.GetProperty("trend"); + + Assert.Equal(75.0, trend[0].GetProperty("total_granted_mb").GetDouble(), precision: 6); + Assert.Equal(JsonValueKind.Null, trend[1].GetProperty("total_granted_mb").ValueKind); + } + + private async Task SeedConnectionAsync() + { + if (_seedConn is null) + { + _seedConn = _duckDb.CreateConnection(); + await _seedConn.OpenAsync(); + } + return _seedConn; + } + + private async Task SeedMemoryAsync(DateTime collectionTimeUtc) + { + using var readLock = _duckDb.AcquireReadLock(); + var connection = await SeedConnectionAsync(); + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +INSERT INTO memory_stats + (collection_id, collection_time, server_id, server_name, + total_physical_memory_mb, available_physical_memory_mb, + target_server_memory_mb, total_server_memory_mb, buffer_pool_mb, plan_cache_mb) +VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10)"; + cmd.Parameters.Add(new DuckDBParameter { Value = _nextId++ }); + cmd.Parameters.Add(new DuckDBParameter { Value = DateTime.SpecifyKind(collectionTimeUtc, DateTimeKind.Unspecified) }); + cmd.Parameters.Add(new DuckDBParameter { Value = _serverId }); + cmd.Parameters.Add(new DuckDBParameter { Value = ServerName }); + cmd.Parameters.Add(new DuckDBParameter { Value = 65536.0 }); + cmd.Parameters.Add(new DuckDBParameter { Value = 8192.0 }); + cmd.Parameters.Add(new DuckDBParameter { Value = 49152.0 }); + cmd.Parameters.Add(new DuckDBParameter { Value = 40000.0 }); + cmd.Parameters.Add(new DuckDBParameter { Value = 35000.0 }); + cmd.Parameters.Add(new DuckDBParameter { Value = 5000.0 }); + await cmd.ExecuteNonQueryAsync(); + } + + private async Task SeedGrantAsync(DateTime collectionTimeUtc, int poolId, double grantedMb) + { + using var readLock = _duckDb.AcquireReadLock(); + var connection = await SeedConnectionAsync(); + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +INSERT INTO memory_grant_stats + (collection_id, collection_time, server_id, server_name, + resource_semaphore_id, pool_id, granted_memory_mb) +VALUES ($1, $2, $3, $4, $5, $6, $7)"; + cmd.Parameters.Add(new DuckDBParameter { Value = _nextId++ }); + cmd.Parameters.Add(new DuckDBParameter { Value = DateTime.SpecifyKind(collectionTimeUtc, DateTimeKind.Unspecified) }); + cmd.Parameters.Add(new DuckDBParameter { Value = _serverId }); + cmd.Parameters.Add(new DuckDBParameter { Value = ServerName }); + cmd.Parameters.Add(new DuckDBParameter { Value = (short)0 }); + cmd.Parameters.Add(new DuckDBParameter { Value = poolId }); + cmd.Parameters.Add(new DuckDBParameter { Value = grantedMb }); + await cmd.ExecuteNonQueryAsync(); + } +} diff --git a/Lite.Tests/TrendEmptyParityToolTests.cs b/Lite.Tests/TrendEmptyParityToolTests.cs index 84d0d08ac..d14c4da96 100644 --- a/Lite.Tests/TrendEmptyParityToolTests.cs +++ b/Lite.Tests/TrendEmptyParityToolTests.cs @@ -116,7 +116,10 @@ public async Task QueryDurationTrend_NeverCollected_AndAQuietWindow_AreDifferent /// /// #3529: the payload used to carry a hardcoded total_granted_mb of 0.0 — an agent investigating /// RESOURCE_SEMAPHORE read "granted was 0 all window" and ruled out memory grants, the exact wrong - /// turn. The field is now an explicit null and the envelope names the real source. + /// turn. With the #3548 join this is now specifically the UNCOVERED window (no memory_grant_stats + /// rows seeded anywhere near these points): every point stays an explicit null and the envelope's + /// note names the real source — never a fabricated zero. The covered arms live in + /// . /// [Fact] public async Task MemoryTrend_GrantedMemoryIsNullWithANoteNamingTheGrantsTool_NeverALiteralZero() @@ -134,14 +137,18 @@ public async Task MemoryTrend_GrantedMemoryIsNullWithANoteNamingTheGrantsTool_Ne } } - /// #3529's description half: the tool promised granted memory it never delivered. + /// #3529's description half, superseded by the #3548 join: the tool now DELIVERS granted + /// memory (joined per point from the grants series), so the description may promise it again — but it + /// must name the null gap rather than promising an always-filled field, and still point at + /// get_memory_grants as the series' own tool. [Fact] - public void MemoryTrend_Description_PointsAtTheGrantsTool_AndDoesNotPromiseGrantedMemory() + public void MemoryTrend_Description_PromisesTheJoinedGrantSeries_AndNamesTheNullGap() { var description = typeof(McpMemoryTools).GetMethod(nameof(McpMemoryTools.GetMemoryTrend))! .GetCustomAttribute()!.Description; - Assert.DoesNotContain("and granted memory", description, StringComparison.Ordinal); + Assert.Contains("granted memory joined per point", description, StringComparison.Ordinal); + Assert.Contains("total_granted_mb is null", description, StringComparison.Ordinal); Assert.Contains("get_memory_grants", description, StringComparison.Ordinal); } diff --git a/Lite/Mcp/McpMemoryTools.cs b/Lite/Mcp/McpMemoryTools.cs index d5851d16f..af766f3ad 100644 --- a/Lite/Mcp/McpMemoryTools.cs +++ b/Lite/Mcp/McpMemoryTools.cs @@ -48,7 +48,7 @@ public static async Task GetMemoryStats( } } - [McpServerTool(Name = "get_memory_trend"), Description("Gets memory usage trend over time: total server memory, target memory, buffer pool, and plan cache. total_granted_mb in this payload is always null — granted memory is a separate series; use get_memory_grants for it. Useful for identifying memory growth patterns or pressure periods.")] + [McpServerTool(Name = "get_memory_trend"), Description("Gets memory usage trend over time: total server memory, target memory, buffer pool, plan cache, and granted memory joined per point from the memory-grant series. total_granted_mb is null on points the grants series does not cover — a granted_note explains any gap; use get_memory_grants for grant detail. Useful for identifying memory growth patterns or pressure periods.")] public static async Task GetMemoryTrend( LocalDataService dataService, ServerManager serverManager, @@ -91,27 +91,43 @@ A bare empty array here told an MCP client nothing at all -- and Darling's twin $"No memory stats have EVER been recorded for {resolved.ServerName}. This is not an empty window — the memory_stats collector has stored nothing at all for this server. Check that collection is running and that the server is enabled; get_memory_stats will be equally empty until it does."); } - var result = points.Select(p => new + var grants = await dataService.GetMemoryGrantTrendAsync(resolved.ServerId, hours_back, asOfUtc: windowEnd); + var granted = AlignGrantSeries( + points.Select(p => p.CollectionTime).ToArray(), + grants.Select(g => (g.CollectionTime, g.TotalGrantedMb)).ToArray()); + + var result = points.Select((p, i) => new { time = p.CollectionTime.ToString("o"), total_server_memory_mb = p.TotalServerMemoryMb, target_server_memory_mb = p.TargetServerMemoryMb, buffer_pool_mb = p.BufferPoolMb, plan_cache_mb = p.PlanCacheMb, - /* GetMemoryTrendAsync is a memory_stats-only read and never assigns TotalGrantedMb — the - grant overlay is a separate series (get_memory_grants). null, not the model's default 0: - a literal zero read as "granted was 0 all window" and steered callers away from memory - grants at exactly the wrong moment (#3529). */ - total_granted_mb = (double?)null + /* Joined from the memory-grant series (#3548): the nearest memory_grant_stats snapshot + within 30 seconds of this memory sample, SUM(granted_memory_mb) across pools — the same + series the Memory Overview overlay charts. null when no snapshot aligns, never a + fabricated 0: a literal zero read as "granted was 0 all window" and steered callers away + from memory grants at exactly the wrong moment (#3529). A genuine 0.0 still appears when + a snapshot exists with nothing granted. */ + total_granted_mb = granted[i] }); - return JsonSerializer.Serialize(new - { - server = resolved.ServerName, - hours_back, - granted_note = "total_granted_mb is not sourced by this tool — the memory_stats series carries no grant data. Use get_memory_grants for the granted-memory series.", - trend = result - }, McpHelpers.JsonOptions); + /* The note exists to explain null points; a fully covered window gets no note at all rather + than a null-valued key (JsonOptions writes nulls). */ + return granted.Any(v => v is null) + ? JsonSerializer.Serialize(new + { + server = resolved.ServerName, + hours_back, + granted_note = GrantGapNote, + trend = result + }, McpHelpers.JsonOptions) + : JsonSerializer.Serialize(new + { + server = resolved.ServerName, + hours_back, + trend = result + }, McpHelpers.JsonOptions); } catch (Exception ex) { @@ -119,6 +135,52 @@ grants at exactly the wrong moment (#3529). */ } } + /// + /// Half the 1-minute cadence floor both collectors share (CollectorScheduleDefaults). The two + /// series each stamp their own capture clock per collector run, so same-cycle rows sit seconds + /// apart and can never be equality-joined — while a grants series on a slower cadence must NOT smear + /// onto every memory point. Within half the finest cadence, at most one snapshot can claim a point. + /// + private static readonly TimeSpan GrantJoinTolerance = TimeSpan.FromSeconds(30); + + /// Why a point is null, stated once per payload — and only when a null point exists. + private const string GrantGapNote = + "total_granted_mb is null where no memory-grant snapshot lies within 30 seconds of the memory sample — the memory_grant_stats series is collected on its own schedule, so a gap means no grant measurement at that moment, not zero granted. Use get_memory_grants for the full grant picture."; + + /// + /// Nearest-match join of the memory-grant series onto the memory-trend points (#3548): for each trend + /// point, the closest grants snapshot within , else null — no grant + /// measurement at that moment, which is not the same claim as a genuine 0.0 from a snapshot with + /// nothing granted. Both inputs are time-ascending (both reads ORDER BY collection_time), so one + /// forward pointer finds every nearest neighbor. Twin of Darling's + /// DarlingMcpTrendTools.AlignGrantSeries — the two must stay in step so both SKUs join the + /// same way. + /// + private static double?[] AlignGrantSeries( + DateTime[] trendTimes, + (DateTime Time, double TotalGrantedMb)[] grants) + { + var aligned = new double?[trendTimes.Length]; + if (grants.Length == 0) return aligned; + + var g = 0; + for (var t = 0; t < trendTimes.Length; t++) + { + var target = trendTimes[t]; + while (g + 1 < grants.Length && (grants[g + 1].Time - target).Duration() <= (grants[g].Time - target).Duration()) + { + g++; + } + + if ((grants[g].Time - target).Duration() <= GrantJoinTolerance) + { + aligned[t] = grants[g].TotalGrantedMb; + } + } + + return aligned; + } + [McpServerTool(Name = "get_memory_clerks"), Description("Gets the top memory consumers by memory clerk type — shows which SQL Server components are using the most memory.")] public static async Task GetMemoryClerks( LocalDataService dataService, diff --git a/Lite/Services/LocalDataService.MemoryGrants.cs b/Lite/Services/LocalDataService.MemoryGrants.cs index 03b1a02e8..9c2543b65 100644 --- a/Lite/Services/LocalDataService.MemoryGrants.cs +++ b/Lite/Services/LocalDataService.MemoryGrants.cs @@ -18,12 +18,12 @@ public partial class LocalDataService /// /// Gets memory grant trend — total granted MB per collection snapshot for the Memory Overview overlay. /// - public async Task> GetMemoryGrantTrendAsync(int serverId, int hoursBack = 4, DateTime? fromDate = null, DateTime? toDate = null) + public async Task> GetMemoryGrantTrendAsync(int serverId, int hoursBack = 4, DateTime? fromDate = null, DateTime? toDate = null, DateTime? asOfUtc = null) { using var connection = await OpenConnectionAsync(); using var command = connection.CreateCommand(); - var (startTime, endTime) = GetTimeRange(hoursBack, fromDate, toDate, asOfUtc: null, SelectedServerTabUtcOffsetMinutes); + var (startTime, endTime) = GetTimeRange(hoursBack, fromDate, toDate, asOfUtc, SelectedServerTabUtcOffsetMinutes); command.CommandText = @" SELECT From e43e8466ad4b9e9fc54dddc7dac6ab74a8b8728b Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 01:39:33 -0400 Subject: [PATCH 12/69] The daily classifier bands deadlocks as a rate through the card band's tiers, not any-deadlock-is-Critical (#3564) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * The daily classifier bands deadlocks as a rate through the card band's tiers, not any-deadlock-is-Critical (#3525) The shared DailyHealthBandCalculator still shipped #3368's un-fixed twin: signals.Deadlocks > 0 banded the whole day Critical. On the measured 43-server fleet that read 87.9% of 24-hour windows Critical, so ~7 of 8 Performance Calendar day cells painted red from deadlocks alone, and the fleet sweep's variable 15-1440min span scaled its Critical rate ~6x from the cadence knob. DailyHealthSignals now carries the Window its counts cover (the ServerHealthMetrics.DeadlockWindow discipline: default(TimeSpan) is unusable, never a divisor), and Classify routes the Deadlocks signal through ServerHealthClassifier.DeadlockSeverity with the store-backed DeadlockRateThresholds (V120) via DailyHealthThresholds.DeadlockRates — Critical/Warning fold into the day's tiers, and the sub-1h/undeclared window arm falls to Warning-not-rate (zero stays out of the trigger). DeadlockRatePerHour/DeadlockSeverity widen int->long for the day-scale counts; every int caller converts implicitly. Producers declare their windows: the calendar-day projections (Darling health reader, viewer, Lite) pass 24h, the fleet sweep passes its own span. The Darling day reads hoist one store-tiers read per range/sweep (the fleet reader's own hot-swap argument), the sweep's Compose takes the tiers as a required parameter, and its would-have-paged deadlock family now fires only when the rate crossed the Critical tier, with rate + tier + count as evidence. The shared deadlock reason/tooltip line reports the rate beside the count ("120 deadlocks (5.0/hr)"), the card reason's own disclosure rule, and the sweep verdict JSON carries deadlock_rate_per_hour + window_minutes additively. Tests: two-window proofs on the day classifier (same rate bands the same over 1h and 24h; same count bands differently), the single-deadlock day control, the sub-hour arms, tier-override pins, sweep ledger pins, a whole-repo census that every production DailyHealthSignals bundle declares its Window (with the calendar projections pinned to 24h and the sweep pinned to its span), and the Lite live-DuckDB calendar now proves the rate path end-to-end with a 480-deadlock storm day. Fixes #3525 Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx * Two live health-tool tests catch up to #3525's rate banding DarlingMcpHealthToolsLivePostgresTests planted one deadlock and asserted the day reads Critical - the any-deadlock-is-Critical semantics #3525 removes. One deadlock across a 24h day is 0.04/hr, far below the measured tiers, so the boundary test now pins the honest Healthy band plus the deadlock_count evidence (its real subject is date-boundary visibility), and the planted-rows test pins Warning (its 85% CPU sample trips the HighCpuWarningSamples tier) plus the same evidence. The Critical-band rate path is covered by the PR's own storm-day tests. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx * The still-forming day clamps its window to elapsed time (#3525 review) The calendar-day projections stamped Window = 24h on every row including today's half-open bucket, so an active storm diluted against hours that had not happened yet: 60 deadlocks in the last hour read 2.5/hr (Healthy) instead of 60/hr (Critical) - a false-negative the pre-#3525 any-deadlock trigger could not produce. The shared CalendarDayWindow helper now clamps the current day to its elapsed portion against the read's own clock: the anchored range reads (both SKUs) hand their resolved as_of window end so a backdated read clamps against itself, the live calendar reads use the wall clock, and a day minutes old falls to the sub-hour Warning-not-rate arm rather than a fabricated multiplied rate. Finished days band over their full 24 hours exactly as before, and the sweep path is untouched. The window census now pins all three projections to the clamp helper so a hand-rolled FromDays(1) cannot reintroduce the dilution, and the twin band suites carry the storm proof at the review's own numbers. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx --------- Co-authored-by: Claude Fable 5 --- .../DailyDeadlockWindowCensusTests.cs | 199 ++++++++++++++++++ Darling/Darling.Tests/DailyHealthBandTests.cs | 193 ++++++++++++++++- .../DarlingMcpHealthToolsTests.cs | 27 ++- .../Darling.Tests/FleetSweepEngineTests.cs | 121 ++++++++++- .../FleetSweepEngine.cs | 84 +++++++- .../Mcp/DarlingHealthReader.cs | 53 ++++- .../Mcp/DarlingMcpHealthTools.cs | 4 +- .../ViewerDataService.DailySummary.cs | 23 +- Lite.Tests/DailyHealthBandTests.cs | 194 ++++++++++++++++- Lite.Tests/PerformanceCalendarDataTests.cs | 27 ++- Lite/Mcp/McpHealthTools.cs | 3 +- .../Services/LocalDataService.DailySummary.cs | 21 +- PerformanceMonitor.Common/DailyHealthBand.cs | 99 ++++++++- .../ServerHealthBands.cs | 12 +- 14 files changed, 986 insertions(+), 74 deletions(-) create mode 100644 Darling/Darling.Tests/DailyDeadlockWindowCensusTests.cs diff --git a/Darling/Darling.Tests/DailyDeadlockWindowCensusTests.cs b/Darling/Darling.Tests/DailyDeadlockWindowCensusTests.cs new file mode 100644 index 000000000..1be13dd6d --- /dev/null +++ b/Darling/Darling.Tests/DailyDeadlockWindowCensusTests.cs @@ -0,0 +1,199 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Linq; +using System.Text.RegularExpressions; +using PerformanceMonitor.Common; +using Xunit; + +namespace Darling.Tests; + +/// +/// #3525: the shared daily classifier bands deadlocks as a RATE over +/// — so every PRODUCTION signals bundle has to declare the window its counts cover, or the band silently +/// takes the unrateable arm and a genuinely storming day maxes out at Warning. This is the +/// DeadlockRateBandRungTests.EveryProductionMetricBundleDeclaresTheWindowAndTheTiers census, one +/// signals type over: same walker, same brace-matching, same whole-repo scope. +/// +public sealed class DailyDeadlockWindowCensusTests +{ + /// + /// Every production bundle that assigns Deadlocks also assigns + /// Window. + /// + /// Keyed on Deadlocks AND HasData assigned in the same initializer — the + /// rung census's own lesson about member names shared across types, met the way it met it: two other + /// production types (InactionFigures, the PostgreSQL DatabaseRow) carry a + /// Deadlocks member, and neither has a HasData, while a signals bundle without + /// HasData is unbuildable in practice (it would band NoData unconditionally). A type-name + /// scan is not an option for the reason the rung census states: three of the four sites are + /// target-typed => new() { ... }. Prefixed members like TotalDeadlocks are excluded + /// by the leading character class. Tests are excluded because a fixture deliberately omitting the + /// window IS the unrateable-arm pin; deprecated/ because the old Dashboard froze its own + /// banding. + /// + [Fact] + public void EveryProductionDailySignalsBundle_DeclaresTheWindow() + { + var offenders = new List(); + var found = 0; + + foreach (var file in ProductionCSharpFiles()) + { + var code = CSharpSourceWalker.StripCommentsAndStrings(System.IO.File.ReadAllText(file)); + + foreach (var initializer in SignalsBundleInitializers(code)) + { + found++; + + if (!AssignsMember(initializer, "Window")) + { + offenders.Add(System.IO.Path.GetFileName(file)); + } + } + } + + /* The scan has to have FOUND the bundles, or "no offenders" is vacuous. Four production sites: + DarlingHealthReader's ToSignals (the calendar/MCP day), the viewer's DailySummaryRow.ToSignals, + Lite's DailySummaryRow.ToSignals, and the fleet sweep's span read. Pinned as an exact count — a + FIFTH bundle is a new surface that has to be looked at, and a floor would let it in silently, + while a count below four means the regex stopped matching, not that the code got better. */ + Assert.Equal(4, found); + + Assert.True( + offenders.Count == 0, + "these production DailyHealthSignals bundles carry a deadlock count with no window beside it, " + + "so the day band's deadlock signal falls to the unrateable arm and can never read Critical: " + + string.Join(", ", offenders.Distinct().OrderBy(f => f, StringComparer.Ordinal))); + } + + /// + /// The three CALENDAR-DAY projections declare the day's window through the ONE clamp helper — stated + /// against the source because the census above can only see that A window was assigned, and a calendar + /// day whose denominator drifted would band the same count differently across the three surfaces that + /// answer the same question. The helper (not a bare 24h constant) is the pin because the still-forming + /// day must clamp to its elapsed portion (#3525 review: an active storm banded over unelapsed hours + /// reads Healthy mid-crisis), and a projection that hand-rolls FromDays(1) reintroduces that dilution. + /// + [Fact] + public void EveryCalendarDayProjection_BandsThroughTheDayWindowClamp() + { + var surfaces = new (string What, string Text)[] + { + ("service daily read", RepoFile.ReadRepoFile( + "Darling", "PerformanceMonitor.Darling.Service", "Mcp", "DarlingHealthReader.cs")), + ("viewer calendar row", RepoFile.ReadRepoFile( + "Darling", "PerformanceMonitor.Darling.Viewer", "ViewerDataService.DailySummary.cs")), + ("Lite calendar row", RepoFile.ReadRepoFile( + "Lite", "Services", "LocalDataService.DailySummary.cs")), + }; + + foreach (var (what, text) in surfaces) + { + Assert.False(string.IsNullOrWhiteSpace(text), $"{what}: read nothing, so this pin would assert nothing"); + Assert.Contains( + "Window = DailyHealthBandCalculator.CalendarDayWindow(SummaryDate, ReferenceUtc)", + text, StringComparison.Ordinal); + Assert.DoesNotContain("Window = TimeSpan.FromDays(1)", text, StringComparison.Ordinal); + } + } + + /// + /// The fleet sweep's window is its OWN span, never a calendar day — the whole #3525 point for the + /// sweep was that a 15-minute span and a 24-hour day must not band the same count the same way. + /// + [Fact] + public void TheSweepWindow_IsTheSpan_NotACalendarDay() + { + var sweep = RepoFile.ReadRepoFile( + "Darling", "PerformanceMonitor.Darling.Service", "FleetSweepEngine.cs"); + + Assert.Contains("Window = spanEndUtc - spanStartUtc", sweep, StringComparison.Ordinal); + Assert.DoesNotContain("Window = TimeSpan.FromDays(1)", sweep, StringComparison.Ordinal); + Assert.DoesNotContain("CalendarDayWindow", sweep, StringComparison.Ordinal); + } + + /* ─────────────────────── helpers (the rung census's own, re-keyed) ─────────────────────── */ + + private static readonly Regex AssignsDeadlocks = + new(@"(?)", RegexOptions.Compiled); + + private static IEnumerable ProductionCSharpFiles() + { + foreach (var file in System.IO.Directory.EnumerateFiles( + RepoFile.Root, "*.cs", System.IO.SearchOption.AllDirectories)) + { + var segments = file.Split(System.IO.Path.DirectorySeparatorChar, System.IO.Path.AltDirectorySeparatorChar); + if (segments.Contains("bin") || segments.Contains("obj") + || segments.Contains("deprecated") + || segments.Any(s => s.EndsWith("Tests", StringComparison.Ordinal))) + { + continue; + } + + yield return file; + } + } + + private static IEnumerable SignalsBundleInitializers(string code) + { + foreach (var initializer in BraceMatchedInitializers(code)) + { + if (AssignsMember(initializer, "HasData")) + { + yield return initializer; + } + } + } + + private static IEnumerable BraceMatchedInitializers(string code) + { + foreach (var match in AssignsDeadlocks.Matches(code).Cast()) + { + var depth = 0; + var open = -1; + + for (var i = match.Index; i >= 0; i--) + { + if (code[i] == '}') + { + depth++; + } + else if (code[i] == '{' && depth-- == 0) + { + open = i; + break; + } + } + + if (open < 0) + { + continue; + } + + depth = 0; + for (var i = open; i < code.Length; i++) + { + if (code[i] == '{') + { + depth++; + } + else if (code[i] == '}' && --depth == 0) + { + yield return code[open..(i + 1)]; + break; + } + } + } + } + + private static bool AssignsMember(string initializer, string member) => + Regex.IsMatch(initializer, $@"(?)"); +} diff --git a/Darling/Darling.Tests/DailyHealthBandTests.cs b/Darling/Darling.Tests/DailyHealthBandTests.cs index 26dc263d9..8f1ec13cb 100644 --- a/Darling/Darling.Tests/DailyHealthBandTests.cs +++ b/Darling/Darling.Tests/DailyHealthBandTests.cs @@ -20,9 +20,13 @@ namespace Darling.Tests; /// public class DailyHealthBandTests { + private static readonly TimeSpan Day = TimeSpan.FromHours(24); + private static readonly TimeSpan Hour = TimeSpan.FromHours(1); + private static DailyHealthSignals Signals( bool hasData = true, long deadlocks = 0, long collectionErrors = 0, long highCpu = 0, - long blocking = 0, long memPressure = 0, long memCritical = 0, long alerts = 0) => new() + long blocking = 0, long memPressure = 0, long memCritical = 0, long alerts = 0, + TimeSpan window = default) => new() { HasData = hasData, Deadlocks = deadlocks, @@ -32,6 +36,7 @@ private static DailyHealthSignals Signals( MemoryPressureEvents = memPressure, MemoryCriticalEvents = memCritical, AlertCount = alerts, + Window = window, }; [Fact] @@ -48,15 +53,104 @@ public void Collected_AndNothingElevated_IsHealthy() } [Theory] - [InlineData("deadlock", 1, 0, 0, 0, 0, 0)] - [InlineData("collection-error", 0, 1, 0, 0, 0, 0)] - [InlineData("memory-critical", 0, 0, 0, 0, 1, 0)] - [InlineData("sustained-cpu", 0, 0, 6, 0, 0, 0)] - [InlineData("heavy-blocking", 0, 0, 0, 11, 0, 0)] - public void CriticalTriggers_EachAloneIsCritical(string _, long deadlocks, long collErrors, long highCpu, long blocking, long memCritical, long alerts) - { - var s = Signals(deadlocks: deadlocks, collectionErrors: collErrors, highCpu: highCpu, blocking: blocking, memCritical: memCritical, alerts: alerts); + [InlineData("collection-error", 1, 0, 0, 0, 0)] + [InlineData("memory-critical", 0, 0, 0, 1, 0)] + [InlineData("sustained-cpu", 0, 6, 0, 0, 0)] + [InlineData("heavy-blocking", 0, 0, 11, 0, 0)] + public void CriticalTriggers_EachAloneIsCritical(string _, long collErrors, long highCpu, long blocking, long memCritical, long alerts) + { + var s = Signals(collectionErrors: collErrors, highCpu: highCpu, blocking: blocking, memCritical: memCritical, alerts: alerts); + Assert.Equal(DailyHealthBand.Critical, DailyHealthBandCalculator.Classify(s)); + } + + /* ── the deadlock RATE trigger (#3525 — #3368's twin, routed through the card band's tiers) ── */ + + [Fact] + public void ACriticalDeadlockRate_AloneIsCritical() + { + // 480 over 24h = 20.0/hr, the Critical tier — the same pair the Overview card's dot bands on. + Assert.Equal( + DailyHealthBand.Critical, + DailyHealthBandCalculator.Classify(Signals(deadlocks: 480, window: Day))); + } + + /// + /// The defect #3525 was filed on: one deadlock in a 24-hour day banded the WHOLE day Critical — the + /// count trigger #3368 removed from the card, still shipping in this classifier. Measured on the same + /// 43-server fleet, count > 0 read 87.9% of 24-hour windows Critical, so ~7 of 8 calendar cells + /// painted red from deadlocks alone. At 0.04/hr the day is Healthy; the deadlock stays countable — the + /// tooltip lists it and the drill is offered — the BAND just stops claiming a crisis. + /// + [Fact] + public void ASingleDeadlockInADay_IsNoLongerCritical() + { + var s = Signals(deadlocks: 1, window: Day); + Assert.Equal(DailyHealthBand.Healthy, DailyHealthBandCalculator.Classify(s)); + + Assert.Contains(DayDrillTarget.Deadlocks, DailyHealthBandCalculator.AvailableDrills(s)); + Assert.Contains("1 deadlock", DailyHealthBandCalculator.Describe(s)); + } + + /// + /// The two-window proof, on the DAY classifier — the same per-hour rate bands the day identically over + /// a 1-hour window (a fleet-sweep span at the cadence ceiling's scale) and a 24-hour one (a calendar + /// day). Counts are integer-rate-times-whole-hours so the asserted rate is exactly the one the band + /// sees (the DeadlockRateBandTests discipline). + /// + [Theory] + [InlineData(4, DailyHealthBand.Healthy)] + [InlineData(5, DailyHealthBand.Warning)] + [InlineData(19, DailyHealthBand.Warning)] + [InlineData(20, DailyHealthBand.Critical)] + [InlineData(50, DailyHealthBand.Critical)] + public void TheSameDeadlockRate_BandsTheDayTheSame_OverAnHourAndADay(long ratePerHour, DailyHealthBand expected) + { + Assert.Equal(expected, DailyHealthBandCalculator.Classify(Signals(deadlocks: ratePerHour, window: Hour))); + Assert.Equal(expected, DailyHealthBandCalculator.Classify(Signals(deadlocks: ratePerHour * 24, window: Day))); + } + + /// + /// And the discriminating converse: the same COUNT over the two windows bands differently, which a + /// count trigger cannot do at all — the pin that goes red on any revert to counting. + /// + [Fact] + public void TheSameDeadlockCount_OverTwoWindows_BandsDifferently() + { + Assert.Equal(DailyHealthBand.Critical, DailyHealthBandCalculator.Classify(Signals(deadlocks: 30, window: Hour))); + Assert.Equal(DailyHealthBand.Healthy, DailyHealthBandCalculator.Classify(Signals(deadlocks: 30, window: Day))); + } + + /// + /// A sub-hour window — a fleet sweep at any cadence under an hour — is not rate-banded (#3368's own + /// arm): a non-zero count reads Warning (deadlocks demonstrably happened; no rate supports Critical, + /// and 1 deadlock in 15 minutes is 4/hr arithmetically but the hour was not observed), and a zero + /// count stays out of the deadlock trigger entirely rather than claiming anything. An undeclared + /// window — default(TimeSpan), a producer that declared nothing — takes the same arm, so no + /// path can rate-multiply or restore count-is-Critical by omission. + /// + [Fact] + public void ASubHourOrUndeclaredWindow_FallsToWarning_NeverCritical() + { + foreach (var window in new[] { default, TimeSpan.FromMinutes(15), TimeSpan.FromMinutes(59) }) + { + Assert.Equal(DailyHealthBand.Warning, DailyHealthBandCalculator.Classify(Signals(deadlocks: 1, window: window))); + Assert.Equal(DailyHealthBand.Warning, DailyHealthBandCalculator.Classify(Signals(deadlocks: 10_000, window: window))); + Assert.Equal(DailyHealthBand.Healthy, DailyHealthBandCalculator.Classify(Signals(deadlocks: 0, window: window))); + } + } + + /// + /// The day bands on the tiers it is handed — the store-backed pair (#3368, V120) travels through + /// , so a Darling store with raised tiers moves the + /// calendar, get_daily_summary and the sweep together with the card. + /// + [Fact] + public void TheDeadlockRateTiers_AreOverridable() + { + var raised = new DailyHealthThresholds { DeadlockRates = new DeadlockRateThresholds(100.0, 500.0) }; + var s = Signals(deadlocks: 480, window: Day); // 20/hr: Critical on the shipped pair Assert.Equal(DailyHealthBand.Critical, DailyHealthBandCalculator.Classify(s)); + Assert.Equal(DailyHealthBand.Healthy, DailyHealthBandCalculator.Classify(s, raised)); } [Theory] @@ -73,7 +167,8 @@ public void WarningTriggers_EachAloneIsWarning(string _, long highCpu, long bloc [Fact] public void CriticalBeatsWarning_WhenBothPresent() { - var s = Signals(deadlocks: 1, highCpu: 3, alerts: 4); + // A critical deadlock rate (480/24h = 20/hr) plus moderate CPU + alerts (warning) still bands Critical. + var s = Signals(deadlocks: 480, highCpu: 3, alerts: 4, window: Day); Assert.Equal(DailyHealthBand.Critical, DailyHealthBandCalculator.Classify(s)); } @@ -175,6 +270,33 @@ public void BuildReasons_BlockingLine_OmitsPeak_WhenZero() Assert.DoesNotContain(reasons, r => r.Contains("peak block", StringComparison.Ordinal)); } + /// + /// The deadlock line carries the per-hour rate the band evaluated (#3525) — the card reason's own + /// disclosure rule: "120 deadlocks" against an amber cell cannot say which tier was crossed, because + /// 120 in an hour and 120 in a day are the same string. The count stays (the countable fact); the rate + /// is added (the banded one). Shared by the tooltip and the day-detail reasons, so both surfaces say it. + /// + [Fact] + public void DeadlockLine_CarriesTheRate_WhenTheWindowIsRateable() + { + var day = Signals(deadlocks: 120, window: Day); // 5.0/hr — the Warning tier exactly + Assert.Contains("120 deadlocks (5.0/hr)", DailyHealthBandCalculator.Describe(day)); + Assert.Contains("120 deadlocks (5.0/hr)", DailyHealthBandCalculator.BuildReasons(day)); + + // A single deadlock still reads singular, rate beside it. + Assert.Contains("1 deadlock (0.0/hr)", DailyHealthBandCalculator.Describe(Signals(deadlocks: 1, window: Day))); + } + + [Fact] + public void DeadlockLine_PrintsTheCountAlone_OnAnUnrateableWindow() + { + // No declared window: no rate is computable, and printing one would claim a measurement nobody + // took — the count alone is exactly what the band had to go on. + var described = DailyHealthBandCalculator.Describe(Signals(deadlocks: 2)); + Assert.Contains("2 deadlocks", described); + Assert.DoesNotContain("/hr", described); + } + [Fact] public void AvailableDrills_NoData_OffersNothing() { @@ -227,4 +349,55 @@ public void BuildKeyMetricsLine_NoWait_ShowsNone_And_LargeWaitInMinutes() Assert.Contains("Top wait: none", line); Assert.Contains("Total wait: 20.0 min", line); // 1200s / 60 } + + /* ─────────── #3525 review: the still-forming day clamps to its elapsed portion ─────────── */ + + [Fact] + public void CalendarDayWindow_FinishedDay_IsTwentyFourHours() + { + var day = new DateTime(2026, 7, 8); + Assert.Equal(TimeSpan.FromDays(1), DailyHealthBandCalculator.CalendarDayWindow(day, new DateTime(2026, 7, 9))); + Assert.Equal(TimeSpan.FromDays(1), DailyHealthBandCalculator.CalendarDayWindow(day, new DateTime(2026, 9, 1, 12, 0, 0))); + } + + [Fact] + public void CalendarDayWindow_TodayClampsToElapsed_AndFutureIsZero() + { + var day = new DateTime(2026, 7, 8); + Assert.Equal(TimeSpan.FromHours(1), DailyHealthBandCalculator.CalendarDayWindow(day, day.AddHours(1))); + Assert.Equal(TimeSpan.FromMinutes(30), DailyHealthBandCalculator.CalendarDayWindow(day, day.AddMinutes(30))); + Assert.Equal(TimeSpan.Zero, DailyHealthBandCalculator.CalendarDayWindow(day, day.AddDays(-1))); + } + + [Fact] + public void TodayCell_ActiveStorm_IsNotDilutedByUnelapsedHours() + { + /* The review's own numbers: 60 deadlocks in the first hour of the still-forming day. Banded over + a full 24h the rate reads 2.5/hr (below the 5/hr Warning tier) and the crisis paints Healthy; + over the elapsed hour it is 60/hr — past the 20/hr Critical tier. The clamp is what keeps an + in-progress storm red on the calendar. The same 60 over a genuinely FINISHED day is honestly + 2.5/hr, and stays sub-Warning by design. */ + var day = new DateTime(2026, 7, 8); + var stormWindow = DailyHealthBandCalculator.CalendarDayWindow(day, day.AddHours(1)); + Assert.Equal( + DailyHealthBand.Critical, + DailyHealthBandCalculator.Classify(Signals(deadlocks: 60, window: stormWindow))); + + var finishedWindow = DailyHealthBandCalculator.CalendarDayWindow(day, day.AddDays(2)); + Assert.Equal( + DailyHealthBand.Healthy, + DailyHealthBandCalculator.Classify(Signals(deadlocks: 60, window: finishedWindow))); + } + + [Fact] + public void TodayCell_MinutesOld_FallsToTheUnrateableArm() + { + /* Sub-hour elapsed lands in DeadlockSeverity's unrateable arm (#3368's Warning-not-rate rule): + minutes into the day a single deadlock reads Warning, never a fabricated multiplied rate. */ + var window = DailyHealthBandCalculator.CalendarDayWindow( + new DateTime(2026, 7, 8), new DateTime(2026, 7, 8, 0, 10, 0)); + Assert.Equal( + DailyHealthBand.Warning, + DailyHealthBandCalculator.Classify(Signals(deadlocks: 1, window: window))); + } } diff --git a/Darling/Darling.Tests/DarlingMcpHealthToolsTests.cs b/Darling/Darling.Tests/DarlingMcpHealthToolsTests.cs index 086dbf613..9a6a02ef7 100644 --- a/Darling/Darling.Tests/DarlingMcpHealthToolsTests.cs +++ b/Darling/Darling.Tests/DarlingMcpHealthToolsTests.cs @@ -158,11 +158,21 @@ public void DailySummaryRow_BandsThroughSharedCalculator() { var date = new DateTime(2026, 7, 9, 0, 0, 0, DateTimeKind.Unspecified); - /* A deadlock day is Critical (the shared calculator's rule). */ - var critical = new Reader.DailySummaryReadRow(date, 0m, "", 0, DeadlockCount: 1, 0, 0, 0, 0, 0, 0, 0, HasData: true); + /* A day at a critical deadlock RATE is Critical (#3525): 480 over the row's 24-hour window is + 20/hr, the card band's Critical tier. One deadlock in a day is 0.04/hr and no longer paints the + cell red — the count trigger this replaced read 87.9% of production days Critical. */ + var critical = new Reader.DailySummaryReadRow(date, 0m, "", 0, DeadlockCount: 480, 0, 0, 0, 0, 0, 0, 0, HasData: true); Assert.Equal(DailyHealthBand.Critical, critical.HealthBand); Assert.Equal("Critical", critical.OverallHealth); + var oneDeadlock = new Reader.DailySummaryReadRow(date, 0m, "", 0, DeadlockCount: 1, 0, 0, 0, 0, 0, 0, 0, HasData: true); + Assert.Equal(DailyHealthBand.Healthy, oneDeadlock.HealthBand); + + /* And the band honours the tiers the read stamped from the store (#3368's knobs): the same 20/hr + day under raised tiers is not Critical. */ + var raised = critical with { RateTiers = new DeadlockRateThresholds(100.0, 500.0) }; + Assert.Equal(DailyHealthBand.Healthy, raised.HealthBand); + /* A collected-but-quiet day is Healthy. */ var healthy = new Reader.DailySummaryReadRow(date, 12m, "CXPACKET", 3, 0, 0, 0, 0, 0, 0, 0, 0, HasData: true); Assert.Equal(DailyHealthBand.Healthy, healthy.HealthBand); @@ -271,7 +281,12 @@ await DarlingMcpTestData.ExecAsync(connection, ct, postgres, ServerName, boundary.ToString("yyyy-MM-dd", CultureInfo.InvariantCulture)); DarlingMcpTestData.AssertEnvelope(onItsOwnDay, ServerName, "overall_health"); - Assert.Contains("Critical", onItsOwnDay, StringComparison.Ordinal); + /* #3525: deadlocks band as a per-hour RATE through the card band's tiers now, so one deadlock + across a 24h day (0.04/hr, far below the measured Warning tier of 5/hr) is a Healthy day — + the old any-deadlock-is-Critical reading is gone by design. The row's VISIBILITY to the + explicit-date read is what this test pins, so assert the evidence and the honest band. */ + Assert.Contains("\"deadlock_count\":1", onItsOwnDay, StringComparison.Ordinal); + Assert.Contains("Healthy", onItsOwnDay, StringComparison.Ordinal); Assert.Contains("2026-07-20", onItsOwnDay, StringComparison.Ordinal); /* The bug: the same rows are invisible to an implicit "today", which is what the sibling test @@ -352,7 +367,11 @@ has already rolled over — the tool then correctly returns its empty envelope a var daily = await DarlingMcpHealthTools.GetDailySummary( postgres, ServerName, when.ToString("yyyy-MM-dd", CultureInfo.InvariantCulture)); DarlingMcpTestData.AssertEnvelope(daily, ServerName, "overall_health"); - Assert.Contains("Critical", daily, StringComparison.Ordinal); /* the deadlock makes the day Critical */ + /* #3525: one deadlock is 0.04/hr against a 24h day — below the rate tiers, so it no longer + makes the day Critical. The 85% CPU sample trips the HighCpuWarningSamples=1 tier, so the + day reads Warning, and the planted deadlock stays visible as evidence. */ + Assert.Contains("\"deadlock_count\":1", daily, StringComparison.Ordinal); + Assert.Contains("Warning", daily, StringComparison.Ordinal); bodySucceeded = true; } diff --git a/Darling/Darling.Tests/FleetSweepEngineTests.cs b/Darling/Darling.Tests/FleetSweepEngineTests.cs index 5de912823..f172a4593 100644 --- a/Darling/Darling.Tests/FleetSweepEngineTests.cs +++ b/Darling/Darling.Tests/FleetSweepEngineTests.cs @@ -41,11 +41,17 @@ the window states its own start instant instead. */ private static FleetSweepInstrumentCounters SteadyInstruments(long passes = 500) => new(StartedLongAgo, passes, AlertReadFailuresTotal: 0); + /* Every fixture's signals carry the sweep span ComposeSimple declares (1h) — the deadlock band is a + RATE over that window (#3525), so a fixture omitting the window would take the unrateable arm and + max out at Warning. */ + private static readonly TimeSpan FixtureSpan = TimeSpan.FromHours(1); + private static FleetSweepServerReading Healthy(int id, string name) => - new(id, name, new DailyHealthSignals { HasData = true }, 0, null); + new(id, name, new DailyHealthSignals { HasData = true, Window = FixtureSpan }, 0, null); - private static FleetSweepServerReading CriticalDeadlocks(int id, string name, long deadlocks = 3) => - new(id, name, new DailyHealthSignals { HasData = true, Deadlocks = deadlocks }, 0, null); + /* 25 deadlocks over the 1-hour span = 25/hr, past the shipped Critical tier (20/hr). */ + private static FleetSweepServerReading CriticalDeadlocks(int id, string name, long deadlocks = 25) => + new(id, name, new DailyHealthSignals { HasData = true, Deadlocks = deadlocks, Window = FixtureSpan }, 0, null); private static FleetSweepServerReading NoData(int id, string name) => new(id, name, default, 0, null); @@ -57,7 +63,8 @@ private static FleetSweepComposition ComposeSimple( IReadOnlyList? previousVerdicts = null, IReadOnlyList? activeItems = null, FleetSweepInstrumentCounters? instruments = null, - DateTime? now = null) + DateTime? now = null, + DeadlockRateThresholds? deadlockTiers = null) { return FleetSweepEngine.Compose( now ?? Now, @@ -68,7 +75,8 @@ private static FleetSweepComposition ComposeSimple( previousRun, previousVerdicts ?? Array.Empty(), activeItems ?? Array.Empty(), - instruments ?? SteadyInstruments()); + instruments ?? SteadyInstruments(), + deadlockTiers ?? DeadlockRateThresholds.Default); } /* ─────────────────────── verdicts: the shared scorer, unforked ─────────────────────── */ @@ -103,9 +111,13 @@ public void Verdicts_ComeFromTheSharedScorer_WithItsOwnReasons() Assert.Null(verdictB.BandReason); /* Verdict beside its inputs: the evidence payload carries the signals, so a reader can - disagree with the band rather than believe it. */ + disagree with the band rather than believe it — including, since #3525, the rate the deadlock + signal banded on and the window it was normalised over, without which the deadlock count is + unfalsifiable. */ Assert.NotNull(verdictA.VerdictJson); - Assert.Contains("\"deadlocks\":3", verdictA.VerdictJson, StringComparison.Ordinal); + Assert.Contains("\"deadlocks\":25", verdictA.VerdictJson, StringComparison.Ordinal); + Assert.Contains("\"deadlock_rate_per_hour\":25", verdictA.VerdictJson, StringComparison.Ordinal); + Assert.Contains("\"window_minutes\":60", verdictA.VerdictJson, StringComparison.Ordinal); } /* ─────────────────────── the diff: sweep N against sweep N−1's rows ─────────────────────── */ @@ -365,9 +377,10 @@ public void UnderMasterOff_TheSweepProducesRows_AndTheLedgerCarriesTheWouldHaveP new FleetSweepServerReading(1, "server-a", new DailyHealthSignals { HasData = true, - Deadlocks = 2, + Deadlocks = 25, // 25/hr over the 1h span — past the Critical tier (#3525) HighCpuEvents = 10, BlockingEvents = 20, + Window = FixtureSpan, }, 1500, null), Healthy(2, "server-b"), }; @@ -398,6 +411,98 @@ crossed their own summary-scoring critical trigger. */ Assert.True(unmuted.Run.AlertsEnabled); } + /* ─────────────────────── the deadlock family is the RATE's, not the count's (#3525) ─────────────────────── */ + + /// + /// The would-have-paged deadlock family fires on the RATE the shared scorer banded Critical with — + /// never on a bare count. A day Critical from another trigger, carrying deadlocks below the Critical + /// tier, writes no deadlock row: under the old count trigger its threshold was literally 1, so every + /// sweep span containing any deadlock claimed a page the new banding does not stand behind. + /// + [Fact] + public void TheDeadlockFamily_FiresOnTheRate_NotTheCount() + { + /* Critical via heavy blocking; 3 deadlocks over the 1h span is 3/hr — Healthy on the deadlock + band, so the ledger must not name the family. */ + var subRate = new FleetSweepServerReading(1, "server-a", new DailyHealthSignals + { + HasData = true, + Deadlocks = 3, + BlockingEvents = 20, + Window = FixtureSpan, + }, 0, null); + + var families = ComposeSimple(new[] { subRate }, alertsEnabled: false) + .WouldHavePaged.Select(w => w.AlertFamily).ToList(); + Assert.Contains(FleetSweepEngine.FamilyBlocking, families); + Assert.DoesNotContain(FleetSweepEngine.FamilyDeadlocks, families); + + /* And when the family DOES fire, its evidence names the rate as the value, the Critical tier as + the threshold, and the raw count beside them — figures an operator can audit against + get_alert_settings and the deadlock grid. */ + var paged = ComposeSimple(new[] { CriticalDeadlocks(1, "server-a") }, alertsEnabled: false) + .WouldHavePaged.Single(w => w.AlertFamily == FleetSweepEngine.FamilyDeadlocks); + using var evidence = JsonDocument.Parse(paged.EvidenceJson); + Assert.Equal("deadlocks per hour over the sweep span", evidence.RootElement.GetProperty("trigger").GetString()); + Assert.Equal(25.0, evidence.RootElement.GetProperty("value").GetDouble()); + Assert.Equal( + ServerHealthThresholds.DeadlockCriticalPerHourDefault, + evidence.RootElement.GetProperty("threshold").GetDouble()); + Assert.Equal(25, evidence.RootElement.GetProperty("deadlock_count").GetInt64()); + } + + /// + /// A sub-hour sweep span is not rateable (#3368's arm), so deadlocks alone cannot band the span + /// Critical — and even when ANOTHER trigger makes the day Critical, the deadlock family stays out of + /// the ledger: 10,000 deadlocks in 15 minutes is 40,000/hr arithmetically, and declining to claim it + /// is the honest reading the whole rate band is built on. + /// + [Fact] + public void ASubHourSpan_NeverPagesTheDeadlockFamily() + { + var reading = new FleetSweepServerReading(1, "server-a", new DailyHealthSignals + { + HasData = true, + Deadlocks = 10_000, + CollectionErrors = 1, + Window = TimeSpan.FromMinutes(15), + }, 0, null); + + var composition = ComposeSimple(new[] { reading }, alertsEnabled: false); + + Assert.Equal("Critical", composition.Verdicts.Single().Band); + var families = composition.WouldHavePaged.Select(w => w.AlertFamily).ToList(); + Assert.Contains(FleetSweepEngine.FamilyCollectionErrors, families); + Assert.DoesNotContain(FleetSweepEngine.FamilyDeadlocks, families); + + /* Deadlocks alone on the same span: Warning, not Critical — the unrateable arm. */ + var alone = new FleetSweepServerReading(1, "server-a", new DailyHealthSignals + { + HasData = true, + Deadlocks = 10_000, + Window = TimeSpan.FromMinutes(15), + }, 0, null); + Assert.Equal("Warning", ComposeSimple(new[] { alone }).Verdicts.Single().Band); + } + + /// + /// The tiers handed to Compose are the tiers the verdicts band on (#3525) — the store's pair + /// travels into the shared scorer, so a fleet whose knobs were raised sweeps on the raised pair + /// rather than the shipped one while get_alert_settings reports the raised numbers. + /// + [Fact] + public void TheVerdicts_BandOnTheTiersHandedIn() + { + var reading = CriticalDeadlocks(1, "server-a"); // 25/hr: Critical on the shipped pair + + Assert.Equal("Critical", ComposeSimple(new[] { reading }).Verdicts.Single().Band); + + var raised = new DeadlockRateThresholds(100.0, 500.0); + var onRaised = ComposeSimple(new[] { reading }, alertsEnabled: false, deadlockTiers: raised); + Assert.Equal("Healthy", onRaised.Verdicts.Single().Band); + Assert.Empty(onRaised.WouldHavePaged); + } + /// /// The stored DOCUMENT carries the ledger key exactly when the check ran (#3478): absent on an /// alerts-on sweep, because the would-have-paged derivation never executed there and a stored diff --git a/Darling/PerformanceMonitor.Darling.Service/FleetSweepEngine.cs b/Darling/PerformanceMonitor.Darling.Service/FleetSweepEngine.cs index 36a303a32..eede5a929 100644 --- a/Darling/PerformanceMonitor.Darling.Service/FleetSweepEngine.cs +++ b/Darling/PerformanceMonitor.Darling.Service/FleetSweepEngine.cs @@ -175,13 +175,21 @@ public static FleetSweepComposition Compose( FleetSweepRun? previousRun, IReadOnlyList previousVerdicts, IReadOnlyList activeWatchItems, - FleetSweepInstrumentCounters instruments) + FleetSweepInstrumentCounters instruments, + DeadlockRateThresholds deadlockRateTiers) { ArgumentNullException.ThrowIfNull(readings); ArgumentNullException.ThrowIfNull(previousVerdicts); ArgumentNullException.ThrowIfNull(activeWatchItems); ArgumentNullException.ThrowIfNull(instruments); + /* #3525: the deadlock-rate tiers travel INTO the shared scorer, so the sweep's verdicts band on the + pair get_alert_settings reports — required rather than defaulted, the DeadlockSeverity discipline: + a caller that kept the old call would compile and silently band on the shipped pair while the + Overview card used the store's. Built once, because the banding thresholds must be one + configuration for the whole sweep. */ + var banding = new DailyHealthThresholds { DeadlockRates = deadlockRateTiers }; + var sweepId = nowUtc.Ticks; var inSettleWindow = nowUtc - instruments.ServiceStartedUtc < PostRestartSettleWindow; var previousByServer = previousVerdicts.ToDictionary(v => v.ServerId); @@ -209,7 +217,7 @@ header cannot read green over this card. */ } else { - var classified = DailyHealthBandCalculator.Classify(reading.Signals); + var classified = DailyHealthBandCalculator.Classify(reading.Signals, banding); band = DailyHealthBandCalculator.Label(classified); reason = classified switch { @@ -246,12 +254,12 @@ header cannot read green over this card. */ { foreach (var reading in readings.Where(r => r.ReadFault is null)) { - if (DailyHealthBandCalculator.Classify(reading.Signals) != DailyHealthBand.Critical) + if (DailyHealthBandCalculator.Classify(reading.Signals, banding) != DailyHealthBand.Critical) { continue; } - foreach (var (family, evidence) in DecomposeCriticalTriggers(reading)) + foreach (var (family, evidence) in DecomposeCriticalTriggers(reading, banding)) { wouldHavePaged.Add(new FleetSweepWouldHavePagedEntry(reading.ServerId, family, evidence)); } @@ -693,14 +701,23 @@ private static FleetSweepWatchItem BuildItemImage( /// decided. Each row carries the trigger, the measured figure and the threshold it crossed, so an /// operator auditing a mute reads evidence rather than an assertion. /// - private static IEnumerable<(string Family, string Evidence)> DecomposeCriticalTriggers(FleetSweepServerReading reading) + private static IEnumerable<(string Family, string Evidence)> DecomposeCriticalTriggers( + FleetSweepServerReading reading, DailyHealthThresholds thresholds) { - var thresholds = DailyHealthThresholds.Default; var signals = reading.Signals; - if (signals.Deadlocks > 0) - { - yield return (FamilyDeadlocks, Evidence("deadlocks in span", signals.Deadlocks, 1)); + /* #3525: the deadlock family fires on the RATE the scorer banded Critical with, never on a bare + count — the same DeadlockSeverity call Classify makes, so the ledger cannot page on a trigger the + verdict did not band. A sub-hour span's unrateable arm maxes out at Warning, so it can never + reach this. */ + if (ServerHealthClassifier.DeadlockSeverity(signals.Deadlocks, signals.Window, thresholds.DeadlockRates) + == HealthSeverity.Critical) + { + /* Critical implies a rateable window (the unrateable arm returns Warning or Unknown), so the + rate is present by construction. */ + var ratePerHour = ServerHealthClassifier.DeadlockRatePerHour(signals.Deadlocks, signals.Window)!.Value; + yield return (FamilyDeadlocks, DeadlockRateEvidence( + ratePerHour, thresholds.DeadlockRates.CriticalPerHour, signals.Deadlocks)); } if (signals.CollectionErrors > 0) @@ -733,10 +750,30 @@ private static string Evidence(string what, long value, long threshold) => threshold, }); + /// The deadlock family's evidence (#3525): the shared shape with the RATE as the value — + /// because the rate is what banded — plus the raw count as its own member, because the count is the + /// countable fact an operator reconciles against the deadlock grid. + private static string DeadlockRateEvidence(double ratePerHour, double criticalPerHour, long count) => + JsonSerializer.Serialize(new + { + derivation = "summary-scoring critical trigger under alerts_enabled: false — not an alert-engine replay", + trigger = "deadlocks per hour over the sweep span", + value = ratePerHour, + threshold = criticalPerHour, + deadlock_count = count, + }); + private static string SerializeSignals(FleetSweepServerReading reading) => JsonSerializer.Serialize(new { deadlocks = reading.Signals.Deadlocks, + /* #3525: the rate the deadlock signal banded on, beside the count it was derived from (null on + an unrateable span) — the card's own disclosure rule: evidence a reader can disagree with has + to include the figure the band read. Additive members on NEW rows only; stored verdicts are + immutable and their readers key on the members that were always here. */ + deadlock_rate_per_hour = ServerHealthClassifier.DeadlockRatePerHour( + reading.Signals.Deadlocks, reading.Signals.Window), + window_minutes = reading.Signals.Window.TotalMinutes, collection_errors = reading.Signals.CollectionErrors, high_cpu_events = reading.Signals.HighCpuEvents, blocking_events = reading.Signals.BlockingEvents, @@ -781,6 +818,11 @@ public static async Task RunAsync( : await FleetSweepStore.GetServerVerdictsForEngineAsync(postgres, previousRun.SweepId, cancellationToken).ConfigureAwait(false); var activeItems = await FleetSweepStore.GetActiveWatchItemsAsync(postgres, cancellationToken).ConfigureAwait(false); + /* #3525: the deadlock-rate tiers, read once per sweep off the fleet reader's own published SQL + — an engine-seam read, so a fault here loudly costs this sweep slot rather than quietly + banding the fleet on the shipped pair. */ + var deadlockRateTiers = await ReadDeadlockRateThresholdsAsync(postgres, cancellationToken).ConfigureAwait(false); + var nowUtc = DateTime.UtcNow; var spanStartUtc = ComputeSpanStart(nowUtc, interval, previousRun); @@ -796,7 +838,7 @@ at a time and a herd all at once — and the sweep is a background errand racing var composition = Compose( nowUtc, spanStartUtc, alertsEnabled, servers.Count, readings, - previousRun, previousVerdicts, activeItems, ReadInstrumentCounters()); + previousRun, previousVerdicts, activeItems, ReadInstrumentCounters(), deadlockRateTiers); await FleetSweepStore.RecordSweepAsync( postgres, composition.Run, composition.Verdicts, @@ -850,6 +892,10 @@ private static async Task ReadServerSignalsAsync( MemoryPressureEvents = rows.Sum(r => r.MemoryPressureEvents), MemoryCriticalEvents = rows.Sum(r => r.MemoryCriticalEvents), AlertCount = rows.Sum(r => r.AlertCount), + /* #3525: the sweep's own span, NOT a calendar day — the denominator the deadlock rate + bands on. At the floor cadence (15 min) this is sub-hour and the band's unrateable arm + applies: deadlocks read Warning, never a rate-multiplied Critical. */ + Window = spanEndUtc - spanStartUtc, }; var peakBlock = rows.Count == 0 ? 0L : rows.Max(r => r.MaxBlockDurationMs); @@ -866,6 +912,24 @@ private static async Task ReadServerSignalsAsync( } } + /// The deadlock band's tiers from the store's singleton settings row (#3368, V120) — the + /// fleet reader's read, off its own published SQL, hoisted here once per sweep (#3525). A store with + /// no row yet bands on the shipped pair, which is what such a store would seed anyway; values come + /// back RAW and clamps on read. + private static async Task ReadDeadlockRateThresholdsAsync( + NpgsqlDataSource postgres, CancellationToken cancellationToken) + { + await using var command = postgres.CreateCommand(DarlingFleetReader.FleetDeadlockRateThresholdSql); + command.CommandTimeout = McpCommandDeadlines.ReadSeconds; + await using var reader = await command.ExecuteReaderAsync(cancellationToken).ConfigureAwait(false); + if (await reader.ReadAsync(cancellationToken).ConfigureAwait(false)) + { + return new DeadlockRateThresholds(reader.GetDouble(0), reader.GetDouble(1)); + } + + return DeadlockRateThresholds.Default; + } + /// The in-process instrument counters, read at compose time: the process-global alert /// read-health counter's start instant (the restart detector), the pass total summed across every /// per-server bucket, and the instance-wide swallowed-read total. diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingHealthReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingHealthReader.cs index c006a8aa0..8bca98a24 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingHealthReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingHealthReader.cs @@ -172,6 +172,17 @@ public sealed record DailySummaryReadRow( long BlockingEvents, long HighCpuEvents, long CollectionErrors, long MemoryPressureEvents, long MemoryCriticalEvents, long AlertCount, long MaxBlockDurationMs, bool HasData) { + /// The store's deadlock-rate tiers (#3368/#3525) — stamped by the calendar-day reads so + /// bands on the pair get_alert_settings reports rather than the + /// shipped defaults. The default is the shipped pair, which is what a store at its V120 column + /// defaults holds anyway. + public DeadlockRateThresholds RateTiers { get; init; } = DeadlockRateThresholds.Default; + + /// The clock the still-forming day's window clamps against (#3525 review): anchored + /// reads hand their resolved window end so a backdated as_of clamps against its own "now"; + /// unanchored reads (the explicit-date tool, the viewer path) band against the wall clock. + public DateTime ReferenceUtc { get; init; } = DateTime.UtcNow; + public DailyHealthSignals ToSignals() => new() { HasData = HasData, @@ -182,9 +193,15 @@ public sealed record DailySummaryReadRow( MemoryPressureEvents = MemoryPressureEvents, MemoryCriticalEvents = MemoryCriticalEvents, AlertCount = AlertCount, + /* #3525: a finished calendar day bands over its full 24 hours; the still-forming day clamps + to its elapsed portion against ReferenceUtc, or an active storm dilutes against hours that + have not happened yet (review finding on #3525). The fleet sweep does NOT read this + projection: it sums this row type's raw counts into signals windowed to its own span. */ + Window = DailyHealthBandCalculator.CalendarDayWindow(SummaryDate, ReferenceUtc), }; - public DailyHealthBand HealthBand => DailyHealthBandCalculator.Classify(ToSignals()); + public DailyHealthBand HealthBand => + DailyHealthBandCalculator.Classify(ToSignals(), new DailyHealthThresholds { DeadlockRates = RateTiers }); /// Human label for the band ("Healthy" / "Warning" / "Critical" / "No Data"). public string OverallHealth => DailyHealthBandCalculator.Label(HealthBand); @@ -205,7 +222,8 @@ public sealed record DailySummaryReadRow( /// report different query counts for the same day and there would be no way to tell which was right. /// public static async Task> GetDailySummaryRangeAsync( - NpgsqlDataSource postgres, int serverId, DateTime fromDate, DateTime toDate, CancellationToken cancellationToken = default) + NpgsqlDataSource postgres, int serverId, DateTime fromDate, DateTime toDate, + DateTime? referenceUtc = null, CancellationToken cancellationToken = default) { /* #1664: gate the age decision on the rollups actually existing — a plain-PostgreSQL store has none (and never drops raw, so raw is complete there). #1759: and on what they have MATERIALIZED, which is @@ -220,6 +238,11 @@ runs at human/model cadence and these are two small lookups. */ DateTime.UtcNow, fromDate, rollups.QueryGrainHourly, rollups.QueryGrainDaily, coverage.For(TimescaleSupport.QueryStatsHourlyView, TimescaleSupport.QueryStatsDailyView)); + /* #3525: the deadlock-rate tiers the day band evaluates, read ONCE per range rather than per row — + DarlingFleetReader's own hoist argument: a settings write mid-read must not band some days on the + old pair and the rest on the new one. */ + var rateTiers = await ReadDeadlockRateThresholdsAsync(postgres, cancellationToken); + var results = new List(); await using var command = postgres.CreateCommand(DailySummarySql.RangeSqlFor(tier)); command.CommandTimeout = McpCommandDeadlines.ReadSeconds; @@ -230,12 +253,34 @@ runs at human/model cadence and these are two small lookups. */ await using var reader = await command.ExecuteReaderAsync(cancellationToken); while (await reader.ReadAsync(cancellationToken)) { - results.Add(ReadDailySummaryRow(reader)); + results.Add(ReadDailySummaryRow(reader) with + { + RateTiers = rateTiers, + ReferenceUtc = referenceUtc ?? DateTime.UtcNow, + }); } return results; } + /// The deadlock band's tiers from the store's singleton settings row (#3368, V120), or the + /// shipped pair when the row is absent — DarlingFleetReader.ReadDeadlockRateThresholdsAsync's + /// read, off the same published SQL, for the DAY surfaces (#3525). Values come back RAW; + /// clamps on read. + private static async Task ReadDeadlockRateThresholdsAsync( + NpgsqlDataSource postgres, CancellationToken cancellationToken) + { + await using var command = postgres.CreateCommand(DarlingFleetReader.FleetDeadlockRateThresholdSql); + command.CommandTimeout = McpCommandDeadlines.ReadSeconds; + await using var reader = await command.ExecuteReaderAsync(cancellationToken); + if (await reader.ReadAsync(cancellationToken)) + { + return new DeadlockRateThresholds(reader.GetDouble(0), reader.GetDouble(1)); + } + + return DeadlockRateThresholds.Default; + } + /// /// The daily-summary signals for one server over an EXACT half-open window — the fleet sweep's /// per-server read (#3466 lane 2), which is minus two @@ -281,7 +326,7 @@ public static async Task GetDailySummaryAsync( NpgsqlDataSource postgres, int serverId, DateTime? summaryDate = null, CancellationToken cancellationToken = default) { var targetDate = summaryDate?.Date ?? DateTime.UtcNow.Date; - var rows = await GetDailySummaryRangeAsync(postgres, serverId, targetDate, targetDate.AddDays(1), cancellationToken); + var rows = await GetDailySummaryRangeAsync(postgres, serverId, targetDate, targetDate.AddDays(1), cancellationToken: cancellationToken); return rows.Count > 0 ? rows[0] : new DailySummaryReadRow(targetDate, 0m, "", 0, 0, 0, 0, 0, 0, 0, 0, 0, HasData: false); diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpHealthTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpHealthTools.cs index 39564f525..c2bcbf702 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpHealthTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpHealthTools.cs @@ -162,8 +162,10 @@ means the anchor day is the last one included rather than the first one excluded var fromDate = lastDay.AddDays(-(days_back - 1)); var toDate = lastDay.AddDays(1); + /* The anchor is also the clock the still-forming day's window clamps against (#3525 review): + a backdated as_of must clamp its own "today" against ITSELF, not the process clock. */ var rows = await DarlingHealthReader.GetDailySummaryRangeAsync( - postgres, resolved.ServerId, fromDate, toDate); + postgres, resolved.ServerId, fromDate, toDate, referenceUtc: windowEnd); if (rows.Count == 0) { diff --git a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.DailySummary.cs b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.DailySummary.cs index 597406c67..2a4bcc4ca 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.DailySummary.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.DailySummary.cs @@ -57,6 +57,15 @@ public async Task> GetDailySummaryRangeAsync( DateTime.UtcNow, fromDate, rollups.QueryGrainHourly, rollups.QueryGrainDaily, coverage.For(TimescaleSupport.QueryStatsHourlyView, TimescaleSupport.QueryStatsDailyView)); + /* #3525: the deadlock-rate tiers the day band evaluates, read ONCE per range rather than per row — + the fleet roll-up's own hoist argument: a settings save mid-read must not band some days on the + old pair and the rest on the new one. One read per month navigation, off the Overview's existing + settings-row read. */ + var banding = new DailyHealthThresholds + { + DeadlockRates = await GetDeadlockRateThresholdsAsync(cancellationToken), + }; + await using var command = _dataSource.CreateCommand(DailySummaryRangeSqlFor(tier)); command.CommandTimeout = ViewerCommandDeadlines.CurrentInteractiveReadSeconds; command.Parameters.Add(new NpgsqlParameter { TypedValue = serverId }); @@ -67,7 +76,7 @@ public async Task> GetDailySummaryRangeAsync( await using var reader = await command.ExecuteReaderAsync(cancellationToken); while (await reader.ReadAsync(cancellationToken)) { - results.Add(ReadDailySummaryRow(reader)); + results.Add(ReadDailySummaryRow(reader, banding)); } return results; @@ -87,7 +96,7 @@ public async Task> GetDailySummaryRangeAsync( : new DailySummaryRow { SummaryDate = targetDate, HasData = false, HealthBand = DailyHealthBand.NoData }; } - private static DailySummaryRow ReadDailySummaryRow(DbDataReader reader) + private static DailySummaryRow ReadDailySummaryRow(DbDataReader reader, DailyHealthThresholds banding) { var row = new DailySummaryRow { @@ -105,7 +114,7 @@ private static DailySummaryRow ReadDailySummaryRow(DbDataReader reader) MaxBlockDurationMs = reader.IsDBNull(11) ? 0L : Convert.ToInt64(reader.GetValue(11)), HasData = true, }; - row.HealthBand = DailyHealthBandCalculator.Classify(row.ToSignals()); + row.HealthBand = DailyHealthBandCalculator.Classify(row.ToSignals(), banding); return row; } } @@ -117,6 +126,10 @@ private static DailySummaryRow ReadDailySummaryRow(DbDataReader reader) public class DailySummaryRow { public DateTime SummaryDate { get; set; } + + /// The clock the still-forming day's window clamps against (#3525 review). The viewer's + /// calendar reads are always live, so the wall-clock default is the correct reference. + public DateTime ReferenceUtc { get; set; } = DateTime.UtcNow; public decimal TotalWaitTimeSec { get; set; } public string TopWaitType { get; set; } = ""; public long UniqueQueries { get; set; } @@ -160,5 +173,9 @@ public class DailySummaryRow MemoryPressureEvents = MemoryPressureEvents, MemoryCriticalEvents = MemoryCriticalEvents, AlertCount = AlertCount, + /* #3525: a finished calendar day bands over its full 24 hours; the still-forming day clamps to + its elapsed portion so an active storm is not diluted by hours that have not happened yet + (review finding on #3525). The viewer's reads are live, so the wall clock is the reference. */ + Window = DailyHealthBandCalculator.CalendarDayWindow(SummaryDate, ReferenceUtc), }; } diff --git a/Lite.Tests/DailyHealthBandTests.cs b/Lite.Tests/DailyHealthBandTests.cs index 74ab2aa39..68441d6df 100644 --- a/Lite.Tests/DailyHealthBandTests.cs +++ b/Lite.Tests/DailyHealthBandTests.cs @@ -20,9 +20,13 @@ namespace PerformanceMonitorLite.Tests; /// public class DailyHealthBandTests { + private static readonly TimeSpan Day = TimeSpan.FromHours(24); + private static readonly TimeSpan Hour = TimeSpan.FromHours(1); + private static DailyHealthSignals Signals( bool hasData = true, long deadlocks = 0, long collectionErrors = 0, long highCpu = 0, - long blocking = 0, long memPressure = 0, long memCritical = 0, long alerts = 0) => new() + long blocking = 0, long memPressure = 0, long memCritical = 0, long alerts = 0, + TimeSpan window = default) => new() { HasData = hasData, Deadlocks = deadlocks, @@ -32,6 +36,7 @@ private static DailyHealthSignals Signals( MemoryPressureEvents = memPressure, MemoryCriticalEvents = memCritical, AlertCount = alerts, + Window = window, }; [Fact] @@ -49,17 +54,106 @@ public void Collected_AndNothingElevated_IsHealthy() } [Theory] - [InlineData("deadlock", 1, 0, 0, 0, 0, 0)] - [InlineData("collection-error", 0, 1, 0, 0, 0, 0)] - [InlineData("memory-critical", 0, 0, 0, 0, 1, 0)] - [InlineData("sustained-cpu", 0, 0, 6, 0, 0, 0)] - [InlineData("heavy-blocking", 0, 0, 0, 11, 0, 0)] - public void CriticalTriggers_EachAloneIsCritical(string _, long deadlocks, long collErrors, long highCpu, long blocking, long memCritical, long alerts) - { - var s = Signals(deadlocks: deadlocks, collectionErrors: collErrors, highCpu: highCpu, blocking: blocking, memCritical: memCritical, alerts: alerts); + [InlineData("collection-error", 1, 0, 0, 0, 0)] + [InlineData("memory-critical", 0, 0, 0, 1, 0)] + [InlineData("sustained-cpu", 0, 6, 0, 0, 0)] + [InlineData("heavy-blocking", 0, 0, 11, 0, 0)] + public void CriticalTriggers_EachAloneIsCritical(string _, long collErrors, long highCpu, long blocking, long memCritical, long alerts) + { + var s = Signals(collectionErrors: collErrors, highCpu: highCpu, blocking: blocking, memCritical: memCritical, alerts: alerts); Assert.Equal(DailyHealthBand.Critical, DailyHealthBandCalculator.Classify(s)); } + /* ── the deadlock RATE trigger (#3525 — #3368's twin, routed through the card band's tiers) ── */ + + [Fact] + public void ACriticalDeadlockRate_AloneIsCritical() + { + // 480 over 24h = 20.0/hr, the Critical tier — the same pair the Overview card's dot bands on. + Assert.Equal( + DailyHealthBand.Critical, + DailyHealthBandCalculator.Classify(Signals(deadlocks: 480, window: Day))); + } + + /// + /// The defect #3525 was filed on: one deadlock in a 24-hour day banded the WHOLE day Critical — the + /// count trigger #3368 removed from the card, still shipping in this classifier. Measured on the same + /// 43-server fleet, count > 0 read 87.9% of 24-hour windows Critical, so ~7 of 8 calendar cells + /// painted red from deadlocks alone. At 0.04/hr the day is Healthy; the deadlock stays countable — the + /// tooltip lists it and the drill is offered — the BAND just stops claiming a crisis. + /// + [Fact] + public void ASingleDeadlockInADay_IsNoLongerCritical() + { + var s = Signals(deadlocks: 1, window: Day); + Assert.Equal(DailyHealthBand.Healthy, DailyHealthBandCalculator.Classify(s)); + + Assert.Contains(DayDrillTarget.Deadlocks, DailyHealthBandCalculator.AvailableDrills(s)); + Assert.Contains("1 deadlock", DailyHealthBandCalculator.Describe(s)); + } + + /// + /// The two-window proof, on the DAY classifier — the same per-hour rate bands the day identically over + /// a 1-hour window (a fleet-sweep span at the cadence ceiling's scale) and a 24-hour one (a calendar + /// day). Counts are integer-rate-times-whole-hours so the asserted rate is exactly the one the band + /// sees (the DeadlockRateBandTests discipline). + /// + [Theory] + [InlineData(4, DailyHealthBand.Healthy)] + [InlineData(5, DailyHealthBand.Warning)] + [InlineData(19, DailyHealthBand.Warning)] + [InlineData(20, DailyHealthBand.Critical)] + [InlineData(50, DailyHealthBand.Critical)] + public void TheSameDeadlockRate_BandsTheDayTheSame_OverAnHourAndADay(long ratePerHour, DailyHealthBand expected) + { + Assert.Equal(expected, DailyHealthBandCalculator.Classify(Signals(deadlocks: ratePerHour, window: Hour))); + Assert.Equal(expected, DailyHealthBandCalculator.Classify(Signals(deadlocks: ratePerHour * 24, window: Day))); + } + + /// + /// And the discriminating converse: the same COUNT over the two windows bands differently, which a + /// count trigger cannot do at all — the pin that goes red on any revert to counting. + /// + [Fact] + public void TheSameDeadlockCount_OverTwoWindows_BandsDifferently() + { + Assert.Equal(DailyHealthBand.Critical, DailyHealthBandCalculator.Classify(Signals(deadlocks: 30, window: Hour))); + Assert.Equal(DailyHealthBand.Healthy, DailyHealthBandCalculator.Classify(Signals(deadlocks: 30, window: Day))); + } + + /// + /// A sub-hour window — a fleet sweep at any cadence under an hour — is not rate-banded (#3368's own + /// arm): a non-zero count reads Warning (deadlocks demonstrably happened; no rate supports Critical, + /// and 1 deadlock in 15 minutes is 4/hr arithmetically but the hour was not observed), and a zero + /// count stays out of the deadlock trigger entirely rather than claiming anything. An undeclared + /// window — default(TimeSpan), a producer that declared nothing — takes the same arm, so no + /// path can rate-multiply or restore count-is-Critical by omission. + /// + [Fact] + public void ASubHourOrUndeclaredWindow_FallsToWarning_NeverCritical() + { + foreach (var window in new[] { default, TimeSpan.FromMinutes(15), TimeSpan.FromMinutes(59) }) + { + Assert.Equal(DailyHealthBand.Warning, DailyHealthBandCalculator.Classify(Signals(deadlocks: 1, window: window))); + Assert.Equal(DailyHealthBand.Warning, DailyHealthBandCalculator.Classify(Signals(deadlocks: 10_000, window: window))); + Assert.Equal(DailyHealthBand.Healthy, DailyHealthBandCalculator.Classify(Signals(deadlocks: 0, window: window))); + } + } + + /// + /// The day bands on the tiers it is handed — the store-backed pair (#3368, V120) travels through + /// . Lite has no store knobs for these, so it bands + /// on the shipped defaults, but the seam is the shared one and must honour a handed pair identically. + /// + [Fact] + public void TheDeadlockRateTiers_AreOverridable() + { + var raised = new DailyHealthThresholds { DeadlockRates = new DeadlockRateThresholds(100.0, 500.0) }; + var s = Signals(deadlocks: 480, window: Day); // 20/hr: Critical on the shipped pair + Assert.Equal(DailyHealthBand.Critical, DailyHealthBandCalculator.Classify(s)); + Assert.Equal(DailyHealthBand.Healthy, DailyHealthBandCalculator.Classify(s, raised)); + } + [Theory] [InlineData("moderate-cpu", 3, 0, 0, 0)] [InlineData("some-blocking", 0, 5, 0, 0)] @@ -74,8 +168,8 @@ public void WarningTriggers_EachAloneIsWarning(string _, long highCpu, long bloc [Fact] public void CriticalBeatsWarning_WhenBothPresent() { - // A deadlock (critical) plus moderate CPU + alerts (warning) still bands Critical. - var s = Signals(deadlocks: 1, highCpu: 3, alerts: 4); + // A critical deadlock rate (480/24h = 20/hr) plus moderate CPU + alerts (warning) still bands Critical. + var s = Signals(deadlocks: 480, highCpu: 3, alerts: 4, window: Day); Assert.Equal(DailyHealthBand.Critical, DailyHealthBandCalculator.Classify(s)); } @@ -178,6 +272,33 @@ public void BuildReasons_BlockingLine_OmitsPeak_WhenZero() Assert.DoesNotContain(reasons, r => r.Contains("peak block", StringComparison.Ordinal)); } + /// + /// The deadlock line carries the per-hour rate the band evaluated (#3525) — the card reason's own + /// disclosure rule: "120 deadlocks" against an amber cell cannot say which tier was crossed, because + /// 120 in an hour and 120 in a day are the same string. The count stays (the countable fact); the rate + /// is added (the banded one). Shared by the tooltip and the day-detail reasons, so both surfaces say it. + /// + [Fact] + public void DeadlockLine_CarriesTheRate_WhenTheWindowIsRateable() + { + var day = Signals(deadlocks: 120, window: Day); // 5.0/hr — the Warning tier exactly + Assert.Contains("120 deadlocks (5.0/hr)", DailyHealthBandCalculator.Describe(day)); + Assert.Contains("120 deadlocks (5.0/hr)", DailyHealthBandCalculator.BuildReasons(day)); + + // A single deadlock still reads singular, rate beside it. + Assert.Contains("1 deadlock (0.0/hr)", DailyHealthBandCalculator.Describe(Signals(deadlocks: 1, window: Day))); + } + + [Fact] + public void DeadlockLine_PrintsTheCountAlone_OnAnUnrateableWindow() + { + // No declared window: no rate is computable, and printing one would claim a measurement nobody + // took — the count alone is exactly what the band had to go on. + var described = DailyHealthBandCalculator.Describe(Signals(deadlocks: 2)); + Assert.Contains("2 deadlocks", described); + Assert.DoesNotContain("/hr", described); + } + [Fact] public void AvailableDrills_NoData_OffersNothing() { @@ -230,4 +351,55 @@ public void BuildKeyMetricsLine_NoWait_ShowsNone_And_LargeWaitInMinutes() Assert.Contains("Top wait: none", line); Assert.Contains("Total wait: 20.0 min", line); // 1200s / 60 } + + /* ─────────── #3525 review: the still-forming day clamps to its elapsed portion ─────────── */ + + [Fact] + public void CalendarDayWindow_FinishedDay_IsTwentyFourHours() + { + var day = new DateTime(2026, 7, 8); + Assert.Equal(TimeSpan.FromDays(1), DailyHealthBandCalculator.CalendarDayWindow(day, new DateTime(2026, 7, 9))); + Assert.Equal(TimeSpan.FromDays(1), DailyHealthBandCalculator.CalendarDayWindow(day, new DateTime(2026, 9, 1, 12, 0, 0))); + } + + [Fact] + public void CalendarDayWindow_TodayClampsToElapsed_AndFutureIsZero() + { + var day = new DateTime(2026, 7, 8); + Assert.Equal(TimeSpan.FromHours(1), DailyHealthBandCalculator.CalendarDayWindow(day, day.AddHours(1))); + Assert.Equal(TimeSpan.FromMinutes(30), DailyHealthBandCalculator.CalendarDayWindow(day, day.AddMinutes(30))); + Assert.Equal(TimeSpan.Zero, DailyHealthBandCalculator.CalendarDayWindow(day, day.AddDays(-1))); + } + + [Fact] + public void TodayCell_ActiveStorm_IsNotDilutedByUnelapsedHours() + { + /* The review's own numbers: 60 deadlocks in the first hour of the still-forming day. Banded over + a full 24h the rate reads 2.5/hr (below the 5/hr Warning tier) and the crisis paints Healthy; + over the elapsed hour it is 60/hr — past the 20/hr Critical tier. The clamp is what keeps an + in-progress storm red on the calendar. The same 60 over a genuinely FINISHED day is honestly + 2.5/hr, and stays sub-Warning by design. */ + var day = new DateTime(2026, 7, 8); + var stormWindow = DailyHealthBandCalculator.CalendarDayWindow(day, day.AddHours(1)); + Assert.Equal( + DailyHealthBand.Critical, + DailyHealthBandCalculator.Classify(Signals(deadlocks: 60, window: stormWindow))); + + var finishedWindow = DailyHealthBandCalculator.CalendarDayWindow(day, day.AddDays(2)); + Assert.Equal( + DailyHealthBand.Healthy, + DailyHealthBandCalculator.Classify(Signals(deadlocks: 60, window: finishedWindow))); + } + + [Fact] + public void TodayCell_MinutesOld_FallsToTheUnrateableArm() + { + /* Sub-hour elapsed lands in DeadlockSeverity's unrateable arm (#3368's Warning-not-rate rule): + minutes into the day a single deadlock reads Warning, never a fabricated multiplied rate. */ + var window = DailyHealthBandCalculator.CalendarDayWindow( + new DateTime(2026, 7, 8), new DateTime(2026, 7, 8, 0, 10, 0)); + Assert.Equal( + DailyHealthBand.Warning, + DailyHealthBandCalculator.Classify(Signals(deadlocks: 1, window: window))); + } } diff --git a/Lite.Tests/PerformanceCalendarDataTests.cs b/Lite.Tests/PerformanceCalendarDataTests.cs index 303aedf72..39e395d55 100644 --- a/Lite.Tests/PerformanceCalendarDataTests.cs +++ b/Lite.Tests/PerformanceCalendarDataTests.cs @@ -114,12 +114,21 @@ private Task SeedAlertAsync(DateTime day, string metric, bool dismissed) => [Fact] public async Task GetDailySummaryRange_BucketsEachDay_AndBandsViaSharedCalculator() { - // 07-02 Critical: a deadlock, plus waits (CXPACKET should win the top-wait ranking). + // 07-02 Healthy despite one deadlock (#3525): deadlocks band as a RATE over the 24-hour day + // through the card band's tiers, and 1/day is 0.04/hr — far under the 5/hr Warning tier. The + // count still lands in the row (the drill and tooltip keep it); waits decide the top-wait ranking + // (CXPACKET should win) but never the band. await SeedCollectionRunAsync(Day(2)); await SeedWaitAsync(Day(2), "CXPACKET", 100_000); await SeedWaitAsync(Day(2), "PAGEIOLATCH_SH", 50_000); await SeedDeadlockAsync(Day(2)); + // 07-03 Critical: a deadlock STORM — 480 over the day is 20/hr, the Critical tier, proving the + // rate path end-to-end through the live aggregate rather than through a hand-built signals struct. + await SeedCollectionRunAsync(Day(3)); + for (var i = 0; i < 480; i++) + await SeedDeadlockAsync(Day(3)); + // 07-05 Critical: 6 sustained high-CPU samples (>= threshold). await SeedCollectionRunAsync(Day(5)); for (var i = 0; i < 6; i++) @@ -157,18 +166,21 @@ public async Task GetDailySummaryRange_BucketsEachDay_AndBandsViaSharedCalculato var rows = await _dataService.GetDailySummaryRangeAsync(ServerId, MonthStart, MonthEnd); var byDate = rows.ToDictionary(r => r.SummaryDate.Date); - // Exactly the eight seeded days appear; unseeded days are absent (calendar renders them No-Data). - Assert.Equal(8, rows.Count); + // Exactly the nine seeded days appear; unseeded days are absent (calendar renders them No-Data). + Assert.Equal(9, rows.Count); Assert.False(byDate.ContainsKey(Day(20))); Assert.Equal(DailyHealthBand.Warning, byDate[Day(22)].HealthBand); Assert.Equal(1, byDate[Day(22)].AlertCount); - Assert.Equal(DailyHealthBand.Critical, byDate[Day(2)].HealthBand); + Assert.Equal(DailyHealthBand.Healthy, byDate[Day(2)].HealthBand); Assert.Equal(1, byDate[Day(2)].DeadlockCount); Assert.Equal("CXPACKET", byDate[Day(2)].TopWaitType); Assert.Equal(150m, byDate[Day(2)].TotalWaitTimeSec); + Assert.Equal(DailyHealthBand.Critical, byDate[Day(3)].HealthBand); + Assert.Equal(480, byDate[Day(3)].DeadlockCount); + Assert.Equal(DailyHealthBand.Critical, byDate[Day(5)].HealthBand); Assert.Equal(6, byDate[Day(5)].HighCpuEvents); @@ -216,8 +228,11 @@ public async Task GetDailySummary_SingleDay_DelegatesToRange_AndReturnsNoDataRow var seeded = await _dataService.GetDailySummaryAsync(ServerId, Day(2)); Assert.NotNull(seeded); Assert.True(seeded!.HasData); - Assert.Equal(DailyHealthBand.Critical, seeded.HealthBand); - Assert.Equal("Critical", seeded.OverallHealth); + /* One deadlock in a day is 0.04/hr — Healthy under the rate band (#3525); the count still rides + the row, which is what distinguishes this from the No-Data arm below. */ + Assert.Equal(1, seeded.DeadlockCount); + Assert.Equal(DailyHealthBand.Healthy, seeded.HealthBand); + Assert.Equal("Healthy", seeded.OverallHealth); var empty = await _dataService.GetDailySummaryAsync(ServerId, Day(25)); Assert.NotNull(empty); diff --git a/Lite/Mcp/McpHealthTools.cs b/Lite/Mcp/McpHealthTools.cs index 2cfba2eca..f1ca84c68 100644 --- a/Lite/Mcp/McpHealthTools.cs +++ b/Lite/Mcp/McpHealthTools.cs @@ -142,7 +142,8 @@ ending ON the anchor day means the anchor day is the last one included. */ var fromDate = lastDay.AddDays(-(days_back - 1)); var toDate = lastDay.AddDays(1); - var rows = await dataService.GetDailySummaryRangeAsync(resolved.ServerId, fromDate, toDate); + /* The anchor is also the clock the still-forming day's window clamps against (#3525 review). */ + var rows = await dataService.GetDailySummaryRangeAsync(resolved.ServerId, fromDate, toDate, asOfUtc: windowEnd); if (rows.Count == 0) { diff --git a/Lite/Services/LocalDataService.DailySummary.cs b/Lite/Services/LocalDataService.DailySummary.cs index dcd904b7e..02ab4c5d0 100644 --- a/Lite/Services/LocalDataService.DailySummary.cs +++ b/Lite/Services/LocalDataService.DailySummary.cs @@ -150,7 +150,7 @@ FROM day_spine s /// Returns one per collected day in the half-open [fromDate, toDate) /// window (dates normalized to their date component). Powers the Performance Calendar month grid. /// - public async Task> GetDailySummaryRangeAsync(int serverId, DateTime fromDate, DateTime toDate) + public async Task> GetDailySummaryRangeAsync(int serverId, DateTime fromDate, DateTime toDate, DateTime? asOfUtc = null) { using var _q = TimeQuery("GetDailySummaryRangeAsync", "daily summary range aggregation"); using var connection = await OpenConnectionAsync(); @@ -161,11 +161,16 @@ public async Task> GetDailySummaryRangeAsync(int serverId, command.Parameters.Add(new DuckDBParameter { Value = fromDate.Date }); command.Parameters.Add(new DuckDBParameter { Value = toDate.Date }); + /* #3525 review: the still-forming day's window clamps against the read's own clock — the anchored + MCP read hands its resolved window end so a backdated as_of never clamps against the process + clock; the live calendar read leaves this null. */ + var referenceUtc = asOfUtc ?? DateTime.UtcNow; + var results = new List(); using var reader = await command.ExecuteReaderAsync(); while (await reader.ReadAsync()) { - results.Add(ReadDailySummaryRow(reader)); + results.Add(ReadDailySummaryRow(reader, referenceUtc)); } return results; @@ -185,10 +190,11 @@ public async Task> GetDailySummaryRangeAsync(int serverId, : new DailySummaryRow { SummaryDate = targetDate, HasData = false, HealthBand = DailyHealthBand.NoData }; } - private static DailySummaryRow ReadDailySummaryRow(System.Data.Common.DbDataReader reader) + private static DailySummaryRow ReadDailySummaryRow(System.Data.Common.DbDataReader reader, DateTime referenceUtc) { var row = new DailySummaryRow { + ReferenceUtc = referenceUtc, SummaryDate = reader.IsDBNull(0) ? DateTime.MinValue : Convert.ToDateTime(reader.GetValue(0)), TotalWaitTimeSec = reader.IsDBNull(1) ? 0m : Convert.ToDecimal(reader.GetValue(1)), TopWaitType = reader.IsDBNull(2) ? "" : reader.GetString(2), @@ -211,6 +217,11 @@ private static DailySummaryRow ReadDailySummaryRow(System.Data.Common.DbDataRead public class DailySummaryRow { public DateTime SummaryDate { get; set; } + + /// The clock the still-forming day's window clamps against (#3525 review): the anchored + /// MCP range read hands its resolved window end so a backdated as_of clamps against its own "now"; + /// the live calendar read leaves the wall-clock default. + public DateTime ReferenceUtc { get; set; } = DateTime.UtcNow; public decimal TotalWaitTimeSec { get; set; } public string TopWaitType { get; set; } = ""; public long UniqueQueries { get; set; } @@ -254,5 +265,9 @@ public class DailySummaryRow MemoryPressureEvents = MemoryPressureEvents, MemoryCriticalEvents = MemoryCriticalEvents, AlertCount = AlertCount, + /* #3525: a finished calendar day bands over its full 24 hours; the still-forming day clamps to + its elapsed portion so an active storm is not diluted by hours that have not happened yet + (review finding on #3525). Anchored MCP reads hand their window end; the calendar is live. */ + Window = DailyHealthBandCalculator.CalendarDayWindow(SummaryDate, ReferenceUtc), }; } diff --git a/PerformanceMonitor.Common/DailyHealthBand.cs b/PerformanceMonitor.Common/DailyHealthBand.cs index 5bc33c1e1..1dd53e6dd 100644 --- a/PerformanceMonitor.Common/DailyHealthBand.cs +++ b/PerformanceMonitor.Common/DailyHealthBand.cs @@ -29,8 +29,8 @@ public enum DailyHealthBand /// Elevated but not critical (moderate CPU, some blocking, memory pressure, or alerts) — amber. Warning = 2, - /// A serious day (deadlocks, collection failures, sustained high CPU, heavy blocking, or - /// severe memory pressure) — red. + /// A serious day (a critical deadlock rate, collection failures, sustained high CPU, heavy + /// blocking, or severe memory pressure) — red. Critical = 3, } @@ -64,9 +64,23 @@ public readonly record struct DailyHealthSignals /// True when any collection ran that day. When false the band is always . public bool HasData { get; init; } - /// Deadlocks captured that day. Any (> 0) is Critical. + /// Deadlocks captured in the window. Banded as a RATE over through the + /// card band's own tiers (#3525) — see . public long Deadlocks { get; init; } + /// + /// How long the window these counts cover (#3525) — the denominator the deadlock rate is computed + /// over. A calendar day is 24 hours; a fleet-sweep span is previous-sweep-to-now (sub-day). + /// + /// A measurement property, so its default has to be unusable — the + /// discipline, verbatim: + /// means no window was declared, and the deadlock band then declines to compute a rate rather than + /// dividing by zero. An undeclared (or sub-hour) window fails away from Healthy, never into + /// Critical: a non-zero count reads Warning, a zero count simply stays out of the deadlock + /// trigger. + /// + public TimeSpan Window { get; init; } + /// Collector runs that ended in ERROR that day. Any (> 0) is Critical — a monitoring blind spot is itself serious. public long CollectionErrors { get; init; } @@ -115,6 +129,16 @@ public sealed record DailyHealthThresholds /// Actionable alerts at or above which the day is at least Warning. Default 1 (any alert). public int AlertWarningCount { get; init; } = 1; + /// + /// The deadlock RATE tiers the day bands with (#3525) — the SAME store-backed pair the Overview + /// card's deadlock dot reads (#3368, V120), so a day cell and the card cannot disagree about what a + /// deadlock count over a window means. Defaults to the shipped pair + /// (); a Darling caller with a store hands the + /// config_alert_settings pair in, and Lite — which has no such knobs — bands on the + /// default, which is its store's own future value should it ever grow them. + /// + public DeadlockRateThresholds DeadlockRates { get; init; } = DeadlockRateThresholds.Default; + /// The shipped defaults. Use this everywhere unless a caller has an explicit reason to override. public static DailyHealthThresholds Default { get; } = new(); } @@ -141,9 +165,23 @@ public static DailyHealthBand Classify(in DailyHealthSignals signals, DailyHealt var t = thresholds ?? DailyHealthThresholds.Default; - // Critical: anything that makes the day genuinely serious. Deadlocks, a monitoring gap - // (collection errors), severe memory pressure, sustained high CPU, or heavy blocking. - if (signals.Deadlocks > 0 + /* Deadlocks band as a RATE over the window, through the SAME band the Overview card's deadlock + dot reads (#3525) — not the "any deadlock is Critical" count trigger this replaced, #3368's + un-fixed twin. Measured on the same 43-server fleet: count > 0 read 13.4% of 1-hour windows + Critical and 87.9% of 24-hour windows Critical, and the calendar IS a 24-hour window, so ~7 of + 8 day cells painted red from deadlocks alone and the label stopped discriminating. Delegating + to DeadlockSeverity (rather than re-stating its ladder) is what keeps the day cell, the card + dot and the fleet sweep one banding: Critical/Warning fold into the day's matching tier, and + its unrateable-window arm (a sub-hour or undeclared span) is exactly the fallback this + classifier wants — a non-zero count reads Warning, a zero count (Unknown) stays out of the + deadlock trigger entirely. */ + var deadlockSeverity = ServerHealthClassifier.DeadlockSeverity( + signals.Deadlocks, signals.Window, t.DeadlockRates); + + // Critical: anything that makes the day genuinely serious. A critical deadlock rate, a + // monitoring gap (collection errors), severe memory pressure, sustained high CPU, or heavy + // blocking. + if (deadlockSeverity == HealthSeverity.Critical || signals.CollectionErrors > 0 || signals.MemoryCriticalEvents > 0 || signals.HighCpuEvents >= t.HighCpuCriticalSamples @@ -152,9 +190,10 @@ public static DailyHealthBand Classify(in DailyHealthSignals signals, DailyHealt return DailyHealthBand.Critical; } - // Warning: elevated but not critical — moderate CPU, some blocking, (non-severe) memory - // pressure, or any actionable alert fired that day. - if (signals.HighCpuEvents >= t.HighCpuWarningSamples + // Warning: elevated but not critical — an elevated deadlock rate, moderate CPU, some blocking, + // (non-severe) memory pressure, or any actionable alert fired that day. + if (deadlockSeverity == HealthSeverity.Warning + || signals.HighCpuEvents >= t.HighCpuWarningSamples || signals.BlockingEvents >= t.BlockingWarningEvents || signals.MemoryPressureEvents > 0 || signals.AlertCount >= t.AlertWarningCount) @@ -165,6 +204,28 @@ public static DailyHealthBand Classify(in DailyHealthSignals signals, DailyHealt return DailyHealthBand.Healthy; } + /// + /// The window a CALENDAR-DAY cell bands its counts over: a finished day is its full 24 hours, but + /// the still-forming day is only the portion that has elapsed — otherwise an active storm dilutes + /// against hours that have not happened yet (60 deadlocks in the last hour ÷ 24h reads 2.5/hr and + /// Healthy while the true in-progress rate is 60/hr; the pre-#3525 any-deadlock trigger could not + /// under-read this way, so the clamp is part of the rate change, per its review). The reference + /// clock is the CALLER's: an anchored (as_of) read hands its window end so a backdated read clamps + /// against its own "now", and the live calendars hand the wall clock. An elapsed portion under an + /// hour lands in 's unrateable-window arm + /// (Warning-not-rate), which is exactly right for a day cell minutes old; a reference before the + /// day starts (a future cell) returns zero for the same reason. + /// + public static TimeSpan CalendarDayWindow(DateTime summaryDate, DateTime referenceUtc) + { + var dayStart = summaryDate.Date; + if (referenceUtc >= dayStart.AddDays(1)) + return TimeSpan.FromDays(1); + + var elapsed = referenceUtc - dayStart; + return elapsed > TimeSpan.Zero ? elapsed : TimeSpan.Zero; + } + /// A short human label for the band ("No Data" / "Healthy" / "Warning" / "Critical"). public static string Label(DailyHealthBand band) => band switch { @@ -275,7 +336,7 @@ public static string BuildKeyMetricsLine(string? topWaitType, long highCpuEvents private static List BuildSignalLines(in DailyHealthSignals signals, long peakBlockMs) { var lines = new List(); - AppendCount(lines, signals.Deadlocks, "deadlock", "deadlocks"); + AppendDeadlocks(lines, signals); AppendCount(lines, signals.CollectionErrors, "collection error", "collection errors"); AppendCount(lines, signals.HighCpuEvents, "high-CPU sample", "high-CPU samples"); AppendBlocking(lines, signals.BlockingEvents, peakBlockMs); @@ -297,6 +358,24 @@ private static void AppendCount(List lines, long count, string singular, lines.Add(count.ToString("N0", CultureInfo.InvariantCulture) + " " + noun); } + /// The deadlock line — like but appends the per-hour rate the + /// band evaluated when the window is rateable, e.g. "120 deadlocks (5.0/hr)" — the card reason's own + /// format (#3368/#3525): the count is the countable fact, the rate is the banded one, and a line + /// that shows only the count cannot say which tier was crossed. An unrateable window prints the + /// count alone, which is exactly what the band had to go on. + private static void AppendDeadlocks(List lines, in DailyHealthSignals signals) + { + if (signals.Deadlocks <= 0) + return; + + var noun = signals.Deadlocks == 1 ? "deadlock" : "deadlocks"; + var line = signals.Deadlocks.ToString("N0", CultureInfo.InvariantCulture) + " " + noun; + var rate = ServerHealthClassifier.DeadlockRatePerHour(signals.Deadlocks, signals.Window); + if (rate.HasValue) + line += " (" + rate.Value.ToString("0.0", CultureInfo.InvariantCulture) + "/hr)"; + lines.Add(line); + } + /// The blocking line — like but appends the day's peak block /// duration when one is known (> 0), e.g. "42 blocking events (peak block 12.5 s)". private static void AppendBlocking(List lines, long count, long peakBlockMs) diff --git a/PerformanceMonitor.Common/ServerHealthBands.cs b/PerformanceMonitor.Common/ServerHealthBands.cs index ae2502bf9..dd7285c4f 100644 --- a/PerformanceMonitor.Common/ServerHealthBands.cs +++ b/PerformanceMonitor.Common/ServerHealthBands.cs @@ -697,8 +697,11 @@ public static HealthSeverity BlockingSeverity(int? blockingCountOrNullWhenUnmeas /// Public because both cards REPORT the rate beside the count: a card that bands on a figure /// it does not show leaves an operator reading "Deadlocks 3" against a Critical dot with no way to /// see which number crossed which tier. + /// + /// The count is a long because the daily classifier's signals carry day-scale + /// roll-ups as longs (#3525); every int caller widens implicitly. /// - public static double? DeadlockRatePerHour(int deadlockCount, TimeSpan window) => + public static double? DeadlockRatePerHour(long deadlockCount, TimeSpan window) => window >= ServerHealthThresholds.DeadlockRateMinimumWindow ? deadlockCount / window.TotalHours : null; @@ -743,7 +746,10 @@ public static HealthSeverity BlockingSeverity(int? blockingCountOrNullWhenUnmeas /// keeps reading healthy; that reasoning is untouched here. A PostgreSQL target has no /// SQL-Server-deadlock reading at any rate, so it bands off none. /// Deadlocks counted in the window, or null where the engine has no - /// source behind the reading. + /// source behind the reading. A long for 's reason (#3525): + /// the shared daily classifier routes its day-scale Deadlocks roll-up through this same band, + /// so the calendar, get_daily_summary, the fleet sweep and the Overview card cannot disagree + /// about what the same count over the same window means. /// How long that count covers. Required rather than defaulted, for the reason /// 's source is: a caller that kept the old single-argument call would /// compile and silently band a bare count again, which is the entire defect. @@ -751,7 +757,7 @@ public static HealthSeverity BlockingSeverity(int? blockingCountOrNullWhenUnmeas /// would let a surface band on the shipped pair while get_alert_settings reported the /// store's. public static HealthSeverity DeadlockSeverity( - int? deadlockCount, + long? deadlockCount, TimeSpan window, DeadlockRateThresholds thresholds) { From dcd7cb24971b601d6aacd4f3eff4ba3f7588d0b6 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 01:40:01 -0400 Subject: [PATCH 13/69] Procedures slicer plots the physical series on a physical sort; Query Store slice-total labels say Total (#3567) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * Procedures slicer: physical sort plots the physical series, and slice totals stop calling themselves averages Fixes #3556 — the #3547 bug's verified siblings, found during that fix and kept out of its file boundary. The procedures grid's physical-reads sort mapped to the LOGICAL aggregate under a "Total Physical Reads" label, with no physical case in the value switch — while the slicer's SELECT computed total_physical_reads at ordinal 6 and the reader dropped it on the floor, the #3530 shape exactly. The reader now maps TotalPhysicalReads, and the handler mirrors the Query Store fix: metric TotalPhysReads, the value-switch case, honest label. The overlay side was already wired — ComputeProcOverlayPoints has carried a dormant TotalPhysReads -> DeltaPhysicalReads arm all along, and the handler re-triggers SelectionChanged on metric change, so bars and overlay go physical together with no further threading. The Query Store switch's two remaining dishonest labels — "Avg Reads" and "Avg Writes" over execution-weighted slice TOTALS — now say Total, matching the physical arm beside them and the query-stats grid's vocabulary. Pinned three ways: a distinct-per-column reader test (110/30/50 — equal fixture values are how the drop stayed invisible), and a source-text pin over both sorting handlers' switch arms in the AvailabilityGroupsGridSortTests style, since the handlers are private WPF event handlers. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx * Darling viewer: port the #3547/#3556 grid fixes the review found unported PR #3567's claude-review flagged that ViewerServerTab.Queries.cs — the viewer's copy of Lite's sort-driven slicer metric mapping — still carried both bug classes this PR fixes in Lite. The procedures grid's physical sort mapped to the LOGICAL series with no physical value case, and the Query Store grid had both the "Avg"-over-totals labels AND the un-ported #3547 physical/logical swap. Both handlers now match Lite: TotalPhysReads mapping, value case, honest Total labels. No reader threading was needed anywhere — the shared slicer reader (ReadQueryStatsSlicerAsync) has always mapped ordinal 6's total_physical_reads, all three slicer SELECTs compute it, and the three item-timeline reads already carry physical_reads into ItemTimelinePoint, with ComputeOverlayPoints' TotalPhysReads arm sitting dormant — so only the handlers were lying. One genuine port gap beyond the switch arms: Lite's sorting handlers re-run SelectionChanged after a metric change so a selected row's overlay re-projects onto the new metric; the viewer's handlers never did, which would have drawn the newly-honest physical bars under a stale-metric overlay — the exact mismatch the #3550 commit warned about. All three handlers gain the re-trigger. Pins mirror Lite's: a source-text pin file over both fixed handlers plus the re-trigger (the omission was per handler, so pinned per handler), the slicer SQL shape theory now asserts the reads/writes/physical alias order the shared reader's ordinals depend on, and the live slicer + dedup tests gain distinct-per-column fixture values (Lite's 110/30/50 signature; 41x100/41x10 for the Query Store bucket and 40x100/40x10 for the timeline point) — equal fixture values are exactly how a swap stays invisible. Verified against a throwaway PG 18.4 + TimescaleDB cluster: 11/11 live, 48/48 ungated. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx --------- Co-authored-by: Claude Fable 5 --- Darling/Darling.Tests/ViewerQueriesTests.cs | 41 ++++++- .../ViewerSlicerMetricMapTests.cs | 86 ++++++++++++++ .../ViewerServerTab.Queries.cs | 28 ++++- Lite.Tests/ProcStatsSlicerReadTests.cs | 111 ++++++++++++++++++ Lite.Tests/QuerySlicerMetricMapTests.cs | 67 +++++++++++ Lite/Controls/ServerTab.Grids.cs | 12 +- Lite/Services/LocalDataService.QueryStats.cs | 5 + 7 files changed, 338 insertions(+), 12 deletions(-) create mode 100644 Darling/Darling.Tests/ViewerSlicerMetricMapTests.cs create mode 100644 Lite.Tests/ProcStatsSlicerReadTests.cs create mode 100644 Lite.Tests/QuerySlicerMetricMapTests.cs diff --git a/Darling/Darling.Tests/ViewerQueriesTests.cs b/Darling/Darling.Tests/ViewerQueriesTests.cs index 06ce9e2a2..e332f8422 100644 --- a/Darling/Darling.Tests/ViewerQueriesTests.cs +++ b/Darling/Darling.Tests/ViewerQueriesTests.cs @@ -317,6 +317,15 @@ public void SlicerSql_BucketsByHour_SevenColumnShape(string sqlName, string tabl /* The bucket key and the window filter agree, stated as the invariant rather than left implicit in two InlineData columns that a future edit could change one of. */ Assert.Contains(windowFilter, bucketExpression, StringComparison.Ordinal); + + /* #3556: ReadQueryStatsSlicerAsync maps the IO columns by ORDINAL (4 reads, 5 writes, 6 physical) + for all three slicers, so every SELECT must keep the three aliases in that relative order — a + reorder would silently swap series under the sort-driven metric labels. */ + var reads = sql.IndexOf("AS total_reads", StringComparison.Ordinal); + var writes = sql.IndexOf("AS total_writes", StringComparison.Ordinal); + var physical = sql.IndexOf("AS total_physical_reads", StringComparison.Ordinal); + Assert.True(reads >= 0 && writes > reads && physical > writes, + $"{sqlName}: expected total_reads, then total_writes, then total_physical_reads in the SELECT."); } /// @@ -902,6 +911,12 @@ await InsertQueryStoreAsync(connection, DedupServerId, bucketStart.AddMinutes(g. Assert.Equal(13.0, bucket.TotalCpu, 3); Assert.Equal(282.0, bucket.TotalElapsed, 3); + /* #3556: the insert helper's fixed per-execution averages (logical 100, physical 10) make the + two IO series distinct, so the QS slicer feeding physical from the logical ordinal — or vice + versa — goes red. 41 deduped executions x 100 / x 10. */ + Assert.Equal(4100.0, bucket.TotalReads, 3); + Assert.Equal(410.0, bucket.TotalPhysicalReads, 3); + /* ── the comparison ── this read groups by (database, query_hash), which is COARSER than the interval grain, and the seed gives both queries the same hash — so it is also the pin that dedup happens at the INTERVAL grain FIRST and only then re-aggregates up to the hash. @@ -933,6 +948,10 @@ agrees with the deduped bars it is drawn over instead of showing a rising stairc Assert.Equal(bucketStart.AddMinutes(15), point.PointTime); Assert.Equal(280.0, point.ElapsedMs, 3); /* 40 x 7,000us; un-deduped this is 3 points, 50/150/280 */ Assert.Equal(12.0, point.CpuMs, 3); /* 40 x 300us */ + /* #3556's overlay half: the physical-sorted bars now draw under a physical overlay, so the + timeline's logical/physical split gets the same distinct-value pin as the bars'. */ + Assert.Equal(4000.0, point.Reads, 3); /* 40 x 100 logical */ + Assert.Equal(400.0, point.PhysicalReads, 3); /* 40 x 10 physical */ /* ── the MCP / REST surface ── the same dedup, so an agent and the web dashboard see the grid's numbers rather than the inflated ones. */ @@ -1289,9 +1308,11 @@ public async Task QueryStatsSlicer_BucketsByHour_AgainstDevPostgres() try { await InsertQueryStatsAsync(connection, SlicerServerId, hour1.AddMinutes(5), "DB", "0xA", - deltaExec: 1, deltaWorker: 60_000, deltaElapsed: 120_000, deltaReads: 10, queryText: "a"); + deltaExec: 1, deltaWorker: 60_000, deltaElapsed: 120_000, deltaReads: 60, queryText: "a", + deltaWrites: 20, deltaPhysicalReads: 30); await InsertQueryStatsAsync(connection, SlicerServerId, hour1.AddMinutes(35), "DB", "0xB", - deltaExec: 1, deltaWorker: 60_000, deltaElapsed: 120_000, deltaReads: 10, queryText: "b"); + deltaExec: 1, deltaWorker: 60_000, deltaElapsed: 120_000, deltaReads: 50, queryText: "b", + deltaWrites: 10, deltaPhysicalReads: 20); await InsertQueryStatsAsync(connection, SlicerServerId, hour2.AddMinutes(5), "DB", "0xA", deltaExec: 1, deltaWorker: 30_000, deltaElapsed: 60_000, deltaReads: 10, queryText: "a"); @@ -1303,6 +1324,15 @@ await InsertQueryStatsAsync(connection, SlicerServerId, hour2.AddMinutes(5), "DB Assert.Equal(2, first.SessionCount); /* two distinct query hashes in hour1 */ Assert.Equal(120.0, first.TotalCpu, 3); /* (60000 + 60000) us / 1000 -> ms */ + /* #3556's distinct-per-column pin (Lite's ProcStatsSlicerReadTests twin, same 110/30/50 + signature): the three slicers share ReadQueryStatsSlicerAsync, so mapping any IO ordinal to + the wrong bucket field goes red here — equal fixture values are exactly how a swap would stay + invisible. TotalReads and TotalLogicalReads are deliberate aliases of the LOGICAL aggregate. */ + Assert.Equal(110.0, first.TotalReads, 3); + Assert.Equal(110.0, first.TotalLogicalReads, 3); + Assert.Equal(30.0, first.TotalWrites, 3); + Assert.Equal(50.0, first.TotalPhysicalReads, 3); + bodySucceeded = true; } finally @@ -1316,7 +1346,8 @@ await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, async (cleanup, cleanupCt) = private static async Task InsertQueryStatsAsync( NpgsqlConnection connection, int serverId, DateTime collectionTimeUtc, string databaseName, string queryHash, - long deltaExec, long deltaWorker, long deltaElapsed, long deltaReads, string queryText, string sqlHandle = "0xSQLHANDLE") + long deltaExec, long deltaWorker, long deltaElapsed, long deltaReads, string queryText, string sqlHandle = "0xSQLHANDLE", + long deltaWrites = 0, long deltaPhysicalReads = 0) { using var command = new NpgsqlCommand(@" INSERT INTO query_stats @@ -1344,8 +1375,8 @@ INSERT INTO query_stats command.Parameters.AddWithValue(deltaElapsed); command.Parameters.AddWithValue(deltaReads); command.Parameters.AddWithValue(0L); - command.Parameters.AddWithValue(0L); - command.Parameters.AddWithValue(0L); + command.Parameters.AddWithValue(deltaWrites); + command.Parameters.AddWithValue(deltaPhysicalReads); command.Parameters.AddWithValue(0L); command.Parameters.AddWithValue(0L); command.Parameters.AddWithValue(1L); diff --git a/Darling/Darling.Tests/ViewerSlicerMetricMapTests.cs b/Darling/Darling.Tests/ViewerSlicerMetricMapTests.cs new file mode 100644 index 000000000..27d8dcd3a --- /dev/null +++ b/Darling/Darling.Tests/ViewerSlicerMetricMapTests.cs @@ -0,0 +1,86 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System.Text.RegularExpressions; +using Xunit; + +namespace Darling.Tests; + +/// +/// #3547/#3556's Darling half — the twin of Lite.Tests.QuerySlicerMetricMapTests. The viewer's grid +/// sorting handlers translate a sorted column name into a slicer metric key and a label, and the value +/// switch below each translates that key into a bucket field — three hops where a physical-reads sort +/// silently plotted the LOGICAL aggregate (both the procedures and Query Store grids here; Lite fixed its +/// copies in #3550/#3567 and this port stayed broken), and where slice TOTALS were labeled "Avg". The +/// handlers are private event handlers on a WPF UserControl, so this pins the source text rather than +/// instantiating the control; the value-level halves are pinned with distinct per-column fixture values by +/// the live slicer and dedup tests in ViewerQueriesLivePostgresTests, so between them a +/// logical/physical swap on either side of the seam goes red. +/// +public sealed class ViewerSlicerMetricMapTests +{ + private const string QueriesSource = "Darling/PerformanceMonitor.Darling.Viewer/ViewerServerTab.Queries.cs"; + + private static string HandlerBody(string handlerName) + { + var source = ParitySourceLocal.ReadFile(QueriesSource); + var match = Regex.Match( + source, + @"private void " + handlerName + @".*?(?=\n private )", + RegexOptions.Singleline); + Assert.True(match.Success, $"{QueriesSource}: {handlerName} not found — update this pin if the handler moved."); + return match.Value; + } + + [Fact] + public void QueryStore_PhysicalSort_MapsToThePhysicalSeries() + { + var body = HandlerBody("QueryStoreGrid_Sorting"); + + Assert.Matches(@"""AvgPhysicalReads""\s*=>\s*\(""TotalPhysReads"",\s*""Total Physical Reads""\)", body); + Assert.Matches(@"""TotalPhysReads""\s*=>\s*bucket\.TotalPhysicalReads", body); + } + + /// #3556's label half: the plotted bucket values are execution-weighted slice TOTALS, so an + /// "Avg" label under-claimed what the bars showed. + [Fact] + public void QueryStore_SliceTotalMetrics_CarryTotalLabels() + { + var body = HandlerBody("QueryStoreGrid_Sorting"); + + Assert.Matches(@"""AvgLogicalReads""\s*=>\s*\(""TotalReads"",\s*""Total Reads""\)", body); + Assert.Matches(@"""AvgLogicalWrites""\s*=>\s*\(""TotalWrites"",\s*""Total Writes""\)", body); + } + + [Fact] + public void Procedures_PhysicalSort_MapsToThePhysicalSeries() + { + var body = HandlerBody("ProcedureStatsGrid_Sorting"); + + Assert.Matches(@"""TotalPhysicalReads""\s*=>\s*\(""TotalPhysReads"",\s*""Total Physical Reads""\)", body); + Assert.Matches(@"""TotalPhysReads""\s*=>\s*bucket\.TotalPhysicalReads", body); + } + + /// + /// The viewer-specific half of the fix: Lite's sorting handlers re-run the SelectionChanged handler + /// after a metric change so a selected row's overlay is re-projected onto the new metric, and this port + /// shipped without that — so honest bars would have drawn under a stale-metric overlay, the exact + /// mismatch the #3547 fix's commit warned about. Pinned per handler, because the omission was per + /// handler. + /// + [Theory] + [InlineData("QueryStatsGrid_Sorting", "QueryStatsGrid_SelectionChanged(QueryStatsGrid, null!)")] + [InlineData("ProcedureStatsGrid_Sorting", "ProcedureStatsGrid_SelectionChanged(ProcedureStatsGrid, null!)")] + [InlineData("QueryStoreGrid_Sorting", "QueryStoreGrid_SelectionChanged(QueryStoreGrid, null!)")] + public void EverySortingHandler_RecomputesTheOverlayOnMetricChange(string handlerName, string retrigger) + { + var body = HandlerBody(handlerName); + + Assert.Contains(retrigger, body, System.StringComparison.Ordinal); + } +} diff --git a/Darling/PerformanceMonitor.Darling.Viewer/ViewerServerTab.Queries.cs b/Darling/PerformanceMonitor.Darling.Viewer/ViewerServerTab.Queries.cs index b5f1ae3b8..9f698e9fc 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/ViewerServerTab.Queries.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/ViewerServerTab.Queries.cs @@ -328,6 +328,10 @@ private void QueryStatsGrid_Sorting(object sender, DataGridSortingEventArgs e) } QueryStatsSlicer.UpdateMetric(label); + + // Re-compute overlay with new metric if a row is selected + if (QueryStatsGrid.SelectedItem != null) + QueryStatsGrid_SelectionChanged(QueryStatsGrid, null!); } /// Sorting the Top Procedures grid swaps the slicer's aggregate curve to match the sorted column. @@ -344,7 +348,10 @@ private void ProcedureStatsGrid_Sorting(object sender, DataGridSortingEventArgs "AvgElapsedMs" => ("AvgElapsed", "Avg Duration (ms)"), "TotalLogicalReads" or "AvgReads" => ("TotalReads", "Total Reads"), "TotalLogicalWrites" => ("TotalWrites", "Total Writes"), - "TotalPhysicalReads" => ("TotalReads", "Total Physical Reads"), + /* #3556's Darling half: the #3547 bug one grid over — this arm mapped the physical sort to the + LOGICAL series under a physical label. The shared slicer reader has always mapped ordinal 6's + total_physical_reads into TotalPhysicalReads; only this handler pointed at the wrong series. */ + "TotalPhysicalReads" => ("TotalPhysReads", "Total Physical Reads"), _ => ("TotalCpu", "Total CPU (ms)"), }; @@ -362,11 +369,15 @@ private void ProcedureStatsGrid_Sorting(object sender, DataGridSortingEventArgs "AvgElapsed" => bucket.TotalElapsed / n, "TotalReads" => bucket.TotalReads, "TotalWrites" => bucket.TotalWrites, + "TotalPhysReads" => bucket.TotalPhysicalReads, _ => bucket.TotalCpu, }; } ProcStatsSlicer.UpdateMetric(label); + + if (ProcedureStatsGrid.SelectedItem != null) + ProcedureStatsGrid_SelectionChanged(ProcedureStatsGrid, null!); } /// Sorting the Query Store grid swaps the slicer's aggregate curve to match the sorted column. @@ -381,9 +392,14 @@ private void QueryStoreGrid_Sorting(object sender, DataGridSortingEventArgs e) "AvgCpuTimeMs" => ("AvgCpu", "Avg CPU (ms)"), "TotalDurationMs" => ("TotalElapsed", "Total Duration (ms)"), "AvgDurationMs" => ("AvgElapsed", "Avg Duration (ms)"), - "AvgLogicalReads" => ("TotalReads", "Avg Reads"), - "AvgLogicalWrites" => ("TotalWrites", "Avg Writes"), - "AvgPhysicalReads" => ("TotalReads", "Avg Physical Reads"), + /* #3556: the plotted bucket values are execution-weighted slice TOTALS, so the old "Avg" + labels under-claimed what the bars showed. Total labels, matching the physical arm below. */ + "AvgLogicalReads" => ("TotalReads", "Total Reads"), + "AvgLogicalWrites" => ("TotalWrites", "Total Writes"), + /* #3547's Darling half: this arm still mapped physical to the LOGICAL series under a physical + label — the swap Lite fixed in #3550, never ported here. The reader side needed nothing: the + slicer SQL computes total_physical_reads and the shared reader maps it. */ + "AvgPhysicalReads" => ("TotalPhysReads", "Total Physical Reads"), "TotalExecutions" => ("Sessions", "Executions"), _ => ("TotalCpu", "Total CPU (ms)"), }; @@ -402,12 +418,16 @@ private void QueryStoreGrid_Sorting(object sender, DataGridSortingEventArgs e) "AvgElapsed" => bucket.TotalElapsed / n, "TotalReads" => bucket.TotalReads, "TotalWrites" => bucket.TotalWrites, + "TotalPhysReads" => bucket.TotalPhysicalReads, "Sessions" => bucket.SessionCount, _ => bucket.TotalCpu, }; } QueryStoreSlicer.UpdateMetric(label); + + if (QueryStoreGrid.SelectedItem != null) + QueryStoreGrid_SelectionChanged(QueryStoreGrid, null!); } /// The sorted column's member path (SortMemberPath, else the bound Binding path) — Lite's fallback. diff --git a/Lite.Tests/ProcStatsSlicerReadTests.cs b/Lite.Tests/ProcStatsSlicerReadTests.cs new file mode 100644 index 000000000..66eb528cd --- /dev/null +++ b/Lite.Tests/ProcStatsSlicerReadTests.cs @@ -0,0 +1,111 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor Lite. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Threading.Tasks; +using DuckDB.NET.Data; +using PerformanceMonitorLite.Database; +using PerformanceMonitorLite.Services; +using PerformanceMonitorLite.Tests; +using Xunit; + +namespace Lite.Tests; + +/// +/// #3556: the procedures slicer's reader mapped BOTH read fields to ordinal 4 and never read ordinal 6, +/// so the SELECT's total_physical_reads was computed and dropped on the floor — the #3530 bug's twin, one +/// grid over. As there, the seed's three I/O columns carry values no other column can reproduce: equal +/// fixture values are exactly how the slip stayed invisible, so distinct-per-column is the point of this +/// test, not a nicety. +/// +public sealed class ProcStatsSlicerReadTests : IClassFixture, IDisposable +{ + private const int ServerId = 8856; + private const string Db = "ProcSliceDb"; + + private readonly DuckDbInitializer _duckDb; + private DuckDBConnection? _seedConn; + private long _nextId = 1; + + public ProcStatsSlicerReadTests(SharedDuckDbFixture fixture) + { + fixture.ResetData(); + _duckDb = fixture.DuckDb; + } + + public void Dispose() => _seedConn?.Dispose(); + + /* hoursBack (not fromDate/toDate) on purpose: GetTimeRange applies ServerTimeHelper.UtcOffsetMinutes + to an explicit range, which would make a fixed timestamp depend on the machine's server-time + offset. Floored to the hour so the single seeded row sits squarely in one date_trunc bucket. */ + private static readonly DateTime BucketStart = HourFloor(DateTime.UtcNow.AddHours(-3)); + + private static DateTime HourFloor(DateTime t) => + DateTime.SpecifyKind(new DateTime(t.Ticks - (t.Ticks % TimeSpan.TicksPerHour)), DateTimeKind.Unspecified); + + private async Task SeedConnectionAsync() + { + if (_seedConn is null) + { + _seedConn = _duckDb.CreateConnection(); + await _seedConn.OpenAsync(); + } + return _seedConn; + } + + private async Task SeedAsync( + DateTime collectionTime, + string objectName, + long workerUs, + long elapsedUs, + long logicalReads, + long logicalWrites, + long physicalReads) + { + using var readLock = _duckDb.AcquireReadLock(); + var connection = await SeedConnectionAsync(); + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +INSERT INTO procedure_stats + (collection_id, collection_time, server_id, server_name, + database_name, schema_name, object_name, delta_execution_count, + delta_worker_time, delta_elapsed_time, + delta_logical_reads, delta_logical_writes, delta_physical_reads) +VALUES ($1, $2, $3, $4, $5, 'dbo', $6, $7, $8, $9, $10, $11, $12)"; + cmd.Parameters.Add(new DuckDBParameter { Value = _nextId++ }); + cmd.Parameters.Add(new DuckDBParameter { Value = DateTime.SpecifyKind(collectionTime, DateTimeKind.Unspecified) }); + cmd.Parameters.Add(new DuckDBParameter { Value = ServerId }); + cmd.Parameters.Add(new DuckDBParameter { Value = "ProcSliceSrv" }); + cmd.Parameters.Add(new DuckDBParameter { Value = Db }); + cmd.Parameters.Add(new DuckDBParameter { Value = objectName }); + cmd.Parameters.Add(new DuckDBParameter { Value = 10L }); + cmd.Parameters.Add(new DuckDBParameter { Value = workerUs }); + cmd.Parameters.Add(new DuckDBParameter { Value = elapsedUs }); + cmd.Parameters.Add(new DuckDBParameter { Value = logicalReads }); + cmd.Parameters.Add(new DuckDBParameter { Value = logicalWrites }); + cmd.Parameters.Add(new DuckDBParameter { Value = physicalReads }); + await cmd.ExecuteNonQueryAsync(); + } + + [Fact] + public async Task SlicerBucket_MapsWritesAndPhysicalReads_ToTheirOwnColumns() + { + await SeedAsync(BucketStart.AddMinutes(5), "usp_IoMap", + workerUs: 1_000, elapsedUs: 2_000, logicalReads: 110, logicalWrites: 30, physicalReads: 50); + + var bucket = Assert.Single(await new LocalDataService(_duckDb).GetProcStatsSlicerDataAsync(ServerId, hoursBack: 24)); + + /* All pairwise distinct — 110 / 30 / 50. TotalReads and TotalLogicalReads are deliberate aliases + of the LOGICAL aggregate (ordinal 4), the same shape the query-stats and Query Store slicers + map; physical rides its own column at ordinal 6. */ + Assert.Equal(110.0, bucket.TotalReads, precision: 6); + Assert.Equal(110.0, bucket.TotalLogicalReads, precision: 6); + Assert.Equal(30.0, bucket.TotalWrites, precision: 6); + Assert.Equal(50.0, bucket.TotalPhysicalReads, precision: 6); + } +} diff --git a/Lite.Tests/QuerySlicerMetricMapTests.cs b/Lite.Tests/QuerySlicerMetricMapTests.cs new file mode 100644 index 000000000..889c2df1e --- /dev/null +++ b/Lite.Tests/QuerySlicerMetricMapTests.cs @@ -0,0 +1,67 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor Lite. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System.Text.RegularExpressions; +using Xunit; + +namespace Lite.Tests; + +/// +/// #3547/#3556: the grid sorting handlers translate a sorted column name into a slicer metric key and a +/// label, and the value switch below each translates that key into a bucket field — three hops where a +/// physical-reads sort silently plotted the LOGICAL aggregate (Query Store first, then its twin in the +/// procedures grid), and where slice TOTALS were labeled "Avg". The handlers are private event handlers on +/// a WPF UserControl, so like this pins the source +/// text rather than instantiating the control; the value-level halves are pinned with distinct per-column +/// fixture values by QueryStoreDedupReadTests and ProcStatsSlicerReadTests, so between them a +/// logical/physical swap on either side of the seam goes red. +/// +public sealed class QuerySlicerMetricMapTests +{ + private const string GridsSource = "Lite/Controls/ServerTab.Grids.cs"; + + private static string HandlerBody(string handlerName) + { + var source = ParitySource.ReadFile(GridsSource); + var match = Regex.Match( + source, + @"private void " + handlerName + @".*?(?=\n private )", + RegexOptions.Singleline); + Assert.True(match.Success, $"{GridsSource}: {handlerName} not found — update this pin if the handler moved."); + return match.Value; + } + + [Fact] + public void QueryStore_PhysicalSort_MapsToThePhysicalSeries() + { + var body = HandlerBody("QueryStoreGrid_Sorting"); + + Assert.Matches(@"""AvgPhysicalReads""\s*=>\s*\(""TotalPhysReads"",\s*""Total Physical Reads""\)", body); + Assert.Matches(@"""TotalPhysReads""\s*=>\s*bucket\.TotalPhysicalReads", body); + } + + /// #3556's label half: the plotted bucket values are execution-weighted slice TOTALS, so an + /// "Avg" label under-claimed what the bars showed. + [Fact] + public void QueryStore_SliceTotalMetrics_CarryTotalLabels() + { + var body = HandlerBody("QueryStoreGrid_Sorting"); + + Assert.Matches(@"""AvgLogicalReads""\s*=>\s*\(""TotalReads"",\s*""Total Reads""\)", body); + Assert.Matches(@"""AvgLogicalWrites""\s*=>\s*\(""TotalWrites"",\s*""Total Writes""\)", body); + } + + [Fact] + public void Procedures_PhysicalSort_MapsToThePhysicalSeries() + { + var body = HandlerBody("ProcedureStatsGrid_Sorting"); + + Assert.Matches(@"""TotalPhysicalReads""\s*=>\s*\(""TotalPhysReads"",\s*""Total Physical Reads""\)", body); + Assert.Matches(@"""TotalPhysReads""\s*=>\s*bucket\.TotalPhysicalReads", body); + } +} diff --git a/Lite/Controls/ServerTab.Grids.cs b/Lite/Controls/ServerTab.Grids.cs index affc567af..fbb5a2c41 100644 --- a/Lite/Controls/ServerTab.Grids.cs +++ b/Lite/Controls/ServerTab.Grids.cs @@ -367,8 +367,10 @@ private void QueryStoreGrid_Sorting(object sender, DataGridSortingEventArgs e) "AvgCpuTimeMs" => ("AvgCpu", "Avg CPU (ms)"), "TotalDurationMs" => ("TotalElapsed", "Total Duration (ms)"), "AvgDurationMs" => ("AvgElapsed", "Avg Duration (ms)"), - "AvgLogicalReads" => ("TotalReads", "Avg Reads"), - "AvgLogicalWrites" => ("TotalWrites", "Avg Writes"), + /* #3556: the plotted bucket values are execution-weighted slice TOTALS, so the old "Avg" + labels under-claimed what the bars showed. Total labels, matching the physical arm below. */ + "AvgLogicalReads" => ("TotalReads", "Total Reads"), + "AvgLogicalWrites" => ("TotalWrites", "Total Writes"), /* #3547: this arm mapped physical to the LOGICAL series while the reader dropped the physical column; #3530 populates TotalPhysicalReads, so the label and the series finally agree. */ "AvgPhysicalReads" => ("TotalPhysReads", "Total Physical Reads"), @@ -418,7 +420,10 @@ private void ProcedureStatsGrid_Sorting(object sender, DataGridSortingEventArgs "AvgElapsedMs" => ("AvgElapsed", "Avg Duration (ms)"), "TotalLogicalReads" or "AvgReads" => ("TotalReads", "Total Reads"), "TotalLogicalWrites" => ("TotalWrites", "Total Writes"), - "TotalPhysicalReads" => ("TotalReads", "Total Physical Reads"), + /* #3556: the #3547 bug one grid over — this arm mapped the physical sort to the LOGICAL + series while the slicer reader dropped its SELECT's total_physical_reads column on the + floor. The reader maps ordinal 6 now, so the label and the series agree here too. */ + "TotalPhysicalReads" => ("TotalPhysReads", "Total Physical Reads"), _ => ("TotalCpu", "Total CPU (ms)"), }; @@ -436,6 +441,7 @@ private void ProcedureStatsGrid_Sorting(object sender, DataGridSortingEventArgs "AvgElapsed" => bucket.TotalElapsed / n, "TotalReads" => bucket.TotalReads, "TotalWrites" => bucket.TotalWrites, + "TotalPhysReads" => bucket.TotalPhysicalReads, _ => bucket.TotalCpu, }; } diff --git a/Lite/Services/LocalDataService.QueryStats.cs b/Lite/Services/LocalDataService.QueryStats.cs index 3026a8aa7..5cf380f52 100644 --- a/Lite/Services/LocalDataService.QueryStats.cs +++ b/Lite/Services/LocalDataService.QueryStats.cs @@ -861,9 +861,14 @@ GROUP BY date_trunc('hour', collection_time) SessionCount = reader.IsDBNull(1) ? 0 : Convert.ToInt64(reader.GetValue(1)), TotalCpu = reader.IsDBNull(2) ? 0 : ToDouble(reader.GetValue(2)), TotalElapsed = reader.IsDBNull(3) ? 0 : ToDouble(reader.GetValue(3)), + /* Ordinal 4 (total_reads) is the LOGICAL-reads aggregate — TotalReads and TotalLogicalReads + are deliberate aliases of it, same as the query-stats and Query Store slicers. Physical + reads ride separately at ordinal 6; this reader shipped without that mapping, so the + physical column was computed and then dropped on the floor (#3556, the #3530 bug's twin). */ TotalReads = reader.IsDBNull(4) ? 0 : ToDouble(reader.GetValue(4)), TotalWrites = reader.IsDBNull(5) ? 0 : ToDouble(reader.GetValue(5)), TotalLogicalReads = reader.IsDBNull(4) ? 0 : ToDouble(reader.GetValue(4)), + TotalPhysicalReads = reader.IsDBNull(6) ? 0 : ToDouble(reader.GetValue(6)), Value = reader.IsDBNull(2) ? 0 : ToDouble(reader.GetValue(2)), }); } From 7c9e1332ecdd87a8eb8956c6c9e65abeb8922124 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 05:39:02 -0400 Subject: [PATCH 14/69] CHANGELOG: the brains-review wave, one splice (17 entries) (#3568) * CHANGELOG: the brains-review wave, one splice (17 entries) Wave-1 of the 2026-09 brains-review campaign landed sixteen PRs on dev with a buffered changelog protocol - agents reported their entries to the coordinator instead of touching this file, so sixteen PRs merged without a single CHANGELOG conflict. This is the one post-wave splice: seventeen Fixed entries covering #3523-#3537, #3547, #3548, #3551, #3556, plus their reference links. #3549 was documentation-only and carries no entry. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx * CHANGELOG: fold the #3561 Dashboard divisor entry into the wave splice PR #3569 landed after the splice opened; its entry joins the same PR to keep the one-splice-per-wave discipline (18 entries now). Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx * CHANGELOG: fold the #3563 viewer-pass entry into the wave splice (19 entries) Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx --------- Co-authored-by: Claude Fable 5 --- CHANGELOG.md | 45 +++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 45 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index ab726a132..358a3f39b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -19,6 +19,31 @@ cut it is archived and compacted like every other version: Releases before 3.0.0 are not archived: those entries carry no prose to move. +## [Unreleased] + +### Fixed + +- **The by-CPU tools now actually rank by CPU** ([#3523]) - get_top_queries_by_cpu and get_top_procedures_by_cpu ordered by summed elapsed time in both SKUs, so on a wait-bound server the real CPU consumers could be missing from the page entirely - and attributed_cpu_ratio read as "hidden CPU" when it actually meant "wrong sort key". Every ranking site now orders by worker time, including the over-fetch cut that could drop a CPU-heavy query before the final sort ever saw it. The viewer's Duration grids keep their elapsed ranking, which is what they promise. +- **analyze_server no longer answers "all metrics are within normal ranges" when the analysis window collected nothing** ([#3524]) - The analysis gate passes on lifetime history, so a server whose collection died still reached the all-clear path with an empty window. Both SKUs' analysis services now flag the zero-facts window and analyze_server returns the "unavailable" envelope pointing at get_collection_health; the genuine all-clear (facts collected, zero findings) is unchanged. +- **The Performance Calendar, daily summary, and fleet sweep band deadlocks as a measured per-hour rate, not any-deadlock-is-Critical** ([#3525]) - The shared daily classifier routed the Deadlocks signal through the Overview card's store-backed rate tiers (#3368, V120) with each surface's real window as the denominator, so one deadlock no longer paints a calendar day red, sweep verdicts stop scaling with the cadence knob, sub-hour spans fall to Warning instead of a multiplied rate, and the day tooltip/reasons report the rate beside the count. The still-forming day clamps its window to the elapsed portion, so an active storm bands on its true in-progress rate instead of diluting against hours that have not happened yet. +- **Perfmon rates are honest per-second values in analysis** ([#3527]) - The PERFMON_*_SEC facts, the batch-request anomaly window, and both SKUs' batch-request baselines read the per-collection-interval delta as if per-second (60-300x overstatement); every read now divides by the measured sample_interval_seconds (Darling's baseline derives it from collection-time gaps, since the continuous aggregate stores no interval), and interval-0 rows - where no delta was knowable - are skipped instead of read as rates. +- **Floor the SQL count thresholds, give Store Disk Pressure a GB floor, and count measured metrics in the fleet Healthy label** ([#3528]) - The SQL Server deadlock and blocking count thresholds now floor at 1 on read like their PostgreSQL twins, so a store row hand-edited to 0 can't fire on a quiet server; Store Disk Pressure gains a self_alerts.disk_free_warn_gb floor (default 50, 0 disables) so a large store volume at a low percent stops paging CRITICAL with hundreds of GB of runway; and fleet cards carry measured_metric_count/metric_count so a Healthy label built off one measured metric of six says so ("1 of 6 measured" on the web fleet page). +- **get_memory_trend stops reporting granted memory as a hardcoded zero** ([#3529]) - The MCP payload shipped a literal total_granted_mb = 0.0 on both SKUs, steering agents away from memory grants during grant-pressure investigations. The field became an explicit null with the envelope naming get_memory_grants as the grants series' source, and both tool descriptions stopped promising granted memory. +- **Lite's Query Store time slicer reads physical reads from its own column** ([#3530]) - The reader mapped both read fields to the logical-reads ordinal and never read the SELECT's total_physical_reads column, so the physical aggregate was computed and dropped. Pinned with a test whose fixture rows carry distinct values per I/O column. +- **LCK_M_IS advice carries the same RCSI caveats as its LCK_M_S twin** ([#3531]) - The intent-shared-lock advice handed out the READ_COMMITTED_SNAPSHOT ALTER with no caveats. Both lock twins now name the brief exclusive lock the ALTER takes, the tempdb version-store cost, and the test-on-a-copy warning for NOLOCK-dependent code. +- **Collector schedules refuse cadences that would fabricate quiet** ([#3532]) - Delta-family collectors (wait/latch/spinlock/query/procedure/file I/O/memory-grant/perfmon stats, and the PostgreSQL wait/statement collectors) now cap at 30 minutes in both Lite's and Darling's schedule editors, with the store-side resolver ignoring out-of-policy rows: past the shared 60-minute delta gap policy every cycle re-baselined and recorded zeros forever, so the charts flatlined green precisely because collection stopped measuring. +- **get_pg_plans finds the plan you asked for, not just the plans in the top page** ([#3533]) - the queryid filter ran client-side over a fetched top-duration page: the reader had no query_id predicate, so the tool pulled the top limit x 10 shapes by total time and filtered them in C#, which made any plan ranked below that page unfindable at every window size - and the filtered-empty branch then told the caller capture was working, the plan was never captured, and the statement was "not the query to look at", while the plan sat in the store. The predicate now runs in the store's SQL over every capture in the window, the over-fetch is gone, and the miss text says what was actually searched and what a miss can mean, with get_pg_top_queries and get_pg_plan_capture_readiness cross-references. The web dashboard's read dispatch already passed query_id through, so it gains the server-side search with no wiring change. +- **get_pg_autovacuum_health classifies severity from the same axis it ranks by** ([#3534]) - The reader ranks tables by GREATEST(dead ratio, insert ratio) but severity only read the dead side, so an append-only worst_table ten times past its insert-vacuum threshold reported "ok"; severity now comes from the worse of the two ratios, the insert-side ratio is published as insert_threshold_ratio, a one-sample window reports growth as unknown instead of "flat" (with first_seen_at showing the window), and the page-scoped summary counts carry the sibling limit_reached flag. +- **get_pg_replication_slots headlines the worst-classified slot and never spells unknown WAL growth as stable** ([#3535]) - worst_slot was picked by retained size, so an active 45 GB keeping-pace slot ("ok") outranked an inactive growing orphan ("critical_orphan_filling_disk"); slots now rank by severity with size only breaking ties, growth is null when either endpoint is the collector's -1 sentinel or the window holds one sample, new unknown-growth severities say so explicitly, and raw retained_wal_bytes nulls the sentinel like its _gb sibling. +- **get_pg_io_stats no longer renders track_io_timing=off as an impossibly fast disk** ([#3536]) - track_io_timing is OFF by default in PostgreSQL, so on a stock server the read and write time counters are never populated - and this read divided the zeros out to 0.000 ms latencies while its trend sibling had shipped the honest contract all along. The single-window read now mirrors that contract exactly: the setting is read from the collected pg_server_config bounded by the window's end (inferred from the window's data when never collected, with io_timing_source saying which), io_timing_tracked and timing_note are published, and every time-derived figure is null when untracked rather than zero, while the operation counts, hit ratios and byte figures stand untouched. The new busiest_basis field states which key decided the "busiest" ranking. +- **The PostgreSQL xmin-horizon alert catches rotating holders and stops firing on single observations** ([#3537]) - The persistence gate gains a horizon arm that fires when the horizon sits past the age threshold in a majority of the window's real captures (counted from the collector's own collection_log runs) regardless of who holds it, naming the rotating-holder pattern under a stable dedup subject; a minimum-observations floor stops the first holder after quiet hours from reading 1-of-1 as chronic. +- **QueryStore slicer's physical-reads sort plots the physical series** ([#3547]) - Sorting the Query Store grid by physical reads relabeled the slicer but kept plotting logical reads, and the selected-row overlay had no physical series to draw at all. Bars and overlay both plot the real physical-reads aggregate now. +- **get_memory_trend joins the grants series so total_granted_mb carries real data** ([#3548]) - Completing #3529's honest null: both SKUs join the memory-grant series the viewers already chart into the trend payload, matching each memory point to the nearest grants snapshot within 30 seconds. A snapshot measuring nothing granted is a genuine 0.0, an uncovered point stays null, and the granted_note appears only when there is a gap to explain. +- **Viewer Recommendations tabs stop saying "All clear" when the analysis window collected nothing** ([#3551]) - Both viewers only branched on the insufficient-data message, so a server whose collection died still rendered the all-clear on a zero-finding read. Lite's Generate now consumes the analysis service's window-empty state directly; the Darling service persists it into the analysis-state marker (false-with-a-message, no schema change) so the viewer's read and Generate now doors both render a "nothing was measured - check Collection Health" notice. The genuine all-clear (facts measured, zero findings) is unchanged. +- **Viewer pass for the Store Disk Pressure GB floor and fleet measured-metric qualifier** ([#3563]) - The Settings window can now edit the Store Disk Pressure warning's GB floor (V126, 0 disables) next to its percent sibling, and the WPF fleet card's band label carries the web fleet page's "N of M measured" qualifier when a band folded over unmeasured metrics, so an Unknown-heavy server no longer reads as an unqualified green. +- **Deprecated Dashboard analysis reads perfmon counters as true per-second rates** ([#3561]) - The Dashboard's analysis facts, batch-request anomaly window, and batch-request baseline read the per-collection-interval perfmon delta as if it were per-second, overstating by 60-300x at common cadences; all three now divide by the row's measured sample interval and skip rows with no knowable delta, matching the Lite/Darling fix in #3560. +- **Sorting the Procedures or Query Store grid by physical reads plots the physical series, in both apps** ([#3556]) - In Lite and the Darling viewer, the Procedures grid's physical-reads sort plotted the logical aggregate under a physical label, and the Query Store slicer labeled execution-weighted slice totals "Avg". The Darling viewer's Query Store grid also still carried the original #3547 physical/logical swap, and none of its grids re-projected a selected row's overlay when the sort metric changed, so the bars and the overlay could show different metrics. Both apps now plot the series the sorted column names, label slice totals as totals, and keep the overlay on the bars' metric - pinned by distinct-per-column reader tests and source-text pins over the sorting handlers in both apps. + ## [3.8.0] - 2026-09-17 Full entries: [docs/changelog/3.8.md](docs/changelog/3.8.md) @@ -1227,6 +1252,26 @@ Full entries: [docs/changelog/3.0.md](docs/changelog/3.0.md) - **Failed SQL Agent job alert** ([#749]) - **Installer: optional custom data/log file locations** ([#768]) +[#3523]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3523 +[#3524]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3524 +[#3525]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3525 +[#3527]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3527 +[#3528]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3528 +[#3529]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3529 +[#3530]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3530 +[#3531]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3531 +[#3532]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3532 +[#3533]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3533 +[#3534]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3534 +[#3535]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3535 +[#3536]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3536 +[#3537]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3537 +[#3547]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3547 +[#3548]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3548 +[#3551]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3551 +[#3556]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3556 +[#3561]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3561 +[#3563]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3563 [#3514]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3514 [#3477]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3477 [#3495]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3495 From cec535100ddf3d6c827849be8442115b3134153a Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 05:43:57 -0400 Subject: [PATCH 15/69] Dashboard analysis perfmon reads divide by the measured interval (#3561) (#3569) Mirror of #3560 (Lite/Darling) into the deprecated Dashboard's frozen analysis twins: cntr_value_delta spans one collection interval, not one second, so the raw reads overstated by the cadence (60x at 60s, 300x at 5min) against thresholds defined in requests/sec. All three reads move together, or the z-score compares across units: - SqlServerFactCollector PERFMON_*_SEC facts divide the delta by the row's measured sample_interval_seconds; interval <= 0 rows (unknowable delta) are filtered so rn = 1 lands on the newest usable row - a counter with only interval-0 rows emits no fact, never 0. The raw delta and the divisor ride the metadata. - SqlServerAnomalyDetector's batch-request window AVG/MAX divide by NULLIF(sample_interval_seconds, 0) with interval <= 0 rows filtered, keeping the window statistic in the same requests/sec unit as the BatchRequestFloor/Fallback bars. - SqlServerBaselineProvider's batch_requests arm computes the baseline population as the per-second rate; the restart signature stays on the RAW delta (its > 1000 bar predates the division). The window and perfmon SQL move to public consts (the Darling twin's tested shape) and GetBaselineQuery goes internal so Dashboard.Tests can pin the text - the Dashboard has no test database. Semantics verified live against SQL Server via a temp-table clone of collect.perfmon_stats. Fixes #3561 Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx Co-authored-by: Claude Fable 5 --- .../PerfmonPerSecondSqlTests.cs | 102 ++++++++++++++++++ .../Analysis/SqlServerAnomalyDetector.cs | 30 ++++-- .../Analysis/SqlServerBaselineProvider.cs | 14 ++- .../SqlServerFactCollector.Resources.cs | 50 ++++++--- 4 files changed, 165 insertions(+), 31 deletions(-) create mode 100644 deprecated/Dashboard.Tests/PerfmonPerSecondSqlTests.cs diff --git a/deprecated/Dashboard.Tests/PerfmonPerSecondSqlTests.cs b/deprecated/Dashboard.Tests/PerfmonPerSecondSqlTests.cs new file mode 100644 index 000000000..26de1de26 --- /dev/null +++ b/deprecated/Dashboard.Tests/PerfmonPerSecondSqlTests.cs @@ -0,0 +1,102 @@ +/* + * Performance Monitor Dashboard + * Copyright (c) 2026 Darling Data, LLC + * Licensed under the MIT License - see LICENSE file for details + */ + +using PerformanceMonitorDashboard.Analysis; +using Xunit; + +namespace PerformanceMonitorDashboard.Tests; + +/// +/// #3527/#3561: cntr_value_delta spans one COLLECTION INTERVAL, not one second — read raw, the +/// perfmon analysis reads overstate by the cadence (60x at 60s, 300x at 5min). All three reads +/// must divide by the row's measured sample_interval_seconds and skip interval <= 0 rows +/// (no delta was knowable: first sighting, reset, gap) — together or the baseline comparison is +/// cross-unit. The Dashboard has no test database, so these pin the SQL text, the mirror of the +/// Darling twin's #3560 pins (PgFactCollector.PerfmonSql / PgAnomalyDetector.BatchRequestWindowSql). +/// +public class PerfmonPerSecondSqlTests +{ + /* ---------------- read 1: the fact collector ---------------- */ + + [Fact] + public void PerfmonFactSql_SelectsTheMeasuredInterval_AndFiltersUnknowableRows() + { + var sql = SqlServerFactCollector.PerfmonSql; + + /* The interval rides both the CTE and the outer select so the C# division has its divisor. */ + Assert.Contains("sample_interval_seconds,", sql); + Assert.Contains("SELECT counter_name, cntr_value, cntr_value_delta, sample_interval_seconds", sql); + + /* Interval <= 0 rows are filtered INSIDE the CTE, so rn = 1 lands on the newest row a rate + can honestly be derived from — a counter with only interval-0 rows emits no fact, never 0. */ + Assert.Contains("AND sample_interval_seconds > 0", sql); + var filterAt = sql.IndexOf("AND sample_interval_seconds > 0", System.StringComparison.Ordinal); + var rnFilterAt = sql.IndexOf("FROM latest WHERE rn = 1", System.StringComparison.Ordinal); + Assert.True(filterAt >= 0 && filterAt < rnFilterAt, "the interval filter must sit inside the latest CTE, before rn = 1"); + } + + /* ---------------- read 2: the anomaly detector's window ---------------- */ + + [Fact] + public void BatchRequestWindow_DividesByMeasuredInterval_AndSkipsUnknowableRows() + { + var sql = SqlServerAnomalyDetector.BatchRequestWindowSql; + + Assert.Contains("AVG(cntr_value_delta * 1.0 / NULLIF(sample_interval_seconds, 0))", sql); + Assert.Contains("MAX(cntr_value_delta * 1.0 / NULLIF(sample_interval_seconds, 0))", sql); + Assert.Contains("sample_interval_seconds > 0", sql); + + /* A raw AVG/MAX of the delta is exactly the #3527 defect — pin its absence. */ + Assert.DoesNotContain("AVG(cntr_value_delta)", sql); + Assert.DoesNotContain("MAX(cntr_value_delta)", sql); + } + + /* ---------------- read 3: the baseline arm ---------------- */ + + [Fact] + public void BatchRequestBaseline_PopulationIsPerSecond_RestartSignatureStaysRaw() + { + var sql = SqlServerBaselineProvider.GetBaselineQuery(SqlServerMetricNames.BatchRequests); + + Assert.NotNull(sql); + + /* The baseline population is the per-second rate, in the detector's window unit. */ + Assert.Contains("cntr_value_delta * 1.0 / NULLIF(sample_interval_seconds, 0) AS v", sql); + Assert.Contains("AVG(v) AS mean_val", sql); + Assert.Contains("STDEV(v) AS stddev_val", sql); + Assert.Contains("AND sample_interval_seconds > 0", sql); + + /* The restart signature stays on the RAW delta: its > 1000 bar predates the division and + marks a counter reset regardless of cadence. */ + Assert.Contains("LAG(cntr_value_delta) OVER (ORDER BY collection_time) AS prev_value", sql); + Assert.Contains("WHERE NOT (cntr_value_delta = 0 AND ISNULL(prev_value, 0) > 1000)", sql); + + /* A raw AVG/STDEV of the delta is exactly the #3527 defect — pin its absence. */ + Assert.DoesNotContain("AVG(cntr_value_delta)", sql); + Assert.DoesNotContain("STDEV(cntr_value_delta)", sql); + } + + /* ---------------- the consistency constraint ---------------- */ + + /// + /// The whole fix is the three reads agreeing: the window statistic, the baseline population, + /// and the fact value all divide by the measured interval, or the z-score compares across + /// units. The fact collector's division happens in C# (delta / sample_interval_seconds after + /// the SQL delivers the divisor), so its SQL is pinned for the divisor column and the + /// interval filter; the two aggregate reads divide in the SQL itself. + /// + [Fact] + public void AllThreeReads_FilterUnknowableRows() + { + var factSql = SqlServerFactCollector.PerfmonSql; + var windowSql = SqlServerAnomalyDetector.BatchRequestWindowSql; + var baselineSql = SqlServerBaselineProvider.GetBaselineQuery(SqlServerMetricNames.BatchRequests); + + Assert.Contains("sample_interval_seconds > 0", factSql); + Assert.Contains("sample_interval_seconds > 0", windowSql); + Assert.Contains("sample_interval_seconds > 0", baselineSql); + } +} diff --git a/deprecated/Dashboard/Analysis/SqlServerAnomalyDetector.cs b/deprecated/Dashboard/Analysis/SqlServerAnomalyDetector.cs index b0bfc0076..71ea6a89d 100644 --- a/deprecated/Dashboard/Analysis/SqlServerAnomalyDetector.cs +++ b/deprecated/Dashboard/Analysis/SqlServerAnomalyDetector.cs @@ -832,6 +832,24 @@ FROM collect.file_io_stats } } + // Batch-request window: per-second rate per sample (#3527) — cntr_value_delta spans one collection + // interval, so divide by the row's MEASURED sample_interval_seconds. Interval <= 0 marks an + // unknowable delta (first sighting/reset/gap) and the row is skipped, never read as 0. Keeps the + // window statistic in the same requests/sec unit as the baseline and the + // BatchRequestFloor/Fallback thresholds. + public const string BatchRequestWindowSql = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT + AVG(cntr_value_delta * 1.0 / NULLIF(sample_interval_seconds, 0)) AS avg_batch, + MAX(cntr_value_delta * 1.0 / NULLIF(sample_interval_seconds, 0)) AS peak_batch, + COUNT(*) AS sample_count +FROM collect.perfmon_stats +WHERE collection_time >= @windowStart AND collection_time <= @windowEnd +AND counter_name = 'Batch Requests/sec' +AND cntr_value_delta >= 0 +AND sample_interval_seconds > 0;"; + /// /// Detects batch requests/sec anomalies using z-score against time-bucketed baseline. /// @@ -849,17 +867,7 @@ private async Task DetectBatchRequestAnomalies(AnalysisContext context, List= @windowStart AND collection_time <= @windowEnd -AND counter_name = 'Batch Requests/sec' -AND cntr_value_delta >= 0;"; + cmd.CommandText = BatchRequestWindowSql; cmd.Parameters.Add(new SqlParameter("@windowStart", context.TimeRangeStart)); cmd.Parameters.Add(new SqlParameter("@windowEnd", context.TimeRangeEnd)); diff --git a/deprecated/Dashboard/Analysis/SqlServerBaselineProvider.cs b/deprecated/Dashboard/Analysis/SqlServerBaselineProvider.cs index 5ba2c1c20..c24822277 100644 --- a/deprecated/Dashboard/Analysis/SqlServerBaselineProvider.cs +++ b/deprecated/Dashboard/Analysis/SqlServerBaselineProvider.cs @@ -190,7 +190,8 @@ public async Task GetBaselineAsync(string metricName, DateTime a } } - private static string? GetBaselineQuery(string metricName) + // Internal for Dashboard.Tests (InternalsVisibleTo): the #3527 per-second pins read the arm SQL. + internal static string? GetBaselineQuery(string metricName) { // All queries return: hour_of_day, day_of_week, mean_val, stddev_val, sample_count // Day-of-week normalization: (DATEPART(weekday, x) + @@DATEFIRST - 1) % 7 gives Sunday=0 @@ -217,21 +218,28 @@ GROUP BY DATEPART(HOUR, collection_time), // Cumulative counter — restart exclusion via CTE with LAG. // server_start_time is inline in collect.perfmon_stats. // Exclude samples within 5 min of a detected restart. + // #3527: v is the per-second rate — the per-interval delta divided by the row's measured + // sample_interval_seconds — so the baseline population is in the same requests/sec unit as + // the detector's window statistic. Interval <= 0 rows (unknowable delta) are skipped, never + // read as 0. The restart signature stays on the RAW delta: its > 1000 bar predates the + // division and marks a counter reset regardless of cadence. SqlServerMetricNames.BatchRequests => @" SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; ;WITH filtered AS ( SELECT collection_time, cntr_value_delta, + cntr_value_delta * 1.0 / NULLIF(sample_interval_seconds, 0) AS v, LAG(cntr_value_delta) OVER (ORDER BY collection_time) AS prev_value FROM collect.perfmon_stats WHERE collection_time >= @windowStart AND collection_time < @windowEnd AND counter_name = 'Batch Requests/sec' AND cntr_value_delta >= 0 + AND sample_interval_seconds > 0 ) SELECT DATEPART(HOUR, collection_time) AS hour_of_day, (DATEPART(WEEKDAY, collection_time) + @@DATEFIRST - 1) % 7 AS day_of_week, - AVG(cntr_value_delta) AS mean_val, - STDEV(cntr_value_delta) AS stddev_val, + AVG(v) AS mean_val, + STDEV(v) AS stddev_val, COUNT(*) AS sample_count, COUNT(DISTINCT CAST(collection_time AS DATE)) AS distinct_days FROM filtered diff --git a/deprecated/Dashboard/Analysis/SqlServerFactCollector.Resources.cs b/deprecated/Dashboard/Analysis/SqlServerFactCollector.Resources.cs index 2dedac4eb..6a8e08350 100644 --- a/deprecated/Dashboard/Analysis/SqlServerFactCollector.Resources.cs +++ b/deprecated/Dashboard/Analysis/SqlServerFactCollector.Resources.cs @@ -197,19 +197,12 @@ FROM collect.memory_grant_stats } } - /// - /// Collects key perfmon throughput counters: Batch Requests/sec, compilations, recompilations. - /// Unscored context that distinguishes a busy server from a sick one (used by the AI surfaces). - /// - private async Task CollectPerfmonFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" + // #3527: cntr_value_delta spans one COLLECTION INTERVAL, not one second — at a 60s cadence + // the raw delta is 60x the true rate. The honest rate divides by the row's MEASURED + // sample_interval_seconds; interval <= 0 marks an unknowable delta (first sighting, counter + // reset, gap), so those rows are filtered rather than emitted as 0 — rn = 1 lands on the + // newest row a rate can honestly be derived from. + public const string PerfmonSql = @" SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; ;WITH latest AS ( @@ -217,15 +210,33 @@ private async Task CollectPerfmonFactsAsync(AnalysisContext context, List counter_name, cntr_value, cntr_value_delta, + sample_interval_seconds, ROW_NUMBER() OVER (PARTITION BY counter_name ORDER BY collection_time DESC) AS rn FROM collect.perfmon_stats WHERE collection_time >= @startTime AND collection_time <= @endTime AND counter_name IN ('Batch Requests/sec', 'SQL Compilations/sec', 'SQL Re-Compilations/sec') + AND sample_interval_seconds > 0 ) -SELECT counter_name, cntr_value, cntr_value_delta +SELECT counter_name, cntr_value, cntr_value_delta, sample_interval_seconds FROM latest WHERE rn = 1"; + /// + /// Collects key perfmon throughput counters: Batch Requests/sec, compilations, recompilations. + /// Unscored context that distinguishes a busy server from a sick one (used by the AI surfaces). + /// Fact values are per-second rates: the per-interval delta divided by the row's measured + /// sample_interval_seconds (#3527); the raw delta and the divisor ride the metadata. + /// + private async Task CollectPerfmonFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = PerfmonSql; + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); @@ -235,6 +246,7 @@ AND counter_name IN ('Batch Requests/sec', 'SQL Compilations/sec', 'SQL Re-Com var counterName = reader.GetString(0); var cntrValue = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); var deltaValue = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); + var intervalSeconds = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)); var (factKey, source) = counterName switch { @@ -246,8 +258,11 @@ AND counter_name IN ('Batch Requests/sec', 'SQL Compilations/sec', 'SQL Re-Com if (factKey == null) continue; - // All remaining counters are per-second rates — use the delta. - var value = (double)deltaValue; + // The delta spans one collection interval — divide by the measured interval for the + // per-second rate (#3527). The SQL already filters interval <= 0 (unknowable delta); + // this guard keeps a raw or zero value from ever escaping if that filter regresses. + if (intervalSeconds <= 0) continue; + var value = deltaValue / (double)intervalSeconds; facts.Add(new Fact { @@ -258,7 +273,8 @@ AND counter_name IN ('Batch Requests/sec', 'SQL Compilations/sec', 'SQL Re-Com Metadata = new Dictionary { ["cntr_value"] = cntrValue, - ["delta_cntr_value"] = deltaValue + ["delta_cntr_value"] = deltaValue, + ["sample_interval_seconds"] = intervalSeconds } }); } From 53471406001000a8c0a544628d0a51a5d5338a0e Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 05:44:30 -0400 Subject: [PATCH 16/69] Viewer pass for #3528: Settings box for the Store Disk Pressure GB floor, and the measured-metric qualifier on the WPF fleet card (#3571) The V126 knob (self_disk_free_warn_gb) lands in the viewer: the Settings window gets its box one knob over from the percent sibling (prefill, >= 0 save gate matching the MCP write bound and read-side clamp, Restore Defaults at the shipped 50, master-switch follow), and ViewerDataService.AlertSettings.cs appends the column to the select / upsert / bind / reader at ordinal 66 ($67). The SelfDiskWarnGbFloorRungTests abstinence pin flips to pin the wired state, with the bind-order roundtrip and the Settings-window pins; V124's end-anchored bind pin hands its top-of-bind claim on, the V122->V124 precedent. The WPF fleet card also picks up the web fleet page's measured-metric qualifier: ServerSummaryItem carries MeasuredMetricCount / MetricCount through the shared classifier fold, and the card tooltip's band label says "Healthy - 1 of 6 measured" for a band folded over unmeasured metrics instead of claiming every metric is inside its threshold. Fixes #3563 Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx Co-authored-by: Claude Fable 5 --- .../FleetSweepCadenceKnobRungTests.cs | 11 ++- .../SelfDiskWarnGbFloorRungTests.cs | 99 ++++++++++++++++--- .../ViewerOverviewExplainsItselfTests.cs | 72 ++++++++++++++ .../SettingsWindow.xaml | 5 +- .../SettingsWindow.xaml.cs | 10 ++ .../ViewerDataService.AlertSettings.cs | 20 +++- .../ViewerDataService.Fleet.cs | 48 ++++++++- .../ViewerDataService.Overview.cs | 15 +++ .../ViewerDataService.cs | 11 ++- 9 files changed, 264 insertions(+), 27 deletions(-) diff --git a/Darling/Darling.Tests/FleetSweepCadenceKnobRungTests.cs b/Darling/Darling.Tests/FleetSweepCadenceKnobRungTests.cs index 1cfd579fc..66fe91f61 100644 --- a/Darling/Darling.Tests/FleetSweepCadenceKnobRungTests.cs +++ b/Darling/Darling.Tests/FleetSweepCadenceKnobRungTests.cs @@ -344,10 +344,13 @@ would otherwise still present the right value at one of the two positions. */ .Max(); Assert.Equal(command.Parameters.Count, highestPlaceholder); - /* The two new columns ride at the END — appended, the rule every knob rung on this table follows, - so every earlier ordinal keeps its column. */ - Assert.False(Assert.IsType>(command.Parameters[^2]).TypedValue); - Assert.Equal(240, Assert.IsType>(command.Parameters[^1]).TypedValue); + /* The two columns ride at THEIR appended ordinals ($65/$66) — fixed forever by the append rule, + which is what keeps every earlier ordinal on its column. Not `[^1]`/`[^2]` any more: that + end-anchored form asserted these are the NEWEST bound columns, which stopped being true when the + viewer pass appended V126's floor ($67) — the top-of-bind claim moved to + SelfDiskWarnGbFloorRungTests the way the probe's top-arm claims hand off between rung files. */ + Assert.False(Assert.IsType>(command.Parameters[64]).TypedValue); + Assert.Equal(240, Assert.IsType>(command.Parameters[65]).TypedValue); } /// Non-overlapping occurrences of . diff --git a/Darling/Darling.Tests/SelfDiskWarnGbFloorRungTests.cs b/Darling/Darling.Tests/SelfDiskWarnGbFloorRungTests.cs index 094165dcb..2335d0628 100644 --- a/Darling/Darling.Tests/SelfDiskWarnGbFloorRungTests.cs +++ b/Darling/Darling.Tests/SelfDiskWarnGbFloorRungTests.cs @@ -10,6 +10,7 @@ using System.Globalization; using System.Linq; using System.Reflection; +using Npgsql; using PerformanceMonitor.Darling.Service; using PerformanceMonitor.Darling.Service.Mcp; using PerformanceMonitor.Darling.Storage; @@ -168,18 +169,18 @@ file inherited it from FleetSweepCadenceKnobRungTests. */ /* ---- every settings-row surface handles the column ------------------------------------------------ */ /// - /// EVERY wired surface that reads or writes the settings row names the column — and the ONE surface - /// deliberately not wired yet is pinned to its abstinence. The wired lists drive ordinals or parameter - /// positions, so a column added to one and not the others re-maps reads and writes at once. + /// EVERY surface that reads or writes the settings row names the column — the viewer INCLUDED. The + /// wired lists drive ordinals or parameter positions, so a column added to one and not the others + /// re-maps reads and writes at once. /// - /// The viewer's select/upsert is the pinned abstinence, unlike every earlier knob rung: - /// this knob lands backend-first (store plane + the two MCP tools), and the Settings window's box - /// follows in the viewer pass. The viewer's explicit column lists mean its select and save are - /// untouched by the new column — nothing throws, and a viewer Save cannot null the floor out. When the - /// viewer pass wires the box, this assertion is where that decision flips. + /// The viewer's select/upsert was this rung's pinned abstinence, unlike every earlier knob + /// rung: the knob landed backend-first (store plane + the two MCP tools), and this assertion pinned + /// DoesNotContain until the viewer pass (#3563) wired the Settings window's box — the flip that + /// paragraph promised. What the abstinence protected still holds now that it is wired: the viewer's + /// explicit column lists mean a Save writes the floor rather than nulling it out. /// [Fact] - public void EveryWiredSettingsRowSurfaceNamesTheColumn_AndTheViewerAbstains() + public void EverySettingsRowSurfaceNamesTheColumn_TheViewerIncluded() { var service = RepoFile.ReadRepoFile( "Darling", "PerformanceMonitor.Darling.Service", "StoreConfigProvider.cs"); @@ -200,10 +201,84 @@ public void EveryWiredSettingsRowSurfaceNamesTheColumn_AndTheViewerAbstains() "case \"disk_free_warn_gb\": AddInt(\"self_disk_free_warn_gb\", n, \"self_alerts.disk_free_warn_gb\", 0, int.MaxValue); break;", tools, StringComparison.Ordinal); - /* The deliberate abstinence: the viewer's settings surface does not name the column yet. */ + /* The viewer pass (#3563): the viewer's select reads the column, its upsert WRITES it (or Save + silently drops whatever the box held), and its reader maps the appended ordinal. */ var viewerSettings = RepoFile.ReadRepoFile( "Darling", "PerformanceMonitor.Darling.Viewer", "ViewerDataService.AlertSettings.cs"); - Assert.DoesNotContain(FloorColumn, viewerSettings, StringComparison.Ordinal); + Assert.Contains(FloorColumn, ViewerDataService.AlertSettingsSelectSql, StringComparison.Ordinal); + Assert.Contains( + $"{FloorColumn} = EXCLUDED.{FloorColumn}", ViewerDataService.AlertSettingsUpsertSql, StringComparison.Ordinal); + Assert.Contains("SelfDiskFreeWarnGb = reader.GetInt32(66)", viewerSettings, StringComparison.Ordinal); + } + + /// + /// The Settings window's box (#3563): prefilled from the row, saved through the same bound the MCP + /// writer accepts and the read-side clamp keeps (>= 0 — 0 removes the floor), following the + /// alerts master switch like every #2107 sibling, and Restore Defaults writes the shipped constant. + /// + /// The viewer restates the shipped default as a literal — in the row initializer and the + /// Restore Defaults button — because DarlingSelfAlertEvaluator lives on the Service assembly the + /// viewer does not reference (its percent sibling's "10" has the same shape). Both literals are pinned + /// equal to the constant here, the same equality the rung SQL's own default carries, so a moved shipped + /// default cannot leave the window handing out a floor no other surface reports. + /// + [Fact] + public void TheSettingsWindowBox_PrefillsSavesGatesAndRestores_AtTheSharedBoundAndDefault() + { + var window = RepoFile.ReadRepoFile( + "Darling", "PerformanceMonitor.Darling.Viewer", "SettingsWindow.xaml.cs"); + var xaml = RepoFile.ReadRepoFile( + "Darling", "PerformanceMonitor.Darling.Viewer", "SettingsWindow.xaml"); + + /* The box exists, one knob over from its percent sibling. */ + Assert.Contains("x:Name=\"AlertSelfDiskWarnGbBox\"", xaml, StringComparison.Ordinal); + + /* Prefill, save gate (the [0, int.MaxValue) bound's floor), and the master-switch follow. */ + Assert.Contains("AlertSelfDiskWarnGbBox.Text = r.SelfDiskFreeWarnGb.ToString(", window, StringComparison.Ordinal); + Assert.Contains( + "if (int.TryParse(AlertSelfDiskWarnGbBox.Text, out var selfDiskGb) && selfDiskGb >= 0)", + window, StringComparison.Ordinal); + Assert.Contains("row.SelfDiskFreeWarnGb = selfDiskGb;", window, StringComparison.Ordinal); + Assert.Contains("AlertSelfDiskWarnGbBox.IsEnabled = enabled;", window, StringComparison.Ordinal); + + /* Restore Defaults writes the shipped figure, and the viewer row seeds it — both as literals + pinned equal to the constant. */ + var shipped = ((int)DarlingSelfAlertEvaluator.DiskFreeWarnFloorGb).ToString(CultureInfo.InvariantCulture); + Assert.Contains($"AlertSelfDiskWarnGbBox.Text = \"{shipped}\";", window, StringComparison.Ordinal); + Assert.Equal((int)DarlingSelfAlertEvaluator.DiskFreeWarnFloorGb, AlertSettingsRow.Defaults().SelfDiskFreeWarnGb); + Assert.Equal((int)DarlingSelfAlertEvaluator.DiskFreeWarnFloorGb, new AlertsConfig().SelfDiskFreeWarnGb); + } + + /// + /// The viewer row round-trips through the bind at the APPENDED ordinal — the four-parallel-sequences + /// trap (the column list, the upsert's $N placeholders, the bind order, and the reader ordinals), + /// held for the one sequence no SQL text pin can see: the bind. The V124 rung's shape, one knob on. + /// + [Fact] + public void TheViewerRow_RoundTripsThroughTheBind_AtTheAppendedOrdinal() + { + var bind = typeof(ViewerDataService) + .GetMethod("BindAlertSettings", BindingFlags.NonPublic | BindingFlags.Static)!; + + var row = AlertSettingsRow.Defaults(); + /* Deliberately NOT the shipped 50 — a bind that dropped the column and fell back to the default + would otherwise still present the right value at the position. 75 is the MCP round-trip test's + sample, for the same reason. */ + row.SelfDiskFreeWarnGb = 75; + + using var command = new NpgsqlCommand(); + bind.Invoke(null, new object[] { command, row }); + + /* The bind supplies exactly as many parameters as the upsert's highest placeholder. */ + var highestPlaceholder = System.Text.RegularExpressions.Regex + .Matches(ViewerDataService.AlertSettingsUpsertSql, @"\$(\d+)") + .Select(m => int.Parse(m.Groups[1].Value, CultureInfo.InvariantCulture)) + .Max(); + Assert.Equal(command.Parameters.Count, highestPlaceholder); + + /* The new column rides at the END — appended, the rule every knob rung on this table follows, + so every earlier ordinal keeps its column. */ + Assert.Equal(75, Assert.IsType>(command.Parameters[^1]).TypedValue); } /* ---- the seam reaches the gate -------------------------------------------------------------------- */ @@ -212,7 +287,7 @@ public void EveryWiredSettingsRowSurfaceNamesTheColumn_AndTheViewerAbstains() /// The settings adapter defaults to the shipped constant and clamps a hand-edited store value at the /// 0 floor — the raw-in/clamped-out split every knob on this table uses, with 0 IN range because it /// removes the floor (the pvs_floor_gb reading) rather than being nonsense. The write bound in - /// is the same + /// is the same /// [0, int.MaxValue], so no accepted value is one this clamp rewrites. /// [Fact] diff --git a/Darling/Darling.Tests/ViewerOverviewExplainsItselfTests.cs b/Darling/Darling.Tests/ViewerOverviewExplainsItselfTests.cs index 783591671..ca6d1b18c 100644 --- a/Darling/Darling.Tests/ViewerOverviewExplainsItselfTests.cs +++ b/Darling/Darling.Tests/ViewerOverviewExplainsItselfTests.cs @@ -121,6 +121,78 @@ public void TheCardsTooltip_OnAHealthyCard_DoesNotClaimItNeedsAttention() Assert.Contains("Healthy", tooltip, StringComparison.Ordinal); } + // ── The band label carries the measured-metric qualifier (#3528 / #3563) ─────────────────────── + + /// A card whose six severities all carry real readings — the only shape that earns the + /// unqualified all-clear. Deadlocks need a non-zero window to band (a zero window reads Unknown). + private static ServerSummaryItem FullyMeasured(string name = "f1", int id = 7) => + new() + { + DisplayName = name, + ServerId = id, + IsOnline = true, + CpuPercent = 50, + TotalThreads = 512, + CurrentWorkers = 100, + DeadlockWindow = TimeSpan.FromHours(1), + }; + + /// + /// The #3528 card: the band's fold SKIPS Unknown, so an online PostgreSQL target with five of six + /// metrics structurally Unknown still bands Healthy — and this tooltip claimed "every metric on this + /// card is inside its threshold" for it, an affirmative statement about five readings that were never + /// taken. The web fleet card says "1 of 6 measured" (#3562); the WPF card's band label now says the + /// same, wording and gate alike, so the two surfaces read alike. + /// + [Fact] + public void TheCardsTooltip_QualifiesAHealthyBand_ThatFoldedOverUnmeasuredMetrics() + { + /* No CPU/threads snapshot, DMV-sourced memory/blocking/deadlocks nulled by the engine — only the + collector row measured. The same shape DarlingFleetReader's card serializes as 1-of-6. */ + var pg = Healthy(); + pg.IsPostgres = true; + + Assert.Equal(1, pg.MeasuredMetricCount); + Assert.Equal(6, pg.MetricCount); + Assert.StartsWith("Healthy — 1 of 6 measured", pg.StatusTooltip, StringComparison.Ordinal); + Assert.DoesNotContain("every metric on this card", pg.StatusTooltip, StringComparison.Ordinal); + + /* The counts are the shared classifier's own fold over the card's metrics — the service's + measured_metric_count / metric_count pair, not a viewer re-derivation. */ + Assert.Equal( + ServerHealthClassifier.MeasuredMetricCounts(pg.ToHealthMetrics()), + (pg.MeasuredMetricCount, pg.MetricCount)); + } + + /// The all-clear's "every metric" claim survives — but only where it is true. A fully-measured + /// healthy card is unchanged by #3563, which is what keeps the qualifier a qualifier rather than a new + /// line every green card carries. + [Fact] + public void TheCardsTooltip_KeepsTheUnqualifiedAllClear_WhenEveryMetricIsMeasured() + { + var card = FullyMeasured(); + + Assert.Equal(6, card.MeasuredMetricCount); + Assert.Equal(6, card.MetricCount); + Assert.StartsWith( + "Healthy — every metric on this card is inside its threshold", card.StatusTooltip, StringComparison.Ordinal); + Assert.DoesNotContain("measured", card.StatusTooltip, StringComparison.Ordinal); + } + + /// A partially-measured PROBLEM card keeps its reason and gains the qualifier beside it — the + /// web card appends the coverage to every card it is short on, not just the green ones, and the reason + /// must stay the ranking's sentence verbatim (the drift-prevention this file pins). + [Fact] + public void TheCardsTooltip_CarriesTheQualifierBesideTheReason_OnAPartiallyMeasuredProblemCard() + { + /* Busy(): CPU, memory, blocking and collectors measured; threads and deadlocks Unknown. */ + var card = Busy(); + + Assert.Equal(4, card.MeasuredMetricCount); + Assert.StartsWith("Critical — CPU 96%, Blocking 6 · 4 of 6 measured", card.StatusTooltip, StringComparison.Ordinal); + Assert.Contains(FleetRollup.BuildReason(card), card.StatusTooltip, StringComparison.Ordinal); + } + /// Offline and awaiting-first-collection already come back as whole sentences naming themselves, so /// the band label is not stamped in front of them a second time. [Fact] diff --git a/Darling/PerformanceMonitor.Darling.Viewer/SettingsWindow.xaml b/Darling/PerformanceMonitor.Darling.Viewer/SettingsWindow.xaml index 20adee4c9..f52cf3ef1 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/SettingsWindow.xaml +++ b/Darling/PerformanceMonitor.Darling.Viewer/SettingsWindow.xaml @@ -425,7 +425,10 @@ - + + = 0 and <= 100) row.SelfDiskFreeWarnPercent = selfDiskPct; + /* #3528: validated to the same bound DarlingAlertSettings clamps (Math.Max(0, ...)) and the MCP + writer accepts ([0, int.MaxValue]) — 0 is IN range because it removes the floor. */ + if (int.TryParse(AlertSelfDiskWarnGbBox.Text, out var selfDiskGb) && selfDiskGb >= 0) + row.SelfDiskFreeWarnGb = selfDiskGb; if (int.TryParse(AlertCollectionStaleMinutesBox.Text, out var staleMin) && staleMin is >= 5 and <= 1440) row.CollectionStaleMinutes = staleMin; if (int.TryParse(AlertCollectionFailureThresholdBox.Text, out var failThresh) && failThresh is >= 1 and <= 1000) @@ -1013,6 +1018,10 @@ leave this button writing a threshold nobody chose. */ AlertDiskCriticalPercentBox.Text = "3"; AlertDiskCriticalGbBox.Text = "2"; AlertSelfDiskWarnPercentBox.Text = "10"; + /* #3528: the shipped GB floor (DarlingSelfAlertEvaluator.DiskFreeWarnFloorGb) as a literal — the + constant lives on the Service assembly the viewer does not reference, and the V126 rung test + pins this literal equal to it. */ + AlertSelfDiskWarnGbBox.Text = "50"; AlertCollectionStaleMinutesBox.Text = "30"; AlertCollectionFailureThresholdBox.Text = "10"; /* #3060: derived, unlike its neighbours, because this one is not merely a mirrored default — it is @@ -1158,6 +1167,7 @@ private void UpdateAlertControlStates() AlertDiskCriticalPercentBox.IsEnabled = enabled; AlertDiskCriticalGbBox.IsEnabled = enabled; AlertSelfDiskWarnPercentBox.IsEnabled = enabled; + AlertSelfDiskWarnGbBox.IsEnabled = enabled; AlertCollectionStaleMinutesBox.IsEnabled = enabled; AlertCollectionFailureThresholdBox.IsEnabled = enabled; AlertStoreJobCadenceWarnPercentBox.IsEnabled = enabled; diff --git a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.AlertSettings.cs b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.AlertSettings.cs index 7e1521b6a..6102bfde2 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.AlertSettings.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.AlertSettings.cs @@ -74,7 +74,9 @@ notify toggle (V20) are appended so the existing ordinals stay pinned. */ /* #3444: V122's PostgreSQL Deadlocks/Blocking count thresholds, APPENDED for the same reason. */ "pg_deadlock_count_threshold, pg_blocking_count_threshold, " + /* #3466: V124's fleet-sweep cadence knobs, APPENDED for the same reason. */ - "fleet_sweep_enabled, fleet_sweep_interval_minutes"; + "fleet_sweep_enabled, fleet_sweep_interval_minutes, " + + /* #3528: V126's store-disk-warn GB floor, APPENDED for the same reason. */ + "self_disk_free_warn_gb"; /// The single global alert-settings row (id=1), for the Settings window prefill + the migrate-in /// defaults check. Column order matches . @@ -91,7 +93,7 @@ INSERT INTO config_alert_settings (id, " + AlertSettingsColumns + @", modified_a VALUES (1, $1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11, $12, $13, $14, $15, $16, $17, $18, $19, $20, $21, $22, $23, $24, $25, $26, $27, $28, $29, $30, $31, $32, $33, $34, $35, $36, $37, $38, $39, $40, $41, $42, $43, $44, $45, $46, $47, $48, $49, $50, $51, $52, $53, $54, $55, $56, $57, $58, $59, $60, $61, $62, - $63, $64, $65, $66, + $63, $64, $65, $66, $67, (now() AT TIME ZONE 'UTC')) ON CONFLICT (id) DO UPDATE SET enabled = EXCLUDED.enabled, @@ -160,6 +162,7 @@ ON CONFLICT (id) DO UPDATE SET pg_blocking_count_threshold = EXCLUDED.pg_blocking_count_threshold, fleet_sweep_enabled = EXCLUDED.fleet_sweep_enabled, fleet_sweep_interval_minutes = EXCLUDED.fleet_sweep_interval_minutes, + self_disk_free_warn_gb = EXCLUDED.self_disk_free_warn_gb, modified_at = (now() AT TIME ZONE 'UTC')"; /// The two cpu_mode values the service honors (it compares case-insensitively against @@ -257,6 +260,7 @@ private static void BindAlertSettings(NpgsqlCommand command, AlertSettingsRow r) command.Parameters.Add(new NpgsqlParameter { TypedValue = r.PgBlockingCountThreshold }); // $64 (#3444, V122) command.Parameters.Add(new NpgsqlParameter { TypedValue = r.FleetSweepEnabled }); // $65 (#3466, V124) command.Parameters.Add(new NpgsqlParameter { TypedValue = r.FleetSweepIntervalMinutes }); // $66 (#3466, V124) + command.Parameters.Add(new NpgsqlParameter { TypedValue = r.SelfDiskFreeWarnGb }); // $67 (#3528, V126) } private static AlertSettingsRow ReadAlertSettingsRow(NpgsqlDataReader reader) => new() @@ -338,6 +342,8 @@ private static void BindAlertSettings(NpgsqlCommand command, AlertSettingsRow r) /* #3466 fleet-sweep cadence knobs appended (V124) at ordinals 64-65. */ FleetSweepEnabled = reader.GetBoolean(64), FleetSweepIntervalMinutes = reader.GetInt32(65), + /* #3528 store-disk-warn GB floor appended (V126) at ordinal 66. */ + SelfDiskFreeWarnGb = reader.GetInt32(66), }; /// Maps the Settings window's CPU-mode combo tag ("Total"/"SqlOnly") to the store value. @@ -391,6 +397,15 @@ public sealed class AlertSettingsRow /* #2107 (V55): the previously-hardcoded thresholds; defaults are the constants they replaced. */ public int SelfDiskFreeWarnPercent { get; set; } = 10; + + /// #3528 (V126): the Store Disk Pressure warning's GB floor — the percent above additionally + /// requires free space below this many GB before the alert fires (an AND qualifier, the PVS floor's + /// composition); 0 removes the floor. The default mirrors the V126 DDL default and the shipped constant + /// (DarlingSelfAlertEvaluator.DiskFreeWarnFloorGb) as a literal, like its percent sibling above: + /// the constant lives on the SERVICE assembly the viewer does not reference, and + /// SelfDiskWarnGbFloorRungTests pins the two equal so a moved shipped default cannot leave this + /// row seeding a floor no surface reports. + public int SelfDiskFreeWarnGb { get; set; } = 50; public int CollectionStaleMinutes { get; set; } = 30; public int CollectionFailureThreshold { get; set; } = 10; public int DiskCriticalFreePercent { get; set; } = 3; @@ -527,6 +542,7 @@ public bool ValueEquals(AlertSettingsRow other) && AgDisconnectRefireMinutes == other.AgDisconnectRefireMinutes && DatabaseStateEnabled == other.DatabaseStateEnabled && SelfDiskFreeWarnPercent == other.SelfDiskFreeWarnPercent + && SelfDiskFreeWarnGb == other.SelfDiskFreeWarnGb && CollectionStaleMinutes == other.CollectionStaleMinutes && CollectionFailureThreshold == other.CollectionFailureThreshold && DiskCriticalFreePercent == other.DiskCriticalFreePercent diff --git a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Fleet.cs b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Fleet.cs index ec9114b9f..640226d74 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Fleet.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Fleet.cs @@ -746,12 +746,54 @@ the CPU number that WAS collected. */ /* Online and stale: the band is the headline. A healthy card gets an all-clear rather than BuildReason's "Needs attention" fallback, which is written for a ranking that only ever holds problem servers and on a grid showing EVERY server would say the opposite of the truth. */ - _ => band == FleetHealthBand.Healthy - ? "Healthy — every metric on this card is inside its threshold" - : WithReason(ServerHealthClassifier.BandLabel(band), " — ", s), + _ => BandHeadline(band, s), }; } + /// + /// The band label, qualified by measured-metric coverage when the band folded over unmeasured metrics + /// (#3528). The fold behind the band SKIPS Unknown, so an online server with five of six metrics + /// structurally Unknown still bands Healthy — and this headline claimed "every metric on this card is + /// inside its threshold" for it, an affirmative statement about five readings that were never taken. + /// The qualifier is the web fleet card's, wording and gate alike ("1 of 6 measured", only when measured + /// < total), so the two surfaces read alike; a fully-measured card is unchanged. + /// + private static string BandHeadline(FleetHealthBand band, ServerSummaryItem s) + { + var coverage = MeasuredCoveragePhrase(s); + + if (band == FleetHealthBand.Healthy) + { + /* The all-clear's "every metric" claim is earned only at full coverage — at partial coverage + the coverage IS the headline's second half, because the claim it replaces is false. */ + return coverage.Length == 0 + ? "Healthy — every metric on this card is inside its threshold" + : "Healthy — " + coverage; + } + + /* The guarded reason-append stays WithReason's (one copy — its own doc says why); the qualifier + rides after whatever it produced: beside a named reason as a second " · " phrase (the web status + line's list separator), else straight after the bare label. */ + var label = ServerHealthClassifier.BandLabel(band); + var headline = WithReason(label, " — ", s); + if (coverage.Length == 0) + { + return headline; + } + + return headline.Equals(label, StringComparison.Ordinal) + ? label + " — " + coverage + : headline + " · " + coverage; + } + + /// "N of M measured" when the card banded over unmeasured metrics, else "" — the web fleet + /// card's wording and gate (metric_count > 0 && measured_metric_count < + /// metric_count), over the card's own counts. + private static string MeasuredCoveragePhrase(ServerSummaryItem s) => + s.MetricCount > 0 && s.MeasuredMetricCount < s.MetricCount + ? $"{s.MeasuredMetricCount} of {s.MetricCount} measured" + : ""; + /// /// A headline plus what the card can actually name — or the headline alone when it can name nothing. /// diff --git a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Overview.cs b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Overview.cs index fb21a8f21..f3ebcad61 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Overview.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Overview.cs @@ -988,6 +988,21 @@ raw value would show 0.0/hr on a card whose severity says Unknown. */ /// The card's worst metric band (offline handled separately by the border / overlay). public HealthSeverity OverallMetricSeverity => ServerHealthClassifier.OverallMetricSeverity(ToHealthMetrics()); + /// + /// How many of the card's per-metric severities carried a real reading when it banded (#3528), through + /// the SAME shared fold the service's fleet card publishes as measured_metric_count — the fold + /// behind SKIPS Unknown, so a card can read Healthy off one + /// measured metric of six, and this count is what lets the band label say so ("Healthy — 1 of 6 + /// measured") instead of rendering an unqualified green. Purely descriptive: it feeds neither the band + /// nor the worst-first score. + /// + public int MeasuredMetricCount => ServerHealthClassifier.MeasuredMetricCounts(ToHealthMetrics()).Measured; + + /// The denominator for — from the shared classifier rather + /// than a hardcoded six, so a new metric row moves both counts at once (the service's + /// metric_count). + public int MetricCount => ServerHealthClassifier.MeasuredMetricCounts(ToHealthMetrics()).Total; + /// The card's raw per-metric inputs, for the shared classifier (banding + fleet score). public ServerHealthMetrics ToHealthMetrics() => new() { diff --git a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.cs b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.cs index de225b5e4..cc6b2f30a 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.cs @@ -1014,11 +1014,12 @@ qualifier that stops a large store volume at a low percent paging CRITICAL. COLU StorageVersion.SchemaVersion rather than falling through to 125 and showing a spurious upgrade banner on a store that is current. - The reason to gate is that standing invariant rather than a viewer read that would throw: - nothing in the viewer reads this column yet — the knob is backend-first (store plane + the - two MCP tools), and the Settings window's box follows in the viewer pass. The column is - named only in the probe line, not this prose, per the V71 finding: the coverage ratchet - strips information_schema lines but cannot strip a comment. */ + The gate earns its place beyond that standing invariant since the viewer pass (#3563): the + alert-settings select now names the column, so a viewer pointed below this rung would throw + a raw 42703 when the Settings window prefills — the knob landed backend-first, and the + Settings window's box is wired now. The column is named only in the probe line, not this + prose, per the V71 finding: the coverage ratchet strips information_schema lines but cannot + strip a comment. */ if (hasSelfDiskWarnGbFloor) { return 126; From cebb3d80e8721e900b4e48501f28e562feddde2e Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 06:03:01 -0400 Subject: [PATCH 17/69] The store TLS root certificate gets a unique per-generation name, so cached roots from prior rotations can never break Windows chain building (#3557) (#3572) Windows caches every root a TLS client processes into that user's intermediate-CA store, one entry per store rotation, forever. With every rotation minting a root under the same subject, the cached pile grows until CryptoAPI's subject-matched issuer walk fails outright ("unknown chain building error") - measured at ~50 cached roots on a dev box, killing SslStreamCertificateContext.Create server-side in the #2117 handshake test and the default-trust build a viewer's SslStream runs during connect. A random mint tag in the root CN makes every rotation a genuinely distinct CA, so cached copies of other rotations are never issuer candidates and the pile never forms, on any machine. SKI/AKI does NOT prevent the engine error (tested, matrix-controlled); subject uniqueness does (6/6 fresh-process runs green against a ~55-cert pile left in place). The test harness now returns the server-side exception it used to swallow - "handshake completed = false" alone cannot distinguish an Npgsql rejection from the server never reaching TLS at all, which is exactly the misread that kept this failure looking like a certificate-validation verdict. Fixes #3557 Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx Co-authored-by: Claude Fable 5 --- .../NpgsqlRootCertificateValidationTests.cs | 25 ++++++++++++------- .../StoreTlsCertificates.cs | 13 +++++++++- 2 files changed, 28 insertions(+), 10 deletions(-) diff --git a/Darling/Darling.Tests/NpgsqlRootCertificateValidationTests.cs b/Darling/Darling.Tests/NpgsqlRootCertificateValidationTests.cs index bbf7c745d..dbd7dafd9 100644 --- a/Darling/Darling.Tests/NpgsqlRootCertificateValidationTests.cs +++ b/Darling/Darling.Tests/NpgsqlRootCertificateValidationTests.cs @@ -39,11 +39,12 @@ public async Task ChainShape_VerifyFullWithPrintedRoot_CompletesTheHandshake_OnE { var generated = StoreTlsCertificates.Create("localhost", IPAddress.Loopback, validityYears: 2); - var completed = await HandshakeCompletesAsync(generated.ServerCertChainPem, generated.ServerKeyPem, generated.RootCertPem); + var (completed, serverFailure) = await HandshakeCompletesAsync(generated.ServerCertChainPem, generated.ServerKeyPem, generated.RootCertPem); Assert.True(completed, "VerifyFull with the printed root must survive Npgsql's certificate validation on this platform — " + - "this is the exact remote-viewer path #2117 exists to fix."); + "this is the exact remote-viewer path #2117 exists to fix." + + (serverFailure is null ? string.Empty : $" Server side of the harness failed: {serverFailure}")); } [Fact] @@ -69,17 +70,22 @@ public async Task LegacySelfSignedShape_VerifyFullWithItselfAsRoot_TheFieldConfi using var legacy = request.CreateSelfSigned(DateTimeOffset.UtcNow.AddDays(-1), DateTimeOffset.UtcNow.AddYears(2)); var pem = legacy.ExportCertificatePem(); - var completed = await HandshakeCompletesAsync(pem, rsa.ExportPkcs8PrivateKeyPem(), pem); + var (completed, serverFailure) = await HandshakeCompletesAsync(pem, rsa.ExportPkcs8PrivateKeyPem(), pem); /* Recorded, not required: the CHAIN shape's test above is the guarantee. The dynamic skip puts the platform fact in every CI log without inventing a requirement that the legacy shape fail — the first cut asserted that and Windows CI refuted it. */ - Assert.Skip($"legacy self-signed shape at VerifyFull: handshake completed = {completed} on {Environment.OSVersion.Platform}"); + Assert.Skip($"legacy self-signed shape at VerifyFull: handshake completed = {completed} on {Environment.OSVersion.Platform}" + + (serverFailure is null ? string.Empty : $"; server-side failure: {serverFailure.GetType().Name}: {serverFailure.Message}")); } - /// Runs the fake server + a VerifyFull Npgsql connect; true when the server-side TLS - /// handshake completed (the client accepted the certificate). - private static async Task HandshakeCompletesAsync(string serverCertChainPem, string serverKeyPem, string rootPem) + /// Runs the fake server + a VerifyFull Npgsql connect. Completed is true when the + /// server-side TLS handshake completed (the client accepted the certificate); ServerFailure + /// is whatever the server task threw, because "completed = false" alone cannot distinguish an + /// Npgsql rejection from the server never reaching TLS at all — #3557 was exactly that, a + /// chain-build failure swallowed here and misread as a + /// certificate-validation verdict for days. + private static async Task<(bool Completed, Exception? ServerFailure)> HandshakeCompletesAsync(string serverCertChainPem, string serverKeyPem, string rootPem) { var rootPath = Path.Combine(Path.GetTempPath(), $"darling-test-root-{Guid.NewGuid():N}.crt"); await File.WriteAllTextAsync(rootPath, rootPem); @@ -168,9 +174,10 @@ depends on a complete read. */ is handshakeCompleted, not the exception. */ } - try { await serverTask; } catch { /* aborted handshakes land here; the flag says enough */ } + Exception? serverFailure = null; + try { await serverTask; } catch (Exception ex) { serverFailure = ex; /* client-rejection aborts AND pre-TLS failures both land here — return it so the caller can tell them apart */ } try { File.Delete(rootPath); } catch { /* temp file, best-effort */ } - return handshakeCompleted; + return (handshakeCompleted, serverFailure); } } diff --git a/Darling/PerformanceMonitor.Darling.Service/StoreTlsCertificates.cs b/Darling/PerformanceMonitor.Darling.Service/StoreTlsCertificates.cs index 6bab7fc1a..b82a9ea1e 100644 --- a/Darling/PerformanceMonitor.Darling.Service/StoreTlsCertificates.cs +++ b/Darling/PerformanceMonitor.Darling.Service/StoreTlsCertificates.cs @@ -50,8 +50,19 @@ internal static Generated Create(string hostName, IPAddress listenIp, int validi var notAfter = notBefore.AddYears(validityYears); using var caKey = RSA.Create(2048); + /* The random mint tag makes every generated root's SUBJECT unique, and that uniqueness is + load-bearing (#3557): Windows silently caches each root a TLS client ever saw into that + user's intermediate-CA store, one entry per rotation, forever. When rotations all share + one subject, that cached pile grows until the chain engine's subject-matched issuer walk + fails outright — "unknown chain building error" from X509Chain.Build, measured at ~50 + cached same-subject roots, killing SslStreamCertificateContext.Create and the + default-trust build a viewer's SslStream runs. (SKI/AKI does NOT prevent it; tested.) + Distinct subjects mean cached copies of other rotations are never candidate issuers for + this chain, so the pile never forms. Distinct names are also honest X.509: each rotation + IS a different CA, and two CAs sharing a DN with different keys is the pathology. */ + var mintTag = Convert.ToHexStringLower(RandomNumberGenerator.GetBytes(4)); var caRequest = new CertificateRequest( - $"CN=PerformanceMonitor Darling store root ({hostName})", caKey, HashAlgorithmName.SHA256, RSASignaturePadding.Pkcs1); + $"CN=PerformanceMonitor Darling store root ({hostName} {mintTag})", caKey, HashAlgorithmName.SHA256, RSASignaturePadding.Pkcs1); /* pathLenConstraint 0: this root may sign end-entity certs only — even with the key discarded, the constraint documents the intent in the certificate itself. */ caRequest.CertificateExtensions.Add(new X509BasicConstraintsExtension(true, true, 0, true)); From bffca6b80e9ec9e262b939f036ba30a55e340a8e Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 09:43:34 -0400 Subject: [PATCH 18/69] CHANGELOG: the #3557 unique-root-name entry (post-splice follow-up) (#3578) Claude-Session: https://claude.ai/code/session_01QMYFB4qv5tqGwMmdPsa4Zx Co-authored-by: Claude Fable 5 --- CHANGELOG.md | 2 ++ 1 file changed, 2 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 358a3f39b..707d8cb9a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -23,6 +23,7 @@ Releases before 3.0.0 are not archived: those entries carry no prose to move. ### Fixed +- **Each generated store TLS root certificate carries a unique per-generation name** ([#3557]) - Windows caches every root a viewer's TLS stack ever sees into the user's intermediate-CA store, one per certificate rotation; with all rotations sharing one name, that cache eventually breaks certificate chain building outright on long-running viewer machines. Distinct names per rotation mean the pile can never form. - **The by-CPU tools now actually rank by CPU** ([#3523]) - get_top_queries_by_cpu and get_top_procedures_by_cpu ordered by summed elapsed time in both SKUs, so on a wait-bound server the real CPU consumers could be missing from the page entirely - and attributed_cpu_ratio read as "hidden CPU" when it actually meant "wrong sort key". Every ranking site now orders by worker time, including the over-fetch cut that could drop a CPU-heavy query before the final sort ever saw it. The viewer's Duration grids keep their elapsed ranking, which is what they promise. - **analyze_server no longer answers "all metrics are within normal ranges" when the analysis window collected nothing** ([#3524]) - The analysis gate passes on lifetime history, so a server whose collection died still reached the all-clear path with an empty window. Both SKUs' analysis services now flag the zero-facts window and analyze_server returns the "unavailable" envelope pointing at get_collection_health; the genuine all-clear (facts collected, zero findings) is unchanged. - **The Performance Calendar, daily summary, and fleet sweep band deadlocks as a measured per-hour rate, not any-deadlock-is-Critical** ([#3525]) - The shared daily classifier routed the Deadlocks signal through the Overview card's store-backed rate tiers (#3368, V120) with each surface's real window as the denominator, so one deadlock no longer paints a calendar day red, sweep verdicts stop scaling with the cadence knob, sub-hour spans fall to Warning instead of a multiplied rate, and the day tooltip/reasons report the rate beside the count. The still-forming day clamps its window to the elapsed portion, so an active storm bands on its true in-progress rate instead of diluting against hours that have not happened yet. @@ -1272,6 +1273,7 @@ Full entries: [docs/changelog/3.0.md](docs/changelog/3.0.md) [#3556]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3556 [#3561]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3561 [#3563]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3563 +[#3557]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3557 [#3514]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3514 [#3477]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3477 [#3495]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/3495 From a6b269707ddc470c38b33a396bbf1688b6b317be Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 11:17:01 -0400 Subject: [PATCH 19/69] The Locking & Contention grid names the table as schema.table, so the same table name in two schemas no longer reads as one object (#3576) (#3583) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * The Locking & Contention grid names the table as schema.table, so the same table name in two schemas no longer reads as one object (#3576) An Azure SQL DB user reported the FinOps Locking & Contention grid showing the bare table name. On Azure SQL DB — and anywhere per-tenant or per-environment schemas are the pattern — the same table name routinely lives in several schemas, so dbo.Orders and archive.Orders rendered as one indistinguishable "Orders". The store always carried schema_name (the index drill's Identity panel shows it); only the grid dropped it. The row model gains a computed FullName (schema.table, bare name when the schema is empty — the ProcedureStatsRow.FullName idiom) and the Table column binds it, so the qualified name sorts, filters, and exports as one string. The header's filter Tag moves with the binding: the popup filter resolves Tag to a property by reflection, and a Tag naming a property the row lacks filters nothing, silently. Ported to the Darling viewer's copy of the grid and to the deprecated Dashboard's frozen twin (smallest faithful diff). Pinned in Lite.Tests (property, filter mechanism through the real matcher, and both SKUs' grid XAML from source) and Darling.Tests (the viewer's own row model). * The Postgres Index Usage grid names the table as schema.table too (#3576, sibling surface) Same defect class one tab over, Darling-only: the viewer's Postgres Index Usage grid showed the bare table name while the five other table-naming grids on that tab show Schema as a column. The store's reader carried SchemaName from the start; it was dropped in two places — the display row had no SchemaName member for the mapper to fill, and the grid bound TableName. Both halves fixed: the display IndexUsageRow gains SchemaName (mapper copies it) and a computed FullName (schema.table, bare on empty schema), and the Table column binds FullName at width 200. No filter buttons on this grid, so no Tag to move. Pinned through the real PgDisplay.IndexUsage mapper — a pin on the property alone would pass with the mapper still dropping the schema, which is the exact shape that shipped — plus the grid binding from source. Lite has no Postgres tab; the deprecated Dashboard has no such grid. --- .../ViewerIndexLockingRowFullNameTests.cs | 53 +++++++ ...iewerPgIndexUsageGridQualifiedNameTests.cs | 118 ++++++++++++++ .../FinOpsTab.xaml | 4 +- .../ViewerDataService.FinOps.cs | 12 ++ .../ViewerPostgresDisplay.cs | 14 ++ .../ViewerServerTab.xaml | 4 +- .../IndexLockingGridQualifiedNameTests.cs | 149 ++++++++++++++++++ Lite/Controls/FinOpsTab.xaml | 8 +- .../LocalDataService.FinOps.IndexObjects.cs | 11 ++ .../Dashboard/Controls/FinOpsContent.xaml | 6 +- .../DatabaseService.FinOps.IndexObjects.cs | 9 ++ 11 files changed, 382 insertions(+), 6 deletions(-) create mode 100644 Darling/Darling.Tests/ViewerIndexLockingRowFullNameTests.cs create mode 100644 Darling/Darling.Tests/ViewerPgIndexUsageGridQualifiedNameTests.cs create mode 100644 Lite.Tests/IndexLockingGridQualifiedNameTests.cs diff --git a/Darling/Darling.Tests/ViewerIndexLockingRowFullNameTests.cs b/Darling/Darling.Tests/ViewerIndexLockingRowFullNameTests.cs new file mode 100644 index 000000000..a6c193add --- /dev/null +++ b/Darling/Darling.Tests/ViewerIndexLockingRowFullNameTests.cs @@ -0,0 +1,53 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using PerformanceMonitor.Common; +using PerformanceMonitor.Darling.Viewer; +using Xunit; + +namespace Darling.Tests; + +/// +/// #3576, the viewer's half. The FinOps Locking & Contention grid binds +/// (schema.table) instead of the bare table name, so two tables sharing a name across schemas no longer +/// read as one. The viewer's row model is a hand-kept copy of Lite's, so the property is pinned here against +/// the copy the viewer actually compiles; the grid XAML of both SKUs is pinned from Lite.Tests +/// (IndexLockingGridQualifiedNameTests), which reads both files from source. +/// +public sealed class ViewerIndexLockingRowFullNameTests +{ + [Fact] + public void FullName_IsSchemaDotTable_WhenSchemaIsPresent() + { + var row = new IndexLockingRow { SchemaName = "archive", TableName = "Orders" }; + + Assert.Equal("archive.Orders", row.FullName); + } + + [Fact] + public void FullName_FallsBackToBareTable_WhenSchemaIsEmpty() + { + /* The reader writes "" (never null) for a NULL schema_name, so "" is the real fallback input. */ + var row = new IndexLockingRow { SchemaName = "", TableName = "Orders" }; + + Assert.Equal("Orders", row.FullName); + } + + [Fact] + public void ColumnFilter_OnFullName_ActuallyFilters() + { + /* The popup filter resolves its button's Tag to a row property by reflection; a Tag naming a property + the row lacks makes MatchesFilter return true for every row — a filter that filters nothing. */ + Assert.NotNull(typeof(IndexLockingRow).GetProperty("FullName")); + + var filter = new ColumnFilterState { ColumnName = "FullName", Operator = FilterOperator.Contains, Value = "archive." }; + + Assert.True(ColumnFilterMatcher.MatchesFilter(new IndexLockingRow { SchemaName = "archive", TableName = "Orders" }, filter)); + Assert.False(ColumnFilterMatcher.MatchesFilter(new IndexLockingRow { SchemaName = "dbo", TableName = "Orders" }, filter)); + } +} diff --git a/Darling/Darling.Tests/ViewerPgIndexUsageGridQualifiedNameTests.cs b/Darling/Darling.Tests/ViewerPgIndexUsageGridQualifiedNameTests.cs new file mode 100644 index 000000000..3c81e5383 --- /dev/null +++ b/Darling/Darling.Tests/ViewerPgIndexUsageGridQualifiedNameTests.cs @@ -0,0 +1,118 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.IO; +using PerformanceMonitor.Darling.Storage; +using PerformanceMonitor.Darling.Viewer; +using Xunit; +using static Darling.Tests.RepoFile; + +namespace Darling.Tests; + +/// +/// #3576, the Postgres sibling. The viewer's Index Usage grid showed the bare table name while the five other +/// table-naming grids on the Postgres tab show Schema as a column, so public.events and +/// archive.events read as one table. The store's reader carried SchemaName from the start; the +/// drop happened in TWO places — the display row had no SchemaName member for the mapper to fill, and +/// the grid bound TableName — so this pins the schema through the REAL mapper () +/// rather than through a hand-built display row, and then pins the grid's binding from source. A pin on the +/// property alone would pass with the mapper still dropping the schema, which is exactly the shape that was +/// shipped. Darling-only: Lite has no Postgres tab and the deprecated Dashboard has no such grid. +/// +public sealed class ViewerPgIndexUsageGridQualifiedNameTests +{ + /* ---------------- through the mapper ---------------- */ + + [Fact] + public void Mapper_CarriesTheSchema_AndFullNameIsSchemaDotTable() + { + var projected = PgDisplay.IndexUsage(Row(schema: "archive", table: "events")); + + Assert.Equal("archive", projected.SchemaName); + Assert.Equal("events", projected.TableName); + Assert.Equal("archive.events", projected.FullName); + } + + [Fact] + public void Mapper_NullSchema_FallsBackToBareTable() + { + /* The reader's SchemaName is nullable; the mapper writes "" for null, and "" is the fallback input. */ + var projected = PgDisplay.IndexUsage(Row(schema: null, table: "events")); + + Assert.Equal("", projected.SchemaName); + Assert.Equal("events", projected.FullName); + } + + [Fact] + public void FullName_DistinguishesTheSameTableNameAcrossSchemas() + { + /* The reported defect in one assertion: these two rows used to render identically. */ + var publicRow = PgDisplay.IndexUsage(Row(schema: "public", table: "events")); + var archiveRow = PgDisplay.IndexUsage(Row(schema: "archive", table: "events")); + + Assert.Equal(publicRow.TableName, archiveRow.TableName); + Assert.NotEqual(publicRow.FullName, archiveRow.FullName); + } + + /* ---------------- the grid, from source ---------------- */ + + [Fact] + public void IndexUsageGrid_TableColumn_BindsFullName() + { + var xaml = ReadRepoFile(Path.Combine("Darling", "PerformanceMonitor.Darling.Viewer", "ViewerServerTab.xaml")); + + var start = xaml.IndexOf("x:Name=\"PgIndexUsageGrid\"", StringComparison.Ordinal); + Assert.True(start >= 0, "ViewerServerTab.xaml: no element named PgIndexUsageGrid — the grid was renamed or removed."); + var end = xaml.IndexOf("", start, StringComparison.Ordinal); + Assert.True(end > start, "ViewerServerTab.xaml: PgIndexUsageGrid has no closing ."); + var grid = xaml.Substring(start, end - start); + + Assert.Contains("Header=\"Table\" Binding=\"{Binding FullName}\"", grid, StringComparison.Ordinal); + + /* The regression's fingerprint. */ + Assert.DoesNotContain("Binding=\"{Binding TableName}\"", grid, StringComparison.Ordinal); + } + + /// + /// The reader row with only the two facts under test varied. Every other argument is inert for the + /// mapper's schema/table handling; the droppability inputs are the shape DarlingPgIndexUsageReaderTests + /// uses for a plain, valid, non-constraint index. + /// + private static DarlingPgIndexUsageReader.PgIndexUsageRow Row(string? schema, string table) => + new( + DatabaseName: "appdb", + SchemaName: schema, + TableName: table, + IndexName: "events_idx", + MeasuredAt: DateTime.UtcNow, + TotalScans: 0, + ScansInWindow: 0, + TuplesRead: 0, + TuplesFetched: 0, + BlocksRead: 0, + BlocksHit: 0, + IndexBytes: 1_000_000, + TableBytes: 9_000_000, + IsUnique: false, + IsPrimaryKey: false, + IsValid: true, + IsReady: true, + IsReplicaIdentity: false, + IsPartial: false, + IsExpression: false, + SupportsConstraint: false, + IndexMethod: "btree", + ColumnCount: 1, + IndexDefinition: "CREATE INDEX events_idx ON archive.events (a)", + LastScan: null, + StatsReset: null, + FirstSeenAt: DateTime.UtcNow.AddDays(-30), + SampleCount: 5, + StatsWereResetInWindow: false); +} diff --git a/Darling/PerformanceMonitor.Darling.Viewer/FinOpsTab.xaml b/Darling/PerformanceMonitor.Darling.Viewer/FinOpsTab.xaml index fecfcdda4..00e8ac495 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/FinOpsTab.xaml +++ b/Darling/PerformanceMonitor.Darling.Viewer/FinOpsTab.xaml @@ -610,7 +610,9 @@ @@ -104,11 +106,26 @@ public void ScoreAll(List facts) || HasSignificantWait(factsByKey, "SOS_SCHEDULER_YIELD", 0.25) || HasSignificantWait(factsByKey, "RESOURCE_SEMAPHORE", 0.10); + // #3526: the SECOND escape, per-fact and for ANOMALY_* only — extremity. The impact-peer + // escape above is the right shape for CXPACKET (parallelism is only an outage when a thread/CPU/ + // grant peer says so), but applied to every ANOMALY_* it made the baseline engine notification- + // inert at shipped settings: every anomaly ramp saturates its BASE at 1.0, the cap holds the FINAL + // at 1.49, and the notify floor is 1.5 — so a 20σ session spike beside a 15σ batch-request spike + // at 3am produced nothing that could page unless THREADPOOL/SOS/RESOURCE_SEMAPHORE happened to be + // in the picture too. The product's best per-server-calibrated statistics were structurally its + // quietest. An anomaly whose deviation is EXTREME against the cutoff it fired at (IsExtremeAnomaly: + // 3x its own fire threshold — 10.5σ on the 3.5σ robust path, 15σ on the 5.0σ heavy-tail path, 6σ + // classical; 3x the absolute bar on the never-blind fallback path) is released from the cap on its + // own evidence. Releasing the cap does NOT page by itself: base still maxes at 1.0, so the CRITICAL + // band is reached only through the anomaly co-fire amplifiers (AnomalyAmplifiers) — corroboration + // stays the house rule for >= 1.5, the escape merely stops the cap from discarding it. The + // impact-peer escape is unchanged and still releases everything; CXPACKET / CXCONSUMER / + // QUERY_HIGH_DOP have no extremity arm and stay capped however many anomalies co-fire beside them. if (!impactPeerCoFired) { foreach (var fact in facts) { - if (IsTuningClassKey(fact.Key)) + if (IsTuningClassKey(fact.Key) && !IsExtremeAnomaly(fact)) fact.Severity = Math.Min(fact.Severity, TuningClassSeverityCeiling); } } @@ -596,19 +613,122 @@ private static double ScoreBadActorFact(Fact fact) // Layer-3 tuning-class severity ceiling (see ScoreAll). Parallelism/anomaly signals describe a tuning // opportunity, not an outage — their FINAL severity is capped here (bands are >= 1.5 CRITICAL) unless an - // impact peer co-fired. 1.49 keeps a capped fact in the WARNING band without touching SeverityBand. + // impact peer co-fired or (#3526, ANOMALY_* only) the anomaly's own deviation is extreme. 1.49 keeps a + // capped fact in the WARNING band without touching SeverityBand. private const double TuningClassSeverityCeiling = 1.49; + /* #3526: the extremity escape's multiple (see IsExtremeAnomaly). An anomaly leaves the tuning-class cap + when its deviation is this many times the cutoff it FIRED at. Every deviation ramp saturates its base + at 2x its anchor (ScoreAnomalyFact: 0.5 at the anchor, 1.0 at 2x), so 3x sits a full anchor PAST the + point where the ramp stopped distinguishing — "extreme" means "so far out the scorer ran out of + scale", not "the top of the ramp". The arithmetic per path, with the detectors' shipped cutoffs + (AnomalyThresholds): robust modified-z 3.5 → the escape opens at 10.5σ; heavy-tail modified-z 5.0 + (waits, query duration) → 15σ; classical z 2.0 (rollup-bound metrics, pre-#1743 facts) → 6σ — the + #1743 fleet measurement read a busy tenant's REAL 2-3x evening surge at 1.4-2.0 classical sigmas, + so 6 classical sigmas against a stddev that history has already inflated is a genuinely rare + reading, not a busy evening. All three sit under the 25σ display cap (SigmaDisplayCap), so a fact + can actually carry them. The never-blind fallback path (baseline_low_quality) has no meaningful + sigma, so it escapes only at 3x its ABSOLUTE bar (fallback_exceedance >= 3): I/O latency 150 ms, + batch requests 15,000/s, sessions 1,500, query duration 15 s total elapsed — and CPU (bar 90%) and + memory total/target (bar 101%) can never reach 3x their bars, so a young store's CPU or memory + anomaly cannot escape on an untrustworthy baseline at all, which is correct: "we do not know your + normal yet" is not evidence of an outage. An operator who scales a metric's deviation threshold + scales its fire_threshold with it (ModifiedZThresholdFor), so the escape bar tracks the knob — up + to the display cap: AnomalyGate clamps the stored deviation_sigma at SigmaDisplayCap (25σ) BEFORE + the scorer ever sees it, so a bar above 25σ would be unreachable and the escape would go silently + dead for exactly the deployments that tuned a metric hard (a knob past ~8.3x the shipped anchor + puts 3x over 25) — the same "structurally quietest" defect this constant exists to fix, just for a + differently-tuned store. IsExtremeAnomaly therefore takes min(3x anchor, SigmaDisplayCap): a sigma + pinned at the cap means "at least 25σ", which is extreme under any anchor an operator can set. The + wait profile's modified_z is not display-capped (BaselineMath.ModifiedZScore) and needs no such + bound. */ + private const double ExtremeAnomalyMultiple = 3.0; + /// /// Tuning-class keys whose FINAL severity is capped at the WARNING ceiling (Layer 3) unless an /// impact peer co-fired: parallelism (CXPACKET/CXCONSUMER), excessive-DOP queries, and every - /// anomaly fact. Today only CXPACKET can exceed the ceiling on amplifiers (ANOMALY_* and - /// QUERY_HIGH_DOP already max at 1.0) — the rest is forward-safety as those ramps evolve. + /// anomaly fact. CXPACKET and (#3526) the corroborated ANOMALY_* families can exceed the ceiling + /// on amplifiers; an anomaly is released from the cap only when holds. + /// QUERY_HIGH_DOP still maxes at 1.0 — its membership is forward-safety as that ramp evolves. /// private static bool IsTuningClassKey(string key) => key is "CXPACKET" or "CXCONSUMER" or "QUERY_HIGH_DOP" || key.StartsWith("ANOMALY_", StringComparison.OrdinalIgnoreCase); + /// + /// The deviation-scored anomaly families (the z-score / modified-z detectors: peak vs a per-server + /// hour-of-week baseline, graded off deviation_sigma against fire_threshold, or off + /// fallback_exceedance on the low-quality path). Shared by and + /// so the two cannot route a key differently. + /// + private static bool IsDeviationScoredAnomalyKey(string key) => + key.StartsWith("ANOMALY_CPU_SPIKE", StringComparison.OrdinalIgnoreCase) + || key.StartsWith("ANOMALY_READ_LATENCY", StringComparison.OrdinalIgnoreCase) + || key.StartsWith("ANOMALY_WRITE_LATENCY", StringComparison.OrdinalIgnoreCase) + || key.StartsWith("ANOMALY_BATCH_REQUESTS", StringComparison.OrdinalIgnoreCase) + || key.StartsWith("ANOMALY_SESSION_SPIKE", StringComparison.OrdinalIgnoreCase) + || key.StartsWith("ANOMALY_QUERY_DURATION", StringComparison.OrdinalIgnoreCase) + || key.StartsWith("ANOMALY_MEMORY_PRESSURE", StringComparison.OrdinalIgnoreCase); + + /// + /// #1743: the cutoff a deviation-scored anomaly actually FIRED at (carried by the detector as + /// fire_threshold): the knob-scaled classical threshold on the classical path, the knob-scaled + /// modified-z cutoff on the robust path, and the pre-#1743 default of 2.0 when the fact carries none. + /// The severity ramp anchors here and the extremity escape is a multiple of it. + /// + private static double FireAnchor(Fact fact) + { + var anchor = fact.Metadata.GetValueOrDefault("fire_threshold", 2.0); + return anchor <= 0 ? 2.0 : anchor; + } + + /// + /// #3526: whether an ANOMALY_* fact's deviation is extreme enough to leave the Layer-3 tuning-class + /// cap on its own evidence (see for the arithmetic). Routes + /// each family off the SAME metadata its severity ramp grades from, so a fact can never be extreme + /// on one statistic while scored on another: + /// + /// Deviation-scored families, trustworthy baseline: deviation_sigma >= 3 x fire_threshold. + /// Deviation-scored families, low-quality baseline: fallback_exceedance >= 3 — the + /// stored sigma is the real (small, meaningless) z the detector refused to trust, so the escape + /// must not read it either. + /// ANOMALY_WAIT_PROFILE on the robust trigger: modified_z >= 3 x 5.0; on the + /// pre-#1743 / robust-less ratio trigger: ratio >= 3 x 4.0. Never for is_new — a + /// first-occurrence profile's sentinel ratio (NoBaselineRatio, 100) is a scoring device, not a + /// measurement, and "no baseline" cannot be "extreme against baseline". + /// + /// The ratio/count/delta families — blocking and deadlock spikes, day-over-day object growth and + /// contention, the legacy per-type ANOMALY_WAIT_ facts — stay capped: none is graded in sigmas, so + /// "3x the fire threshold" has no calibrated meaning for them, and the blocking/deadlock classes + /// already reach CRITICAL through their never-capped impact keys (BLOCKING_EVENTS, BLOCKING_CHAIN, + /// DEADLOCKS) when the events are real. + /// + private static bool IsExtremeAnomaly(Fact fact) + { + if (!fact.Key.StartsWith("ANOMALY_", StringComparison.OrdinalIgnoreCase)) return false; + + if (IsDeviationScoredAnomalyKey(fact.Key)) + { + if (fact.Metadata.GetValueOrDefault("baseline_low_quality") >= 1.0) + return fact.Metadata.GetValueOrDefault("fallback_exceedance") >= ExtremeAnomalyMultiple; + + // Bounded at the display cap — see ExtremeAnomalyMultiple: the stored sigma can never exceed it. + var escapeBar = Math.Min(ExtremeAnomalyMultiple * FireAnchor(fact), Baselines.AnomalyThresholds.SigmaDisplayCap); + return fact.Metadata.GetValueOrDefault("deviation_sigma") >= escapeBar; + } + + if (fact.Key.StartsWith("ANOMALY_WAIT_PROFILE", StringComparison.OrdinalIgnoreCase)) + { + if (fact.Metadata.GetValueOrDefault("is_new") > 0) return false; + var modifiedZ = fact.Metadata.GetValueOrDefault("modified_z"); + if (modifiedZ > 0) + return modifiedZ >= ExtremeAnomalyMultiple * Baselines.AnomalyThresholds.HeavyTailModifiedZThreshold; + return fact.Metadata.GetValueOrDefault("ratio") >= ExtremeAnomalyMultiple * WaitProfileRatioFloor; + } + + return false; + } + /// /// Scores anomaly facts based on deviation from baseline. /// At 2σ → 0.5, at 4σ → 1.0. Higher deviations are more severe. @@ -616,13 +736,7 @@ key is "CXPACKET" or "CXCONSUMER" or "QUERY_HIGH_DOP" /// private static double ScoreAnomalyFact(Fact fact) { - if (fact.Key.StartsWith("ANOMALY_CPU_SPIKE", StringComparison.OrdinalIgnoreCase) - || fact.Key.StartsWith("ANOMALY_READ_LATENCY", StringComparison.OrdinalIgnoreCase) - || fact.Key.StartsWith("ANOMALY_WRITE_LATENCY", StringComparison.OrdinalIgnoreCase) - || fact.Key.StartsWith("ANOMALY_BATCH_REQUESTS", StringComparison.OrdinalIgnoreCase) - || fact.Key.StartsWith("ANOMALY_SESSION_SPIKE", StringComparison.OrdinalIgnoreCase) - || fact.Key.StartsWith("ANOMALY_QUERY_DURATION", StringComparison.OrdinalIgnoreCase) - || fact.Key.StartsWith("ANOMALY_MEMORY_PRESSURE", StringComparison.OrdinalIgnoreCase)) + if (IsDeviationScoredAnomalyKey(fact.Key)) { // Deviation-based scoring: 2σ = 0.5, 4σ = 1.0 var deviation = fact.Metadata.GetValueOrDefault("deviation_sigma"); @@ -651,8 +765,7 @@ private static double ScoreAnomalyFact(Fact fact) shape for classical fires and for pre-#1743 facts (default 2.0), and the same proportional shape for robust fires at 3.5 or 5.0. Without the anchor, a family firing at 5σ scores saturated-flat 1.0 forever against a ramp built for 2σ fires. */ - var anchor = fact.Metadata.GetValueOrDefault("fire_threshold", 2.0); - if (anchor <= 0) anchor = 2.0; + var anchor = FireAnchor(fact); if (deviation < anchor) return 0.0; var base_score = 0.5 + 0.5 * Math.Min((deviation - anchor) / anchor, 1.0); return base_score * confidence; @@ -771,10 +884,252 @@ private static List GetAmplifiers(Fact fact) "PLAN_REGRESSION" => PlanRegressionAmplifiers(), "DB_CONFIG" => DbConfigAmplifiers(), "DISK_SPACE" => DiskSpaceAmplifiers(), + _ when fact.Key.StartsWith("ANOMALY_", StringComparison.OrdinalIgnoreCase) => AnomalyAmplifiers(fact.Key), _ => [] }; } + /// + /// #3526: the anomaly co-fire arm — corroboration for the baseline engine's findings, which had no + /// amplifiers at all and so could never leave their 1.0 base. Modelled on the impact-peer style of + /// the Layer-3 escape and the PAGEIOLATCH / IO-latency arms: a sibling anomaly family firing in the + /// same window (BaseSeverity > 0 — the scorer zeroes anything under its cutoff, so > 0 means + /// "fired against its own baseline") is a co-fire, and a MEASURED absolute fact confirming the same + /// pressure (SQL CPU >= 80%, the I/O-latency fact at its concerning bar, grant waiters, the + /// buffer-pool / log waits at the bars their own arms use) is a co-fire. + /// + /// Worked numbers — the arm's magnitudes are chosen so corroboration can carry an EXTREME anomaly + /// past the 1.5 notify floor and nothing can carry a routine one there: + /// + /// Base at the fire threshold (0.5) with two co-fires: 0.5 x (1 + 0.3 + 0.3) = 0.8. With every + /// arm lit (the load family's maximum is +1.2): 1.1. Never reaches 1.5, and the Layer-3 cap holds + /// regardless because the anomaly is not extreme. + /// Base saturated but ROUTINE (2x the anchor — 4σ classical, 7σ robust; 1.0) with three co-fires: + /// 1.0 x 1.9 = 1.9 → capped to 1.49. Still WARNING: saturation is not extremity, and the cap is + /// exactly what keeps a busy evening from paging. + /// Base EXTREME (>= 3x the anchor; 1.0, the cap released) alone: 1.0. Below the cap it escaped + /// — a lone 20σ reading with nothing else moving does not page, by design: the CRITICAL band is + /// earned only with corroboration (the same rule every other base fact follows), and a solitary + /// extreme reading is exactly the shape a collector hiccup or a variance-collapsed baseline pinned + /// at the 25σ display cap produces. + /// Base EXTREME with one +0.3 co-fire: 1.3, WARNING. With two: 1.6 → pages. The issue's own + /// 3am shape — a 20σ session spike (root, extreme) beside a 15σ batch-request anomaly (+0.3) and + /// SQL CPU at 85% (+0.3) — scores 1.6 and reaches the operator for the first time at shipped + /// settings. A +0.3 and a +0.2 land on 1.5 exactly: two independent corroborators is the bar. + /// + /// + private static List AnomalyAmplifiers(string key) + { + if (key.StartsWith("ANOMALY_SESSION_SPIKE", StringComparison.OrdinalIgnoreCase) + || key.StartsWith("ANOMALY_BATCH_REQUESTS", StringComparison.OrdinalIgnoreCase) + || key.StartsWith("ANOMALY_CPU_SPIKE", StringComparison.OrdinalIgnoreCase) + || key.StartsWith("ANOMALY_QUERY_DURATION", StringComparison.OrdinalIgnoreCase)) + return LoadAnomalyAmplifiers(key); + + if (key.StartsWith("ANOMALY_READ_LATENCY", StringComparison.OrdinalIgnoreCase)) + return ReadLatencyAnomalyAmplifiers(); + + if (key.StartsWith("ANOMALY_WRITE_LATENCY", StringComparison.OrdinalIgnoreCase)) + return WriteLatencyAnomalyAmplifiers(); + + if (key.StartsWith("ANOMALY_WAIT_PROFILE", StringComparison.OrdinalIgnoreCase)) + return WaitProfileAnomalyAmplifiers(); + + if (key.StartsWith("ANOMALY_MEMORY_PRESSURE", StringComparison.OrdinalIgnoreCase)) + return MemoryPressureAnomalyAmplifiers(); + + // Blocking/deadlock spikes and the object-stats anomalies have no arm: they are not released from + // the cap (IsExtremeAnomaly) and their impact lives in the never-capped BLOCKING_* / DEADLOCKS keys. + return []; + } + + /// + /// A sibling anomaly family fired in the same window against its own baseline. The self-key is + /// skipped by the callers, so a family never corroborates itself. + /// + private static bool AnomalyCoFired(Dictionary facts, string siblingKey) => + facts.TryGetValue(siblingKey, out var sibling) && sibling.BaseSeverity > 0; + + /// + /// The LOAD family — sessions, batch requests, CPU, query duration — corroborate one another (a real + /// surge moves more than one of them) and are confirmed by measured SQL CPU at the 80% bar the SOS + /// and compile-gateway arms already use. Each sibling is +0.3; the root's own key is omitted. + /// + private static List LoadAnomalyAmplifiers(string selfKey) + { + var amplifiers = new List(); + void Sibling(string siblingKey, string description) + { + if (selfKey.StartsWith(siblingKey, StringComparison.OrdinalIgnoreCase)) return; + amplifiers.Add(new() + { + Description = description, + Boost = 0.3, + Predicate = facts => AnomalyCoFired(facts, siblingKey) + }); + } + + Sibling("ANOMALY_SESSION_SPIKE", "Session-count anomaly co-fired — the surge is visible in connections too"); + Sibling("ANOMALY_BATCH_REQUESTS", "Batch-request anomaly co-fired — the surge is visible in throughput too"); + Sibling("ANOMALY_CPU_SPIKE", "CPU anomaly co-fired — the surge is consuming CPU far above this server's norm"); + Sibling("ANOMALY_QUERY_DURATION", "Query-duration anomaly co-fired — the surge is slowing queries"); + amplifiers.Add(new() + { + Description = "SQL Server CPU >= 80% — the surge is consuming real CPU, not just moving a counter", + Boost = 0.3, + Predicate = facts => facts.TryGetValue("CPU_SQL_PERCENT", out var cpu) && cpu.Value >= 80 + }); + return amplifiers; + } + + /// + /// ANOMALY_READ_LATENCY: a per-server read-latency deviation confirmed by the absolute read-latency fact + /// at its concerning bar (20 ms — "bad in absolute terms, not only for you"), by the wait profile + /// shifting (queries are actually waiting on it), by write latency deviating alongside (a storage-side + /// event, not one hot file), and by PAGEIOLATCH at the IO_READ_LATENCY_MS arm's own 10% bar. + /// + private static List ReadLatencyAnomalyAmplifiers() => + [ + new() + { + Description = "Read latency at the absolute concerning bar — slow for any server, not only against this baseline", + Boost = 0.3, + Predicate = facts => facts.TryGetValue("IO_READ_LATENCY_MS", out var io) && io.BaseSeverity >= 0.5 + }, + new() + { + Description = "Wait-profile anomaly co-fired — queries are waiting on the slow reads", + Boost = 0.3, + Predicate = facts => AnomalyCoFired(facts, "ANOMALY_WAIT_PROFILE") + }, + new() + { + Description = "Write-latency anomaly co-fired — the storage path is slow in both directions", + Boost = 0.3, + Predicate = facts => AnomalyCoFired(facts, "ANOMALY_WRITE_LATENCY") + }, + new() + { + Description = "PAGEIOLATCH waits elevated — buffer pool misses confirm the read pressure", + Boost = 0.2, + Predicate = facts => HasSignificantWait(facts, "PAGEIOLATCH_SH", 0.10) + || HasSignificantWait(facts, "PAGEIOLATCH_EX", 0.10) + } + ]; + + /// + /// ANOMALY_WRITE_LATENCY: the write-side twin — the absolute write-latency fact at its concerning bar + /// (10 ms), the wait profile shifting, read latency deviating alongside, and WRITELOG at the + /// IO_WRITE_LATENCY_MS arm's own 5% bar. + /// + private static List WriteLatencyAnomalyAmplifiers() => + [ + new() + { + Description = "Write latency at the absolute concerning bar — slow for any server, not only against this baseline", + Boost = 0.3, + Predicate = facts => facts.TryGetValue("IO_WRITE_LATENCY_MS", out var io) && io.BaseSeverity >= 0.5 + }, + new() + { + Description = "Wait-profile anomaly co-fired — queries are waiting on the slow writes", + Boost = 0.3, + Predicate = facts => AnomalyCoFired(facts, "ANOMALY_WAIT_PROFILE") + }, + new() + { + Description = "Read-latency anomaly co-fired — the storage path is slow in both directions", + Boost = 0.3, + Predicate = facts => AnomalyCoFired(facts, "ANOMALY_READ_LATENCY") + }, + new() + { + Description = "WRITELOG waits elevated — transaction log I/O confirms the write pressure", + Boost = 0.2, + Predicate = facts => HasSignificantWait(facts, "WRITELOG", 0.05) + } + ]; + + /// + /// ANOMALY_WAIT_PROFILE: the all-types wait rate shifting against its baseline, corroborated by WHAT the + /// waiting is costing — I/O latency deviating (+0.3 each side), query duration deviating (+0.3: the + /// waits are landing on user queries), and the load family moving (+0.2 each: a surge is driving it). + /// + private static List WaitProfileAnomalyAmplifiers() => + [ + new() + { + Description = "Read-latency anomaly co-fired — the wait shift is storage-bound", + Boost = 0.3, + Predicate = facts => AnomalyCoFired(facts, "ANOMALY_READ_LATENCY") + }, + new() + { + Description = "Write-latency anomaly co-fired — the wait shift is log/storage-bound", + Boost = 0.3, + Predicate = facts => AnomalyCoFired(facts, "ANOMALY_WRITE_LATENCY") + }, + new() + { + Description = "Query-duration anomaly co-fired — the waits are landing on user queries", + Boost = 0.3, + Predicate = facts => AnomalyCoFired(facts, "ANOMALY_QUERY_DURATION") + }, + new() + { + Description = "CPU anomaly co-fired — a load surge is driving the wait shift", + Boost = 0.2, + Predicate = facts => AnomalyCoFired(facts, "ANOMALY_CPU_SPIKE") + }, + new() + { + Description = "Session-count anomaly co-fired — a connection surge is driving the wait shift", + Boost = 0.2, + Predicate = facts => AnomalyCoFired(facts, "ANOMALY_SESSION_SPIKE") + } + ]; + + /// + /// ANOMALY_MEMORY_PRESSURE: total-over-target deviating against baseline, corroborated by the symptoms + /// real memory pressure produces — ring-buffer pressure notifications (the engine saying so itself), + /// grant waiters at the PAGEIOLATCH arm's bar, RESOURCE_SEMAPHORE in the wait stats, PAGEIOLATCH at + /// the 10% bar (buffer pool churn), and read latency deviating (the churn reaching storage). + /// + private static List MemoryPressureAnomalyAmplifiers() => + [ + new() + { + Description = "Memory-pressure notifications present — the engine itself is reporting pressure", + Boost = 0.3, + Predicate = facts => facts.TryGetValue("MEMORY_PRESSURE_EVENTS", out var mp) && mp.BaseSeverity > 0 + }, + new() + { + Description = "Memory grant waiters present — grants competing for the same memory", + Boost = 0.3, + Predicate = facts => facts.TryGetValue("MEMORY_GRANT_PENDING", out var mg) && mg.Value >= 1 + }, + new() + { + Description = "RESOURCE_SEMAPHORE waits present — grant pressure visible in wait stats", + Boost = 0.2, + Predicate = facts => facts.TryGetValue("RESOURCE_SEMAPHORE", out var rs) && rs.BaseSeverity > 0 + }, + new() + { + Description = "PAGEIOLATCH waits elevated — buffer pool churning under the pressure", + Boost = 0.2, + Predicate = facts => HasSignificantWait(facts, "PAGEIOLATCH_SH", 0.10) + || HasSignificantWait(facts, "PAGEIOLATCH_EX", 0.10) + }, + new() + { + Description = "Read-latency anomaly co-fired — the churn is reaching storage", + Boost = 0.2, + Predicate = facts => AnomalyCoFired(facts, "ANOMALY_READ_LATENCY") + } + ]; + /// /// PARAMETER_SENSITIVITY: a single plan with wildly varying per-execution cost. /// Corroborated by grant/spill divergence and memory-grant pressure. From f85c8012ff779f94c81ba2cb2b47c8e350f50300 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 11:19:44 -0400 Subject: [PATCH 21/69] get_store_metrics: the job_history block now says whose eyes its rows are visible to, and proves rows exist where it can (#3574) (#3585) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The block reported recording: true from the GUC (#3175) and stopped there. The reader's real question is "will I see rows", and timescaledb_information.job_history is a security_barrier view that shows a row only to a member of the job's owner role or of the database owner (2.28.1 views.sql: pg_has_role(current_user, , 'MEMBER') OR pg_has_role(current_user, owner, 'MEMBER')), while the jobs and job_stats views a reader checks first are not filtered. A least-privilege role can list all 110 jobs and read an empty history on a store recording perfectly; that cost a real postmortem. The reader gains a second, failure-isolated statement (JobHistoryEvidenceSql) that evaluates the view's own two pg_has_role tests for the connection doing the reading, counts the rows it sees over a fixed 24-hour window, takes the population half from the unfiltered job_stats, and publishes reader_role, visibility (All/Partial/None), job counts, rows_observed, newest_row_at, jobs_run_in_window, newest_run_started_at and a contradiction flag. The note names the rule with the predicate, the role it read as, and what that role may see; recording on + reader admitted + jobs ran + zero rows is the new CONTRADICTION arm, with its one benign cause and how to settle it. Evaluating the predicate is load-bearing rather than decorative: in managed mode the MCP host connects as the mcp role, which the view filters OUT, so a bare count would have manufactured the very false contradiction the issue is about on every managed store. The block now says "None — the filter, not the table" there and names the role that can see; a bring-your-own owner connection is the self-proving case. Bind is an explicit timestamptz with Kind=Utc — the inverse of the store's naive discipline, because these are TimescaleDB's own TIMESTAMPTZ catalog columns. --- .../DarlingMcpStoreMetricsToolsTests.cs | 512 +++++++++++++++++- .../Mcp/DarlingMcpInstructions.cs | 2 +- .../Mcp/DarlingMcpStoreMetricsTools.cs | 206 ++++++- .../Mcp/DarlingStoreMetricsReader.cs | 339 +++++++++++- .../TimescaleSupport.cs | 11 +- Darling/README.md | 2 + 6 files changed, 1046 insertions(+), 26 deletions(-) diff --git a/Darling/Darling.Tests/DarlingMcpStoreMetricsToolsTests.cs b/Darling/Darling.Tests/DarlingMcpStoreMetricsToolsTests.cs index 03e368fdc..b9fcf866b 100644 --- a/Darling/Darling.Tests/DarlingMcpStoreMetricsToolsTests.cs +++ b/Darling/Darling.Tests/DarlingMcpStoreMetricsToolsTests.cs @@ -8,8 +8,11 @@ using System; using System.ComponentModel; +using System.IO; using System.Linq; using System.Reflection; +using System.Runtime.CompilerServices; +using System.Threading; using System.Threading.Tasks; using ModelContextProtocol.Server; using Npgsql; @@ -102,6 +105,7 @@ grain a growth question wants. */ [Theory] [InlineData(nameof(DarlingStoreMetricsReader.StoreMetricsLatestSql))] [InlineData(nameof(DarlingStoreMetricsReader.StoreMetricsDailySql))] + [InlineData(nameof(DarlingStoreMetricsReader.JobHistoryEvidenceSql))] public void Reads_ArePostgresDialect_NoTsqlIsms_NoBareNow(string sqlName) { var sql = (string)typeof(DarlingStoreMetricsReader).GetField(sqlName, BindingFlags.Public | BindingFlags.Static)!.GetValue(null)!; @@ -216,9 +220,13 @@ public void TheJobHistoryNote_IsDifferentForEveryState_AndSplitsOffOnWhoSetIt() vacuous, and the whole point of four states is that there are four. */ Assert.Equal(4, statuses.Length); + /* NotApplicable evidence on purpose: it contributes NO text (pinned by its own test below), so what + is compared here is the GUC half alone — the #3175 arms, byte-for-byte what they were before the + evidence half was appended to them. */ var notes = statuses .Select(s => DarlingMcpStoreMetricsTools.JobHistoryNote( - new DarlingStoreMetricsReader.JobExecutionLoggingReading(s, null, null, null))) + new DarlingStoreMetricsReader.JobExecutionLoggingReading(s, null, null, null), + DarlingStoreMetricsReader.JobHistoryEvidence.NotApplicable)) .ToArray(); Assert.Equal(notes.Length, notes.Distinct(StringComparer.Ordinal).Count()); @@ -235,8 +243,8 @@ conf append at all (postgresql.auto.conf is read last) and needs the override re Assert.False(offByDefault.OffByExplicitOverride); Assert.True(offByOverride.OffByExplicitOverride); Assert.NotEqual( - DarlingMcpStoreMetricsTools.JobHistoryNote(offByDefault), - DarlingMcpStoreMetricsTools.JobHistoryNote(offByOverride)); + DarlingMcpStoreMetricsTools.JobHistoryNote(offByDefault, DarlingStoreMetricsReader.JobHistoryEvidence.NotApplicable), + DarlingMcpStoreMetricsTools.JobHistoryNote(offByOverride, DarlingStoreMetricsReader.JobHistoryEvidence.NotApplicable)); /* Only On is Recording. A precondition check whose "yes" leaked into any other state would put a maximum question back on an instrument that is off. */ @@ -253,6 +261,384 @@ maximum question back on an instrument that is off. */ .OffByExplicitOverride); } + /* ---------------- #3574: the evidence behind the flag, and who the rows are visible to ---------------- */ + + private static DarlingStoreMetricsReader.JobExecutionLoggingReading On() => new( + DarlingStoreMetricsReader.JobExecutionLoggingStatus.On, "on", "configuration file", null); + + private static DarlingStoreMetricsReader.JobExecutionLoggingReading OffByDefault() => new( + DarlingStoreMetricsReader.JobExecutionLoggingStatus.Off, "off", "default", null); + + /// An Observed reading with every fact stated, so a test flips exactly the one it is + /// about. The defaults are the OWNER's reading on a live store: database-owner member, 110 jobs, 12 + /// rows in the window, 9 jobs started a run in it. + private static DarlingStoreMetricsReader.JobHistoryEvidence Observed( + string role = "darling", + bool dbOwnerMember = true, + long jobs = 110, + long ownerMemberJobs = 110, + long rows = 12, + DateTime? newestRow = null, + long ran = 9, + DateTime? newestRun = null) => new( + DarlingStoreMetricsReader.JobHistoryEvidenceStatus.Observed, + role, dbOwnerMember, jobs, ownerMemberJobs, rows, + newestRow ?? (rows > 0 ? new DateTime(2026, 9, 18, 10, 0, 0, DateTimeKind.Utc) : null), + ran, + newestRun ?? (ran > 0 ? new DateTime(2026, 9, 18, 10, 30, 0, DateTimeKind.Utc) : null)); + + /// The managed-mode mcp role's reading on the same store: a member of nothing, and the + /// view shows it nothing — zero rows, while the unfiltered job_stats still counts the runs. + private static DarlingStoreMetricsReader.JobHistoryEvidence FilteredReader() => + Observed(role: "mcp", dbOwnerMember: false, ownerMemberJobs: 0, rows: 0, newestRow: null); + + /// + /// #3574: the evidence read evaluates the view's OWN predicate for the connection doing the reading, + /// and takes its two counts from the two views that disagree about visibility. + /// + /// Both pg_has_role tests, in the view's own terms. The membership in the database + /// owner (resolved through pg_get_userbyid(datdba), as the view does) and the per-job membership + /// in j.owner. Pinned because a read that counted rows without asking whether it was allowed to + /// see any would report zero on every managed store — the MCP host connects as the mcp role, which + /// the view filters out — and manufacture the contradiction the issue exists to prevent. + /// + /// The population half comes from the UNFILTERED view. job_stats has no ownership + /// clause, so "jobs started a run in the window" holds whatever the reader's standing; taken from the + /// filtered view it would be zero exactly when the count it was meant to qualify is zero, and the pair + /// would agree for the wrong reason. + /// + [Fact] + public void TheEvidenceSql_EvaluatesTheViewsOwnPredicate_AndCountsFromBothViews() + { + var sql = DarlingStoreMetricsReader.JobHistoryEvidenceSql; + + /* The view's two tests, evaluated for this reader. IS TRUE mirrors the view, whose own second test + can meet a NULL owner (a history row whose job was deleted) and must read it as "not a member". */ + Assert.Contains("current_user::text", sql, StringComparison.Ordinal); + Assert.Matches(@"pg_has_role\(\s*current_user,\s*\(SELECT pg_get_userbyid\(datdba\) FROM pg_database WHERE datname = current_database\(\)\),\s*'MEMBER'\) IS TRUE", sql); + Assert.Matches(@"pg_has_role\(current_user, j\.owner, 'MEMBER'\) IS TRUE", sql); + + /* Three views: the filtered one being counted, and the two unfiltered ones the reader checks first. */ + Assert.Contains("FROM timescaledb_information.job_history", sql, StringComparison.Ordinal); + Assert.Contains("FROM timescaledb_information.jobs", sql, StringComparison.Ordinal); + Assert.Contains("FROM timescaledb_information.job_stats", sql, StringComparison.Ordinal); + + /* One window, bound once, applied to both halves — so the count and its population share a + denominator. */ + Assert.Contains("h.start_time >= $1", sql, StringComparison.Ordinal); + Assert.Contains("js.last_run_started_at >= $1", sql, StringComparison.Ordinal); + Assert.DoesNotContain("$2", sql, StringComparison.Ordinal); + + /* The never-ran sentinel is -infinity, not NULL (#1760); the newest start must NULLIF it away or a + store whose jobs have never run reports a start in 4714 BC. */ + Assert.Contains("NULLIF(js.last_run_started_at, '-infinity'::timestamptz)", sql, StringComparison.Ordinal); + + /* The window is fixed and published beside the count. 24 hours: every job this product schedules + runs at least daily, so a live store always has starts inside it, and it sits inside the history + view's own one-month default retention. */ + Assert.Equal(24, DarlingStoreMetricsReader.JobHistoryEvidenceWindowHours); + } + + /// + /// #3574: the bind is timestamptz with Kind = Utc, stated explicitly — the INVERSE of + /// the naive-UTC discipline every other store read follows, because these are TimescaleDB's own + /// TIMESTAMPTZ catalog columns and here the naive bind would be the bug. A source pin, since the + /// only server that would catch the wrong Kind is one running off UTC, which no test store does. + /// + [Fact] + public void TheEvidenceRead_BindsAnExplicitTimestampTz_BecauseTheColumnsAreTimestampTz() + { + var source = File.ReadAllText(ReaderSourcePath()); + var method = source[source.IndexOf("GetJobHistoryEvidenceAsync(", StringComparison.Ordinal)..]; + method = method[..method.IndexOf("/// One object's newest self-metrics row", StringComparison.Ordinal)]; + + Assert.Contains("NpgsqlDbType.TimestampTz", method, StringComparison.Ordinal); + Assert.Contains("DateTimeKind.Utc", method, StringComparison.Ordinal); + /* And NOT the naive idiom, which is correct one method up and wrong here. */ + Assert.DoesNotContain("DateTimeKind.Unspecified", method, StringComparison.Ordinal); + } + + /// + /// #3574: is derived from the + /// view's two membership facts exactly as the view combines them — database-owner membership sees + /// everything and short-circuits the per-job test; otherwise the per-job count decides — and is + /// Unknown wherever a verdict would be vacuous. + /// + [Fact] + public void TheVisibility_IsDerivedFromTheViewsTwoTests_TheWayTheViewCombinesThem() + { + /* The owner: a member of the database owner, so All — regardless of the per-job count, which the + view never reaches for such a reader. */ + Assert.Equal(DarlingStoreMetricsReader.JobHistoryVisibility.All, Observed().Visibility); + Assert.Equal(DarlingStoreMetricsReader.JobHistoryVisibility.All, Observed(ownerMemberJobs: 0).Visibility); + Assert.Equal(110L, Observed(ownerMemberJobs: 0).HistoryVisibleJobCount); + + /* Not the database owner, but a member of every job's owner: still All. */ + Assert.Equal( + DarlingStoreMetricsReader.JobHistoryVisibility.All, + Observed(dbOwnerMember: false, ownerMemberJobs: 110).Visibility); + + /* The managed-mode mcp role: a member of neither. None, and the visible count is zero. */ + Assert.Equal(DarlingStoreMetricsReader.JobHistoryVisibility.None, FilteredReader().Visibility); + Assert.Equal(0L, FilteredReader().HistoryVisibleJobCount); + + /* Some jobs' owner but not all: Partial, with the fraction preserved for the note. */ + var partial = Observed(dbOwnerMember: false, ownerMemberJobs: 3); + Assert.Equal(DarlingStoreMetricsReader.JobHistoryVisibility.Partial, partial.Visibility); + Assert.Equal(3L, partial.HistoryVisibleJobCount); + + /* No jobs at all: None and All are both vacuously true, so neither is claimed. */ + Assert.Equal( + DarlingStoreMetricsReader.JobHistoryVisibility.Unknown, + Observed(jobs: 0, ownerMemberJobs: 0, ran: 0, rows: 0).Visibility); + + /* And nothing was observed: nothing is derived. Every derived field is null, never a plausible zero. */ + foreach (var blank in new[] + { + DarlingStoreMetricsReader.JobHistoryEvidence.Unreadable, + DarlingStoreMetricsReader.JobHistoryEvidence.NotApplicable, + }) + { + Assert.Equal(DarlingStoreMetricsReader.JobHistoryVisibility.Unknown, blank.Visibility); + Assert.Null(blank.HistoryVisibleJobCount); + Assert.Null(blank.RowsObserved); + Assert.Null(blank.NewestRowAt); + Assert.Null(blank.ReaderRole); + } + } + + /// + /// #3574: the contradiction is the conjunction of FOUR conditions, and each one alone is a different, + /// non-finding shape. A zero read through a filtered role is the filter; a zero with no runs in the + /// window is an absence of information; a zero with the GUC off is #3175's arm; rows seen is recording + /// proven. Only all four together say the instrument is not writing what the GUC says it is. Each row of + /// the table flips exactly one condition off the positive control, so the pin cannot pass by a + /// predicate that is simply always false. + /// + [Fact] + public void TheContradiction_NeedsAllFourConditions_AndEachAloneIsNotOne() + { + var positive = Observed(rows: 0, newestRow: null); + Assert.True(positive.ContradictsRecording(recording: true)); + + /* 1. Not recording: the GUC-off arm, not this one. */ + Assert.False(positive.ContradictsRecording(recording: false)); + + /* 2. Reader filtered: the mcp role's zero is the view's doing. Partial is not enough either — the + jobs that ran may be the ones this reader cannot see. */ + Assert.False(FilteredReader().ContradictsRecording(recording: true)); + Assert.False(Observed(dbOwnerMember: false, ownerMemberJobs: 3, rows: 0, newestRow: null).ContradictsRecording(recording: true)); + + /* 3. Rows seen: recording is proven, whatever the run count. */ + Assert.False(Observed(rows: 1).ContradictsRecording(recording: true)); + + /* 4. Nothing ran: nothing to record, so nothing is contradicted. */ + Assert.False(Observed(rows: 0, newestRow: null, ran: 0, newestRun: null).ContradictsRecording(recording: true)); + + /* And not-observed evidence never contradicts anything: a read that did not complete has no + standing to declare a finding. */ + Assert.False(DarlingStoreMetricsReader.JobHistoryEvidence.Unreadable.ContradictsRecording(recording: true)); + Assert.False(DarlingStoreMetricsReader.JobHistoryEvidence.NotApplicable.ContradictsRecording(recording: true)); + } + + /// + /// #3574: the note names the visibility rule and the reader on every arm where the view exists, and + /// says a DIFFERENT thing for each thing the evidence established — including the new contradiction + /// arm, in so many words. + /// + /// Distinctness across evidence shapes with the GUC held constant, the same claim the + /// #3175 pin makes across GUC states with the evidence held constant. The defect class is one absence + /// being read as another; two evidence shapes sharing a sentence would reproduce it. + /// + [Fact] + public void TheVisibilityNote_NamesTheRuleAndTheReader_AndSaysADifferentThingPerShape() + { + var on = On(); + + var filtered = DarlingMcpStoreMetricsTools.JobHistoryNote(on, FilteredReader()); + var contradiction = DarlingMcpStoreMetricsTools.JobHistoryNote(on, Observed(rows: 0, newestRow: null)); + var proven = DarlingMcpStoreMetricsTools.JobHistoryNote(on, Observed()); + var nothingRan = DarlingMcpStoreMetricsTools.JobHistoryNote(on, Observed(rows: 0, newestRow: null, ran: 0, newestRun: null)); + var partial = DarlingMcpStoreMetricsTools.JobHistoryNote(on, Observed(dbOwnerMember: false, ownerMemberJobs: 3)); + var noJobs = DarlingMcpStoreMetricsTools.JobHistoryNote(on, Observed(jobs: 0, ownerMemberJobs: 0, rows: 0, ran: 0)); + var unreadable = DarlingMcpStoreMetricsTools.JobHistoryNote(on, DarlingStoreMetricsReader.JobHistoryEvidence.Unreadable); + + var all = new[] { filtered, contradiction, proven, nothingRan, partial, noJobs, unreadable }; + Assert.Equal(all.Length, all.Distinct(StringComparer.Ordinal).Count()); + + /* The rule, stated with the view's own predicate, and the trap named, on every one of them. */ + foreach (var note in all) + { + Assert.Contains("pg_has_role(current_user, , 'MEMBER') OR pg_has_role(current_user, , 'MEMBER')", note, StringComparison.Ordinal); + Assert.Contains("jobs and job_stats views show every role every job", note, StringComparison.Ordinal); + Assert.Contains("contradicts nothing", note, StringComparison.Ordinal); + /* And the #3175 half is still in front of it, untouched. */ + Assert.StartsWith( + DarlingMcpStoreMetricsTools.JobHistoryNote(on, DarlingStoreMetricsReader.JobHistoryEvidence.NotApplicable), + note, + StringComparison.Ordinal); + } + + /* The reader is named wherever there was one. */ + Assert.Contains("read as 'mcp'", filtered, StringComparison.Ordinal); + Assert.Contains("read as 'darling'", contradiction, StringComparison.Ordinal); + + /* The filtered arm: the zero is the filter, the role that can see is named, and the unfiltered + population is reported so the store does not read as idle. */ + Assert.Contains("NOTHING by construction", filtered, StringComparison.Ordinal); + Assert.Contains("the filter, not the table", filtered, StringComparison.Ordinal); + Assert.Contains("owner role", filtered, StringComparison.Ordinal); + Assert.Contains("9 job(s) started a run", filtered, StringComparison.Ordinal); + Assert.DoesNotContain("CONTRADICTION", filtered, StringComparison.Ordinal); + + /* The contradiction arm: named as such, with the benign cause and how to settle it, and never + 'quiet' or 'clean'. */ + Assert.Contains("CONTRADICTION", contradiction, StringComparison.Ordinal); + Assert.Contains("do not read this zero as quiet", contradiction, StringComparison.Ordinal); + Assert.Contains("switched on AFTER", contradiction, StringComparison.Ordinal); + Assert.Contains("re-read after", contradiction, StringComparison.Ordinal); + Assert.Contains("a finding", contradiction, StringComparison.Ordinal); + + /* Rows seen: the flag is a measurement. */ + Assert.Contains("12 row(s)", proven, StringComparison.Ordinal); + Assert.Contains("a measurement here, not a GUC echo", proven, StringComparison.Ordinal); + Assert.DoesNotContain("CONTRADICTION", proven, StringComparison.Ordinal); + + /* Zero rows, zero runs: not clean — an absence of information, said so. */ + Assert.Contains("proves nothing either way", nothingRan, StringComparison.Ordinal); + Assert.Contains("absence of information", nothingRan, StringComparison.Ordinal); + Assert.DoesNotContain("CONTRADICTION", nothingRan, StringComparison.Ordinal); + + /* Partial: the fraction, and no verdict. */ + Assert.Contains("3 of 110 jobs", partial, StringComparison.Ordinal); + Assert.Contains("THOSE jobs only", partial, StringComparison.Ordinal); + Assert.DoesNotContain("CONTRADICTION", partial, StringComparison.Ordinal); + + /* Unreadable: the flag is a GUC echo and the note says so, instead of a zero. */ + Assert.Contains("UNKNOWN", unreadable, StringComparison.Ordinal); + Assert.Contains("GUC's word alone", unreadable, StringComparison.Ordinal); + Assert.DoesNotContain("rows_observed = ", unreadable, StringComparison.Ordinal); + + /* The GUC-off arm with an admitted reader: says what it will see once on, and that anything seen + now is failures — never that logging is secretly on. */ + var offAdmitted = DarlingMcpStoreMetricsTools.JobHistoryNote(OffByDefault(), Observed(rows: 2)); + Assert.Contains("FAILED runs", offAdmitted, StringComparison.Ordinal); + Assert.DoesNotContain("CONTRADICTION", offAdmitted, StringComparison.Ordinal); + Assert.DoesNotContain("not a GUC echo", offAdmitted, StringComparison.Ordinal); + + /* The GUC unreadable but the view readable: the rows are counted and NOT classified — neither + proof of recording nor a failure census, because that split is the setting's to make. */ + var gucUnknownAdmitted = DarlingMcpStoreMetricsTools.JobHistoryNote( + new DarlingStoreMetricsReader.JobExecutionLoggingReading( + DarlingStoreMetricsReader.JobExecutionLoggingStatus.Unreadable, null, null, null), + Observed(rows: 2)); + Assert.Contains("not classified", gucUnknownAdmitted, StringComparison.Ordinal); + Assert.Contains("2 row(s)", gucUnknownAdmitted, StringComparison.Ordinal); + Assert.DoesNotContain("CONTRADICTION", gucUnknownAdmitted, StringComparison.Ordinal); + Assert.DoesNotContain("not a GUC echo", gucUnknownAdmitted, StringComparison.Ordinal); + Assert.DoesNotContain("any it sees now are FAILED", gucUnknownAdmitted, StringComparison.Ordinal); + } + + /// + /// #3574: on a plain-PostgreSQL connection there is no view to be filtered, so the evidence half adds + /// NOTHING and the #3175 NotRegistered note is byte-for-byte what it was. The one arm where a sentence + /// about who may read the view would be noise. + /// + [Fact] + public void TheNotApplicableEvidence_AddsNoText_SoTheNotRegisteredNoteIsUnchanged() + { + var notRegistered = new DarlingStoreMetricsReader.JobExecutionLoggingReading( + DarlingStoreMetricsReader.JobExecutionLoggingStatus.NotRegistered, null, null, null); + + Assert.Equal( + "", + DarlingMcpStoreMetricsTools.JobHistoryVisibilityNote(notRegistered, DarlingStoreMetricsReader.JobHistoryEvidence.NotApplicable)); + + var note = DarlingMcpStoreMetricsTools.JobHistoryNote(notRegistered, DarlingStoreMetricsReader.JobHistoryEvidence.NotApplicable); + Assert.DoesNotContain("WHO CAN SEE", note, StringComparison.Ordinal); + Assert.DoesNotContain("pg_has_role", note, StringComparison.Ordinal); + Assert.Contains("does not exist on this connection", note, StringComparison.Ordinal); + } + + /// + /// #3574: the two reads fail SEPARATELY, and the evidence read is not attempted where there is no view. + /// + /// No store is needed. The data source points at a port nothing listens on, so any read that + /// reaches the network fails at connect — which is exactly the failure the isolation has to absorb. The + /// NotRegistered half goes further: it hands the read an ALREADY-CANCELLED token, and the method's own + /// contract is that cancellation is never isolated, so a NotApplicable coming back (rather than an + /// ) proves the read returned before touching the connection + /// at all. + /// + [Fact] + public async Task TheEvidenceRead_FailsSeparatelyFromTheGucRead_AndIsSkippedWhereThereIsNoView() + { + await using var nowhere = NpgsqlDataSource.Create( + "Host=127.0.0.1;Port=1;Username=nobody;Password=nobody;Database=nowhere;Timeout=1;Command Timeout=1"); + + /* A registered GUC reading, handed in from outside: the evidence read fails at connect and reports + Unreadable, and the GUC reading it was given is untouched — the count timing out cannot make the + GUC unknown. */ + var on = On(); + var evidence = await DarlingStoreMetricsReader.GetJobHistoryEvidenceAsync(nowhere, on, TestContext.Current.CancellationToken); + Assert.Equal(DarlingStoreMetricsReader.JobHistoryEvidenceStatus.Unreadable, evidence.Status); + Assert.Null(evidence.RowsObserved); + Assert.Null(evidence.ReaderRole); + Assert.True(on.Recording); + + /* And the note on that pair still carries the GUC's answer, qualified as an echo. */ + var note = DarlingMcpStoreMetricsTools.JobHistoryNote(on, evidence); + Assert.Contains("is ON", note, StringComparison.Ordinal); + Assert.Contains("GUC's word alone", note, StringComparison.Ordinal); + + /* NotRegistered: not attempted. A fired token would throw out of any read that started. */ + var notRegistered = new DarlingStoreMetricsReader.JobExecutionLoggingReading( + DarlingStoreMetricsReader.JobExecutionLoggingStatus.NotRegistered, null, null, null); + var skipped = await DarlingStoreMetricsReader.GetJobHistoryEvidenceAsync( + nowhere, notRegistered, new CancellationToken(canceled: true)); + Assert.Same(DarlingStoreMetricsReader.JobHistoryEvidence.NotApplicable, skipped); + + /* The GUC read's own isolation, for the pairing: it too becomes Unreadable at connect rather than + throwing or inventing an Off. */ + var guc = await DarlingStoreMetricsReader.GetJobExecutionLoggingAsync(nowhere, TestContext.Current.CancellationToken); + Assert.Equal(DarlingStoreMetricsReader.JobExecutionLoggingStatus.Unreadable, guc.Status); + } + + /// + /// #3574: the description names the ownership filter and the fields the block now carries for it, + /// asserted TOGETHER with the shipped SQL evaluating the predicate — the sibling pins' reason: a + /// sentence about a read that does not exist is advice to look somewhere this tool declines to look, + /// and a read with no sentence sits in the JSON unexplained. + /// + [Fact] + public void TheDescription_NamesTheOwnershipFilter_AndTheEvidenceFieldsBehindIt() + { + var description = ToolMethods().Single().GetCustomAttribute()?.Description; + Assert.NotNull(description); + + /* The rule, the trap, and the fact that the tool's own managed-mode reader is on the wrong side of + it — in one ordered match each, so a rewording that kept the words but dropped the claim goes + red. */ + Assert.Matches(@"job_history is ownership-filtered[^.]*members of the job's owner role or of the database owner[^.]*jobs and job_stats views show every role every job", description!); + Assert.Matches(@"managed mode[^.]*mcp role[^.]*filters out", description!); + + /* Every evidence field the block publishes is named where a caller reads about the block. */ + foreach (var field in new[] { "reader_role", "visibility", "rows_observed", "newest_row_at", "jobs_run_in_window", "contradiction" }) + { + Assert.Contains(field, description!, StringComparison.Ordinal); + } + + /* And the window the counts are over, so the number never travels without its denominator. */ + Assert.Contains($"{DarlingStoreMetricsReader.JobHistoryEvidenceWindowHours}-hour window", description!, StringComparison.Ordinal); + + /* The read behind the sentence. */ + Assert.Contains("pg_has_role", DarlingStoreMetricsReader.JobHistoryEvidenceSql, StringComparison.Ordinal); + } + + private static string ReaderSourcePath([CallerFilePath] string thisFile = "") + => Path.GetFullPath(Path.Combine( + Path.GetDirectoryName(thisFile)!, "..", "PerformanceMonitor.Darling.Service", "Mcp", "DarlingStoreMetricsReader.cs")); + [Fact] public void GetStoreMetrics_IsInTheServerInstructions() { @@ -338,9 +724,10 @@ public void ComputeDailyGrowth_EmptyAndSingleDay_YieldNothing() } /// -/// The job_history precondition probe against a LIVE server (#3175). Two things no text assertion -/// can reach: that the shipped SQL parses and binds its one positional parameter, and that the GUC name the -/// product writes into postgresql.conf is a name PostgreSQL actually knows. +/// The job_history precondition probe against a LIVE server (#3175), and the evidence probe behind it +/// (#3574). Things no text assertion can reach: that the shipped SQL parses and binds its positional +/// parameter, that the GUC name the product writes into postgresql.conf is a name PostgreSQL actually knows, +/// and that the view's ownership predicate evaluates for the connection the way the record derives it. /// /// A PAIRED control, because a zero-row result is the whole subject. The reader maps "no /// pg_settings row" to NotRegistered, so a probe that only ever saw zero rows — because the name was @@ -354,10 +741,11 @@ public void ComputeDailyGrowth_EmptyAndSingleDay_YieldNothing() /// change look like a defect. What is pinned is that the reading is INTERNALLY CONSISTENT — a registered /// state, a value PostgreSQL renders for a bool, and Recording true for exactly on. /// -/* #1776 own-store: this class reads only pg_settings — a server-scoped catalog view, no store tables, no - DDL, no rows written — but it takes [Collection("live-postgres")] anyway because it shares the cluster - whose GUCs it reads with every other class that has it, and a class reading the shared store must either - carry the attribute or record why not. */ +/* #1776 own-store: this class reads only pg_settings and TimescaleDB's own information views (#3574) — + server- and catalog-scoped, no store tables, no DDL, no rows written — but it takes + [Collection("live-postgres")] anyway because it shares the cluster whose GUCs and jobs it reads with every + other class that has it, and a class reading the shared store must either carry the attribute or record + why not. */ [Collection("live-postgres")] public sealed class DarlingStoreMetricsJobLoggingLivePostgresTests { @@ -390,4 +778,108 @@ name resolves — not which way the setting happens to be pointing on this clust await using var reader = await command.ExecuteReaderAsync(ct); Assert.False(await reader.ReadAsync(ct)); } + + /// + /// #3574: the evidence read against a live TimescaleDB — the shipped SQL parses, binds its one + /// timestamptz parameter, evaluates the view's predicate for the connection, and its fields are + /// INTERNALLY CONSISTENT with each other and with direct reads of the same views. + /// + /// Inserts nothing and pins no count. How many jobs exist, whether any ran in the last day + /// and whether the GUC is on are properties of whatever cluster DARLING_TEST_PG points at. What + /// is pinned is the null-versus-zero contract: RowsObserved is a COUNT and never null once the + /// read completes (a zero is a zero), while NewestRowAt is null exactly when the view showed this + /// connection no row at all — the two are different facts, and the whole issue is one being read as the + /// other. The all-time total is read DIRECTLY first, then the shipped read; rows are only ever added + /// between two reads (the history retention job runs monthly), so a direct total above zero must be + /// matched by a non-null newest row, and a direct total of zero by a null one and a zero count. + /// + /// The predicate is cross-checked, not trusted. The database-owner test is re-evaluated + /// directly on the same connection and must agree with the record; when it holds and jobs exist, the + /// derived Visibility must be All and the visible count the job count — which is the + /// managed-mode OWNER's reading, and the shape under which a zero would be a real contradiction. + /// + [Fact] + public async Task JobHistoryEvidenceProbe_EvaluatesThePredicateForThisConnection_AndItsFieldsAgree() + { + var cs = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(cs), "Set DARLING_TEST_PG to a Postgres connection string to run the live job-history evidence probe test."); + + var ct = TestContext.Current.CancellationToken; + await using var postgres = NpgsqlDataSource.Create(cs!); + + /* Direct facts first, through the same views the shipped read uses. */ + long directTotal; + bool directDbOwnerMember; + string directRole; + await using (var direct = postgres.CreateCommand( + "SELECT (SELECT count(*) FROM timescaledb_information.job_history), " + + "pg_has_role(current_user, (SELECT pg_get_userbyid(datdba) FROM pg_database WHERE datname = current_database()), 'MEMBER') IS TRUE, " + + "current_user::text")) + await using (var directReader = await direct.ExecuteReaderAsync(ct)) + { + Assert.True(await directReader.ReadAsync(ct)); + directTotal = directReader.GetInt64(0); + directDbOwnerMember = directReader.GetBoolean(1); + directRole = directReader.GetString(2); + } + + var logging = await DarlingStoreMetricsReader.GetJobExecutionLoggingAsync(postgres, ct); + Assert.NotEqual(DarlingStoreMetricsReader.JobExecutionLoggingStatus.NotRegistered, logging.Status); + + var evidence = await DarlingStoreMetricsReader.GetJobHistoryEvidenceAsync(postgres, logging, ct); + + Assert.Equal(DarlingStoreMetricsReader.JobHistoryEvidenceStatus.Observed, evidence.Status); + Assert.Equal(directRole, evidence.ReaderRole); + Assert.Equal(directDbOwnerMember, evidence.ReaderIsDatabaseOwnerMember); + + /* Counts, never nulls, once observed. */ + Assert.NotNull(evidence.JobCount); + Assert.NotNull(evidence.OwnerMemberJobCount); + Assert.NotNull(evidence.RowsObserved); + Assert.NotNull(evidence.JobsRunInWindow); + Assert.InRange(evidence.OwnerMemberJobCount!.Value, 0, evidence.JobCount!.Value); + Assert.InRange(evidence.HistoryVisibleJobCount!.Value, 0, evidence.JobCount.Value); + + /* The null-versus-zero contract. */ + if (directTotal == 0) + { + Assert.Null(evidence.NewestRowAt); + Assert.Equal(0L, evidence.RowsObserved); + } + else + { + Assert.NotNull(evidence.NewestRowAt); + Assert.Equal(DateTimeKind.Utc, evidence.NewestRowAt!.Value.Kind); + } + + /* A row inside the window implies a newest row; a run inside the window implies a newest run. */ + if (evidence.RowsObserved > 0) + { + Assert.NotNull(evidence.NewestRowAt); + } + + if (evidence.JobsRunInWindow > 0) + { + Assert.NotNull(evidence.NewestRunStartedAt); + Assert.Equal(DateTimeKind.Utc, evidence.NewestRunStartedAt!.Value.Kind); + } + + /* The predicate's verdict, derived as the view combines its two tests. */ + if (evidence.JobCount > 0 && directDbOwnerMember) + { + Assert.Equal(DarlingStoreMetricsReader.JobHistoryVisibility.All, evidence.Visibility); + Assert.Equal(evidence.JobCount, evidence.HistoryVisibleJobCount); + } + + if (evidence.JobCount == 0) + { + Assert.Equal(DarlingStoreMetricsReader.JobHistoryVisibility.Unknown, evidence.Visibility); + } + + /* And the note built on a live reading names the role that read. */ + Assert.Contains( + $"read as '{directRole}'", + DarlingMcpStoreMetricsTools.JobHistoryNote(logging, evidence), + StringComparison.Ordinal); + } } diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs index 062ac5b78..2986f8667 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs @@ -250,7 +250,7 @@ public static string Build(DarlingPeerDirectory.Snapshot peers) ### Store self-metrics - `get_store_metrics` reads the MONITORING STORE's own growth series — not a monitored SQL Server's. The service records an hourly self-metrics snapshot into the store: per-hypertable total size, pre/post-compression bytes and chunk count; the query-text / query-plan payload dimension tables' total size (the store's dominant payloads) and row counts; and the whole store's size with the enabled-server count. The tool returns the latest snapshot per object plus a daily series, including the whole-store daily growth in bytes and the derived per-server ingest rate (daily growth ÷ enabled servers) — the number to multiply when onboarding N servers. Each daily point is that day's LAST snapshot rather than its maximum, so a maximum question needs the per-run route the tool's own description names. Use it for capacity forecasting: what is driving store growth and how fast. 400 days of history; no `server_name` parameter, because the store is the subject. + `get_store_metrics` reads the MONITORING STORE's own growth series — not a monitored SQL Server's. The service records an hourly self-metrics snapshot into the store: per-hypertable total size, pre/post-compression bytes and chunk count; the query-text / query-plan payload dimension tables' total size (the store's dominant payloads) and row counts; and the whole store's size with the enabled-server count. The tool returns the latest snapshot per object plus a daily series, including the whole-store daily growth in bytes and the derived per-server ingest rate (daily growth ÷ enabled servers) — the number to multiply when onboarding N servers. Each daily point is that day's LAST snapshot rather than its maximum, so a maximum question needs the per-run route the tool's own description names — `timescaledb_information.job_history`, which records only while its GUC is on AND shows rows only to members of the job's owner role or the database owner (the `jobs` view is not filtered, which is the trap); the response's `job_history` block reports the GUC's state, which role it read as, whether the view lets that role see anything, and the rows it observed over the last 24 hours, so read it before treating an empty history as an answer. Use it for capacity forecasting: what is driving store growth and how fast. 400 days of history; no `server_name` parameter, because the store is the subject. | Tool | Purpose | Key Parameters | |------|---------|----------------| diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpStoreMetricsTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpStoreMetricsTools.cs index 34b8bb99b..ac67cf4fa 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpStoreMetricsTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpStoreMetricsTools.cs @@ -9,6 +9,7 @@ using System; using System.Collections.Generic; using System.ComponentModel; +using System.Globalization; using System.Linq; using System.Text.Json; using System.Threading.Tasks; @@ -36,7 +37,7 @@ public sealed class DarlingMcpStoreMetricsTools public const int MaxDaysBack = StoreSelfMetrics.RetentionDays; [McpServerTool(Name = "get_store_metrics"), Description( - "Gets the monitoring store's OWN size and growth metrics — not a monitored SQL Server's. The service records an hourly self-metrics snapshot: per-hypertable total size, pre/post-compression bytes and chunk count; the query-text and query-plan payload dimension tables' total size (the store's dominant payloads) and row counts; the whole store's size with the enabled-server count; and one row per TimescaleDB background job (CAGG refresh, compression, retention) with its last run duration, schedule interval, duration-vs-cadence percent, and run/failure totals — the jobs whose runtimes scale with fleet size. Returns the latest snapshot per object plus a daily series over the window, with the whole-store daily growth in bytes and the derived per-server ingest rate (daily growth / enabled servers). Each daily point is that day's LAST snapshot, never its maximum or its mean — the settled figure a growth question wants, but it means a MAXIMUM question (what was this job's longest run that day, did it enter its warning band) cannot be answered from this series: the day's peak is DROPPED rather than smoothed, so a day whose worst run crossed a threshold reads as a day that never approached it. The route that carries one row per run is TimescaleDB's own job history (timescaledb_information.job_history) — but ONLY while timescaledb.enable_job_execution_logging is on, and it defaults OFF, so on a store that has never had it turned on a maximum over that table returns zero rows, which reads as 'no run exceeded the line' rather than 'this instrument is off'. The job_history block in every response reports that setting's EFFECTIVE value and its source (plus the file that set it, where the connection is privileged enough to see it), so this redirect is never issued blind: read it before treating an empty job_history as an answer, and note that logging covers runs only from the point it was switched on because nothing earlier was recorded to recover. The hourly snapshot behind the series has the same limit one grain down: it samples last_run_duration once an hour at a fixed offset, so a run longer than that offset is never recorded at all. Also reports, read LIVE from the catalog rather than from the recorded series, every retention policy the rollup-coverage gate is holding PAUSED, with the tier's actual data span and how many times its configured drop_after horizon it is really holding — a held policy records zero failures and a normal-looking last run, so it is invisible in the stored job telemetry and is a common cause of unexplained store growth. Use for capacity forecasting: what is driving store growth, how fast, what adding N servers would multiply, which background job is closest to outgrowing its own cadence, and whether retention is actually running.")] + "Gets the monitoring store's OWN size and growth metrics — not a monitored SQL Server's. The service records an hourly self-metrics snapshot: per-hypertable total size, pre/post-compression bytes and chunk count; the query-text and query-plan payload dimension tables' total size (the store's dominant payloads) and row counts; the whole store's size with the enabled-server count; and one row per TimescaleDB background job (CAGG refresh, compression, retention) with its last run duration, schedule interval, duration-vs-cadence percent, and run/failure totals — the jobs whose runtimes scale with fleet size. Returns the latest snapshot per object plus a daily series over the window, with the whole-store daily growth in bytes and the derived per-server ingest rate (daily growth / enabled servers). Each daily point is that day's LAST snapshot, never its maximum or its mean — the settled figure a growth question wants, but it means a MAXIMUM question (what was this job's longest run that day, did it enter its warning band) cannot be answered from this series: the day's peak is DROPPED rather than smoothed, so a day whose worst run crossed a threshold reads as a day that never approached it. The route that carries one row per run is TimescaleDB's own job history (timescaledb_information.job_history) — but ONLY while timescaledb.enable_job_execution_logging is on, and it defaults OFF, so on a store that has never had it turned on a maximum over that table returns zero rows, which reads as 'no run exceeded the line' rather than 'this instrument is off'. The job_history block in every response reports that setting's EFFECTIVE value and its source (plus the file that set it, where the connection is privileged enough to see it), so this redirect is never issued blind: read it before treating an empty job_history as an answer, and note that logging covers runs only from the point it was switched on because nothing earlier was recorded to recover. The setting answers 'is it on'; whether YOU will see rows is a second question, because job_history is ownership-filtered — its rows are visible only to members of the job's owner role or of the database owner, while the jobs and job_stats views show every role every job — so a role that can list all the jobs can still read an empty history on a store that is recording perfectly. The same block therefore also reports which role it read as (reader_role), that role's standing under the view's own predicate (visibility: All, Partial or None, with the job counts behind it), the rows it actually observed over a fixed 24-hour window (rows_observed, newest_row_at) beside how many jobs job_stats says started a run in that window (jobs_run_in_window), and a contradiction flag that is true only when recording is on, the reader can see every job's history, jobs ran, and the view showed no rows — the finding to investigate. In managed mode this tool reads as the least-privilege mcp role, which the view filters out (visibility None), so its rows_observed is zero by construction and the note names the role that can see; only a connection with owner standing makes the flag self-proving. The hourly snapshot behind the series has the same limit one grain down: it samples last_run_duration once an hour at a fixed offset, so a run longer than that offset is never recorded at all. Also reports, read LIVE from the catalog rather than from the recorded series, every retention policy the rollup-coverage gate is holding PAUSED, with the tier's actual data span and how many times its configured drop_after horizon it is really holding — a held policy records zero failures and a normal-looking last run, so it is invisible in the stored job telemetry and is a common cause of unexplained store growth. Use for capacity forecasting: what is driving store growth, how fast, what adding N servers would multiply, which background job is closest to outgrowing its own cadence, and whether retention is actually running.")] public static async Task GetStoreMetrics( NpgsqlDataSource postgres, [Description("Days of daily-series history. Default 30; max 400 (the series' own retention).")] int days_back = 30) @@ -97,8 +98,18 @@ the code does is worse than none. */ to. Read here rather than left to the caller because an empty job_history and a quiet fleet are the same result set, so a reader who follows the redirect cannot tell whether the answer they get back is a census or an artefact. Failure-isolated inside the reader, so this cannot - fail the response it qualifies. */ + fail the response it qualifies. + + #3574: and then the EVIDENCE behind that state — what this connection actually sees in the + view, and whether the view's ownership predicate would show it anything. The GUC answers "is + it on"; the reader's question is "will I see rows", and job_history filters by role membership + where the jobs view a reader checks first does not. On a managed store this very connection is + the least-privilege mcp role, which that predicate filters OUT, so the second read has to + evaluate the predicate for itself rather than count and assume. A second statement rather than + a column on the first: the two fail independently, and a count that timed out must not make + the GUC read unknown. */ var jobLogging = await DarlingStoreMetricsReader.GetJobExecutionLoggingAsync(postgres); + var jobEvidence = await DarlingStoreMetricsReader.GetJobHistoryEvidenceAsync(postgres, jobLogging); return JsonSerializer.Serialize(new { @@ -154,7 +165,29 @@ different readings of an empty job_history. */ setting = jobLogging.Setting, source = jobLogging.Source, source_file = jobLogging.SourceFile, - note = JobHistoryNote(jobLogging), + /* #3574. The evidence fields, all null unless evidence = Observed. reader_role first, + because every count that follows is a count through THAT role's eyes and the view + decides per role what it shows; visibility is the predicate's verdict on that role, + derived from the two membership facts beside it rather than asserted. rows_observed + travels with its window so the number never leaves without its denominator, and + jobs_run_in_window is the population half from the UNFILTERED job_stats view — the + proof that there was something to see. contradiction is the new finding class as one + bool, true only when all four of its conditions hold; the note spells them out. */ + evidence = jobEvidence.Status.ToString(), + reader_role = jobEvidence.ReaderRole, + reader_is_database_owner_member = jobEvidence.ReaderIsDatabaseOwnerMember, + visibility = jobEvidence.Status == DarlingStoreMetricsReader.JobHistoryEvidenceStatus.Observed + ? jobEvidence.Visibility.ToString() + : null, + job_count = jobEvidence.JobCount, + history_visible_job_count = jobEvidence.HistoryVisibleJobCount, + observed_window_hours = DarlingStoreMetricsReader.JobHistoryEvidenceWindowHours, + rows_observed = jobEvidence.RowsObserved, + newest_row_at = jobEvidence.NewestRowAt?.ToString("o", CultureInfo.InvariantCulture), + jobs_run_in_window = jobEvidence.JobsRunInWindow, + newest_run_started_at = jobEvidence.NewestRunStartedAt?.ToString("o", CultureInfo.InvariantCulture), + contradiction = jobEvidence.ContradictsRecording(jobLogging.Recording), + note = JobHistoryNote(jobLogging, jobEvidence), }, objects = latest .Where(r => r.ObjectKind != StoreSelfMetrics.StoreObjectKind) @@ -217,13 +250,24 @@ number an onboarding wave moves first. */ } /// - /// What an empty timescaledb_information.job_history means on THIS store (#3175). One sentence - /// per state, and the states deliberately do not share one: the whole defect is that "off" and "nothing - /// happened" produce the same empty result, so a note that hedged across both would reproduce it in - /// prose. Off splits again on whether anything SET it off, because the two need different - /// actions — one heals itself, the other needs an override removed. + /// What an empty timescaledb_information.job_history means on THIS store (#3175), and for WHOM + /// (#3574). Two halves, concatenated. The first is the GUC's: one sentence per state, and the states + /// deliberately do not share one — the whole defect is that "off" and "nothing happened" produce the + /// same empty result, so a note that hedged across both would reproduce it in prose; Off splits + /// again on whether anything SET it off, because the two need different actions (one heals itself, the + /// other needs an override removed). The second half is the evidence's, from + /// : the ownership rule the view enforces, which role this block + /// read as, what that role is allowed to see, what it saw, and whether the four conditions of the + /// contradiction hold. It is appended to every arm on which the view exists — including the GUC-off + /// arms, because a reader who heals the GUC and then checks as the wrong role walks into the same trap + /// one step later — and omitted only where there is no view to be filtered. /// - internal static string JobHistoryNote(DarlingStoreMetricsReader.JobExecutionLoggingReading reading) + internal static string JobHistoryNote( + DarlingStoreMetricsReader.JobExecutionLoggingReading reading, + DarlingStoreMetricsReader.JobHistoryEvidence evidence) + => GucNote(reading) + JobHistoryVisibilityNote(reading, evidence); + + private static string GucNote(DarlingStoreMetricsReader.JobExecutionLoggingReading reading) => reading.Status switch { DarlingStoreMetricsReader.JobExecutionLoggingStatus.On => @@ -266,4 +310,148 @@ internal static string JobHistoryNote(DarlingStoreMetricsReader.JobExecutionLogg + "timescaledb_information.job_history is recording is UNKNOWN — which is not the same as off. Treat " + "an empty job_history on this store as unexplained rather than as a clean result until this reads.", }; + + /// + /// The visibility rule, stated with a measurement (#3574). The rule itself is one sentence and it is the + /// same on every arm: history rows are visible only to members of the job's owner role or of the database + /// owner, and the jobs / job_stats views a reader checks first are NOT filtered, which is + /// the trap. What follows it depends on what the evidence read established about THIS reader: + /// + /// None — the managed-mode mcp role's reading on every store: the view shows this + /// connection nothing by construction, so its zero is the filter and not the table, and the note names + /// the role that can see and says what the unfiltered job_stats saw in the meantime. + /// All — the count is a census, and the flag is self-proving: rows seen means recording is + /// a measurement; zero rows with runs in the window and the GUC on is the CONTRADICTION, named as such, + /// with the one benign cause and how to settle it; zero rows with no runs proves nothing either way and + /// the note says so rather than calling it clean. + /// Partial — a census of the jobs this reader owns, stated as a fraction, no verdict. + /// Unknown after an Observed read — no jobs exist, so there is no owner to be a + /// member of; after an Unreadable one — the flag stays a GUC echo and the note says so. + /// + /// Empty for : the view + /// does not exist on that connection, and a rule about who may read a view that is not there would be + /// noise on the one arm whose existing text already says everything true. + /// + /// Counts are formatted invariant and timestamps as ISO 8601 UTC so the sentence a caller reads + /// agrees byte-for-byte with the fields beside it. + /// + internal static string JobHistoryVisibilityNote( + DarlingStoreMetricsReader.JobExecutionLoggingReading reading, + DarlingStoreMetricsReader.JobHistoryEvidence evidence) + { + const string rule = + " WHO CAN SEE THE ROWS: timescaledb_information.job_history shows a row only to a member of the job's " + + "owner role or of the database owner (its own WHERE clause: pg_has_role(current_user, , 'MEMBER') OR pg_has_role(current_user, , 'MEMBER')), while the jobs and " + + "job_stats views show every role every job — so a role that can list all the jobs can still read an " + + "empty history on a store that is recording perfectly, and a zero from a role outside those " + + "memberships contradicts nothing."; + + switch (evidence.Status) + { + case DarlingStoreMetricsReader.JobHistoryEvidenceStatus.NotApplicable: + return ""; + + case DarlingStoreMetricsReader.JobHistoryEvidenceStatus.Unreadable: + return rule + + " The evidence read behind this block did not complete, so what THIS connection sees in " + + "job_history is UNKNOWN and 'recording' above is the GUC's word alone, not a measurement."; + } + + var role = evidence.ReaderRole ?? "(unknown role)"; + var window = Invariant(DarlingStoreMetricsReader.JobHistoryEvidenceWindowHours); + var rows = Invariant(evidence.RowsObserved ?? 0); + var ran = Invariant(evidence.JobsRunInWindow ?? 0); + var newestRow = evidence.NewestRowAt is { } nr ? nr.ToString("o", CultureInfo.InvariantCulture) : null; + var newestRun = evidence.NewestRunStartedAt is { } ns ? ns.ToString("o", CultureInfo.InvariantCulture) : null; + + switch (evidence.Visibility) + { + case DarlingStoreMetricsReader.JobHistoryVisibility.None: + return rule + + $" This block read as '{role}', which is a member of NEITHER, so the view shows this connection " + + $"NOTHING by construction: rows_observed = {rows} here is the filter, not the table, and says " + + "nothing about whether recording works. The service's owner role — the role that created the " + + "jobs — sees the rows; read job_history as that role before concluding anything from an empty " + + $"result. Meanwhile job_stats, which is not filtered, says {ran} job(s) started a run in the " + + $"last {window} hours" + + (newestRun is null ? "." : $" (newest start {newestRun})."); + + case DarlingStoreMetricsReader.JobHistoryVisibility.Partial: + return rule + + $" This block read as '{role}', which is a member of the owner of " + + $"{Invariant(evidence.HistoryVisibleJobCount ?? 0)} of {Invariant(evidence.JobCount ?? 0)} jobs " + + "and not of the database owner, so the count is a census of THOSE jobs only: " + + $"{rows} row(s) with a start in the last {window} hours" + + (newestRow is null ? ", none ever." : $", newest {newestRow}.") + + " A job outside that set shows this role nothing, whatever it recorded."; + + case DarlingStoreMetricsReader.JobHistoryVisibility.All: + { + var standing = evidence.ReaderIsDatabaseOwnerMember == true + ? "a member of the database owner" + : "a member of every job's owner"; + var census = + $" This block read as '{role}', which is {standing}, so the count is a census: {rows} row(s) with a " + + $"start in the last {window} hours" + + (newestRow is null ? ", none ever" : $", newest {newestRow}") + + $"; job_stats says {ran} job(s) started a run in the same window" + + (newestRun is null ? "." : $" (newest start {newestRun})."); + + if (evidence.ContradictsRecording(reading.Recording)) + { + return rule + census + + " CONTRADICTION: the GUC says recording, this role can see every job's history, jobs " + + "started runs inside the window, and the view showed NONE of them — do not read this zero " + + "as quiet. The one benign cause is logging switched on AFTER the last of those starts (the " + + "managed v11 heal lands on a service-owned server start and writes nothing for earlier " + + "runs), which the next hourly run settles: re-read after it. A zero that persists while " + + "jobs_run_in_window climbs means the instrument is not writing what the GUC says it is, " + + "and that is a finding, not a quiet hour."; + } + + if (reading.Recording && evidence.RowsObserved is > 0) + { + return rule + census + + " 'recording' is therefore a measurement here, not a GUC echo."; + } + + if (reading.Recording) + { + /* Zero rows, zero runs: nothing happened for the instrument to catch, so the zero is not + evidence either way — and saying "clean" here would be the exact mis-reading #3175 and + #3574 exist to prevent, one level down. */ + return rule + census + + " No job started a run in the window, so there was nothing to record and this zero " + + "proves nothing either way; it is not a clean result, it is an absence of information."; + } + + if (reading.Status == DarlingStoreMetricsReader.JobExecutionLoggingStatus.Unreadable) + { + /* The GUC could not be read but the view could: the rows are real and this reader sees + them all, but whether they are the successes logging records or the failures TimescaleDB + writes regardless cannot be said without the setting — so it is not said. */ + return rule + census + + " Whether logging is on could not be read, so these rows are not classified: TimescaleDB " + + "writes a FAILED run's row regardless of the setting and a successful run's only while " + + "it is on."; + } + + /* GUC off, reader admitted: say what it will see once logging is on, and what it sees even + now — TimescaleDB writes a FAILED run's row regardless of the setting (the + bgw_job_stat_history_update path in 2.28.1 logs failures unconditionally), so a non-zero + count with the GUC off is failures, not a sign that logging is secretly on. */ + return rule + census + + " Once logging is on this connection will see the rows; any it sees now are FAILED runs, " + + "which TimescaleDB writes regardless of the setting — successes need it on."; + } + + default: + return rule + + $" This block read as '{role}'; timescaledb_information.jobs lists no jobs on this connection, " + + "so there is no owner to be a member of and nothing for history to record yet."; + } + } + + private static string Invariant(long value) => value.ToString(CultureInfo.InvariantCulture); } diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingStoreMetricsReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingStoreMetricsReader.cs index e7bd1d593..bd2ae3397 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingStoreMetricsReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingStoreMetricsReader.cs @@ -27,10 +27,12 @@ namespace PerformanceMonitor.Darling.Service.Mcp; /// per-server daily ingest rate (whole-store daily growth divided by the enabled-server count), computed in /// — pure, so it is unit-tested without a store. /// -/// Plus one read that is not a metric at all: asks whether +/// Plus two reads that are not metrics at all: asks whether /// timescaledb_information.job_history — the route the tool's description sends a maximum question -/// to, since this series cannot answer one — is actually recording (#3175). It qualifies the redirect, so -/// a caller who follows it can tell a census from an empty table. +/// to, since this series cannot answer one — is switched on (#3175), and +/// asks whether the connection doing the asking would SEE its rows and how many it does see (#3574). Together +/// they qualify the redirect, so a caller who follows it can tell a census from an empty table — and can +/// tell an empty table from a table the view is hiding from them. /// internal static class DarlingStoreMetricsReader { @@ -134,6 +136,121 @@ FROM collect.store_metrics FROM pg_settings WHERE name = $1"; + /// + /// The evidence behind : what THIS connection actually + /// sees in timescaledb_information.job_history, and whether the view would show it anything at all + /// (#3574). The GUC read above answers "is the instrument switched on"; the reader's real question + /// is "will I see its output", and between those two sits a role-membership filter nothing else on + /// this surface mentioned. + /// + /// THE FALSE-QUIET ARM #3175 DID NOT KNOW ABOUT: recording on, rows present, reader filtered. + /// job_history is a security_barrier view whose definition on TimescaleDB 2.28.1 ends with + /// + /// WHERE (pg_catalog.pg_has_role(current_user, + /// (SELECT pg_catalog.pg_get_userbyid(datdba) + /// FROM pg_catalog.pg_database + /// WHERE datname = current_database()), + /// 'MEMBER') IS TRUE + /// OR pg_catalog.pg_has_role(current_user, owner, 'MEMBER') IS TRUE); + /// + /// so a row is visible only to a member of the database owner's role or of the job's owner role + /// (_timescaledb_catalog.bgw_job.owner is regrole NOT NULL DEFAULT current_role — a job + /// belongs to whoever created it, which for every policy this product adds is the service's owner role). + /// The base table is not a back door: pre_install/tables.sql ends with + /// REVOKE ALL ON _timescaledb_internal.bgw_job_stat_history FROM PUBLIC, so the two filtered views + /// (job_history, job_errors) are the only way in. The trap is that the two views a reader + /// checks FIRST are not filtered. timescaledb_information.jobs has no WHERE clause at all and + /// job_stats has none either, so a role that can see all 110 jobs and every one of their + /// total_runs reasonably assumes it can see their history — and reads zero rows, forever, on a + /// store that is recording perfectly. Measured on a production store on 2.28.1: the GUC effective + /// on (sighup context, pending_restart = false), all 110 jobs owned by the service's owner + /// role, two independent reads as the least-privilege admin role counted 0 rows ever, and + /// the owner role's read of the same view returned every hourly run. That zero was declared "unknowable" + /// in a real postmortem before anyone read the view's definition. Recording was never broken and the GUC + /// never lied; the block just never said whose eyes the rows are visible to, and never proved rows exist. + /// + /// THIS READ IS ITSELF SUBJECT TO THE FILTER, and that is why it evaluates the predicate rather + /// than assuming it passes. In managed mode the MCP host connects as the dedicated least-privilege + /// mcp role (DarlingMcpHostService; DarlingManagedRoles grants it SELECT and a few + /// narrow writes, never membership in the owner role) — so on a managed store THIS connection is exactly + /// the kind of reader the view shows nothing to, and a bare count(*) here would have reported + /// rows_observed = 0 on every managed store and manufactured the very contradiction this issue is + /// about. The read therefore also returns current_user and evaluates the view's own two + /// pg_has_role tests for it: membership in the database owner (which sees everything) and, per + /// job, membership in that job's owner. From those the caller knows whether the count that follows is a + /// census, a partial census, or a zero the view produced by construction — and the note says which, in + /// so many words, naming the role. On a bring-your-own store whose connection string is the owner role + /// the count IS the census and the flag becomes self-proving; on a managed store the block says it + /// cannot see, and says who can, which is the sentence that would have ended the postmortem in a minute. + /// + /// The population half rides in the same statement, from the UNFILTERED view. A zero is + /// readable only when the instrument would have caught the event AND the event had a chance to occur, so + /// beside the history count the read takes, from job_stats, how many jobs started a run inside the + /// same window and the newest start it knows of. TimescaleDB writes the history row at job START when + /// the GUC is on (bgw_job_stat_history_mark_start inserts it with finish, pid and outcome NULL and + /// the finish updates it; a failure is written regardless of the GUC), so a job that started inside the + /// window while logging was on has a row with start_time inside the window — no waiting on a + /// finish. recording = on, a reader the predicate admits, zero rows, and jobs that started in the + /// window is the contradiction, and the note calls it one. It does not manufacture certainty about the + /// cause: logging switched on AFTER the last of those starts is the benign shape (the v11 heal lands on a + /// service-owned server start, and nothing before that point was written to recover), and the note says + /// how to settle it — re-read after the next hourly run — rather than pronouncing. + /// + /// A FIXED 24-HOUR WINDOW, not the tool's days_back. Three reasons, in order of + /// weight. The question this answers is CURRENT — is the instrument writing now — and a 30-to-400-day + /// forecasting window would let ten days of rows written after a heal hide a recording that stopped + /// yesterday. Every job this product schedules runs at least daily (CAGG refreshes and the compression + /// tick hourly, retention daily), so any 24-hour window on a live store contains starts, which is what + /// lets the job_stats count prove the population half instead of assuming it. And the view's + /// own Job History Log Retention Policy drops rows after one month by default, so a window past + /// that would count a table the retention job had already trimmed and call the trimming "no rows". + /// $1 is the window start. + /// + /// $1 IS BOUND AS timestamptz WITH Kind = Utc, WHICH IS THE INVERSE OF THIS + /// CODEBASE'S RULE, AND DELIBERATELY SO. Every collector column in the store is naive UTC and the + /// discipline everywhere else is to strip Kind before binding, because a timestamptz parameter against a + /// naive column makes PostgreSQL convert the naive side at the session's TimeZone. These columns are the + /// other way round: bgw_job_stat_history.execution_start and bgw_job_stat.last_start are + /// declared TIMESTAMPTZ in TimescaleDB's own catalog, so here a NAIVE bind would be the bug — the + /// parameter, not the column, would be converted at the session zone and the window would skew by the + /// host's offset. The parameter type is stated explicitly rather than inferred so the intent survives a + /// caller passing a DateTime of the wrong Kind. + /// + /// '-infinity' is TimescaleDB's never-ran sentinel in last_run_started_at (not NULL — + /// the lesson from #1760), so the newest start + /// NULLIFs it away; the window predicate needs no guard because -infinity >= $1 is simply false. + /// IS TRUE on each pg_has_role mirrors the view, whose second test can meet a NULL owner + /// (it LEFT JOINs the job catalog, so a history row whose job has since been deleted has none, and the + /// strict function yields NULL) — a NULL must read as "not a member" rather than poison a count. Here + /// the owner comes from the jobs view and cannot be NULL; the guard is kept so the two predicates + /// stay textually the view's own. + /// + public const string JobHistoryEvidenceSql = @" +SELECT + current_user::text AS reader_role, + pg_has_role( + current_user, + (SELECT pg_get_userbyid(datdba) FROM pg_database WHERE datname = current_database()), + 'MEMBER') IS TRUE AS reader_is_database_owner_member, + (SELECT count(*) FROM timescaledb_information.jobs) AS job_count, + (SELECT count(*) + FROM timescaledb_information.jobs AS j + WHERE pg_has_role(current_user, j.owner, 'MEMBER') IS TRUE) AS owner_member_job_count, + (SELECT count(*) + FROM timescaledb_information.job_history AS h + WHERE h.start_time >= $1) AS rows_observed, + (SELECT max(h.start_time) FROM timescaledb_information.job_history AS h) AS newest_row_at, + (SELECT count(*) + FROM timescaledb_information.job_stats AS js + WHERE js.last_run_started_at >= $1) AS jobs_run_in_window, + (SELECT max(NULLIF(js.last_run_started_at, '-infinity'::timestamptz)) + FROM timescaledb_information.job_stats AS js) AS newest_run_started_at"; + + /// The evidence window counts over, in hours. Fixed, not + /// days_back — the paragraph on that constant says why. Published in the response beside the + /// count so the number never travels without its denominator. + public const int JobHistoryEvidenceWindowHours = 24; + /// /// The four distinguishable states of the job_history precondition. Four rather than a bool /// because three of them would otherwise collapse into "not on", and the whole defect being reported @@ -244,6 +361,222 @@ and reporting that as Unreadable would put a measurement-shaped word on an act o } } + /// + /// Whether produced a reading (#3574). Three states, not a nullable + /// count, for the reason has four: a count this read did not + /// obtain, a count there was nothing to obtain, and a count of zero are three different facts about an + /// empty job_history, and the whole defect class is one of them being read as another. + /// + public enum JobHistoryEvidenceStatus + { + /// The read did not complete. Every evidence field is null and the flag stays a GUC echo + /// — reported as such, never as "zero rows". + Unreadable, + + /// Not attempted, because the GUC read said the view does not exist on this connection + /// (). A plain-PostgreSQL store has no + /// job_history to count, and a failed read of an absent view would report as Unreadable — + /// a fault-shaped word for a store that has no fault. + NotApplicable, + + /// The read completed. The counts are what this connection saw, and + /// says whether what it saw is what is there. + Observed, + } + + /// + /// How much of job_history the view's ownership predicate lets THIS reader see (#3574) — derived + /// from the two pg_has_role facts the read evaluates, never assumed. + /// + public enum JobHistoryVisibility + { + /// Not established: the evidence read did not complete, or there are no jobs to be a + /// member of the owner of. + Unknown, + + /// The reader is a member of neither the database owner nor any job's owner. The view + /// returns it NOTHING by construction, so a zero count here is the filter, not the table. This is + /// the managed-mode mcp role's reading on every store. + None, + + /// The reader is a member of some jobs' owners but not all, and not of the database owner. + /// The count is a census of those jobs only. + Partial, + + /// The reader is a member of the database owner, or of every job's owner. The count is a + /// census — the one state in which a zero says something about recording. + All, + } + + /// + /// One reading of : who read, what the view's predicate lets them + /// see, what they saw, and whether anything happened for them to see (#3574). One value, so a caller + /// cannot take the count and drop the role it was counted through — which is precisely the omission + /// this exists to close. + /// + /// Whether the read completed; every other field is null unless + /// . + /// current_user on the connection that counted. + /// The view's first pg_has_role test, evaluated for + /// this reader: membership in the database owner's role, which sees every row regardless of job owner. + /// Every job in timescaledb_information.jobs — the UNFILTERED view, so + /// this is what any role sees and the denominator the visibility fraction is stated against. + /// The view's second test, evaluated per job: how many jobs' owner + /// roles this reader is a member of. + /// History rows with a start inside the window that the view showed this + /// reader. A census only when is . + /// The newest history start the view showed this reader, over all time — + /// null when it showed none. UTC. + /// Jobs whose job_stats.last_run_started_at falls inside the + /// window — the population half, from the unfiltered view, so it holds whatever the reader's + /// visibility. + /// The newest job start job_stats knows of, over all time — + /// null when no job has ever run. UTC. + public sealed record JobHistoryEvidence( + JobHistoryEvidenceStatus Status, + string? ReaderRole, + bool? ReaderIsDatabaseOwnerMember, + long? JobCount, + long? OwnerMemberJobCount, + long? RowsObserved, + DateTime? NewestRowAt, + long? JobsRunInWindow, + DateTime? NewestRunStartedAt) + { + /// The reading for a connection on which the view does not exist — every field null, + /// status . + public static JobHistoryEvidence NotApplicable { get; } = + new(JobHistoryEvidenceStatus.NotApplicable, null, null, null, null, null, null, null, null); + + /// The reading for a read that did not complete — every field null, status + /// . + public static JobHistoryEvidence Unreadable { get; } = + new(JobHistoryEvidenceStatus.Unreadable, null, null, null, null, null, null, null, null); + + /// + /// How many jobs' history the view lets this reader see: every job when the reader is a member of + /// the database owner (the view's first test short-circuits the second), otherwise the per-job + /// membership count. Null unless observed. + /// + public long? HistoryVisibleJobCount => + Status != JobHistoryEvidenceStatus.Observed ? null + : ReaderIsDatabaseOwnerMember == true ? JobCount + : OwnerMemberJobCount; + + /// + /// The reader's standing under the view's predicate, derived from the two membership facts and the + /// job count. when the read did not complete or there are + /// no jobs — with nothing to be an owner of, "none" and "all" would both be vacuously true, and a + /// note built on either would be inventing a measurement. + /// + public JobHistoryVisibility Visibility + { + get + { + if (Status != JobHistoryEvidenceStatus.Observed || JobCount is not > 0) + { + return JobHistoryVisibility.Unknown; + } + + if (ReaderIsDatabaseOwnerMember == true) + { + return JobHistoryVisibility.All; + } + + return OwnerMemberJobCount switch + { + null or 0 => JobHistoryVisibility.None, + var n when n >= JobCount => JobHistoryVisibility.All, + _ => JobHistoryVisibility.Partial, + }; + } + } + + /// + /// The new finding class (#3574): the GUC says recording, the view admits this reader to every job's + /// history, jobs started runs inside the window, and the reader saw NO rows for them. True only when + /// all four hold — a zero read through a filtered role contradicts nothing, a zero with no runs in + /// the window proves nothing, and a zero with the GUC off is the #3175 arm, not this one. The caller + /// supplies because this record deliberately does not carry the GUC + /// reading; the two are read separately and fail separately. + /// + public bool ContradictsRecording(bool recording) => + recording + && Status == JobHistoryEvidenceStatus.Observed + && Visibility == JobHistoryVisibility.All + && RowsObserved == 0 + && JobsRunInWindow is > 0; + } + + /// + /// Reads over the window ending now and starting + /// ago. Failure-isolated to + /// , independently of the GUC read: the two are separate + /// statements on separate checkouts, so either can fail while the other answers, and a reading that + /// collapsed both into one status would report the GUC as unknown because a count timed out. Takes the + /// GUC reading only to skip the view a plain-PostgreSQL store does not have — the read is not attempted + /// for , and the response says NotApplicable rather + /// than dressing an absent view up as a failed read. No logger, for the reason + /// gives. + /// + public static async Task GetJobHistoryEvidenceAsync( + NpgsqlDataSource postgres, + JobExecutionLoggingReading logging, + CancellationToken cancellationToken = default) + { + if (logging is null) + { + throw new ArgumentNullException(nameof(logging)); + } + + if (logging.Status == JobExecutionLoggingStatus.NotRegistered) + { + return JobHistoryEvidence.NotApplicable; + } + + try + { + await using var command = postgres.CreateCommand(JobHistoryEvidenceSql); + command.CommandTimeout = McpCommandDeadlines.ReadSeconds; + + /* Kind = Utc and an EXPLICIT timestamptz, against timestamptz columns — the inverse of every + other bind on this surface, and the paragraph on JobHistoryEvidenceSql says why. Stated rather + than inferred so that a caller's DateTime of another Kind cannot quietly turn this into the + naive bind that would skew the window by the session zone. */ + var windowStartUtc = DateTime.SpecifyKind( + DateTime.UtcNow.AddHours(-JobHistoryEvidenceWindowHours), DateTimeKind.Utc); + command.Parameters.Add(new NpgsqlParameter + { + NpgsqlDbType = NpgsqlTypes.NpgsqlDbType.TimestampTz, + Value = windowStartUtc, + }); + + await using var reader = await command.ExecuteReaderAsync(cancellationToken); + if (!await reader.ReadAsync(cancellationToken)) + { + /* A single-row SELECT of scalar subqueries always yields one row; no row is a shape this + read does not understand, and Unreadable is the only honest word for it. */ + return JobHistoryEvidence.Unreadable; + } + + return new JobHistoryEvidence( + JobHistoryEvidenceStatus.Observed, + reader.IsDBNull(0) ? null : reader.GetString(0), + reader.IsDBNull(1) ? null : reader.GetBoolean(1), + reader.IsDBNull(2) ? null : reader.GetInt64(2), + reader.IsDBNull(3) ? null : reader.GetInt64(3), + reader.IsDBNull(4) ? null : reader.GetInt64(4), + reader.IsDBNull(5) ? null : DateTime.SpecifyKind(reader.GetDateTime(5), DateTimeKind.Utc), + reader.IsDBNull(6) ? null : reader.GetInt64(6), + reader.IsDBNull(7) ? null : DateTime.SpecifyKind(reader.GetDateTime(7), DateTimeKind.Utc)); + } + catch (Exception ex) when (ex is not OperationCanceledException) + { + /* Same cancellation discipline as the GUC read: a caller's own stop is not a measurement. */ + return JobHistoryEvidence.Unreadable; + } + } + /// One object's newest self-metrics row. The four job fields (#2136, V56) are non-null only /// on background_job rows — every other kind leaves them NULL, as the sweep writes them. public sealed record StoreMetricRow( diff --git a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs index fd84c8622..498f2b29b 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs @@ -2429,9 +2429,14 @@ internal static int RefreshPhaseMinutesFor(IReadOnlyList order, string v /// is not a fleet reading. #3175/#3177 has since given the GUC its own marker, so existing stores /// heal; that does not widen this population, because the read predates the heal. A later census /// could be broader, and would have to say so rather than inherit this one's scope. No SHIPPED read - /// touches job_history (every product surface uses job_stats, - /// deliberately — see ), so the gap is in what an investigation can - /// ask, not in what the product reports. Read at 2026-09-08 01:37Z, so the + /// takes a DURATION from job_history (every product duration surface uses job_stats, + /// deliberately — see ); the one shipped read that touches the view + /// at all is DarlingStoreMetricsReader.JobHistoryEvidenceSql (#3574), a bounded row COUNT that + /// proves the instrument is writing and evaluates the view's ownership filter for its own connection — + /// because job_history shows rows only to members of the job's owner or the database owner, a + /// census like this one must be read as such a role, and a zero read as any other role is the filter + /// and not the table. So the gap is in what an investigation can ask, not in what the product reports. + /// Read at 2026-09-08 01:37Z, so the /// window is one that has ENDED and stays true rather than a scope read against a clock a doc comment /// does not have. Each side of the boundary, since a bound is only as good as what it excludes: 304 /// runs at or before it, median 1081.7 s, maximum 13300.7 s; 57 runs after it, diff --git a/Darling/README.md b/Darling/README.md index b1a4b573c..da6bbec16 100644 --- a/Darling/README.md +++ b/Darling/README.md @@ -922,6 +922,8 @@ The first means TimescaleDB's own pool is full; the second means PostgreSQL's is SELECT sum(total_runs), sum(total_successes), sum(total_failures) FROM timescaledb_information.job_stats; ``` +**Before you read the per-run history by hand, know who it shows itself to.** The view behind those totals — `timescaledb_information.job_history`, one row per run with its start, finish, outcome and error — has two conditions `jobs` and `job_stats` do not. It records successful runs only while `timescaledb.enable_job_execution_logging` is on (the managed conf block turns it on; TimescaleDB writes a *failed* run's row regardless), and it is **ownership-filtered**: its own `WHERE` clause shows a row only to a member of the job's owner role or of the database owner, while `jobs` and `job_stats` show every role every job. So the `admin` and `viewer` roles — and the `mcp` role the MCP host reads as in managed mode — can list all the jobs and every one of their `total_runs`, then read an **empty** `job_history` on a store that is recording perfectly. That is not a broken instrument; it is the filter. The jobs are owned by the role that created them, which is the service's owner (`darling` in managed mode; whatever your connection string names in bring-your-own), so read `job_history` as that role before concluding anything from a zero. `get_store_metrics`' `job_history` block reports which role it read as, whether the view's predicate lets that role see anything (`visibility`), and the rows it actually saw over the last 24 hours beside the jobs that started a run in the same window — on a managed store that block reads as `mcp` and says so, and only an owner-standing connection makes the count a census. + ### Retention A purge runs on the first sweep after startup and then daily, and it does not draw from a single source. Every **collector** table takes its horizon from the shared per-collector `CollectorScheduleDefaults` Lite also uses (fleet-wide overrides layer on top); the non-collector tables purged alongside them each carry their own horizon, listed beneath the table. From 552a7bf821ddd67e9ca93ee7e28411b001bbc536 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 11:22:44 -0400 Subject: [PATCH 22/69] The forced-plan-failures alert read gets the covering index the planner will actually take, instead of walking the whole fleet's two-hour slice per server (#3573) (#3586) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The read's (server_id, collection_time) composite was on the production store all along — V1 generates it — and the planner priced it out: server_id's physical correlation is 0.022, so the cost model charged one random page per row and preferred streaming the entire fleet's two-hour slice through the time index (Rows Removed by Filter: 691,058 to keep 37,878; 57,307 buffers; 422 ms warm, 10.3 s cold against a 10 s deadline). Forced under random_page_cost = 1.1 the same statement took the existing composite at 5,063 buffers, so a second plain composite would have been priced and ignored identically. Covering removes the heap term the model got wrong: (server_id, collection_time DESC) INCLUDE every other column the read touches, so it plans as an Index Only Scan over one server's rows (rig: 50 buffers vs 1,514, Heap Fetches: 0). It lives in PgTableTuning, not the ladder — results-invariant perf, autocommit, failure-isolated — as a plain CREATE INDEX: on TimescaleDB 2.28.1 compressed chunks get an 8 KB shell so only the live chunks carry it; transaction_per_chunk was measured leaving an invalid parent index after a mid-build cancel that IF NOT EXISTS then skips forever; CONCURRENTLY is refused on hypertables. Tests pin the index's column list against the read's own column references, the statement text against that list, the two rejected shapes, and (DARLING_TEST_PG) EXPLAIN the shipped statement on a seeded store for the Index Only Scan. The 1,744.9 ms deadline anchors keep their history and gain the arc; the 10 s deadline does not move. --- .../ForcePlanFailuresAccessPathTests.cs | 284 ++++++++++++++++++ Darling/Darling.Tests/PgTableTuningTests.cs | 16 +- .../DarlingAlertReadAdapter.cs | 33 +- .../ServiceCommandDeadlines.cs | 15 +- .../PgTableTuning.cs | 101 ++++++- 5 files changed, 437 insertions(+), 12 deletions(-) create mode 100644 Darling/Darling.Tests/ForcePlanFailuresAccessPathTests.cs diff --git a/Darling/Darling.Tests/ForcePlanFailuresAccessPathTests.cs b/Darling/Darling.Tests/ForcePlanFailuresAccessPathTests.cs new file mode 100644 index 000000000..01d407624 --- /dev/null +++ b/Darling/Darling.Tests/ForcePlanFailuresAccessPathTests.cs @@ -0,0 +1,284 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Globalization; +using System.Linq; +using System.Text; +using System.Text.RegularExpressions; +using System.Threading; +using System.Threading.Tasks; +using Npgsql; +using PerformanceMonitor.Darling.Service; +using PerformanceMonitor.Darling.Storage; +using Xunit; + +namespace Darling.Tests; + +/// +/// #3573: the alerting pass's forced-plan-failures read and the covering index that gives it an access path +/// the planner will actually take. +/// +/// What went wrong is not that an index was missing. V1 generates +/// idx_query_store_stats_time (server_id, collection_time), the exact composite the read's predicate +/// wants, and the production catalog carried it on the hypertable and on the chunk in the failing plan. The +/// planner priced it out: server_id has near-zero physical correlation (43 servers interleaved by +/// collection pass), so the cost model charged one random page per tuple for the composite's heap fetches +/// and preferred streaming the entire fleet's two-hour slice through the perfectly-correlated time index +/// and filtering 95% of it away. Forced under random_page_cost = 1.1 the same statement took the +/// composite and touched 5,063 buffers instead of 57,307 \u2014 the composite was right all along. A second plain +/// composite would have been priced, and ignored, identically. +/// +/// Covering is the fix because it deletes the term the cost model got wrong. With every column +/// the read touches in the key or INCLUDE, the plan is an Index Only Scan with no heap component to misprice, +/// at any random_page_cost and at any share of the fleet the busiest server grows into. That makes the +/// INCLUDE list load-bearing in a way most index definitions are not: a column added to the read and not to +/// the index does not fail anything \u2014 it silently hands the read back to the fleet-wide plan. The ungated pins +/// below hold the two against each other from source; the gated one asks the planner. +/// +/* Live-fixture tests share one Postgres store; the collection serializes them so cross-test row churn + cannot race another class's assertions. */ +[Collection("live-postgres")] +public sealed class ForcePlanFailuresAccessPathTests +{ + /// Distinctive fake ids \u2014 a real server_id is a storage-name hash, never these. The target is + /// one of six so its rows are ~17% of the seeded chunk, a share at which a seq scan is not competitive. + private const int TargetServerId = -735730; + private static readonly int[] OtherServerIds = { -735731, -735732, -735733, -735734, -735735 }; + private const string TestServerName = "force-plan-access-path-e2e"; + + /// + /// Every qs.<column> the shipped statement references, read from the statement itself. The + /// alias is fixed by the SQL, so this is the complete set of columns the scan must produce. + /// + private static IReadOnlyCollection ColumnsTheReadReferences() + { + return Regex.Matches(DarlingAlertReadAdapter.ForcePlanFailuresSql, @"\bqs\.([a-z_]+)") + .Select(m => m.Groups[1].Value) + .Distinct(StringComparer.Ordinal) + .OrderBy(c => c, StringComparer.Ordinal) + .ToList(); + } + + /// + /// The read's column references and the index's column list are the SAME set, both ways. A column the + /// read touches that the index lacks degrades the Index Only Scan to the heap plan it replaced, silently; + /// a column the index carries that the read no longer touches is dead weight on every insert into the + /// largest table in the store (INCLUDE disables deduplication, so each is real bytes per row). + /// + [Fact] + public void TheCoveringIndex_CarriesExactlyTheColumnsTheReadReferences() + { + var referenced = ColumnsTheReadReferences(); + var carried = PgTableTuning.ForcePlanFailuresIndexColumns.OrderBy(c => c, StringComparer.Ordinal).ToList(); + + Assert.Equal(carried, referenced); + + /* The predicate columns are the KEY, in predicate order: the equality column first so one server's + rows are one contiguous index range, the range column second. Everything else is INCLUDE. */ + Assert.Equal("server_id", PgTableTuning.ForcePlanFailuresIndexColumns[0]); + Assert.Equal("collection_time", PgTableTuning.ForcePlanFailuresIndexColumns[1]); + } + + /// + /// The statement text is BUILT from the same column list the pin above holds, so the two cannot drift: the + /// literal SQL carries the key as (server_id, collection_time DESC) and the remaining columns, in + /// order, as INCLUDE. Also pins the decisions the rig measured (see ): idempotent + /// IF NOT EXISTS; NOT CONCURRENTLY, which hypertables refuse; NOT + /// timescaledb.transaction_per_chunk, whose mid-build cancel leaves an invalid parent index that the + /// idempotent re-run then skips forever; and collect.-qualified like every neighbour. + /// + [Fact] + public void TheStatement_IsBuiltFromTheColumnList_AndTakesNeitherMeasuredTrap() + { + var statements = PgTableTuning.Statements + .Where(s => s.Contains(PgTableTuning.ForcePlanFailuresIndexName, StringComparison.Ordinal)) + .ToList(); + var statement = Assert.Single(statements); + + var columns = PgTableTuning.ForcePlanFailuresIndexColumns; + var expected = + "CREATE INDEX IF NOT EXISTS " + PgTableTuning.ForcePlanFailuresIndexName + + " ON collect.query_store_stats (" + columns[0] + ", " + columns[1] + " DESC)" + + " INCLUDE (" + string.Join(", ", columns.Skip(2)) + ")"; + Assert.Equal(expected, statement); + + Assert.DoesNotContain("CONCURRENTLY", statement, StringComparison.OrdinalIgnoreCase); + Assert.DoesNotContain("transaction_per_chunk", statement, StringComparison.OrdinalIgnoreCase); + Assert.DoesNotContain(" WHERE ", statement, StringComparison.Ordinal); /* not partial \u2014 see the statement's remarks */ + } + + /// + /// The evidence no string pin can give: that the planner TAKES the index for the shipped statement. The + /// production failure was a plan choice, not a missing object \u2014 the right composite was in the catalog and + /// the plan walked past it \u2014 so a test that only checked pg_indexes would have passed on the broken + /// store. This builds the store the way the service does (ladder, then the hypertable conversion where + /// TimescaleDB is present, then ), seeds six servers' rows in the + /// collector's per-pass contiguous batches, and EXPLAINs the exact shipped SQL with its real bound + /// parameters: the plan must be an Index Only Scan on the covering index and must NOT be the time-index + /// scan filtering on server_id that the issue's plan showed. + /// + [Fact] + public async Task TheShippedRead_PlansAsAnIndexOnlyScanOnTheCoveringIndex_AgainstDevPostgres() + { + var connectionString = Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + Assert.SkipWhen(string.IsNullOrEmpty(connectionString), + "Set DARLING_TEST_PG to a Postgres connection string to run the live access-path test."); + + var ct = TestContext.Current.CancellationToken; + + using var connection = new NpgsqlConnection(connectionString); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + + /* The service's own order: hypertables first (where the extension exists), so the index is created on + a hypertable and propagates to chunks, then the tuning list. On a plain-PostgreSQL store the index is + an ordinary btree and the plan assertion below holds the same way. #1922: probe on its own connection. */ + var timescaleEnabled = await LiveTimescaleProbe.TryEnableAsync(connectionString!, ct); + if (timescaleEnabled) + { + await TimescaleSupport.ConvertToHypertablesAsync(connection, null, ct); + } + + await PgTableTuning.ApplyAsync(connection, null, ct); + + var bodySucceeded = false; + try + { + await DeleteTestRowsAsync(connection, ct); + + /* The index exists on the hypertable with the shipped definition. */ + using (var indexDef = new NpgsqlCommand( + "SELECT indexdef FROM pg_indexes WHERE schemaname = 'collect' AND tablename = 'query_store_stats' AND indexname = $1", connection)) + { + indexDef.Parameters.AddWithValue(PgTableTuning.ForcePlanFailuresIndexName); + var def = await indexDef.ExecuteScalarAsync(ct) as string; + Assert.NotNull(def); + Assert.Contains("(server_id, collection_time DESC)", def, StringComparison.Ordinal); + Assert.Contains("INCLUDE (" + string.Join(", ", PgTableTuning.ForcePlanFailuresIndexColumns.Skip(2)) + ")", def, StringComparison.Ordinal); + } + + /* Eight passes fifteen minutes apart, all inside the read's two-hour window; each pass writes the six + servers in turn, each server's batch contiguous \u2014 the write pattern that makes server_id's + correlation near zero, which is the condition the production plan was chosen under. All + Kind-Unspecified: naive-UTC storage, see PgCollectorRowWriter. */ + var utcNow = DateTime.SpecifyKind(DateTime.UtcNow, DateTimeKind.Unspecified); + for (var pass = 7; pass >= 0; pass--) + { + var collectionTime = utcNow.AddMinutes(-2 - pass * 15); + foreach (var serverId in OtherServerIds.Take(3).Append(TargetServerId).Concat(OtherServerIds.Skip(3))) + { + /* The target's forced plan climbs one failure per pass (pass 7 = 0 ... pass 0 = 7). */ + var forcedFailures = serverId == TargetServerId ? 7L - pass : (long?)null; + await SeedPassAsync(connection, serverId, collectionTime, rows: 400, forcedFailures, ct); + } + } + + /* The planner needs the visibility map current (the product keeps it so with the insert-autovacuum + override; a test cannot wait for autovacuum) and statistics for the rows just written. */ + using (var vacuum = new NpgsqlCommand("VACUUM ANALYZE collect.query_store_stats", connection)) + { + await vacuum.ExecuteNonQueryAsync(ct); + } + + var plan = await ExplainShippedReadAsync(connection, utcNow - DarlingAlertReadAdapter.ForcePlanFailureWindow, ct); + + /* The chunk copy of a hypertable index is named _, truncated to 63 characters, so match + on the name's stable prefix rather than its whole. */ + Assert.Contains("Index Only Scan", plan, StringComparison.Ordinal); + Assert.Contains("idx_query_store_stats_server_time", plan, StringComparison.Ordinal); + Assert.DoesNotContain("collection_time_idx", plan, StringComparison.Ordinal); + /* The issue's signature: server_id applied as a post-scan Filter (any alias, any parenthesisation) + rather than inside the Index Cond. */ + Assert.False(Regex.IsMatch(plan, @"Filter: \(+(qs(_\d+)?\.)?server_id"), + "server_id is being applied as a Filter after the scan — the fleet-wide plan is back:\n" + plan); + + /* And the read still answers correctly through the new path: the target's forced plan's counter + rose between its two newest sightings (pass 1 -> pass 0), and nothing else did. */ + await using var postgres = NpgsqlDataSource.Create(connectionString!); + var adapter = new DarlingAlertReadAdapter(postgres); + var failures = await adapter.GetForcePlanFailuresAsync(TargetServerId.ToString(CultureInfo.InvariantCulture), ct); + var failure = Assert.Single(failures); + Assert.Equal("ForcedDb", failure.DatabaseName); + Assert.Equal(1L, failure.QueryId); + Assert.Equal(10L, failure.PlanId); + Assert.Equal(1L, failure.FailureDelta); + Assert.Equal(7L, failure.TotalFailures); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(connectionString!, bodySucceeded, async (cleanup, cleanupCt) => + await DeleteTestRowsAsync(cleanup, cleanupCt)); + } + } + + /// + /// One server's batch for one pass: ordinary rows plus, when + /// is given, one forced plan carrying that force_failure_count, so + /// the caller can make the newest two sightings differ by exactly one. Written as a single multi-row INSERT + /// so the batch lands as one contiguous run of heap pages, the way the collector's COPY does. + /// + private static async Task SeedPassAsync(NpgsqlConnection connection, int serverId, DateTime collectionTime, int rows, long? forcedFailures, CancellationToken ct) + { + var sql = new StringBuilder( + "INSERT INTO collect.query_store_stats (collection_id, collection_time, server_id, server_name, database_name, query_id, plan_id, " + + "execution_count, avg_duration_us, plan_forcing_type, is_forced_plan, force_failure_count, last_force_failure_reason) " + + "SELECT $1, $2, $3, $4, 'db_' || (g % 4), 1000 + g, (1000 + g) * 10, 10 + g, 500 + g, 'NONE', FALSE, 0, NULL " + + "FROM generate_series(1, $5) AS g"); + using (var insert = new NpgsqlCommand(sql.ToString(), connection)) + { + insert.Parameters.AddWithValue(1L); + insert.Parameters.AddWithValue(collectionTime); + insert.Parameters.AddWithValue(serverId); + insert.Parameters.AddWithValue(TestServerName); + insert.Parameters.AddWithValue(rows); + await insert.ExecuteNonQueryAsync(ct); + } + + if (forcedFailures is null) + { + return; + } + + using var forced = new NpgsqlCommand( + "INSERT INTO collect.query_store_stats (collection_id, collection_time, server_id, server_name, database_name, query_id, plan_id, " + + "execution_count, avg_duration_us, plan_forcing_type, is_forced_plan, force_failure_count, last_force_failure_reason) " + + "VALUES ($1, $2, $3, $4, 'ForcedDb', 1, 10, 5, 900, 'MANUAL', TRUE, $5, 'GENERAL_FAILURE')", connection); + forced.Parameters.AddWithValue(1L); + forced.Parameters.AddWithValue(collectionTime); + forced.Parameters.AddWithValue(serverId); + forced.Parameters.AddWithValue(TestServerName); + forced.Parameters.AddWithValue(forcedFailures.Value); + await forced.ExecuteNonQueryAsync(ct); + } + + private static async Task ExplainShippedReadAsync(NpgsqlConnection connection, DateTime windowStart, CancellationToken ct) + { + using var explain = new NpgsqlCommand("EXPLAIN (COSTS OFF) " + DarlingAlertReadAdapter.ForcePlanFailuresSql, connection); + explain.Parameters.AddWithValue(TargetServerId); + explain.Parameters.AddWithValue(DateTime.SpecifyKind(windowStart, DateTimeKind.Unspecified)); + var plan = new StringBuilder(); + using var reader = await explain.ExecuteReaderAsync(ct); + while (await reader.ReadAsync(ct)) + { + plan.AppendLine(reader.GetString(0)); + } + + return plan.ToString(); + } + + private static async Task DeleteTestRowsAsync(NpgsqlConnection connection, CancellationToken ct) + { + var ids = string.Join(", ", OtherServerIds.Append(TargetServerId).Select(id => id.ToString(CultureInfo.InvariantCulture))); + using var cleanup = new NpgsqlCommand($"DELETE FROM collect.query_store_stats WHERE server_id IN ({ids})", connection); + await cleanup.ExecuteNonQueryAsync(ct); + } +} diff --git a/Darling/Darling.Tests/PgTableTuningTests.cs b/Darling/Darling.Tests/PgTableTuningTests.cs index 4e0a5d15d..ea9642c01 100644 --- a/Darling/Darling.Tests/PgTableTuningTests.cs +++ b/Darling/Darling.Tests/PgTableTuningTests.cs @@ -17,7 +17,9 @@ namespace Darling.Tests; /// Pins the composer performance-tuning statements (covering indexes + per-table autovacuum-insert override, /// Erik's EXPLAIN-backed field fix) so the tested SQL can never silently drift. These are applied as idempotent /// RUNTIME setup (), NOT a versioned migration, so they do not bump StorageVersion or -/// gate the Viewer — the reason there is no schema-version change to pin here. +/// gate the Viewer — the reason there is no schema-version change to pin here. The #3573 alerting-read covering +/// index rides the same list for the same reason: results-invariant, so the Viewer's connect gate must not +/// learn about it. /// public sealed class PgTableTuningTests { @@ -39,8 +41,16 @@ LATERAL probes (server_id, sql_handle, newest-first — bounded by raw retention Assert.Contains("idx_query_stats_server_hash_time ON collect.query_stats (server_id, query_hash, collection_time DESC)", sql, StringComparison.Ordinal); Assert.Contains("idx_query_store_stats_server_db_query_plan_time ON collect.query_store_stats (server_id, database_name, query_id, plan_id, collection_time DESC)", sql, StringComparison.Ordinal); + /* #3573: the alerting pass's forced-plan-failures read gets the fourth COVERING index — the read's + (server_id, collection_time) predicate as the key and every other column it touches as INCLUDE, so + it runs as an Index Only Scan. The plain (server_id, collection_time) composite V1 already generates + was measured on the production store being priced out by the planner in favour of streaming the + fleet's whole two-hour slice; covering is what removes the heap component the cost model mispriced. + The column list is pinned against the read itself in ForcePlanFailuresAccessPathTests. */ + Assert.Contains("idx_query_store_stats_server_time_forcing ON collect.query_store_stats (server_id, collection_time DESC) INCLUDE (database_name, query_id, plan_id, force_failure_count, is_forced_plan, plan_forcing_type, last_force_failure_reason)", sql, StringComparison.Ordinal); + /* Every index is idempotent (no-op where a field box already hand-applied it, or a prior start made it). */ - Assert.Equal(7, CountOccurrences(sql, "CREATE INDEX IF NOT EXISTS")); /* +1: the #1981 handle index */ + Assert.Equal(8, CountOccurrences(sql, "CREATE INDEX IF NOT EXISTS")); /* +1: the #1981 handle index; +1: the #3573 forced-plan covering index */ Assert.DoesNotContain("CREATE INDEX ON", sql, StringComparison.Ordinal); /* Per-table autovacuum-insert override on exactly the FOUR high-rate insert tables (NOT a global GUC @@ -59,7 +69,7 @@ differ by one word and the wrong one would be silently inert. */ Assert.Contains("ALTER TABLE collect.query_plan_dim SET (autovacuum_vacuum_scale_factor = 0.02, autovacuum_vacuum_threshold = 10000)", sql, StringComparison.Ordinal); Assert.DoesNotContain("collect.query_plan_dim SET (autovacuum_vacuum_insert_scale_factor", sql, StringComparison.Ordinal); - Assert.Equal(12, PgTableTuning.Statements.Count); /* +1 #1981 query_stats handle index, +1 pg_statement_stats, +1 #2402 query_plan_dim */ + Assert.Equal(13, PgTableTuning.Statements.Count); /* +1 #1981 query_stats handle index, +1 pg_statement_stats, +1 #2402 query_plan_dim, +1 #3573 forced-plan covering index */ } /// diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingAlertReadAdapter.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingAlertReadAdapter.cs index 3cea4b118..fa21be745 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingAlertReadAdapter.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingAlertReadAdapter.cs @@ -62,10 +62,21 @@ public sealed class DarlingAlertReadAdapter : IAlertReadAdapter /// one read — the forced-plan check at 1,744.9 ms cold, scanning ~6.0 GB of /// query_store_stats; every other read in the family lands under 3 ms. Ten seconds is 5.7x /// that worst case, so it absorbs a substantial stall rather than only the happy path. That margin - /// has been measured being eaten once: the collection-signals read's whole-history top-N sort grew - /// with the 90-day retention fill until its cold excursions clocked ~12 s against this deadline — - /// the first measured breach — and #3496 restored the margin by making that read chunk-orderable - /// rather than by moving this number. + /// has been measured being eaten twice, and both times restored by fixing the read rather than by + /// moving this number. First the collection-signals read's whole-history top-N sort grew with the + /// 90-day retention fill until its cold excursions clocked ~12 s — the first measured breach — and + /// #3496 made that read chunk-orderable. Then the 1,744.9 ms read itself: the ~6.0 GB it was measured + /// over became 23 GB, and its plan turned out to have been paying a fleet-width tax the whole time — + /// the planner walked the entire fleet's two-hour slice through the time index and filtered 95% of it + /// away per server (Rows Removed by Filter: 691,058, 57,307 buffers) rather than take the + /// (server_id, collection_time) composite that was already there — so the cold tail crossed + /// this deadline at 10.3 s while the fixed #3496 site sat at zero (#3573). The covering index in + /// PgTableTuning made that read an Index Only Scan over one server's rows (50 buffers against + /// 1,514 for the same statement on the rig; see ). The 1,744.9 ms + /// figure therefore stands as the measured floor this number was derived from and as the recorded + /// cost of the access path that has since been replaced, not as a current cost — and 10 s stays: the + /// cadence bound below has not moved, and a deadline re-fitted to a read that now costs milliseconds + /// would only mean the next drift is caught later. /// /// Bounded above by the cadence this pass runs on: s_alertSweepInterval is /// 30 s, so one stalled read must still leave the pass able to finish inside the interval that @@ -1092,6 +1103,20 @@ one carried over from a restart (#2166). A database cleared here is one that is /// /// The > comparison is what makes this a delta read: equal counters are silence, and a /// LOWER counter (unforce/re-force reset) is silence too rather than a negative delta. + /// + /// The access path is a covering index, and the column list here is what it covers (#3573). + /// PgTableTuning.ForcePlanFailuresIndexName is (server_id, collection_time DESC) INCLUDE + /// every other column this statement touches, so it runs as an Index Only Scan over one server's two + /// hours. That is not a nicety. V1's plain (server_id, collection_time) composite was on the + /// production store the whole time and the planner refused it: server_id's physical correlation is + /// ~0.02 (forty-three servers interleaved by collection pass), so the cost model priced the composite's + /// heap fetches as one random page per row and preferred streaming the ENTIRE fleet's two-hour slice through + /// the time index — Rows Removed by Filter: 691,058 to keep 37,878, 57,307 buffers, 422 ms warm and + /// a 10.3 s cold tail against the 10 s deadline. Covering deletes the heap term the model got wrong. The + /// consequence for anyone editing this SQL: reference a column of qs that the index does not carry + /// and nothing fails — the plan silently reverts to the fleet-wide scan. ForcePlanFailuresAccessPathTests + /// pins the two lists against each other and EXPLAINs the shipped statement against a live store; add the + /// column to the index in the same change or that test tells you. /// public const string ForcePlanFailuresSql = @" WITH per_collection AS ( diff --git a/Darling/PerformanceMonitor.Darling.Service/ServiceCommandDeadlines.cs b/Darling/PerformanceMonitor.Darling.Service/ServiceCommandDeadlines.cs index 5f17c5614..f4d871862 100644 --- a/Darling/PerformanceMonitor.Darling.Service/ServiceCommandDeadlines.cs +++ b/Darling/PerformanceMonitor.Darling.Service/ServiceCommandDeadlines.cs @@ -86,7 +86,11 @@ public static class ServiceCommandDeadlines /// /// It happens to land on the same 10 s the alert pass uses. That is two derivations meeting, /// not a number being reused: the alert pass is bounded above by its own 30 s sweep interval and - /// below by a 1,744.9 ms forced-plan read, neither of which appears anywhere above. + /// below by a 1,744.9 ms forced-plan read, neither of which appears anywhere above. (That read has + /// since been re-pathed — #3573 found it streaming the whole fleet's window per server, and the + /// covering index in PgTableTuning made it an Index Only Scan — so its 1,744.9 ms is now the + /// recorded cost of a replaced plan rather than a live figure. Nothing here rested on it, which is + /// the point of this paragraph.) /// public const int CollectionSweepSeconds = 10; @@ -129,7 +133,14 @@ public static class ServiceCommandDeadlines /// deliberately pathological 2,925-row window (the volume a per-query cooldown of zero would produce). /// Those figures are from a local container, so the floor is anchored on the one COLD store read this /// sweep has measured in production instead: #2882's forced-plan read at 1,744.9 ms over ~6.0 GB. - /// A value under ~2 s could fire on a cold read; 5 s carries ~2.9x over it. + /// A value under ~2 s could fire on a cold read; 5 s carries ~2.9x over it. That anchor has since aged + /// in both directions and still holds as a floor: the table it was measured over grew to 23 GB and the + /// read's cold tail reached 10.3 s (#3573) — because its plan walked the entire fleet's two-hour slice and + /// filtered one server out of it, a fleet-width tax that scales with servers, not with this table — and + /// then the covering index in PgTableTuning made it an Index Only Scan over one server's rows. So + /// 1,744.9 ms is the recorded cost of a cold store read under an access path that no longer exists; as + /// a stand-in for "what a cold store read can cost here" it is, if anything, generous, and the + /// arithmetic above is unchanged. /// /// So the CEILING is what fixes this number, not the floor. The pass budget puts it at /// 6 s or less and the floor only rules out the bottom two, which is worth saying plainly rather than diff --git a/Darling/PerformanceMonitor.Darling.Storage/PgTableTuning.cs b/Darling/PerformanceMonitor.Darling.Storage/PgTableTuning.cs index cc88a2be8..f07ad6d7c 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/PgTableTuning.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/PgTableTuning.cs @@ -38,6 +38,12 @@ namespace PerformanceMonitor.Darling.Storage; /// rollover, degrading the Index Only Scan back to heap fetches). Verified: the two panels that timed out at 15 s /// then ran in 139 ms / 514 ms with 0 heap fetches on vacuumed chunks. DarlingStoredPlanReader and ComposeCompiler /// are unchanged — correct as-is, they just needed these indexes + the vacuum state to perform. +/// +/// The alerting pass joined the composer here (#3573), for the same reason and with the same +/// EXPLAIN-backed shape: DarlingAlertReadAdapter.ForcePlanFailuresSql takes the fourth covering index +/// below, an Index Only Scan replacing a plan that streamed the whole fleet's two-hour slice of +/// query_store_stats to keep one server's rows. Its derivation is on the statement itself — including +/// why it is here and not a ladder rung, and why it is a plain CREATE INDEX and not the per-chunk form. /// public static class PgTableTuning { @@ -46,13 +52,30 @@ public static class PgTableTuning conversion uses. */ private const int SetupTimeoutSeconds = 300; + /// + /// The #3573 covering index's name, and the columns it carries in key-then-INCLUDE order. Exposed so the + /// pin (ForcePlanFailuresAccessPathTests) can hold the read's column references against THIS list + /// rather than against a second copy of the statement text, and so the live test can find the index by + /// name in pg_indexes. The first two are the key; the rest are INCLUDE. Every column the read + /// references must appear here or the Index Only Scan silently degrades to the heap plan it replaced. + /// + public const string ForcePlanFailuresIndexName = "idx_query_store_stats_server_time_forcing"; + + public static IReadOnlyList ForcePlanFailuresIndexColumns { get; } = new[] + { + "server_id", "collection_time", + "database_name", "query_id", "plan_id", "force_failure_count", "is_forced_plan", "plan_forcing_type", "last_force_failure_reason", + }; + /// /// The idempotent tuning statements, applied in order, each on its own command (failure-isolated). Three /// COVERING composer indexes (INCLUDE the aggregate columns the Procedures / Queries / Query Store measures /// SUM/AVG, for an Index Only Scan), three (server_id, handle/hash/id, collection_time DESC) lookup - /// indexes for the single-row analyze_*_plan reads (no INCLUDE — one heap fetch is cheap), then the per-table - /// autovacuum-insert override on exactly the four growing tables. Bare collect-qualified names; every - /// identifier is a compile-time constant, never user input, so interpolation is not a concern. + /// indexes for the single-row analyze_*_plan reads (no INCLUDE — one heap fetch is cheap), the #3573 + /// covering index for the alerting pass's forced-plan-failures read (INCLUDE exactly that read's columns, + /// for the same Index Only Scan), then the per-table autovacuum-insert override on exactly the four growing + /// tables. Bare collect-qualified names; every identifier is a compile-time constant, never user input, so + /// interpolation is not a concern. /// public static IReadOnlyList Statements { get; } = new[] { @@ -66,6 +89,78 @@ above. Bounded by raw retention (4 days of chunks), so the build is cheap on any "CREATE INDEX IF NOT EXISTS idx_query_stats_server_hash_time ON collect.query_stats (server_id, query_hash, collection_time DESC)", "CREATE INDEX IF NOT EXISTS idx_query_store_stats_query_hash ON collect.query_store_stats (query_hash, collection_time) INCLUDE (database_name, module_name, execution_count, avg_duration_us, max_duration_us, avg_cpu_time_us, max_cpu_time_us)", "CREATE INDEX IF NOT EXISTS idx_query_store_stats_server_db_query_plan_time ON collect.query_store_stats (server_id, database_name, query_id, plan_id, collection_time DESC)", + /* #3573: the alerting pass's forced-plan-failures read (DarlingAlertReadAdapter.ForcePlanFailuresSql, + WHERE server_id = $1 AND collection_time > $2, a two-hour window, per server, every 30 s pass) outgrew + its 10 s deadline on the largest production store: 1,744.9 ms cold when the deadline was derived over + ~6 GB of query_store_stats, 10.3 s excursions at 23 GB. The live plan named the mechanism — + + Index Scan Backward using _hyper_.._chunk_query_store_stats_collection_time_idx + Index Cond: (collection_time > now() - '02:00:00') + Filter: (server_id = ...) Rows Removed by Filter: 691,058 actual rows: 37,878 + Buffers: shared hit=54,664 read=2,643 + + — the read walks the ENTIRE fleet's two-hour slice through the TimescaleDB default time index and + discards 95% of it to keep one server. Forty-three servers deep, that is the whole slice re-read + forty-three times per pass cycle; warm it is 422 ms, and the excursions are the cold tail whenever + cache pressure evicts a slice ~4x the size it had on measurement day. + + THE INDEX THAT READ WANTED ALREADY EXISTED. V1's generated idx_query_store_stats_time is exactly + (server_id, collection_time), it is on the hypertable and on the very chunk in that plan, and the + planner declined it. Read from the production catalog: server_id's physical correlation is 0.022 + (43 servers interleaved by collection pass) while collection_time's is 0.99999, so the cost model + prices the composite's heap fetches as one random page per tuple — ~50K pages at random_page_cost + 4 — and the perfectly-correlated time index's 54,664-page stream wins on paper at 58,677. It loses + in fact by 11x: forced under random_page_cost = 1.1 the SAME statement took the existing composite + (Bitmap Index Scan, Index Cond on both columns) and touched 5,063 buffers (Heap Blocks: exact=5001) + instead of 57,307, because a server's rows land in one contiguous run per collection pass (~13 + rows a page) that the planner's single correlation statistic cannot see. A second plain composite, + however ordered, would be priced identically and ignored identically — which is why this is not + the (server_id, collection_time DESC) rung the issue first proposed. + + COVERING, so the choice stops depending on the cost model. With every column the read touches in + the key or INCLUDE, the plan is an Index Only Scan whose cost is the index pages for ONE server's + two hours and nothing else — no heap component to misprice, at any random_page_cost and at any + share of the fleet the busiest server grows into. Both uncompressed production chunks read + relallvisible = 100% of relpages (the insert-autovacuum override below is what keeps them there), + so heap fetches for visibility are the newest pass's pages at most. Measured on a PG18 / + TimescaleDB 2.28.1 rig seeded in the production's write pattern: the shipped statement went from + the identical time-index-plus-Filter plan at 1,514 buffers to Index Only Scan, Heap Fetches: 0, + 50 buffers. The INCLUDE list IS the read's column list, deliberately and exactly — a column added + to the read and not to this list silently degrades it back to the heap plan, so + ForcePlanFailuresAccessPathTests pins the two against each other. + + THE COST, stated rather than implied: INCLUDE disables btree deduplication, so this index is + ~86 bytes/row on the rig (V1's deduplicated composite is ~7) — roughly 0.7-0.9 GB per day-chunk on + the largest store's 8-16 M rows/day, against a 4.7-9.4 GB heap per chunk. It is self-limiting: + on 2.28.1 a compressed chunk's uncompressed relation is an empty shell, and CREATE INDEX on the + hypertable builds an 8 KB page for each one (measured: 4 compressed chunks at 8192 bytes each, + the 2 live chunks at 46 MB and 27 MB), so the footprint is the one or two uncompressed chunks and + the compression policy erases the rest a day later. That is also why the owner's "index only the + uncompressed/new chunks" needs no mechanism: it is what the engine does. New chunks inherit the + index at creation; a decompressed chunk fills it and recompression empties it (both measured). + + PLAIN CREATE INDEX, ONE TRANSACTION, deliberately. The build takes a ShareLock on the hypertable + root for its duration — reads proceed, every INSERT into any chunk queues behind it — which is + why this list runs BEFORE collectors start and why this statement belongs here rather than in the + ladder (whose applier wraps each rung in a transaction block, the one place the per-chunk form + below is refused outright). A cancel at SetupTimeoutSeconds rolls the whole build back and the next + start retries with nothing left behind. The per-chunk form, WITH (timescaledb.transaction_per_chunk), + was measured and rejected: it takes the same root ShareLock first, buying no write concurrency here, + and a cancel MID-build — exactly what the command timeout is — commits the chunks built so far, + leaves the parent index indisvalid = false and the live chunk unindexed, after which this very + idempotent statement reports "already exists, skipping" on every start forever. CREATE INDEX + CONCURRENTLY is refused on hypertables ("hypertables do not support concurrent index creation"). + A collision with the compression job on yesterday's chunk makes one of them wait for the other; + if that is this build past its budget, the cancel-and-retry above is the outcome. + + NOT random_page_cost, though it flipped the plan: at 1.1 the composite won by 53,095 to 58,677 for + a server holding ~8% of the fleet's rows, and the ratio scales with that share, so the next-busiest + store or the same one a month on flips back. A store-wide planner setting is also an owner's + ruling for every read at once, not a lane's fix for one. NOT a partial index WHERE is_forced_plan, + which would be a few MB: the read aggregates unforced rows on purpose (a plan's previous sighting + may be its unforced one), so that index needs the statement reshaped and its first-sighting + semantics changed — the cheaper long-run shape, deferred rather than smuggled in. */ + "CREATE INDEX IF NOT EXISTS " + ForcePlanFailuresIndexName + " ON collect.query_store_stats (server_id, collection_time DESC) INCLUDE (database_name, query_id, plan_id, force_failure_count, is_forced_plan, plan_forcing_type, last_force_failure_reason)", "ALTER TABLE collect.procedure_stats SET (" + InsertTuningOptions + ")", "ALTER TABLE collect.query_stats SET (" + InsertTuningOptions + ")", "ALTER TABLE collect.query_store_stats SET (" + InsertTuningOptions + ")", From bebdf55f6b34a1b85dfd710f85da1d09f40cdee8 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 11:24:41 -0400 Subject: [PATCH 23/69] The viewer's tray toast honors mute rules itself, so a Snooze stops the toasts on the next poll instead of waiting on the service's reload (#3570) (#3587) The tray toast is the viewer's own alert channel, and until now the viewer applied no mute rule to it: a polled alert row toasted unless the SERVICE had stamped it muted. A Snooze wrote the rule to the store and bumped the reload beacon, and then the toast's suppression depended on a four-link chain in another process - beacon observed on the next 15 s tick, a monolithic config re-read succeeding in full, the mute cache refreshed, the alert re-firing THROUGH that cache - none of which the viewer observed. When a link was slow or broken the rule sat in Manage Mute Rules while the toasts kept coming. AlertToastCoordinator.SelectToasts now also takes the viewer's own rule set and skips any row a rule covers, judged with the shared MuteRule.MatchesAt over the row's own ToMuteContext (server + metric as the row spells them, plus the detail-text dimensions the Mute This Alert pre-fill already read). The set is the per-poll mute-rule read that already drives the sidebar bell, plus any rule this viewer just wrote (tray Snooze, server Silence) added persist-then-cache. Keep- last-known on a failed read, and a rule written mid-read is carried over so the guarantee has no one-tick hole. The status line says what happens on each clock, and a failed snooze is reported rather than only logged. Lite never had the gap: its snooze lands in the same in-process MuteRuleService its deliverer consults before the balloon. --- .../ViewerAlertToastCoordinatorTests.cs | 156 ++++++++++ .../Darling.Tests/ViewerTraySnoozeTests.cs | 290 ++++++++++++++++++ .../AlertToastCoordinator.cs | 66 +++- .../AlertsHistoryTab.xaml.cs | 15 +- .../MainWindow.ServerManagement.cs | 64 +++- .../MainWindow.xaml.cs | 71 ++++- .../ViewerDataService.AlertHistory.cs | 28 ++ .../ViewerDataService.MuteRules.cs | 35 +++ .../ViewerDataService.ServiceConfig.cs | 11 + 9 files changed, 694 insertions(+), 42 deletions(-) create mode 100644 Darling/Darling.Tests/ViewerTraySnoozeTests.cs diff --git a/Darling/Darling.Tests/ViewerAlertToastCoordinatorTests.cs b/Darling/Darling.Tests/ViewerAlertToastCoordinatorTests.cs index 618724075..62f54b0a4 100644 --- a/Darling/Darling.Tests/ViewerAlertToastCoordinatorTests.cs +++ b/Darling/Darling.Tests/ViewerAlertToastCoordinatorTests.cs @@ -10,6 +10,7 @@ using System.Collections.Generic; using System.Linq; using PerformanceMonitor.Darling.Viewer; +using PerformanceMonitor.Notifications; using Xunit; namespace Darling.Tests; @@ -224,4 +225,159 @@ public void SelectToasts_SeenRowBeyondRetention_IsPruned_ThenReToasts() Assert.Single(toasts); } + + /* ---------------- #3570: the viewer honors mute rules for its own channel ---------------- */ + + /// + /// The report, as a pin. The service re-fires "Agent Not Running" with muted = false — because it has + /// not reloaded its cache yet, or its reload failed, or the beacon never reached it — and the viewer holds + /// the rule the operator's Snooze just wrote. The toast must not appear: the tray is the viewer's channel and + /// the viewer's rule set decides. Before #3570 only row.Muted was consulted and this toasted. + /// + [Fact] + public void SelectToasts_RowCoveredByAViewerRule_IsNotToasted_EvenWhenTheServiceLeftItUnmuted() + { + var coordinator = new AlertToastCoordinator(Retention); + var refire = Row(T0.AddMinutes(5), 1, "Agent Not Running", muted: false); + var snooze = ViewerDataService.BuildTraySnoozeRule("Server1", "Agent Not Running", TimeSpan.FromHours(4), T0); + + var toasts = coordinator.SelectToasts(new[] { refire }, T0.AddMinutes(5), Cooldown, new[] { snooze }); + + Assert.Empty(toasts); + } + + /// Null and empty rule sets are the pre-#3570 behavior exactly: the service's flag alone decides. + [Fact] + public void SelectToasts_NoViewerRules_LeavesAnUnmutedRowToasting() + { + var row = Row(T0, 1, "Agent Not Running"); + + Assert.Single(new AlertToastCoordinator(Retention).SelectToasts(new[] { row }, T0, Cooldown, muteRules: null)); + Assert.Single(new AlertToastCoordinator(Retention).SelectToasts(new[] { row }, T0, Cooldown, Array.Empty())); + } + + /// + /// The viewer-side judgement is on the coordinator's injected clock: a rule + /// whose expiry has passed suppresses nothing, judged at the SAME instant the rest of the decision is. + /// + [Fact] + public void SelectToasts_ExpiredViewerRule_DoesNotSuppress() + { + var coordinator = new AlertToastCoordinator(Retention); + var snooze = ViewerDataService.BuildTraySnoozeRule("Server1", "Agent Not Running", TimeSpan.FromMinutes(15), T0); + + /* Judged one second after the 15 m snooze lapsed. */ + var toasts = coordinator.SelectToasts( + new[] { Row(T0.AddMinutes(16), 1, "Agent Not Running") }, T0.AddMinutes(15).AddSeconds(1), Cooldown, new[] { snooze }); + + Assert.Single(toasts); + } + + [Fact] + public void SelectToasts_DisabledViewerRule_DoesNotSuppress() + { + var coordinator = new AlertToastCoordinator(Retention); + var rule = ViewerDataService.BuildTraySnoozeRule("Server1", "Agent Not Running", TimeSpan.FromHours(4), T0); + rule.Enabled = false; + + var toasts = coordinator.SelectToasts(new[] { Row(T0, 1, "Agent Not Running") }, T0, Cooldown, new[] { rule }); + + Assert.Single(toasts); + } + + /// + /// Scope is the shared matcher's: server name case-insensitive, metric exact (by name), a rule for another + /// server or another metric leaves this row alone. Pinned here so the tray can never be broader OR narrower + /// than the channels the service mutes with the same rule. + /// + [Fact] + public void SelectToasts_ViewerRuleScope_IsTheSharedMatchersScope() + { + var row = Row(T0, 1, "Agent Not Running"); + + var otherServer = ViewerDataService.BuildTraySnoozeRule("Server2", "Agent Not Running", TimeSpan.FromHours(4), T0); + Assert.Single(new AlertToastCoordinator(Retention).SelectToasts(new[] { row }, T0, Cooldown, new[] { otherServer })); + + var otherMetric = ViewerDataService.BuildTraySnoozeRule("Server1", "Failed Agent Job", TimeSpan.FromHours(4), T0); + Assert.Single(new AlertToastCoordinator(Retention).SelectToasts(new[] { row }, T0, Cooldown, new[] { otherMetric })); + + var differentCase = ViewerDataService.BuildTraySnoozeRule("SERVER1", "agent not running", TimeSpan.FromHours(4), T0); + Assert.Empty(new AlertToastCoordinator(Retention).SelectToasts(new[] { row }, T0, Cooldown, new[] { differentCase })); + + /* A whole-server silence (no metric) covers every metric on that server — the sidebar's one-click rule. */ + var silence = ViewerDataService.BuildServerSilenceRule("Server1"); + Assert.Empty(new AlertToastCoordinator(Retention).SelectToasts(new[] { row }, T0, Cooldown, new[] { silence })); + } + + /// + /// A pattern-scoped rule is judged over the dimensions parses out + /// of the row's detail text — the same pre-fill the "Mute This Alert" dialog reads — so a database-scoped + /// mute the operator authored FROM a row covers that row's toast, and only rows about that database. + /// + [Fact] + public void SelectToasts_PatternScopedViewerRule_IsJudgedOverTheRowsDetailText() + { + var rule = new MuteRule { MetricName = "Blocking Detected", DatabasePattern = "Sales" }; + + var salesRow = Row(T0, 1, "Blocking Detected", detail: "Blocking detected\n Database: SalesDb\n Wait Type: LCK_M_S"); + Assert.Empty(new AlertToastCoordinator(Retention).SelectToasts(new[] { salesRow }, T0, Cooldown, new[] { rule })); + + var otherDbRow = Row(T0, 1, "Blocking Detected", detail: "Blocking detected\n Database: Payroll"); + Assert.Single(new AlertToastCoordinator(Retention).SelectToasts(new[] { otherDbRow }, T0, Cooldown, new[] { rule })); + + /* No detail text at all: the pattern dimension is unknown, the rule cannot claim it, the row toasts. */ + var bareRow = Row(T0, 1, "Blocking Detected"); + Assert.Single(new AlertToastCoordinator(Retention).SelectToasts(new[] { bareRow }, T0, Cooldown, new[] { rule })); + } + + /// + /// A row the viewer's rule suppressed is marked seen exactly like a service-muted one, so when the snooze + /// lapses the rows it covered do not replay as a storm — only rows that arrive AFTER expiry can toast. + /// + [Fact] + public void SelectToasts_ViewerSuppressedRow_IsMarkedSeen_SoRuleExpiryDoesNotReplayIt() + { + var coordinator = new AlertToastCoordinator(Retention); + var snooze = ViewerDataService.BuildTraySnoozeRule("Server1", "Agent Not Running", TimeSpan.FromMinutes(15), T0); + var covered = Row(T0.AddMinutes(5), 1, "Agent Not Running"); + + Assert.Empty(coordinator.SelectToasts(new[] { covered }, T0.AddMinutes(5), Cooldown, new[] { snooze })); + + /* The snooze has lapsed and the same row is re-read (the poll window overlaps): still nothing. */ + Assert.Empty(coordinator.SelectToasts(new[] { covered }, T0.AddMinutes(16), Cooldown, new[] { snooze })); + + /* A NEW row after expiry toasts — the condition is live again and the operator asked for 15 m, not forever. */ + Assert.Single(coordinator.SelectToasts(new[] { Row(T0.AddMinutes(17), 1, "Agent Not Running") }, T0.AddMinutes(17), Cooldown, new[] { snooze })); + } + + /// + /// The viewer's rule set and the service's flag are ORed: either alone suppresses, and a rule covering row A + /// says nothing about row B on another server in the same poll. + /// + [Fact] + public void SelectToasts_ViewerRulesAndServiceFlag_AreIndependentPerRow() + { + var coordinator = new AlertToastCoordinator(Retention); + var snooze = ViewerDataService.BuildTraySnoozeRule("Server1", "Agent Not Running", TimeSpan.FromHours(4), T0); + + var coveredByRule = Row(T0, 1, "Agent Not Running"); + var mutedByService = Row(T0, 2, "High CPU", muted: true); + var neither = Row(T0, 3, "Agent Not Running"); + + var toasts = coordinator.SelectToasts(new[] { coveredByRule, mutedByService, neither }, T0, Cooldown, new[] { snooze }); + + var only = Assert.Single(toasts); + Assert.Equal(3, only.ServerId); + } + + /// A null entry in the rule list is skipped rather than thrown on — the filter runs inside the refresh loop. + [Fact] + public void IsMutedByViewerRules_SkipsNullEntries() + { + var row = Row(T0, 1, "Agent Not Running"); + var rules = new MuteRule[] { null!, ViewerDataService.BuildTraySnoozeRule("Server1", "Agent Not Running", TimeSpan.FromHours(1), T0) }; + + Assert.True(AlertToastCoordinator.IsMutedByViewerRules(row, rules, T0)); + Assert.False(AlertToastCoordinator.IsMutedByViewerRules(row, new MuteRule[] { null! }, T0)); + } } diff --git a/Darling/Darling.Tests/ViewerTraySnoozeTests.cs b/Darling/Darling.Tests/ViewerTraySnoozeTests.cs new file mode 100644 index 000000000..e96a78de1 --- /dev/null +++ b/Darling/Darling.Tests/ViewerTraySnoozeTests.cs @@ -0,0 +1,290 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Text.RegularExpressions; +using PerformanceMonitor.Darling.Viewer; +using PerformanceMonitor.Notifications; +using Xunit; + +namespace Darling.Tests; + +/// +/// Pins for #3570 — "Snoozing an alert in the tray does nothing". The tray toast is the viewer's own alert +/// channel, and until this issue the viewer never applied a mute rule to it: it toasted every polled row the +/// SERVICE had not stamped muted, so a Snooze suppressed the next toast only after the service noticed +/// the reload beacon, re-read its whole config view, refreshed its rule cache, and re-fired the alert through +/// it. Nothing viewer-side observed that chain; when it was slow or broken the rule sat in +/// config_mute_rules while the toasts kept coming. +/// +/// The decision logic that closes it — judging polled +/// rows against the viewer's own rule set — is pinned in ViewerAlertToastCoordinatorTests. This file +/// pins the two ends of the loop that feed it: the rule the Snooze WRITES +/// () and the context a row is JUDGED as +/// (), and that the one matches the other by construction — the +/// spelling question. Plus the honesty of the status line's "within N s", which restates a service constant the +/// viewer cannot reference. The WPF surfaces themselves (the balloon, the status bar) are not unit-testable; +/// everything that decides what they say is. +/// +public sealed class ViewerTraySnoozeTests +{ + private static readonly DateTime T0 = new(2026, 9, 18, 7, 22, 53, DateTimeKind.Utc); + + private static ViewerAlertRow Row(string serverName, string metric, string? detail = null, bool muted = false) => new() + { + AlertTime = T0, + ServerId = 7, + ServerName = serverName, + MetricName = metric, + CurrentValue = 0, + ThresholdValue = 0, + AlertSent = false, + NotificationType = AlertDelivery.ChannelNoneConfigured, + Muted = muted, + DetailText = detail, + }; + + /* ---------------- the rule the Snooze writes ---------------- */ + + [Fact] + public void BuildTraySnoozeRule_ScopesToTheRowsServerAndMetric_ExpiringAfterTheDuration() + { + var rule = ViewerDataService.BuildTraySnoozeRule("sql-prod-01", "Agent Not Running", TimeSpan.FromHours(4), T0); + + Assert.Equal("sql-prod-01", rule.ServerName); + Assert.Equal("Agent Not Running", rule.MetricName); + Assert.True(rule.Enabled); + Assert.Equal(T0, rule.CreatedAtUtc); + Assert.Equal(T0.AddHours(4), rule.ExpiresAtUtc); + Assert.Equal("Snoozed from tray (4h)", rule.Reason); + + /* A snooze is "this alert on this server" — never narrower. */ + Assert.Null(rule.DatabasePattern); + Assert.Null(rule.QueryTextPattern); + Assert.Null(rule.WaitTypePattern); + Assert.Null(rule.JobNamePattern); + + /* And never broader: a snooze must not be the blanket rule the Stale Mute Rules self-alert flags CRITICAL. */ + Assert.False(rule.MatchesEveryAlert); + Assert.False(string.IsNullOrEmpty(rule.Id)); + } + + [Theory] + [InlineData(15, "15m")] + [InlineData(60, "1h")] + [InlineData(240, "4h")] + public void BuildTraySnoozeRule_ReasonNamesTheButtonThatWroteIt(int minutes, string label) + { + var rule = ViewerDataService.BuildTraySnoozeRule("s", "m", TimeSpan.FromMinutes(minutes), T0); + + Assert.Equal($"{ViewerDataService.TraySnoozeReasonPrefix} ({label})", rule.Reason); + Assert.Equal(label, ViewerDataService.FormatSnoozeDuration(TimeSpan.FromMinutes(minutes))); + } + + /// A row with no server name yields a rule for every server — the only honest scope for a row + /// that did not say — never an empty-string server that would match nothing at all. + [Theory] + [InlineData(null)] + [InlineData("")] + public void BuildTraySnoozeRule_EmptyServerName_BecomesEveryServer(string? serverName) + { + var rule = ViewerDataService.BuildTraySnoozeRule(serverName, "Agent Not Running", TimeSpan.FromHours(1), T0); + + Assert.Null(rule.ServerName); + Assert.Equal("Agent Not Running", rule.MetricName); + Assert.False(rule.MatchesEveryAlert); /* still metric-scoped */ + } + + /* ---------------- the closed loop: the rule the toast writes matches the row the toast came from ---------------- */ + + /// + /// The spelling question, answered by construction. Alert rows on a Darling store spell server_name + /// two ways — the self-alert family writes the server's display name, the shared engine writes the name its + /// snapshot carries — and a mute rule must match the row's OWN spelling to suppress that row's producer. The + /// toast captures and off the + /// row, the rule is built from exactly those, and the viewer's filter judges the same row through + /// : whatever the spelling, the loop closes. Pinned over both + /// shapes so a future "normalize the server name" on one side but not the other fails here. + /// + [Theory] + [InlineData("SQL01", "Agent Not Running")] /* self-alert family: display name */ + [InlineData("sql01.corp.example.internal", "Failed Agent Job")] /* engine family: the snapshot's name */ + [InlineData("sql01,1433", "High CPU")] /* a display name with a port */ + public void TheRuleASnoozeWrites_MatchesTheRowItWasSnoozedFrom(string serverName, string metric) + { + var row = Row(serverName, metric, detail: " Job Name: Nightly ETL\n"); + var rule = ViewerDataService.BuildTraySnoozeRule(row.ServerName, row.MetricName, TimeSpan.FromHours(4), T0); + + Assert.True(rule.MatchesAt(row.ToMuteContext(), T0.AddMinutes(5))); + Assert.True(AlertToastCoordinator.IsMutedByViewerRules(row, new[] { rule }, T0.AddMinutes(5))); + + /* And the SAME rule, judged the way the service judges it — a bare server+metric context, no detail + text — also matches, so the tray and the service's channels agree about this snooze. */ + Assert.True(rule.MatchesAt(new AlertMuteContext { ServerName = serverName, MetricName = metric }, T0.AddMinutes(5))); + } + + /// The rule stops matching the instant it expires, on the clock it is judged with — no ambient UtcNow. + [Fact] + public void TheRuleASnoozeWrites_LapsesExactlyAtItsExpiry() + { + var row = Row("SQL01", "Agent Not Running"); + var rule = ViewerDataService.BuildTraySnoozeRule(row.ServerName, row.MetricName, TimeSpan.FromMinutes(15), T0); + + Assert.True(rule.MatchesAt(row.ToMuteContext(), T0.AddMinutes(15).AddTicks(-1))); + Assert.False(rule.MatchesAt(row.ToMuteContext(), T0.AddMinutes(15))); + } + + /* ---------------- the context a row is judged as ---------------- */ + + [Fact] + public void ToMuteContext_CarriesTheRowsServerAndMetricVerbatim() + { + var context = Row("SQL01", "Agent Not Running").ToMuteContext(); + + Assert.Equal("SQL01", context.ServerName); + Assert.Equal("Agent Not Running", context.MetricName); + Assert.Null(context.DatabaseName); + Assert.Null(context.WaitType); + Assert.Null(context.JobName); + Assert.Null(context.QueryText); + } + + /// The pattern dimensions come from the stored detail text, the same way the "Mute This Alert" + /// pre-fill has always read them — so a rule authored from a row covers that row's toast. + [Fact] + public void ToMuteContext_ParsesThePatternDimensionsOutOfDetailText() + { + var detail = "2 job failure(s)\n Job Name: Nightly ETL\n Database: SalesDb\n Wait Type: LCK_M_X\n Query: SELECT 1"; + var context = Row("SQL01", "Failed Agent Job", detail).ToMuteContext(); + + Assert.Equal("Nightly ETL", context.JobName); + Assert.Equal("SalesDb", context.DatabaseName); + Assert.Equal("LCK_M_X", context.WaitType); + Assert.Equal("SELECT 1", context.QueryText); + + var jobRule = new MuteRule { MetricName = "Failed Agent Job", JobNamePattern = "Nightly" }; + Assert.True(jobRule.MatchesAt(context, T0)); + } + + /// #3309 holds here too: a custom alert's detail text is a user-authored name, never parsed for + /// labels, so a crafted "Database: master" line cannot pre-fill or match a pattern dimension. + [Fact] + public void ToMuteContext_CustomAlert_DoesNotParseDetailText() + { + var context = Row("SQL01", "Custom:42", detail: "Database: master").ToMuteContext(); + + Assert.Equal("Custom:42", context.MetricName); + Assert.Null(context.DatabaseName); + } + + /* ---------------- the status line's "within N s" is the service's real cadence ---------------- */ + + /// + /// The snooze status line tells the operator the service's channels stop "within N s (the service's next + /// sweep)". N is , a restatement of + /// DarlingWorker.s_sweepInterval — the tick at whose top the service polls the reload beacon — which + /// the viewer cannot reference. Read out of the service's source (the field is private) so the sentence + /// cannot outlive the cadence it describes. + /// + [Fact] + public void ServiceReloadTickSeconds_IsTheServicesSweepTick() + { + var sweepTick = CommandPlaneCommandTimeoutTests.SecondsOfPrivateTimeSpan("DarlingWorker.cs", "s_sweepInterval"); + + Assert.Equal(ViewerDataService.ServiceReloadTickSeconds, sweepTick); + } + + /* ---------------- the wiring that makes the coordinator's new argument reach it ---------------- */ + + /// + /// 's rule-set parameter is OPTIONAL (null = the pre-#3570 + /// behavior), so a refactor that drops the argument at the one production call site compiles clean and + /// silently reverts the fix. This reads MainWindow.xaml.cs and asserts the call passes the viewer's + /// rule set, and that the rule-set refresh (UpdateServerSilencedAsync, which also drives the sidebar + /// bell) is awaited BEFORE it in the same poll — the ordering the rules-then-toasts contract rests on. + /// + [Fact] + public void PollAlertsAsync_PassesTheViewersRuleSet_AfterRefreshingIt() + { + var body = MemberBody(ViewerSource("MainWindow.xaml.cs"), "PollAlertsAsync"); + + var refresh = body.IndexOf("await UpdateServerSilencedAsync()", StringComparison.Ordinal); + var select = Regex.Match(body, @"SelectToasts\s*\([^;]*?_viewerMuteRules\s*\)", RegexOptions.Singleline); + + Assert.True(refresh >= 0, "PollAlertsAsync no longer awaits UpdateServerSilencedAsync — the toast filter's rule set is never refreshed"); + Assert.True(select.Success, "PollAlertsAsync's SelectToasts call no longer passes _viewerMuteRules — the tray has stopped honoring mute rules (#3570 regressed)"); + Assert.True(refresh < select.Index, "PollAlertsAsync selects toasts BEFORE refreshing the rule set — a rule read this poll reaches the tray a poll late"); + } + + /// + /// The two local writers (tray Snooze, server Silence) add their rule to the viewer's set only AFTER the + /// store write succeeds — persist-then-cache, 's ordering — so a + /// snooze that did not persist never suppresses toasts on this seat while every other surface says no such + /// rule exists. + /// + [Fact] + public void LocalRuleWriters_AddToTheViewersSet_OnlyAfterTheStoreWrite() + { + var snooze = MemberBody(ViewerSource("MainWindow.xaml.cs"), "SnoozeAlertAsync"); + AssertPersistThenCache(snooze, "SnoozeAlertAsync"); + + var silence = MemberBody(ViewerSource("MainWindow.ServerManagement.cs"), "ServerContextMenu_Silence_Click"); + AssertPersistThenCache(silence, "ServerContextMenu_Silence_Click"); + } + + private static void AssertPersistThenCache(string body, string member) + { + var insert = body.IndexOf("InsertMuteRuleAsync(", StringComparison.Ordinal); + var cache = body.IndexOf("_viewerMuteRules.Add(", StringComparison.Ordinal); + + Assert.True(insert >= 0, $"{member} no longer writes the rule to the store"); + Assert.True(cache >= 0, $"{member} no longer hands its rule to the toast filter (#3570 regressed for this writer)"); + Assert.True(insert < cache, $"{member} caches the rule before persisting it — a failed write would suppress toasts for a rule that does not exist"); + } + + /* ---------------- helpers ---------------- */ + + /// The text of one viewer source file, through the shared root resolver (worktree-safe). + private static string ViewerSource(string file) => + RepoFile.ReadRepoFile("Darling", "PerformanceMonitor.Darling.Viewer", file); + + /// + /// The CODE of one method, from its declaration to the brace that closes it, over the comment/string-stripped + /// source ( preserves offsets) so a brace or a name + /// inside a comment or a status-line string can neither close the body early nor satisfy an assertion meant + /// for code. Anchored on the DECLARATION (Task NAME( / void NAME() rather than the bare name: + /// both methods this file reads are also CALLED earlier in their files, and the first bare match would hand + /// back whatever method happens to follow that call. + /// + private static string MemberBody(string source, string member) + { + var stripped = CSharpSourceWalker.StripCommentsAndStrings(source); + var signature = Regex.Match( + stripped, @"\b(?:Task|void)\s+" + Regex.Escape(member) + @"\s*\(", RegexOptions.CultureInvariant); + Assert.True(signature.Success, $"could not find the declaration of {member}( in the viewer source"); + + var open = stripped.IndexOf('{', signature.Index); + Assert.True(open >= 0, $"could not find the opening brace of {member}"); + + var depth = 0; + for (var i = open; i < stripped.Length; i++) + { + if (stripped[i] == '{') + { + depth++; + } + else if (stripped[i] == '}' && --depth == 0) + { + return stripped.Substring(signature.Index, i - signature.Index + 1); + } + } + + Assert.Fail($"unbalanced braces while reading {member}"); + return ""; + } +} diff --git a/Darling/PerformanceMonitor.Darling.Viewer/AlertToastCoordinator.cs b/Darling/PerformanceMonitor.Darling.Viewer/AlertToastCoordinator.cs index 8a668c998..1a9216de1 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/AlertToastCoordinator.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/AlertToastCoordinator.cs @@ -9,6 +9,7 @@ using System; using System.Collections.Generic; using System.Linq; +using PerformanceMonitor.Notifications; namespace PerformanceMonitor.Darling.Viewer; @@ -34,6 +35,22 @@ namespace PerformanceMonitor.Darling.Viewer; /// /// Muted rows never toast (Lite parity: the service still logs a muted row, flagged, with channels /// skipped). Bookkeeping is pruned each call so neither map grows without bound over a long-lived viewer. +/// +/// The tray is the VIEWER's channel, and the viewer honors mute rules for it itself (#3570). Before +/// this, a row toasted unless the SERVICE had stamped it muted — a flag that records the service's +/// decision about the service's channels (email/webhook), made against the service's in-memory rule cache, +/// which only a control-plane reload refreshes. So a Snooze from a toast (a config_mute_rules row) +/// suppressed the next toast only after a four-link cross-process chain completed: the reload beacon +/// observed on the service's next 15 s tick, a monolithic store re-read that succeeds in full, the mute +/// cache refreshed, and the alert re-firing THROUGH that cache. Nothing on the viewer side observed any of +/// it; when a link was slow or broken the rule sat in the table — visible in Manage Mute Rules — while the +/// toasts kept coming, which is exactly the report. Lite never had the gap: its Snooze lands in the same +/// in-process its deliverer consults before showing the balloon. This is the +/// headless equivalent: takes the viewer's own read of the active rules and skips +/// any row one of them covers, judged with the SAME the service uses, over +/// the context builds — so a rule the snooze just wrote stops +/// the toasts on the very next poll, whatever the service has or has not done with it yet. The service's +/// muted flag is still honored too; the two are ORed. /// public sealed class AlertToastCoordinator { @@ -89,8 +106,9 @@ public void Prime(IEnumerable existingRows) /// Filters a freshly-polled batch to the rows that should toast now, updating the seen-set and cooldown /// state. Rows are considered oldest-first so, within a burst that shares a condition, the EARLIEST row /// wins the cooldown slot (the rest are suppressed until the window elapses). A row is emitted only when - /// it is new (not seen/primed), not muted, and its condition is outside the cooldown window; every - /// processed row is marked seen regardless of outcome so it is considered exactly once. + /// it is new (not seen/primed), not muted — neither by the service's flag nor by a rule in + /// — and its condition is outside the cooldown window; every processed row + /// is marked seen regardless of outcome so it is considered exactly once. /// /// The latest read of recent, non-dismissed alert rows (any order). /// The current time (injected for testability). @@ -98,8 +116,19 @@ public void Prime(IEnumerable existingRows) /// The per-condition cooldown window (the "Tray notification cooldown" setting). /// or negative disables the cooldown so every new, unmuted row toasts. /// + /// + /// The mute rules as THIS viewer currently knows them (#3570): its latest read of config_mute_rules + /// plus any rule it has just written itself (a tray Snooze, a server Silence) and not yet re-read. A row + /// covered by any rule that is enabled and unexpired at is skipped exactly as a + /// service-muted row is: marked seen, never toasted — so a rule that later expires does not replay the + /// rows it covered. Null or empty means "no viewer-side rules", which leaves the pre-#3570 behavior (the + /// service's flag alone). The judgement is over + /// — the shared matcher and the shared context, so the tray + /// agrees with the service's channels about what a rule covers rather than approximating it. + /// public IReadOnlyList SelectToasts( - IEnumerable polledRows, DateTime nowUtc, TimeSpan cooldown) + IEnumerable polledRows, DateTime nowUtc, TimeSpan cooldown, + IReadOnlyList? muteRules = null) { ArgumentNullException.ThrowIfNull(polledRows); @@ -120,6 +149,11 @@ public IReadOnlyList SelectToasts( continue; /* Lite parity: a muted row is logged but never toasted */ } + if (IsMutedByViewerRules(row, muteRules, nowUtc)) + { + continue; /* #3570: a rule this viewer holds covers the row — the tray honors it without waiting on the service */ + } + var conditionKey = ConditionKey(row); if (cooldown > TimeSpan.Zero && _lastToast.TryGetValue(conditionKey, out var last) @@ -136,6 +170,32 @@ public IReadOnlyList SelectToasts( return toasts; } + /// + /// True when any rule in covers at + /// (#3570). The context is built ONCE per row and only when there are rules to test, so a fleet with no + /// mute rules pays nothing for the detail-text parse. Pure: same matcher () + /// the service's applies, judged on the coordinator's injected + /// clock rather than the ambient one so a test can place a rule's expiry on either side of "now". + /// + internal static bool IsMutedByViewerRules(ViewerAlertRow row, IReadOnlyList? muteRules, DateTime nowUtc) + { + if (muteRules is null || muteRules.Count == 0) + { + return false; + } + + var context = row.ToMuteContext(); + foreach (var rule in muteRules) + { + if (rule is not null && rule.MatchesAt(context, nowUtc)) + { + return true; + } + } + + return false; + } + /// /// Drops bookkeeping that can no longer affect a decision: a seen row older than the retention window /// (the poll can never re-read it) and a cooldown entry older than the effective cooldown window. diff --git a/Darling/PerformanceMonitor.Darling.Viewer/AlertsHistoryTab.xaml.cs b/Darling/PerformanceMonitor.Darling.Viewer/AlertsHistoryTab.xaml.cs index 37e638d28..e603f8473 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/AlertsHistoryTab.xaml.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/AlertsHistoryTab.xaml.cs @@ -455,17 +455,10 @@ private async void MuteThisAlert_Click(object sender, RoutedEventArgs e) return; } - var context = new AlertMuteContext - { - ServerName = item.ServerName, - MetricName = item.MetricName - }; - /* #3309: pass the metric name so a custom alert ("Custom:") skips detail_text pre-fill parsing - - a custom rule has no Database/Wait Type/Job/Query dimension to pre-fill, and its user-authored name - must not be able to forge a mute-context label line. */ - context.PopulateFromDetailText(item.DetailText, item.MetricName); - - await CreateMuteRuleAsync(context); + /* The row's own mute context (server + metric + the dimensions parsed from detail_text; #3309's + custom-alert skip lives inside it). Shared with the tray-toast filter (#3570) so a rule authored + from this row is judged against the same context the toast filter will later judge it with. */ + await CreateMuteRuleAsync(item.ToMuteContext()); } private async void MuteSimilarAlerts_Click(object sender, RoutedEventArgs e) diff --git a/Darling/PerformanceMonitor.Darling.Viewer/MainWindow.ServerManagement.cs b/Darling/PerformanceMonitor.Darling.Viewer/MainWindow.ServerManagement.cs index 942d07cb1..b506e5127 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/MainWindow.ServerManagement.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/MainWindow.ServerManagement.cs @@ -477,11 +477,14 @@ entry nothing reads rather than losing anything the operator would miss. The sto /// /// "Silence This Server" — writes a whole-server mute rule (see /// ) the running Darling service honors on its next - /// config reload: every future alert for this server is flagged muted (channels skipped for email/Teams/Slack) - /// and so never toasts. The Darling shortcut over the multi-step Manage Mute Rules dialog, mirroring Lite's - /// one-click "Silence This Server". Idempotent: an existing active silence is reported, not duplicated. Keyed - /// on the server's DISPLAY name (what the alert engine's mute context + the alert rows carry). A read-only - /// seat / schema-skew / failure degrades to the friendly status message like the other server-row writes. + /// config reload: every future alert for this server is flagged muted (channels skipped for email/Teams/Slack). + /// The tray does not wait for that: the rule joins on success, so this viewer's + /// toast filter honors it from the next poll (#3570 — before, "never toasts" held only once the service had + /// reloaded and stamped the next row muted). The Darling shortcut over the multi-step Manage Mute Rules + /// dialog, mirroring Lite's one-click "Silence This Server". Idempotent: an existing active silence is + /// reported, not duplicated. Keyed on the server's DISPLAY name (what the alert engine's mute context + the + /// alert rows carry). A read-only seat / schema-skew / failure degrades to the friendly status message like + /// the other server-row writes. /// private async void ServerContextMenu_Silence_Click(object sender, RoutedEventArgs e) { @@ -500,10 +503,15 @@ private async void ServerContextMenu_Silence_Click(object sender, RoutedEventArg return; } - await _dataService.InsertMuteRuleAsync(ViewerDataService.BuildServerSilenceRule(server.DisplayName)); - /* #2031: flip the sidebar's muted-bell immediately — the poll would catch up anyway. */ + var silence = ViewerDataService.BuildServerSilenceRule(server.DisplayName); + await _dataService.InsertMuteRuleAsync(silence); + /* #2031: flip the sidebar's muted-bell immediately — the poll would catch up anyway. #3570: and + hand the rule to the toast filter now (persist-then-cache), for the same reason. */ server.SetSilenced(true); - StatusText.Text = $"Silenced all alerts for '{server.DisplayName}'. Right-click → Unsilence to restore."; + _viewerMuteRules.Add(silence); + StatusText.Text = + $"Silenced all alerts for '{server.DisplayName}' — tray toasts stop now; the service applies it within " + + $"{ViewerDataService.ServiceReloadTickSeconds} s. Right-click → Unsilence to restore."; } catch (ViewerReadOnlyException ex) { @@ -550,8 +558,11 @@ private async void ServerContextMenu_Unsilence_Click(object sender, RoutedEventA await _dataService.DeleteMuteRuleAsync(rule.Id); } - /* #2031: flip the sidebar's muted-bell immediately — the poll would catch up anyway. */ + /* #2031: flip the sidebar's muted-bell immediately — the poll would catch up anyway. #3570: and + drop the silences from the toast filter's set too, so an un-silenced server can toast on the + next poll rather than after the next re-read. */ server.SetSilenced(false); + _viewerMuteRules.RemoveAll(r => silences.Any(s => s.Id == r.Id)); StatusText.Text = $"Unsilenced '{server.DisplayName}'."; } catch (ViewerReadOnlyException ex) @@ -572,9 +583,16 @@ private async void ServerContextMenu_Unsilence_Click(object sender, RoutedEventA /// /// Refreshes every server's whole-server-silence indicator (#2031) from the store's mute rules — the /// sidebar muted-bell and the context menu's Silence/Unsilence exclusivity both read the resulting - /// flag. Rides the alert poll (the same cadence as the badge), over - /// the WHOLE fleet like the badge does, so a silence created from another seat (or over MCP) surfaces here - /// within a poll tick. Never throws — a broken read must not disturb the refresh loop. + /// flag — and, from the SAME read, replaces + /// , the rule set the tray-toast filter judges polled rows against (#3570). + /// One store read serves both because they want the same thing: the rules in force right now, as the store + /// has them. Rides the alert poll (the same cadence as the badge), over the WHOLE fleet like the badge does, + /// so a rule created from another seat (or over MCP) reaches both the bell and the tray within a poll tick. + /// + /// Never throws — a broken read must not disturb the refresh loop — and on a failed read the rule set + /// is LEFT AS IT WAS rather than emptied: the previous read's rules, plus whatever this viewer wrote since, + /// stay in force for the tray until a read succeeds. Emptying it would let one store blip un-mute every + /// snoozed toast on this seat, which is the #3354 defect one process over. /// private async Task UpdateServerSilencedAsync() { @@ -585,9 +603,31 @@ private async Task UpdateServerSilencedAsync() try { + var readStartedUtc = DateTime.UtcNow; var rules = await _dataService.GetMuteRulesAsync(); var active = rules.Where(r => r.Enabled && !r.IsExpired).ToList(); + /* Replace, not merge: the store is the authority. A rule this viewer added locally before this read + began is in the store too (persist-then-cache), so it comes back in `rules`; a rule deleted from + another seat is absent from `rules` and drops out here, which is the un-mute direction working. + + The one exception is a rule this viewer wrote WHILE the read was in flight. The click handlers + run on this same UI thread, so a Snooze can land between the await above and this line; that + rule is not in `rules` (the SELECT began before the INSERT) and a plain replace would drop it + until the next poll — a one-tick hole in the very guarantee #3570 adds. Every local writer + stamps CreatedAtUtc at the click, so "written during this read" is exactly "created at or after + the read began", and only THOSE are carried over. A stale rule from a previous read can never + pass that test, so a remote delete still lands on this poll. */ + foreach (var local in _viewerMuteRules) + { + if (local.CreatedAtUtc >= readStartedUtc && !active.Any(a => a.Id == local.Id)) + { + active.Add(local); + } + } + + _viewerMuteRules = active; + foreach (var server in _fleet.All) { server.SetSilenced(active.Any(r => ViewerDataService.IsWholeServerSilence(r, server.DisplayName))); diff --git a/Darling/PerformanceMonitor.Darling.Viewer/MainWindow.xaml.cs b/Darling/PerformanceMonitor.Darling.Viewer/MainWindow.xaml.cs index 04a133179..ee2a3c825 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/MainWindow.xaml.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/MainWindow.xaml.cs @@ -121,6 +121,20 @@ public partial class MainWindow : Window private AlertToastCoordinator? _toastCoordinator; private bool _alertPollInFlight; + /// + /// The mute rules as THIS viewer knows them (#3570): the enabled, unexpired rules from its latest read of + /// config_mute_rules (, once per alert poll — the same read + /// that drives the sidebar's muted-bell), plus any rule this viewer has itself just written and not yet + /// re-read (a tray Snooze, a server Silence). Fed to so + /// the tray honors a rule the moment the viewer holds it, instead of waiting for the service to reload its + /// own cache and stamp the next row muted. + /// + /// Keep-last-known on a failed read, deliberately: a store blip must not un-mute the tray for a tick + /// any more than lets one un-mute the service (#3354). Only the UI + /// thread touches it (the poll and the click handlers both run there), so it is a plain list. + /// + private List _viewerMuteRules = new(); + /// /// Restores the window from the tray if a sleep-/lock-driven minimize hid it (Lite #1050). Reachable now /// that minimize-to-tray can hide the viewer; paired with the App-startup SoftwareOnly render mode. @@ -699,14 +713,18 @@ private async Task PollAlertsAsync() /* Per-server badges: always, independent of the toast master switch. */ UpdateServerAttention(rows); - /* The sidebar's muted-bell (#2031): same cadence, its own small read (mute rules, not history). */ + /* The sidebar's muted-bell (#2031): same cadence, its own small read (mute rules, not history). The + same read refreshes _viewerMuteRules for the toast filter below (#3570), so it MUST run before + SelectToasts: a rule written from another seat, over MCP, or by this viewer's own dialogs reaches + the tray on the poll that reads it. */ await UpdateServerSilencedAsync(); - /* Tray toasts: only when notifications are enabled and the tray exists. */ + /* Tray toasts: only when notifications are enabled and the tray exists. The viewer's own rule set + rides along so a snoozed/silenced condition stops toasting NOW, not after the service's reload. */ if (_alertsEnabled && _trayService is not null) { var cooldown = TimeSpan.FromMinutes(Math.Max(0, _alertCooldownMinutes)); - foreach (var row in _toastCoordinator.SelectToasts(rows, DateTime.UtcNow, cooldown)) + foreach (var row in _toastCoordinator.SelectToasts(rows, DateTime.UtcNow, cooldown, _viewerMuteRules)) { ShowAlertToast(row); } @@ -776,8 +794,24 @@ private static string ToastBody(ViewerAlertRow row) /// /// Persists a snooze from a tray toast: a temporary mute rule scoped to the alert's server + metric - /// (Lite's SnoozeBalloon semantics), written straight to config_mute_rules. The running Darling - /// service re-reads it on its next config load, so the condition stops re-alerting until it expires. + /// (Lite's SnoozeBalloon semantics, ), written straight + /// to config_mute_rules with the reload beacon bumped, then — persist-then-cache, the + /// ordering — added to so the + /// very next poll's toast filter honors it (#3570). + /// + /// Two channels, two clocks, and the status line says both. The TRAY stops now: the rule is in this + /// viewer's hand and judges every polled row against it. The SERVICE's + /// channels (email/webhook) stop once it notices the beacon on its next sweep tick — within + /// — and from then on it stamps the condition's + /// rows muted. Before this change the line said only "Snoozed …", and the toast's suppression + /// depended entirely on that second clock plus a reload the viewer could not see succeed or fail; when it + /// did not, the operator had been told "snoozed" and was toasted again five minutes later (the report). + /// + /// A failed write is REPORTED, not just logged: the balloon has already closed by the time the + /// callback returns, so the status line is the only place the operator can learn the snooze did not + /// happen. A read-only seat's carries its own friendly text. The + /// local set is touched only on success — a snooze that did not persist must not suppress toasts on this + /// seat while every other surface says no such rule exists. /// private async Task SnoozeAlertAsync(string serverName, string metricName, TimeSpan duration) { @@ -786,21 +820,26 @@ private async Task SnoozeAlertAsync(string serverName, string metricName, TimeSp return; } - var rule = new MuteRule + var rule = ViewerDataService.BuildTraySnoozeRule(serverName, metricName, duration, DateTime.UtcNow); + var label = ViewerDataService.FormatSnoozeDuration(duration); + + try { - ServerName = string.IsNullOrEmpty(serverName) ? null : serverName, - MetricName = metricName, - ExpiresAtUtc = DateTime.UtcNow + duration, - Reason = $"Snoozed from tray ({FormatSnoozeDuration(duration)})", - }; + await _dataService.InsertMuteRuleAsync(rule); + } + catch (Exception ex) + { + ViewerLogger.Warn("AlertToasts", $"Snooze of {metricName} on {serverName} was not saved: {ex.Message}"); + StatusText.Text = $"Snooze not saved — {metricName} on {serverName} will keep alerting: {ex.Message}"; + return; + } - await _dataService.InsertMuteRuleAsync(rule); - StatusText.Text = $"Snoozed {metricName} on {serverName} for {FormatSnoozeDuration(duration)} — {DateTime.Now:HH:mm:ss}"; + _viewerMuteRules.Add(rule); + StatusText.Text = + $"Snoozed {metricName} on {serverName} for {label} — tray toasts stop now; the service applies it within " + + $"{ViewerDataService.ServiceReloadTickSeconds} s — {DateTime.Now:HH:mm:ss}"; } - private static string FormatSnoozeDuration(TimeSpan d) => - d.TotalHours >= 1 ? $"{(int)d.TotalHours}h" : $"{(int)d.TotalMinutes}m"; - /// /// Single-clicking a sidebar server drives the server-scoped aggregate tabs to it: it syncs the /// Recommendations and FinOps server pickers to the selected server (each remains independently diff --git a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.AlertHistory.cs b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.AlertHistory.cs index 0743d6779..517fa5cd1 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.AlertHistory.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.AlertHistory.cs @@ -83,6 +83,34 @@ public sealed class ViewerAlertRow public bool IsWarning => AlertMetricClassifier.IsWarning(MetricName); + /// + /// This row as the a is judged against — the SAME + /// server name and metric name the row carries (so a rule keyed on them matches this row by construction, + /// whatever spelling the producing family used for server_name), plus the Database / Wait Type / + /// Job Name / Query dimensions parsed out of the stored by + /// . + /// + /// The ONE definition of "what does this alert row look like to a mute rule" on the viewer side + /// (#3570). Both viewer consumers go through it: the Alert History tab's "Mute This Alert" pre-fill, which + /// builds the rule an operator authors FROM a row, and the tray-toast filter in + /// , which decides whether a rule already in force covers a row. Two + /// hand-built contexts could drift — one parsing the detail text and one not — and then a rule authored + /// from a row would suppress the service's channels but not the very toast it was authored from. + /// + /// The metric name is passed to the detail-text parse so a custom alert ("Custom:<id>") + /// skips it (#3309): its detail text is a user-authored name, not DMV label/value lines, and must not be + /// able to forge a label line. + /// + public AlertMuteContext ToMuteContext() + { + var context = new AlertMuteContext + { + ServerName = ServerName, + MetricName = MetricName, + }; + context.PopulateFromDetailText(DetailText, MetricName); + return context; + } } public sealed partial class ViewerDataService diff --git a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.MuteRules.cs b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.MuteRules.cs index d7403ece8..a2c5f757b 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.MuteRules.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.MuteRules.cs @@ -97,6 +97,41 @@ UPDATE config_mute_rules SET /* MetricName/DatabasePattern/QueryTextPattern/WaitTypePattern/JobNamePattern stay null → matches all. */ }; + /// The prefix a tray-toast Snooze stamps ("Snoozed from tray (4h)"), so + /// the Manage Mute Rules list says where the rule came from. Lite's balloon writes "Snoozed from popup (…)". + public const string TraySnoozeReasonPrefix = "Snoozed from tray"; + + /// + /// Builds the temporary rule a tray toast's Snooze writes (#3570): scoped to the toasted alert's server + + /// metric exactly as the alert row spells them, expiring after + /// , every pattern field left null. Lite's SnoozeBalloon semantics, as a + /// pure function so the shape can be pinned without WPF. + /// + /// is the toasted — the row's own + /// server_name — and that is what makes the rule match. Alert rows on this store spell a server two + /// ways (the self-alert family and the shared engine each write the name they evaluate with), and a rule + /// keyed on the row's spelling matches that row's producer by construction, because the producer's mute + /// context and its history row carry the same string. It is also the spelling + /// feeds the viewer's own toast filter, so the rule stops the + /// tray on the next poll too. An empty name (a row with none) becomes null = every server, which is the + /// only honest scope for a row that did not say. + /// + public static MuteRule BuildTraySnoozeRule(string? serverName, string metricName, TimeSpan duration, DateTime nowUtc) => new() + { + ServerName = string.IsNullOrEmpty(serverName) ? null : serverName, + MetricName = metricName, + Enabled = true, + CreatedAtUtc = nowUtc, + ExpiresAtUtc = nowUtc + duration, + Reason = $"{TraySnoozeReasonPrefix} ({FormatSnoozeDuration(duration)})", + /* DatabasePattern/QueryTextPattern/WaitTypePattern/JobNamePattern stay null — a snooze is "this + alert on this server", not a narrower shape. */ + }; + + /// "4h" / "1h" / "15m" — the snooze duration as the buttons label it (Lite's FormatDuration). + public static string FormatSnoozeDuration(TimeSpan d) => + d.TotalHours >= 1 ? $"{(int)d.TotalHours}h" : $"{(int)d.TotalMinutes}m"; + /// /// True when is a WHOLE-SERVER silence for — scoped to /// that server (case-insensitive) with no narrowing pattern on any other field. This is the shape diff --git a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.ServiceConfig.cs b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.ServiceConfig.cs index ff1d6e9e9..038ccaed1 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.ServiceConfig.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.ServiceConfig.cs @@ -128,6 +128,17 @@ public async Task UpdateServiceFlagsAsync( public const string ConfigReloadSignalSql = "UPDATE config_service SET updated_at = (now() AT TIME ZONE 'UTC') WHERE id = 1"; + /// + /// How long, at most, the running service takes to NOTICE a beacon bump: it polls config_version at + /// the top of every sweep tick, and the tick is DarlingWorker.s_sweepInterval (15 s). The viewer + /// does not reference the Service project, so the figure is restated here for the status line a mute + /// write shows (#3570: "the service picks it up within N s") and pinned against the service's source by + /// Darling.Tests.ViewerTraySnoozeTests, so the sentence cannot quietly outlive the cadence it + /// describes. What it bounds is the service's OWN channels (email/webhook); the tray toast the viewer + /// raises honors the rule on the viewer's next poll without waiting on this. + /// + public const int ServiceReloadTickSeconds = 15; + /// Bumps the reload beacon (see ) so the service re-reads the /// store on its next sweep — used after a write to a config table without its own bump trigger. public async Task SignalConfigReloadAsync(CancellationToken cancellationToken = default) From 9df2a9136b421767ee1681aa6433ff5349d12de9 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 11:27:12 -0400 Subject: [PATCH 24/69] The compression-stuck check confirms a -infinity read before it pages, and samples off the policies' run instant (#3575) (#3588) TimescaleDB's job_stats view assembles job_status from pg_stat_activity and next_start from the bgw_job_stat row, so at both edges of every healthy run one SELECT reads -infinity AND Scheduled - the predicate's dead-job arm. A production store paged on exactly that, 53 ms into a 63 ms run that succeeded. - ReadStuckCompressionJobsAsync re-reads StuckCompressionJobsSql five seconds after a -infinity trip and reports the job only if the arm still trips; a failed confirm defers judgement to the next hour rather than paging. The stuck-Running arm is reported from the first read as before. Injectable seam + pure merge (ClassifyCompressionJob / ConfirmStuckCompressionJobs) so the decision table pins without sleeping. - The worker snaps each hourly due time to :30 past the minute (TimescaleSupport.NextCompressionCheckUtc) instead of UtcNow + 1 h, which slipped a few seconds an hour across the :MM:00 instants the fixed-schedule policies fire on. - Both edges captured on a PG18 + TimescaleDB 2.28.1 rig; the crashed-worker state (the persistent -infinity row) reproduced and still reported. Tests: CompressionStuckConfirmReadTests (new), TimescaleSupportTests pins annotated. Census counts unchanged. --- .../CompressionStuckConfirmReadTests.cs | 499 ++++++++++++++++++ .../Darling.Tests/TimescaleSupportTests.cs | 22 +- .../DarlingSelfAlertEvaluator.cs | 16 + .../DarlingWorker.cs | 54 +- .../TimescaleSupport.cs | 442 +++++++++++++++- 5 files changed, 994 insertions(+), 39 deletions(-) create mode 100644 Darling/Darling.Tests/CompressionStuckConfirmReadTests.cs diff --git a/Darling/Darling.Tests/CompressionStuckConfirmReadTests.cs b/Darling/Darling.Tests/CompressionStuckConfirmReadTests.cs new file mode 100644 index 000000000..d60442c66 --- /dev/null +++ b/Darling/Darling.Tests/CompressionStuckConfirmReadTests.cs @@ -0,0 +1,499 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading; +using System.Threading.Tasks; +using Microsoft.Extensions.Logging; +using PerformanceMonitor.Darling.Storage; +using Xunit; + +namespace Darling.Tests; + +/// +/// #3575: the compression-stuck check's -infinity arm is a NON-ATOMIC read of TimescaleDB's own view, +/// and the fix is to read it twice. +/// +/// The defect these pin against. already +/// guarded its dead-job arm with !isRunning, and a production store paged through the guard anyway — +/// the alert's stamp 53 ms inside a 63 ms scheduled run that succeeded. The 2.28.1 view definition explains +/// it: job_status is CASE WHEN pgs.state = 'active' THEN 'Running' … END over a +/// LEFT JOIN pg_stat_activity on application_name, and next_start is the +/// bgw_job_stat row. The scheduler commits -infinity before the worker exists; the worker is +/// gone before its mark_end is visible to a snapshot taken a moment earlier. A tight poll of +/// across a 10-second-cadence policy on a PG18 + +/// TimescaleDB 2.28.1 rig caught -infinity + Scheduled at BOTH edges of every one of seven runs. +/// The predicate stays pure and single-shot; +/// re-reads after and reports only what +/// persists. These tests script the two reads through the internal seam, so an edge and a dead row are +/// each a pair of result sets and nothing here sleeps. +/// +/// What is NOT pinned here, deliberately. The stuck-Running arm's six-hour bound and the +/// evaluator's re-arm-once/escalate machine are unchanged by #3575 and keep their own pins +/// (TimescaleSupportTests, DarlingSelfAlertTests). The only claims this file makes are about the +/// second read: when it is taken, what it ratifies, what it clears, and what a failed one does. +/// +public sealed class CompressionStuckConfirmReadTests +{ + private static readonly DateTime s_now = new(2026, 9, 18, 8, 47, 0, DateTimeKind.Utc); + + /* The three row shapes the view can hand the predicate for one job, named for what they are. */ + + /// The dead-job shape and the run-instant edge: identical on one read — that is the defect. + private static CompressionJobStatRow NegInfinityScheduled(long jobId, string hypertable = "file_io_stats") => + new(jobId, NextStartIsNegativeInfinity: true, JobStatus: "Scheduled", + LastRunStartedAtUtc: s_now.AddHours(-1), ScheduleInterval: TimeSpan.FromHours(1), HypertableName: hypertable); + + /// A healthy job between runs: finite next_start, not running. + private static CompressionJobStatRow Healthy(long jobId, string hypertable = "file_io_stats") => + new(jobId, NextStartIsNegativeInfinity: false, JobStatus: "Scheduled", + LastRunStartedAtUtc: s_now.AddMinutes(-1), ScheduleInterval: TimeSpan.FromHours(1), HypertableName: hypertable); + + /// The mid-run marker: -infinity WITH Running. Belongs to the elapsed arm, never the dead-job arm. + private static CompressionJobStatRow MidRun(long jobId, string hypertable = "file_io_stats") => + new(jobId, NextStartIsNegativeInfinity: true, JobStatus: "Running", + LastRunStartedAtUtc: s_now.AddMilliseconds(-40), ScheduleInterval: TimeSpan.FromHours(1), HypertableName: hypertable); + + /// A hung run: Running since eight hours ago against the six-hour floor. + private static CompressionJobStatRow HungRun(long jobId, string hypertable = "query_stats") => + new(jobId, NextStartIsNegativeInfinity: true, JobStatus: "Running", + LastRunStartedAtUtc: s_now.AddHours(-8), ScheduleInterval: TimeSpan.FromHours(1), HypertableName: hypertable); + + /// + /// A scripted read: hands back each result set in turn, records how many times it was asked, and throws + /// the scripted exception in place of a result set when one is planted. + /// + private sealed class ScriptedReads + { + private readonly Queue _script = new(); + public int Calls { get; private set; } + + public ScriptedReads Then(params CompressionJobStatRow[] rows) + { + _script.Enqueue((IReadOnlyList)rows); + return this; + } + + public ScriptedReads ThenThrow(Exception ex) + { + _script.Enqueue(ex); + return this; + } + + public Task> Read(CancellationToken ct) + { + Calls++; + Assert.True(_script.Count > 0, $"the read was asked a {Calls}th time with nothing scripted for it"); + var next = _script.Dequeue(); + return next is Exception ex + ? Task.FromException>(ex) + : Task.FromResult((IReadOnlyList)next); + } + } + + /// A recording delay: never sleeps, remembers every span it was asked to wait. + private sealed class RecordedDelay + { + public List Waits { get; } = new(); + + public Task Wait(TimeSpan span, CancellationToken ct) + { + Waits.Add(span); + return Task.CompletedTask; + } + } + + /* ---------------- the arm projection ---------------- */ + + [Fact] + public void ClassifyCompressionJob_NamesTheArm_AndIsCompressionJobStuck_IsItsProjection() + { + /* The boolean the existing pins hold is the classifier's projection, so the two cannot disagree — the + reason the classifier is the implementation and not a sibling copy of the same branches. */ + foreach (var (row, expected) in new (CompressionJobStatRow Row, StuckCompressionJobArm Arm)[] + { + (NegInfinityScheduled(1), StuckCompressionJobArm.NextStartNegativeInfinity), + (Healthy(2), StuckCompressionJobArm.None), + (MidRun(3), StuckCompressionJobArm.None), + (HungRun(4), StuckCompressionJobArm.RunningPastBound), + }) + { + var arm = TimescaleSupport.ClassifyCompressionJob( + row.NextStartIsNegativeInfinity, row.JobStatus, row.LastRunStartedAtUtc, row.ScheduleInterval, s_now, out var reason); + var stuck = TimescaleSupport.IsCompressionJobStuck( + row.NextStartIsNegativeInfinity, row.JobStatus, row.LastRunStartedAtUtc, row.ScheduleInterval, s_now, out var boolReason); + + Assert.Equal(expected, arm); + Assert.Equal(arm != StuckCompressionJobArm.None, stuck); + Assert.Equal(reason, boolReason); + } + } + + /* ---------------- the confirm-read: when the second read is taken ---------------- */ + + [Fact] + public async Task HealthyPass_ReadsOnce_AndNeverWaits() + { + /* The common hourly pass: nothing on the racing arm, so the pass costs exactly what it did before + #3575 — one read, no delay. A confirm taken unconditionally would hold the serial sweep loop five + seconds every hour for nothing. */ + var reads = new ScriptedReads().Then(Healthy(1), Healthy(2), MidRun(3)); + var delay = new RecordedDelay(); + + var result = await TimescaleSupport.ReadStuckCompressionJobsAsync( + reads.Read, delay.Wait, s_now, logger: null, TestContext.Current.CancellationToken); + + Assert.Empty(result); + Assert.Equal(1, reads.Calls); + Assert.Empty(delay.Waits); + } + + [Fact] + public async Task HungRunOnly_ReportsFromTheFirstRead_AndNeverWaits() + { + /* The stuck-Running arm is judged on hours of elapsed time; a second look five seconds later could not + change it, so it neither triggers the confirm nor waits on one. */ + var reads = new ScriptedReads().Then(HungRun(4), Healthy(1)); + var delay = new RecordedDelay(); + + var result = await TimescaleSupport.ReadStuckCompressionJobsAsync( + reads.Read, delay.Wait, s_now, logger: null, TestContext.Current.CancellationToken); + + var job = Assert.Single(result); + Assert.Equal(4L, job.JobId); + Assert.Contains("Running", job.Reason, StringComparison.Ordinal); + Assert.Equal(1, reads.Calls); + Assert.Empty(delay.Waits); + } + + [Fact] + public async Task NegInfinityTrip_WaitsTheConfirmDelay_ThenReadsAgain_Once() + { + /* One confirm per pass, not per job: two jobs on the racing arm still cost one delay and one second + read. The span waited is the published constant, so the budgeting argument on it is the one the + code actually honours. */ + var reads = new ScriptedReads() + .Then(NegInfinityScheduled(1), NegInfinityScheduled(2)) + .Then(NegInfinityScheduled(1), NegInfinityScheduled(2)); + var delay = new RecordedDelay(); + + var result = await TimescaleSupport.ReadStuckCompressionJobsAsync( + reads.Read, delay.Wait, s_now, logger: null, TestContext.Current.CancellationToken); + + Assert.Equal(2, result.Count); + Assert.Equal(2, reads.Calls); + Assert.Equal(new[] { TimescaleSupport.StuckCompressionConfirmDelay }, delay.Waits); + } + + /* ---------------- the confirm-read: what the second read decides ---------------- */ + + [Fact] + public async Task TransientEdge_ClearsOnConfirm_ReportsNothing_AndSaysSoAtInformation() + { + /* THE production shape: the first read lands on the run instant and sees -infinity + Scheduled; five + seconds later the run is long over and the job reads healthy. Nothing is reported, so nothing is + re-armed and nothing is paged — and the near miss is written down where a person reading the log + after this alert family fires would look for it. */ + var reads = new ScriptedReads() + .Then(NegInfinityScheduled(1011, "file_io_stats")) + .Then(Healthy(1011, "file_io_stats")); + var delay = new RecordedDelay(); + var log = new CapturingTestLogger(); + + var result = await TimescaleSupport.ReadStuckCompressionJobsAsync( + reads.Read, delay.Wait, s_now, log, TestContext.Current.CancellationToken); + + Assert.Empty(result); + Assert.Equal(2, reads.Calls); + Assert.Contains("Information:", log.Joined, StringComparison.Ordinal); + Assert.Contains("1011", log.Joined, StringComparison.Ordinal); + Assert.Contains("file_io_stats", log.Joined, StringComparison.Ordinal); + Assert.Contains("run-instant edge", log.Joined, StringComparison.Ordinal); + Assert.Contains("#3575", log.Joined, StringComparison.Ordinal); + Assert.DoesNotContain("Warning:", log.Joined, StringComparison.Ordinal); + } + + [Fact] + public async Task PersistentNegInfinity_IsConfirmed_AndReported() + { + /* A row the scheduler has abandoned — or a crashed run sitting out its five-minute backoff, reproduced + on the rig by SIGKILLing a worker — reads the same on both passes. Detection is unchanged in kind; + it is five seconds later in time. */ + var reads = new ScriptedReads() + .Then(NegInfinityScheduled(7, "wait_stats")) + .Then(NegInfinityScheduled(7, "wait_stats")); + + var result = await TimescaleSupport.ReadStuckCompressionJobsAsync( + reads.Read, new RecordedDelay().Wait, s_now, logger: null, TestContext.Current.CancellationToken); + + var job = Assert.Single(result); + Assert.Equal(7L, job.JobId); + Assert.Equal("wait_stats", job.HypertableName); + Assert.Contains("-infinity", job.Reason, StringComparison.Ordinal); + } + + [Fact] + public async Task EdgeThatBecameMidRunOnConfirm_Clears() + { + /* The start edge seen twice in one run, at different stages: first -infinity + Scheduled (the worker + not yet Running), then -infinity + Running (mid-run). The second read's arm is None, so the trip + clears — mid-run belongs to the elapsed arm and always did. */ + var reads = new ScriptedReads() + .Then(NegInfinityScheduled(1)) + .Then(MidRun(1)); + + var result = await TimescaleSupport.ReadStuckCompressionJobsAsync( + reads.Read, new RecordedDelay().Wait, s_now, logger: null, TestContext.Current.CancellationToken); + + Assert.Empty(result); + } + + [Fact] + public async Task MixedPass_KeepsTheHungRun_ConfirmsOneTrip_ClearsTheOther() + { + /* Every row class in one pass, so the merge is pinned as a table rather than one row at a time: the + hung run from the first read, the confirmed trip from the second, the transient trip dropped, and + the always-healthy job never mentioned. */ + var reads = new ScriptedReads() + .Then(HungRun(4), NegInfinityScheduled(1), NegInfinityScheduled(2), Healthy(3)) + .Then(HungRun(4), NegInfinityScheduled(1), Healthy(2), Healthy(3)); + + var result = await TimescaleSupport.ReadStuckCompressionJobsAsync( + reads.Read, new RecordedDelay().Wait, s_now, logger: null, TestContext.Current.CancellationToken); + + Assert.Equal(new[] { 4L, 1L }, result.Select(r => r.JobId).ToArray()); + } + + [Fact] + public async Task JobThatOnlyTripsOnTheConfirm_IsNotReported() + { + /* The confirm ratifies the first pass; it does not widen it. A job that went -infinity between the two + reads has been seen once, which is the count this issue proved insufficient, and it gets its own two + reads next hour. */ + var reads = new ScriptedReads() + .Then(NegInfinityScheduled(1), Healthy(2)) + .Then(NegInfinityScheduled(1), NegInfinityScheduled(2)); + + var result = await TimescaleSupport.ReadStuckCompressionJobsAsync( + reads.Read, new RecordedDelay().Wait, s_now, logger: null, TestContext.Current.CancellationToken); + + Assert.Equal(new[] { 1L }, result.Select(r => r.JobId).ToArray()); + } + + /* ---------------- the confirm-read: failure isolation ---------------- */ + + [Fact] + public async Task ConfirmReadFails_DropsTheTrips_KeepsTheHungRun_WarnsOnce_DoesNotThrow() + { + /* A confirm that fails confirms nothing. The -infinity trips are not reported on the strength of the + single read this issue proved insufficient; the hung run, judged from the first read, still is. The + failure is a Warning naming the count and the consequence — the views were readable seconds ago, so + this is a hiccup on the one read that decides whether to page — and it never reaches the sweep. */ + var reads = new ScriptedReads() + .Then(HungRun(4), NegInfinityScheduled(1), NegInfinityScheduled(2)) + .ThenThrow(new InvalidOperationException("connection reset by peer")); + var log = new CapturingTestLogger(); + + var result = await TimescaleSupport.ReadStuckCompressionJobsAsync( + reads.Read, new RecordedDelay().Wait, s_now, log, TestContext.Current.CancellationToken); + + var job = Assert.Single(result); + Assert.Equal(4L, job.JobId); + Assert.Contains("Warning:", log.Joined, StringComparison.Ordinal); + Assert.Contains("2 job(s)", log.Joined, StringComparison.Ordinal); + Assert.Contains("confirm read", log.Joined, StringComparison.Ordinal); + Assert.Contains("next hour", log.Joined, StringComparison.Ordinal); + Assert.Contains("connection reset by peer", log.Joined, StringComparison.Ordinal); + Assert.Contains("#3575", log.Joined, StringComparison.Ordinal); + } + + [Fact] + public async Task FirstReadFails_ReturnsEmpty_LogsDebug_NeverWaits_DoesNotThrow() + { + /* The pre-#3575 posture, unchanged: a failed first read is "no signal this check" at Debug — the views + may simply be absent on a plain-PG store — and there is nothing to confirm, so no delay is taken. */ + var reads = new ScriptedReads().ThenThrow(new InvalidOperationException("relation does not exist")); + var delay = new RecordedDelay(); + var log = new CapturingTestLogger(); + + var result = await TimescaleSupport.ReadStuckCompressionJobsAsync( + reads.Read, delay.Wait, s_now, log, TestContext.Current.CancellationToken); + + Assert.Empty(result); + Assert.Empty(delay.Waits); + Assert.StartsWith("Debug:", log.Joined, StringComparison.Ordinal); + Assert.DoesNotContain("Warning:", log.Joined, StringComparison.Ordinal); + } + + [Fact] + public async Task Cancellation_DuringTheConfirmDelay_Propagates() + { + /* Shutdown during the five-second wait is cancellation, not a read failure: it propagates to the worker's + own quiet catch rather than being swallowed into "confirm failed" and logged as a store fault. */ + var reads = new ScriptedReads().Then(NegInfinityScheduled(1)); + using var cts = new CancellationTokenSource(); + + Task CancelInsteadOfWaiting(TimeSpan span, CancellationToken ct) + { + cts.Cancel(); + return Task.FromCanceled(cts.Token); + } + + await Assert.ThrowsAnyAsync(() => + TimescaleSupport.ReadStuckCompressionJobsAsync(reads.Read, CancelInsteadOfWaiting, s_now, null, cts.Token)); + } + + /* ---------------- the pure merge, pinned directly ---------------- */ + + [Fact] + public void ConfirmStuckCompressionJobs_NullConfirm_DropsEveryNegInfinityTrip_KeepsRunningPastBound() + { + var first = TimescaleSupport.ClassifyStuckCompressionJobs( + new[] { HungRun(4), NegInfinityScheduled(1), NegInfinityScheduled(2) }, s_now); + Assert.Equal(3, first.Count); + + var merged = TimescaleSupport.ConfirmStuckCompressionJobs(first, confirm: null, s_now, logger: null); + + Assert.Equal(new[] { 4L }, merged.Select(m => m.JobId).ToArray()); + } + + [Fact] + public void ConfirmStuckCompressionJobs_CarriesTheConfirmPassRow() + { + /* The reported job is built from the CONFIRM pass's row — the later read is the one that stood. Today the + two reasons are the same string; the pin is on which row is carried, using the hypertable name the two + passes would only ever disagree on in a test. */ + var first = TimescaleSupport.ClassifyStuckCompressionJobs(new[] { NegInfinityScheduled(1, "first") }, s_now); + var merged = TimescaleSupport.ConfirmStuckCompressionJobs( + first, new[] { NegInfinityScheduled(1, "confirm") }, s_now, logger: null); + + var job = Assert.Single(merged); + Assert.Equal("confirm", job.HypertableName); + } + + /* ---------------- the delay constant, against what it has to clear and what it costs ---------------- */ + + [Fact] + public void ConfirmDelay_ClearsTheMeasuredEdges_AndIsSmallAgainstEveryCadenceItSitsInside() + { + var delay = TimescaleSupport.StuckCompressionConfirmDelay; + + /* Quoted measurements, not derived: the rig's whole run was ~7.5 ms end to end and its start edge + ~3 ms; the production store's hourly no-op runs were 40–100 ms; a Windows backend start is realistically + tens of milliseconds. The delay must outlast all of them by an order of magnitude at least. */ + Assert.True(delay >= TimeSpan.FromMilliseconds(100 * 10), + $"the confirm delay ({delay}) must clear a 100 ms run-instant edge ten times over"); + + /* ...and it must stay far below the shortest PERSISTENT -infinity state there is, TimescaleDB's + MIN_WAIT_AFTER_CRASH_MS of five minutes, or a real crash could slip between the two reads. */ + Assert.True(delay < TimeSpan.FromMinutes(1), + $"the confirm delay ({delay}) must be far below the five-minute crash backoff a crashed row sits at -infinity for"); + + /* ...and it must be negligible against the cadences it sits inside — the hourly check and the six-hour + stuck-Running floor — so a genuinely dead job is detected on the same pass it always was. */ + Assert.True(delay.TotalSeconds * 100 < TimeSpan.FromHours(1).TotalSeconds, + $"the confirm delay ({delay}) must be under 1 % of the hourly check cadence"); + Assert.True(delay * 100 < TimescaleSupport.StuckRunningBound(TimeSpan.FromHours(1)), + $"the confirm delay ({delay}) must be under 1 % of the stuck-Running bound"); + } + + /* ---------------- de-alignment: the check's wall-clock phase ---------------- */ + + [Fact] + public void NextCompressionCheckUtc_SnapsToThePhase_NeverToAMinuteBoundary() + { + var interval = TimeSpan.FromHours(1); + + /* The production fire: 08:47:00.053, 53 ms into file_io_stats' :47:00 run. Scheduled from this fire the + old way, the next sample would have been 09:47:00.053 + the loop's latency — on the boundary again. + Snapped, it is 09:47:30 exactly. */ + var fired = new DateTime(2026, 9, 18, 8, 47, 0, 53, DateTimeKind.Utc); + var next = TimescaleSupport.NextCompressionCheckUtc(fired, interval); + Assert.Equal(new DateTime(2026, 9, 18, 9, 47, 30, DateTimeKind.Utc), next); + Assert.Equal(DateTimeKind.Utc, next.Kind); + + /* From any second of the minute, the result sits on the phase, and the phase is not zero — the whole + point is to be OFF the :00 instant the compression policies fire on. */ + Assert.NotEqual(0, TimescaleSupport.CompressionCheckPhaseSeconds); + for (var second = 0; second < 60; second++) + { + var now = new DateTime(2026, 9, 18, 8, 47, second, 500, DateTimeKind.Utc); + var due = TimescaleSupport.NextCompressionCheckUtc(now, interval); + Assert.Equal(TimescaleSupport.CompressionCheckPhaseSeconds, due.Second); + Assert.Equal(0, due.Millisecond); + /* And the cadence stays hourly to within half a minute either way — never two hours, never zero. */ + var spacing = due - now; + Assert.InRange(spacing, interval - TimeSpan.FromSeconds(30), interval + TimeSpan.FromSeconds(30)); + } + } + + [Fact] + public void NextCompressionCheckUtc_LocksThePhase_AcrossSimulatedFires_UnderLoopLatency() + { + /* The steady state the field will see: each fire happens at the first 15-second sweep pass at or after + the due time, so the fire is 0–15 s late; the next due must still snap back to :30, so the phase does + not creep the way "UtcNow + 1 h" did. Simulated over a day with every latency the loop can produce. */ + var interval = TimeSpan.FromHours(1); + var due = TimescaleSupport.NextCompressionCheckUtc(new DateTime(2026, 9, 18, 1, 46, 13, DateTimeKind.Utc), interval); + + for (var hour = 0; hour < 24; hour++) + { + var latency = TimeSpan.FromSeconds(hour % 16); /* 0..15 s, the loop's whole range */ + var fired = due + latency; + due = TimescaleSupport.NextCompressionCheckUtc(fired, interval); + + Assert.Equal(TimescaleSupport.CompressionCheckPhaseSeconds, due.Second); + Assert.Equal(46, due.Minute); /* the minute never moves while the loop keeps under the half-minute */ + } + } + + [Fact] + public void NextCompressionCheckUtc_ALatePass_SnapsBackToThePhase_OrMovesAWholeMinute_NeverTowardTheBoundary() + { + /* A sweep pass delayed past the half-minute (the #2327 store-metrics worst case can hold the loop for + minutes) fires late. Re-anchored from the fire the old way, the next due would carry that lateness + forward and creep toward a boundary. Snapped, there are only two outcomes, and neither is nearer :00. + + Late inside its own minute (:46:42): the next due snaps BACK to :46:30 — twelve seconds short of a + full hour, still on the phase. */ + var lateInMinute = new DateTime(2026, 9, 18, 9, 46, 42, DateTimeKind.Utc); + Assert.Equal( + new DateTime(2026, 9, 18, 10, 46, 30, DateTimeKind.Utc), + TimescaleSupport.NextCompressionCheckUtc(lateInMinute, TimeSpan.FromHours(1))); + + /* Late into the NEXT minute (:47:05, a pass more than 35 s behind its :46:30 due): the check moves one + whole minute later, to :47:30, and stays on the phase there. */ + var lateIntoNextMinute = new DateTime(2026, 9, 18, 9, 47, 5, DateTimeKind.Utc); + Assert.Equal( + new DateTime(2026, 9, 18, 10, 47, 30, DateTimeKind.Utc), + TimescaleSupport.NextCompressionCheckUtc(lateIntoNextMinute, TimeSpan.FromHours(1))); + } + + [Fact] + public void ThePhase_IsHalfTheCompressionGridStep_AndClearOfEveryPolicyStart() + { + /* The compression policies start at :MM:00 for every MM in CompressionPhaseMinutes (the fixed-schedule + initial_start carries whole minutes and nothing smaller). Half the one-minute grid step is the point + furthest from every start instant in both directions. Derived from the grid's step, not restated. */ + Assert.Equal(30, TimescaleSupport.CompressionCheckPhaseSeconds); + Assert.Equal(TimeSpan.FromMinutes(1) / 2, TimeSpan.FromSeconds(TimescaleSupport.CompressionCheckPhaseSeconds)); + + /* And the AddCompressionPolicySql initial_start really is whole minutes: any policy this product owns + puts its job on :MM:00, so the :30 phase is 30 s from every one of them. */ + foreach (var table in TimescaleSupport.CompressionPhaseOrder) + { + Assert.True(TimescaleSupport.TryCompressionPhaseMinutesFor(table, out var minute)); + Assert.Contains( + $"INTERVAL '{minute} minutes'", + TimescaleSupport.AddCompressionPolicySql(table), StringComparison.Ordinal); + Assert.DoesNotContain("seconds", TimescaleSupport.AddCompressionPolicySql(table), StringComparison.Ordinal); + } + } +} diff --git a/Darling/Darling.Tests/TimescaleSupportTests.cs b/Darling/Darling.Tests/TimescaleSupportTests.cs index dd8ab8586..58b3a0697 100644 --- a/Darling/Darling.Tests/TimescaleSupportTests.cs +++ b/Darling/Darling.Tests/TimescaleSupportTests.cs @@ -164,7 +164,11 @@ managed store compact (#1458). */ public void IsCompressionJobStuck_NextStartNegativeInfinity_IsStuck() { /* The dominant failure mode: next_start = -infinity on a job that is NOT running — the scheduler - abandoned it and never re-fires it. */ + abandoned it and never re-fires it. ONE read says so here, and one read is what the predicate + judges; since #3575 the reader (ReadStuckCompressionJobsAsync) asks twice five seconds apart before + it believes this arm, because the view assembles "not running" and "-infinity" from independent + sources and reads this exact shape for a few milliseconds at either edge of every healthy run. + The predicate itself stays single-shot — CompressionStuckConfirmReadTests pins the second read. */ Assert.True(TimescaleSupport.IsCompressionJobStuck( nextStartIsNegativeInfinity: true, jobStatus: "Scheduled", lastRunStartedAtUtc: null, scheduleInterval: TimeSpan.FromHours(12), nowUtc: s_now, out var reason)); @@ -178,7 +182,13 @@ public void IsCompressionJobStuck_NegativeInfinityWhileRunning_IsTheMidRunMarker -infinity WITH job_status = 'Running' — the engine only computes the real next start when the run finishes. An unconditioned -infinity arm flagged every healthy job caught mid-run (the field's transient stuck→self-healed alert noise, and the CI flake where the live test caught - its own re-arm-triggered run). Mid-run belongs to the elapsed-bound arm: */ + its own re-arm-triggered run). Mid-run belongs to the elapsed-bound arm. + + This guard was necessary and was not sufficient (#3575): 'Running' is pg_stat_activity, read + live, and -infinity is the bgw_job_stat row, read under the statement's snapshot, so the guard + is blind for the milliseconds between the scheduler committing -infinity and the worker + reporting itself active, and again between the worker leaving and its mark_end becoming + visible. That is the reader's problem to close (it re-reads), not this predicate's: */ Assert.False(TimescaleSupport.IsCompressionJobStuck( nextStartIsNegativeInfinity: true, jobStatus: "Running", lastRunStartedAtUtc: s_now.AddMinutes(-3), scheduleInterval: TimeSpan.FromHours(12), nowUtc: s_now, out _)); @@ -2051,7 +2061,13 @@ DETECTION logic is covered by the pure IsCompressionJobStuck unit tests. */ on one snapshot": next_start => now() makes the job immediately due, the scheduler picks it up, and from pickup to completion job_stats reads next_start = -infinity with status Running — the mid-run marker (measured live; the detector now defers that state to its elapsed-bound arm). A single - un-settled read raced the very run the re-arm triggered, which was this test's own flake. */ + un-settled read raced the very run the re-arm triggered, which was this test's own flake. + + Since #3575 the detector also re-reads five seconds later before it reports the -infinity arm, so + the run-instant EDGES (-infinity while the worker is not yet, or no longer, visible as Running — + the shape that paged a production store) clear inside one call rather than surfacing as a flagged + poll here. The wait stays: it is the assertion's contract, and a poll that lands on the edge now + costs five seconds of confirm rather than a flagged iteration. */ await WaitUntilDetectorReportsHealthyAsync(connection, jobId, ct); } diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingSelfAlertEvaluator.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingSelfAlertEvaluator.cs index bf5e4c3c9..d2fef0863 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingSelfAlertEvaluator.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingSelfAlertEvaluator.cs @@ -3872,6 +3872,22 @@ await RecordResolutionAsync(new AlertResolution( /// resolution row is written (the sibling conditions' edge shape). Gated on the master alerts switch. /// Re-arm happens at most ONCE per job per check (only on the first-detection transition). Internal so it /// pins directly with a recording deliverer, a controllable clock, and a fake re-arm delegate. + /// + /// What this machine trusts, and what it cost when the trust was misplaced (#3575). This takes + /// as settled fact: first sight re-arms and pages Critical, absence an hour + /// later posts Recovered. So one false row in the list is not one false message but three — the page, the + /// idempotent re-arm it narrates, and the recovery of a job that was never unwell — on the alert family + /// that reports the store's own health. A production store produced exactly that set from a healthy job: + /// the detector's -infinity arm already guarded on job_status, but TimescaleDB's + /// job_stats view assembles that status from pg_stat_activity and next_start from the + /// job-stat row, and for a few milliseconds at either edge of every run the two disagree in exactly the + /// dead-job shape; the check's sample landed 53 ms into a 63 ms run that succeeded. The fix is upstream + /// of here and deliberately so: TimescaleSupport.ReadStuckCompressionJobsAsync now confirms a + /// -infinity trip with a second read five seconds later before a job reaches this list, and the + /// worker pins its samples to :30 past the minute, off the policies' :MM:00 run instants. + /// This method keeps its single-sample semantics — first sight IS first sight — because the input is now + /// worth that trust, and adding hysteresis here instead would have bought the same protection for an + /// hour of detection latency on a genuinely dead job. /// internal async Task ApplyCompressionJobsStuckAsync( IReadOnlyList stuckJobs, diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs index 95446c8ef..69bc00c6e 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs @@ -114,7 +114,14 @@ so matching its #3304 neighbour costs nothing. */ /* The compression-job self-heal check's cadence (fleet-level, #1581). Compression is a slow archival tier and a stuck policy job takes hours to matter, so hourly is ample and cheap (one job_stats read + at most - one alter_job per stuck job) — no need for the 15s sweep or the 30s alert cadence. */ + one alter_job per stuck job) — no need for the 15s sweep or the 30s alert cadence. + + The PHASE of that hour is not this constant's to choose and is not chosen by "UtcNow + interval" any + more (#3575): the compression policies this check watches fire at :MM:00 of the wall clock on a fixed + schedule (#3035), and a check scheduled from the instant of its previous fire slips a few seconds + every hour and eventually samples one of those :00 instants — which is where a production store's + false page came from. TimescaleSupport.NextCompressionCheckUtc snaps each due time to :30 past its + minute, so this interval sets how OFTEN and that phase sets WHEN in the minute. */ private static readonly TimeSpan s_compressionCheckInterval = TimeSpan.FromHours(1); /* The store self-metrics sweep's cadence (fleet-level, #2068). Store growth is a slow signal — the @@ -475,8 +482,12 @@ a best-effort errand and there is nothing in it that a later hour cannot do. */ private Task? _oversizedPlanSweep; /* MinValue = the first sweep after startup evaluates the compression-job self-heal check (#1581), then - every s_compressionCheckInterval. Fleet-level (one shared store), so it is a single field, not - per-server; only consulted when _timescaleAvailable. */ + every s_compressionCheckInterval, pinned to :30 past the minute by TimescaleSupport.NextCompressionCheckUtc + (#3575) so no steady-state sample lands on the :MM:00 instant the compression policies fire on. The + first sample is deliberately left unpinned — a restart is when an operator is reading the log and wants + the store's job health now — and the confirm-read inside ReadStuckCompressionJobsAsync covers it like + every other sample. Fleet-level (one shared store), so it is a single field, not per-server; only + consulted when _timescaleAvailable. */ private DateTime _nextCompressionCheckUtc = DateTime.MinValue; /* MinValue = the first loop pass after startup runs the fleet sweep (#3466 lane 2), then on the @@ -2155,10 +2166,21 @@ await _selfAlerts.EvaluateWebTlsCertificateAsync( /* #1581: the compression-job self-heal backstop. TimescaleDB compression policy jobs can silently die (next_start = -infinity) or hang, halting the store's archival tier so uncompressed data grows without bound until the disk fills and collection stops for the WHOLE fleet (the field incident). - Timescale-only; own hourly cadence; failure-isolated inside EvaluateCompressionJobHealthAsync. */ + Timescale-only; own hourly cadence; failure-isolated inside EvaluateCompressionJobHealthAsync. + + The next due time is SNAPPED to the wall clock rather than taken from this fire (#3575). The + dead-job arm the check judges reads next_start = -infinity, which is also what the scheduler + writes for the few milliseconds at either edge of every healthy run before the worker is, or + after it stops being, visible as Running — and the policies run at :MM:00 on a fixed schedule, + so a check that re-anchored itself as "UtcNow + 1 h" on every fire crept a few seconds per hour + across those instants until, on a production store, it sampled one 53 ms into a 63 ms run and + paged. NextCompressionCheckUtc puts every steady-state sample at :30 past its minute instead, + half the grid step from every policy's start in both directions. The read itself now confirms + a -infinity trip with a second read five seconds later (ReadStuckCompressionJobsAsync), so the + phase is hardening on top of the fix, not the fix. */ if (_timescaleAvailable && DateTime.UtcNow >= _nextCompressionCheckUtc) { - _nextCompressionCheckUtc = DateTime.UtcNow.Add(s_compressionCheckInterval); + _nextCompressionCheckUtc = TimescaleSupport.NextCompressionCheckUtc(DateTime.UtcNow, s_compressionCheckInterval); await EvaluateCompressionJobHealthAsync(stoppingToken); } @@ -5138,12 +5160,22 @@ evidence the alert is judged on - that is freeBytes/totalBytes above. Losing it /// /// The #1581 compression-job self-heal check (fleet-level, hourly, Timescale-only): read every stuck - /// COMPRESSION-policy job () and hand them to the - /// self-alert evaluator's re-arm-once/escalate machine, wired to - /// on the SAME open connection. One stuck job whose next_start went -infinity silently halts the - /// store's archival tier — the field incident — so this makes it visible AND self-heals it. Failure-isolated - /// at the worker level too (the connection open is OUTSIDE the evaluator's own isolation): a store hiccup logs - /// and skips this check, never aborting the sweep — mirroring the purge / disk-check isolation. + /// COMPRESSION-policy job () + /// and hand them to the self-alert evaluator's re-arm-once/escalate machine, wired to + /// on the SAME open connection. One stuck job whose + /// next_start went -infinity silently halts the store's archival tier — the field incident — so + /// this makes it visible AND self-heals it. Failure-isolated at the worker level too (the connection open is + /// OUTSIDE the evaluator's own isolation): a store hiccup logs and skips this check, never aborting the sweep — + /// mirroring the purge / disk-check isolation. + /// + /// The stuck-job read may hold this method for + /// (#3575), and only on a pass where a job's -infinity arm tripped: the read re-executes its query + /// after that delay and reports the job only if the arm still trips, because TimescaleDB's view assembles + /// next_start and job_status from independent sources and reads the dead-job shape for a few + /// milliseconds at either edge of every healthy run. This method is awaited on the serial sweep loop, so + /// that five seconds is a once-an-hour worst case paid only when there was something to confirm; the + /// budgeting argument is on the constant. The evaluator downstream receives a list that has already been + /// confirmed and does not second-guess it. /// private async Task EvaluateCompressionJobHealthAsync(CancellationToken cancellationToken) { diff --git a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs index 498f2b29b..24e034ded 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs @@ -5634,6 +5634,122 @@ public static TimeSpan StuckRunningBound(TimeSpan? scheduleInterval) return s_stuckRunningFloor; } + /// + /// How long + /// waits before it RE-READS a job whose -infinity arm tripped, and requires the trip to persist + /// (#3575). Only taken when that arm trips; a pass with nothing to confirm costs nothing. + /// + /// Why a confirm-read exists at all. The -infinity arm already carried a running + /// guard (nextStartIsNegativeInfinity && !isRunning) and a false page came through it + /// anyway, on a production store, with the alert's stamp 53 ms inside a 63 ms scheduled run that + /// succeeded. The guard's two inputs are read from INDEPENDENT sources inside TimescaleDB's own view, + /// and the 2.28.1 definition (pg_get_viewdef('timescaledb_information.job_stats'), read live) + /// says so exactly: job_status is CASE WHEN pgs.state = 'active' THEN 'Running' WHEN + /// j.scheduled = false THEN 'Paused' ELSE 'Scheduled' END over a LEFT JOIN pg_stat_activity pgs + /// ON pgs.application_name = j.application_name, while next_start is + /// _timescaledb_internal.bgw_job_stat.next_start. The scheduler's mark_start writes + /// next_start = -infinity (and last_finish = -infinity) in its OWN transaction and commits + /// it BEFORE the worker process is even registered; the worker then has to start, initialise its + /// connection, run one catalog transaction, report its application_name and finally call + /// pgstat_report_activity(STATE_RUNNING) before the join can say Running. Every read that + /// lands in that START EDGE sees -infinity AND Scheduled, which is this predicate's + /// dead-job arm. There is an END EDGE too, with a different cause: the catalog row is read under the + /// statement's MVCC snapshot while pg_stat_activity is read live from shared memory, so one SELECT + /// can pair a pre-mark_end row (-infinity) with a post-exit activity view (no backend, so + /// Scheduled). Both edges were captured on a PG18 + TimescaleDB 2.28.1 rig by polling + /// in a tight loop across a 10-second-cadence policy: every run + /// showed ~3 ms of -infinity + Scheduled before the first Running sample and one more such + /// sample after the last, in a run ~7.5 ms long end to end. On a Windows store — where the production + /// page came from — backend process creation is far slower than a Linux fork, so the start edge is a + /// larger share of a run that is itself only tens of milliseconds when there is nothing to compress. + /// + /// Why a confirm-read and not a same-source running signal. The stat row DOES carry its own + /// mid-run marker — mark_start sets last_finish = -infinity, which the view surfaces as + /// last_run_status IS NULL and last_run_duration IS NULL (the duration is + /// CASE WHEN js.last_finish > js.last_start in every sql/views.sql from 2.14 through + /// 2.28.1, so it reads NULL mid-run and never negative; the belief that it "goes negative" is not borne + /// out by any version checked) — and deriving "running" from it would make both inputs one row. It was + /// rejected because that marker is IDENTICAL for a run whose worker died + /// before mark_end ever ran, which is precisely the state this arm exists to catch: reproduced on + /// the rig by SIGKILLing a job's worker, after which the row read -infinity + Scheduled + + /// last_run_status NULL for the whole five-minute crash backoff. A same-source guard would have read + /// that as "running" and stayed silent on the one state it is for. Re-reading after a delay makes no such + /// assumption: an edge is over in milliseconds, a dead row is still dead seconds later. + /// + /// Why five seconds. The transient it has to outlast is bounded by the run's own edges: the + /// start edge is one process start plus one catalog transaction (~3 ms measured on Linux; tens of + /// milliseconds is the realistic Windows figure, and a pathological second is still covered five times + /// over), and a short run bounds the whole exposure at its own duration (40–100 ms is what a production + /// store's hourly no-op compressions measure). Against what it costs, five seconds is 1/720 of the hourly + /// cadence and 1/4,320 of the six-hour floor, so a genuinely dead job is + /// detected on the same hourly pass it always was, five seconds later. And it is far below the shortest + /// PERSISTENT -infinity state there is: TimescaleDB's crash backoff holds the row at + /// -infinity for at least MIN_WAIT_AFTER_CRASH_MS (five minutes) before the scheduler + /// re-runs a crashed job, so a real crash cannot slip between the two reads. The check is AWAITED on the + /// worker's serial sweep loop (#2327's concern), so the delay is bounded, cancellable, and paid only on + /// the rare pass where the arm tripped at all. + /// + public static readonly TimeSpan StuckCompressionConfirmDelay = TimeSpan.FromSeconds(5); + + /// + /// The compression-job health check's wall-clock phase (#3575): how many seconds past a minute boundary + /// the hourly check is pinned to. applies it. + /// + /// Why the check needs a phase at all, stated against how the two schedules are really + /// anchored — which is not how the postmortem first described them. Every compression policy this + /// product owns runs on a FIXED schedule whose initial_start is date_trunc('hour', now()) + + /// 1 hour + <phase minutes> (, #3035), so its runs + /// begin at :MM:00.000 of the wall clock for the 24 minutes of + /// — three jobs to a minute, every hour, on every store. The health check, by contrast, was anchored to + /// nothing in the wall clock: its first sample was the first sweep pass after service start and each + /// later one was scheduled as UtcNow + 1 hour at the moment of the previous fire, which lands on + /// the first 15-second sweep pass at or after that instant. So its second-of-the-hour SLIPPED forward by + /// the loop's latency every hour — a few seconds to fifteen — and swept across every minute boundary in + /// the band in turn, roughly one boundary every twelve hours. The service behind the production page + /// started at :46:13; seven hourly slips later its sample sat 53 ms past :47:00, the minute + /// file_io_stats compresses on. The schedules were therefore NOT aligned by construction — a + /// service-start anchor drifts, a wall-clock anchor does not — and the collision was the ~1-in-a-hundred + /// draw a drifting sample takes each time it crosses a boundary: the crossing is certain, the landing + /// inside a ~60 ms window is chance. That is still a false page every month or two per store, forever, + /// which is what "structural" correctly meant. + /// + /// Why thirty seconds and not an offset from the service start. Any offset from a DRIFTING + /// anchor drifts with it, so "+N minutes from start" would cross the same boundaries N minutes later. + /// The only offset that holds is one measured from the jobs' own grid, and the grid step is one minute + /// with every job at :00 of its minute — so half a step, :30, is the point furthest from + /// every job's start instant in both directions: 30 s from the previous boundary and 30 s from the next, + /// against edges measured in milliseconds and runs measured in tens of milliseconds when there is nothing + /// to compress. (A run that IS compressing a chunk lasts minutes and straddles :30, but a job that + /// long reports Running, which the -infinity arm already yields to.) Pinning to the wall + /// clock rather than the previous fire is what stops the drift: + /// floors the hourly due time to its minute and adds this phase, so a fire at :47:3x schedules + /// the next at exactly :47:30. A fire that lands LATE in its minute (a slow sweep pass) snaps the + /// next due back to :30 of that same minute, a few seconds short of a full hour; one that lands in + /// the NEXT minute moves the check one whole minute later. Either way the sample is at :30, and + /// nothing the loop does can walk it toward :00. + /// + /// The first check after a restart is deliberately NOT phased — it runs on the first sweep pass, + /// because a restart is when an operator is reading the log and wants the store's job health now — so + /// that single sample keeps the pre-#3575 odds (24 minutes × ~0.1 s of edge in 3,600 s, under 0.1 %), + /// and covers it the same way it covers every other sample. + /// The phase is the hardening; the confirm-read is the fix. + /// + public const int CompressionCheckPhaseSeconds = 30; + + /// + /// When the compression-job health check should next run (#3575): one after + /// , snapped to past that minute so no + /// steady-state sample is ever taken on the :MM:00 instant the compression policies fire on. + /// Pure, so it pins. The snap moves the due time by at most 30 s either way, so the cadence stays + /// hourly to within the loop's own latency; what it never does is land on :00. + /// + public static DateTime NextCompressionCheckUtc(DateTime nowUtc, TimeSpan interval) + { + var due = nowUtc + interval; + var minute = new DateTime(due.Ticks - (due.Ticks % TimeSpan.TicksPerMinute), DateTimeKind.Utc); + return minute.AddSeconds(CompressionCheckPhaseSeconds); + } + /// /// The pure stuck-compression-job decision (#1581). A compression policy job is STUCK when either: /// @@ -5645,6 +5761,8 @@ public static TimeSpan StuckRunningBound(TimeSpan? scheduleInterval) /// /// A job with neither condition is healthy and is NOT flagged. No I/O, so it pins directly with a /// controllable clock. Scoping to compression jobs happens in the query — this decides only "stuck". + /// is the same decision naming WHICH arm fired, for the caller that + /// has to treat the two arms differently. /// /// -infinity is ALSO the engine's mid-run marker, measured live on TimescaleDB /// 2.x (pg17): from the moment the scheduler picks up a due job until its run completes, @@ -5656,6 +5774,17 @@ public static TimeSpan StuckRunningBound(TimeSpan? scheduleInterval) /// left to the second arm, whose elapsed bound is what actually distinguishes a hung run from a /// healthy one. /// + /// And the running guard is itself a non-atomic read (#3575). job_status comes from + /// pg_stat_activity and next_start from bgw_job_stat; the scheduler commits + /// -infinity before the worker exists, and the worker is gone before its mark_end is + /// visible to a snapshot taken a moment earlier, so at both edges of every run the view reads + /// -infinity AND Scheduled — this arm, on a healthy job, for a few milliseconds an hour. + /// A production store paged on exactly that: the alert stamp sat 53 ms inside a 63 ms run that + /// succeeded. This predicate stays pure and single-shot on purpose; the caller closes the race by + /// re-reading after and requiring the -infinity arm + /// to persist (), + /// and the worker keeps its samples off the jobs' run instant (). + /// /// A of counts as NEVER RAN, /// not as "started in year 1" (#1760). already NULLIFs TimescaleDB's /// -infinity never-ran sentinel, so this is the second line of defence: the sentinel maps to @@ -5669,13 +5798,30 @@ public static bool IsCompressionJobStuck( TimeSpan? scheduleInterval, DateTime nowUtc, out string reason) + => ClassifyCompressionJob(nextStartIsNegativeInfinity, jobStatus, lastRunStartedAtUtc, scheduleInterval, nowUtc, out reason) + != StuckCompressionJobArm.None; + + /// + /// with the arm named (#3575): the confirm-read applies ONLY to + /// , because that is the arm whose inputs + /// race; is judged on six hours of elapsed time and + /// a second read five seconds later could not change it. Same decision, same reason text — this is the + /// implementation and the boolean is its projection, so the two cannot drift. + /// + public static StuckCompressionJobArm ClassifyCompressionJob( + bool nextStartIsNegativeInfinity, + string? jobStatus, + DateTime? lastRunStartedAtUtc, + TimeSpan? scheduleInterval, + DateTime nowUtc, + out string reason) { var isRunning = string.Equals(jobStatus, "Running", StringComparison.OrdinalIgnoreCase); if (nextStartIsNegativeInfinity && !isRunning) { reason = "next_start is -infinity — the scheduler will never run it again"; - return true; + return StuckCompressionJobArm.NextStartNegativeInfinity; } if (isRunning @@ -5690,12 +5836,12 @@ public static bool IsCompressionJobStuck( CultureInfo.InvariantCulture, "stuck in the Running state for {0:F0} minutes (over the {1:F0}-minute bound) — the run hung and never finished", elapsed.TotalMinutes, bound.TotalMinutes); - return true; + return StuckCompressionJobArm.RunningPastBound; } } reason = ""; - return false; + return StuckCompressionJobArm.None; } /// @@ -5716,6 +5862,16 @@ public static bool IsCompressionJobStuck( /// while last_run_started_at is bgw_job_stat.last_start. A job's FIRST run therefore reads /// Running while its start time is still the sentinel, and that window flagged a perfectly healthy /// job as stuck — which the self-heal then "fixed" by re-arming a job that was running fine. + /// + /// The same two sources are why one execution of this statement cannot be trusted alone on the + /// -infinity arm (#3575). next_start_neg_infinity is the stat row under the statement's + /// snapshot; job_status is pg_stat_activity read live. At the start of every run the row + /// already says -infinity while no backend yet says Running, and at the end the backend can + /// be gone while the snapshot still holds the pre-mark_end row. No rewrite of this SELECT closes + /// that — the skew is between a catalog snapshot and live shared memory inside TimescaleDB's own view — + /// which is why + /// executes it TWICE, apart, when that arm trips. The text is + /// unchanged from #1760; what changed is how many times it is asked. /// public const string StuckCompressionJobsSql = @" SELECT @@ -5739,8 +5895,32 @@ WHERE j.proc_name LIKE '%compression%' /// every other job type are untouched. Failure-isolated: a store hiccup, or the views being absent (a /// plain-PostgreSQL store — the caller also gates on the extension), yields an empty list and a Debug line, /// never a throw. - /// - public static async Task> ReadStuckCompressionJobsAsync( + /// + /// The -infinity arm is CONFIRMED before it is reported (#3575). When, and only when, + /// the first read flags a job on that arm, this waits , runs + /// once more, and reports the job only if the same arm trips + /// again. A run-instant edge — the view pairing the scheduler's already-committed -infinity with a + /// worker that is not yet, or no longer, visible as Running — is over in milliseconds and clears; + /// a row the scheduler has genuinely abandoned, or a crashed run sitting out its five-minute backoff, + /// reads the same on both passes and is reported with the latency it always had plus five seconds. One + /// confirm per pass, not per job: the delay is taken once however many jobs tripped. The + /// arm is reported from the first read as before — + /// a six-hour elapsed bound has nothing to gain from a second look five seconds later. + /// + /// A confirm read that FAILS confirms nothing. Its -infinity trips are dropped for + /// this pass, with a Warning naming the jobs and the consequence, rather than reported on the strength of + /// the one read this issue proved insufficient: compression is a slow archival tier where a dead job + /// takes hours to matter, so deferring a real detection to the next hourly pass costs little, while a + /// false page on the family that reports the store's own health is the very thing being fixed. The + /// stuck-Running results from the first read are still returned. Like the first read's own catch, this + /// swallow is logged but not counted by the #3013 read-failure surface — that census covers the worker's + /// catch blocks, and both reads sit one level below it. + /// + /// The gated live test polls this method until a job it just re-armed reads healthy, and the + /// confirm only makes that settle sooner: the mid-run marker it used to have to wait out is now judged + /// twice and cleared inside one call instead of surfacing as a flagged poll. + /// + public static Task> ReadStuckCompressionJobsAsync( NpgsqlConnection connection, DateTime nowUtc, ILogger? logger, CancellationToken cancellationToken = default) { if (connection is null) @@ -5748,39 +5928,206 @@ public static async Task> ReadStuckCompressio throw new ArgumentNullException(nameof(connection)); } - var stuck = new List(); + return ReadStuckCompressionJobsAsync( + ct => ReadCompressionJobStatRowsAsync(connection, ct), + Task.Delay, + nowUtc, + logger, + cancellationToken); + } + + /// + /// The seam + /// is built on, with the two things a test needs to control injected: the read (so a transient edge and + /// a persistent dead row can each be scripted as a pair of result sets, and a failing confirm as a throw) + /// and the delay (so the pin can assert it is taken exactly when the -infinity arm tripped and + /// never otherwise, without sleeping). Internal rather than private for that reason alone; production + /// reaches it only through the connection overload. + /// + internal static async Task> ReadStuckCompressionJobsAsync( + Func>> readRows, + Func delay, + DateTime nowUtc, + ILogger? logger, + CancellationToken cancellationToken) + { + if (readRows is null) + { + throw new ArgumentNullException(nameof(readRows)); + } + + if (delay is null) + { + throw new ArgumentNullException(nameof(delay)); + } + try { - using var command = new NpgsqlCommand(StuckCompressionJobsSql, connection) { CommandTimeout = JobCatalogReadTimeoutSeconds }; + var first = ClassifyStuckCompressionJobs(await readRows(cancellationToken), nowUtc); + if (!first.Any(f => f.Arm == StuckCompressionJobArm.NextStartNegativeInfinity)) + { + /* Nothing on the racing arm: no delay, no second read. The common hourly pass costs exactly + what it did before #3575. */ + return first.Select(f => f.ToJob()).ToList(); + } - await using var reader = await command.ExecuteReaderAsync(cancellationToken); - while (await reader.ReadAsync(cancellationToken)) + await delay(StuckCompressionConfirmDelay, cancellationToken); + + IReadOnlyList? confirm; + try { - long jobId = Convert.ToInt64(reader.GetValue(0), CultureInfo.InvariantCulture); - bool negInfinity = !reader.IsDBNull(1) && reader.GetBoolean(1); - string? jobStatus = reader.IsDBNull(2) ? null : reader.GetString(2); - DateTime? lastRunStartedAt = reader.IsDBNull(3) - ? null - : DateTime.SpecifyKind(reader.GetDateTime(3), DateTimeKind.Utc); - TimeSpan? scheduleInterval = reader.IsDBNull(4) - ? null - : TimeSpan.FromSeconds(Convert.ToDouble(reader.GetValue(4), CultureInfo.InvariantCulture)); - string? hypertable = reader.IsDBNull(5) ? null : reader.GetString(5); - - if (IsCompressionJobStuck(negInfinity, jobStatus, lastRunStartedAt, scheduleInterval, nowUtc, out var reason)) - { - stuck.Add(new StuckCompressionJob(jobId, hypertable, reason)); - } + confirm = await readRows(cancellationToken); + } + catch (Exception ex) when (ex is not OperationCanceledException) + { + /* The views were readable seconds ago, so this is a store hiccup on the one read that decides + whether to page — say so at Warning, name what is deferred, and judge it next hour. */ + confirm = null; + logger?.LogWarning( + "Compression-job health check: {Count} job(s) read next_start = -infinity while not Running, but the confirm read {Delay:F0} s later failed — not judged this pass, re-checked next hour (#3575): {Message}", + first.Count(f => f.Arm == StuckCompressionJobArm.NextStartNegativeInfinity), + StuckCompressionConfirmDelay.TotalSeconds, + ex.Message); } + + return ConfirmStuckCompressionJobs(first, confirm, nowUtc, logger); } catch (Exception ex) when (ex is not OperationCanceledException) { /* The views are absent (a plain-PG store or the extension was removed) or the store hiccuped — no signal this check. The caller already gates on the extension; this is belt-and-suspenders. */ logger?.LogDebug("Compression-job health check: could not read job stats: {Message}", ex.Message); + return Array.Empty(); + } + } + + /// + /// The pure merge of a first pass with its confirm pass (#3575), separated so the decision table pins + /// without a clock or a store: + /// + /// from the first pass — reported, untouched by + /// the confirm. + /// from the first pass, and the SAME arm + /// on the confirm pass — reported, carrying the confirm pass's reason (the two are identical today; the + /// later read is the one that stood). + /// That arm on the first pass but not on the confirm — the run-instant edge; cleared, and logged at + /// Information because a person reading the log after this alert family fires deserves to find the near + /// miss, and it is rare enough (a few a year per store) never to be noise. + /// null (the confirm read failed) — every -infinity trip is + /// dropped; the caller has already logged why. + /// + /// A job that appears on the confirm pass but not the first is not reported either: the confirm exists to + /// ratify the first pass, not to widen it, and a job that only just went -infinity gets its own + /// two reads next hour. + /// + internal static IReadOnlyList ConfirmStuckCompressionJobs( + IReadOnlyList first, + IReadOnlyList? confirm, + DateTime nowUtc, + ILogger? logger) + { + if (first is null) + { + throw new ArgumentNullException(nameof(first)); } - return stuck; + var result = new List(first.Count); + var confirmed = confirm is null + ? null + : ClassifyStuckCompressionJobs(confirm, nowUtc) + .Where(c => c.Arm == StuckCompressionJobArm.NextStartNegativeInfinity) + .ToDictionary(c => c.Row.JobId); + + foreach (var flagged in first) + { + switch (flagged.Arm) + { + case StuckCompressionJobArm.RunningPastBound: + result.Add(flagged.ToJob()); + break; + + case StuckCompressionJobArm.NextStartNegativeInfinity: + if (confirmed is null) + { + break; + } + + if (confirmed.TryGetValue(flagged.Row.JobId, out var still)) + { + result.Add(still.ToJob()); + } + else + { + logger?.LogInformation( + "Compression-job health check: job {JobId}{Hypertable} read next_start = -infinity while not Running, and {Delay:F0} s later it was scheduled normally — the run-instant edge of TimescaleDB's job_stats view, not a stuck job; nothing re-armed, nothing alerted (#3575)", + flagged.Row.JobId, + string.IsNullOrEmpty(flagged.Row.HypertableName) ? "" : " on " + flagged.Row.HypertableName, + StuckCompressionConfirmDelay.TotalSeconds); + } + + break; + } + } + + return result; + } + + /// + /// One pass of the pure predicate over a result set: every row it flags, with the arm that fired. Rows + /// the predicate clears are not returned. + /// + internal static List ClassifyStuckCompressionJobs( + IReadOnlyList rows, DateTime nowUtc) + { + if (rows is null) + { + throw new ArgumentNullException(nameof(rows)); + } + + var flagged = new List(); + foreach (var row in rows) + { + var arm = ClassifyCompressionJob( + row.NextStartIsNegativeInfinity, row.JobStatus, row.LastRunStartedAtUtc, row.ScheduleInterval, nowUtc, out var reason); + if (arm != StuckCompressionJobArm.None) + { + flagged.Add(new ClassifiedCompressionJob(row, arm, reason)); + } + } + + return flagged; + } + + /// + /// One execution of , mapped row for row and NOT failure-isolated: + /// the isolation belongs to the caller, which has to tell a failed FIRST read (no signal, Debug) from a + /// failed CONFIRM read (a deferred judgement, Warning). Both -infinity tests already ran in SQL; + /// the #1760 sentinel arrives here as a NULL. + /// + private static async Task> ReadCompressionJobStatRowsAsync( + NpgsqlConnection connection, CancellationToken cancellationToken) + { + var rows = new List(); + using var command = new NpgsqlCommand(StuckCompressionJobsSql, connection) { CommandTimeout = JobCatalogReadTimeoutSeconds }; + + await using var reader = await command.ExecuteReaderAsync(cancellationToken); + while (await reader.ReadAsync(cancellationToken)) + { + long jobId = Convert.ToInt64(reader.GetValue(0), CultureInfo.InvariantCulture); + bool negInfinity = !reader.IsDBNull(1) && reader.GetBoolean(1); + string? jobStatus = reader.IsDBNull(2) ? null : reader.GetString(2); + DateTime? lastRunStartedAt = reader.IsDBNull(3) + ? null + : DateTime.SpecifyKind(reader.GetDateTime(3), DateTimeKind.Utc); + TimeSpan? scheduleInterval = reader.IsDBNull(4) + ? null + : TimeSpan.FromSeconds(Convert.ToDouble(reader.GetValue(4), CultureInfo.InvariantCulture)); + string? hypertable = reader.IsDBNull(5) ? null : reader.GetString(5); + + rows.Add(new CompressionJobStatRow(jobId, negInfinity, jobStatus, lastRunStartedAt, scheduleInterval, hypertable)); + } + + return rows; } /// @@ -6647,6 +6994,51 @@ public static async Task TryRearmJobAsync( /// public sealed record StuckCompressionJob(long JobId, string? HypertableName, string Reason); +/// +/// WHICH arm of fired (#3575), from +/// . Exists because the two arms need different +/// treatment downstream: is judged on two inputs that TimescaleDB's +/// view reads from independent sources and is therefore CONFIRMED by a second read before it is reported; +/// is judged on hours of elapsed time and is reported from the first read. +/// +public enum StuckCompressionJobArm +{ + /// Healthy — neither arm fired. + None, + + /// next_start = -infinity while the job is not reporting Running: the dead-job arm, + /// and the one that races the run instant. + NextStartNegativeInfinity, + + /// Running since before : the hung-run + /// arm. + RunningPastBound, +} + +/// +/// One row of as the predicate consumes it (#3575): +/// the two -infinity tests already applied in SQL, the #1760 sentinel already NULLIFed. Internal because +/// it is the seam the confirm-read pins through, not a product surface; the product's result type is +/// . +/// +internal sealed record CompressionJobStatRow( + long JobId, + bool NextStartIsNegativeInfinity, + string? JobStatus, + DateTime? LastRunStartedAtUtc, + TimeSpan? ScheduleInterval, + string? HypertableName); + +/// +/// A the predicate flagged, with the arm that fired and its reason — +/// the unit merges two passes of (#3575). +/// +internal sealed record ClassifiedCompressionJob(CompressionJobStatRow Row, StuckCompressionJobArm Arm, string Reason) +{ + /// The product-facing shape of this flag. + public StuckCompressionJob ToJob() => new(Row.JobId, Row.HypertableName, Reason); +} + /// /// One background job's cadence reading (#2136): the last SUCCESSFUL run's duration against the job's own /// schedule interval, from . From 91029fb6ae0fd2c41cf80c09b237a5e5945e3ad6 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 11:29:38 -0400 Subject: [PATCH 25/69] Text on an accent fill gets a measured ink in every theme, and the FinOps row marks become theme brushes Dark can actually see (#3577) (#3589) Two contrast defects from a community report, both measured rather than eyeballed (WCAG relative-luminance contrast; AA wants 4.5:1 for text, 3:1 for a non-text marker). The selected tab on Cool Breeze read 2.71:1 - and not because the theme chose that. Every TabItem style set a light Foreground for the selected state, but a string header becomes a TextBlock, and the themes' app-level implicit TextBlock style out-ranks the Foreground that TextBlock inherits from the TabItem, so the setter never reached the text: the header always rendered ForegroundBrush on the accent. The same dead setter meant Dark's selected tab was 1.99:1 (visible in the README's plan-viewer screenshot) and Light's would have been light-on-cyan had it ever applied. An empty implicit TextBlock style inside the header presenter shadows the app-level one, so the header inherits the TabItem's Foreground the way stock WPF intends, and a new per-theme AccentForegroundColor/Brush is the ink for text on any accent fill: white on Cool Breeze (5.39:1), the darkest surface tone on Dark (7.52:1), the page text on Light (6.79:1, no change). The same ForegroundBrush-on-accent pair recurred on the highlighted combo item, the accent button, the selected calendar day and Dark's grid selection (all 1.99:1 on Dark, 2.71:1 on Cool Breeze); they take the same ink through the same key. Cool Breeze's AccentHoverColor moves one step toward the accent (#2B87C8 -> #267BB8) because no ink passed on the old shade (white 3.89:1); white is 4.56:1 on the new one. TabCloseButton drops its hard-coded White so the x inherits the header ink - it was invisible on the light themes' unselected tabs. The FinOps "Mark Done / To Do / Do Not Do" tints were three fixed 20% overlays in DataGridRowMarks, which composite to 1.2-1.4:1 against Dark's near-black rows - the operator could not find the rows they had marked. They are theme brushes now (RowMarkDoneBrush / RowMarkToDoBrush / RowMarkDoNotBrush, painted by resource reference so a marked row follows a theme switch). Light and Cool Breeze keep the shipped tints; Dark builds its own from the theme's Success / Warning / Error colors at the opacity that pushes the mark as far from the row as it can go while the row text holds >= 4.5:1 on both row backgrounds (marks 2.9-3.1:1, text 4.5-5.1:1). The literals stay as the fallback for a host without the keys, and a test pins the keys in every theme file of both apps. Darling viewer theme copies ported 1:1; the deprecated Dashboard twin carries the identical tab/accent defect and gets the same fix (no marks there - it has none). Arm (B) of the report - user-maintainable colors with reset - is not in this change. --- .../Themes/CoolBreezeTheme.xaml | 104 +++++++++++++++--- .../Themes/DarkTheme.xaml | 103 ++++++++++++++--- .../Themes/LightTheme.xaml | 98 ++++++++++++++--- Lite.Tests/DataGridRowMarkTests.cs | 24 ++++ Lite/Themes/CoolBreezeTheme.xaml | 104 +++++++++++++++--- Lite/Themes/DarkTheme.xaml | 103 ++++++++++++++--- Lite/Themes/LightTheme.xaml | 98 ++++++++++++++--- PerformanceMonitor.Ui/DataGridRowMarks.cs | 44 ++++++-- .../Dashboard/Themes/CoolBreezeTheme.xaml | 96 +++++++++++++--- deprecated/Dashboard/Themes/DarkTheme.xaml | 90 ++++++++++++--- deprecated/Dashboard/Themes/LightTheme.xaml | 90 ++++++++++++--- 11 files changed, 817 insertions(+), 137 deletions(-) diff --git a/Darling/PerformanceMonitor.Darling.Viewer/Themes/CoolBreezeTheme.xaml b/Darling/PerformanceMonitor.Darling.Viewer/Themes/CoolBreezeTheme.xaml index 4c42dc07f..654786044 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/Themes/CoolBreezeTheme.xaml +++ b/Darling/PerformanceMonitor.Darling.Viewer/Themes/CoolBreezeTheme.xaml @@ -10,8 +10,17 @@ #1E6FA8 - #2B87C8 + + #267BB8 #155A8A + + #FFFFFF #CFDDE9 @@ -54,6 +63,7 @@ + @@ -86,6 +96,14 @@ + + + + + @@ -162,6 +180,8 @@ - - + + + use the theme accent and the matching readable text color (the on-accent ink, #3577). --> - + - + + @@ -1099,7 +1159,12 @@ + VerticalAlignment="{TemplateBinding VerticalContentAlignment}"> + + + - - + + + use the theme accent and the matching readable text color (the on-accent ink, #3577). --> - + - + + @@ -1098,7 +1159,12 @@ + VerticalAlignment="{TemplateBinding VerticalContentAlignment}"> + + + - - + + + use the theme accent and the matching readable text color (the on-accent ink, #3577). --> - + - + + @@ -1099,7 +1155,12 @@ + VerticalAlignment="{TemplateBinding VerticalContentAlignment}"> + + + - - + + + use the theme accent and the matching readable text color (the on-accent ink, #3577). --> - + - + + @@ -1099,7 +1159,12 @@ + VerticalAlignment="{TemplateBinding VerticalContentAlignment}"> + + + - - + + + use the theme accent and the matching readable text color (the on-accent ink, #3577). --> - + - + + @@ -1098,7 +1159,12 @@ + VerticalAlignment="{TemplateBinding VerticalContentAlignment}"> + + + - - + + + use the theme accent and the matching readable text color (the on-accent ink, #3577). --> - + - + + @@ -1099,7 +1155,12 @@ + VerticalAlignment="{TemplateBinding VerticalContentAlignment}"> + + + @@ -1146,7 +1198,12 @@ + VerticalAlignment="{TemplateBinding VerticalContentAlignment}"> + + + @@ -1145,7 +1193,12 @@ + VerticalAlignment="{TemplateBinding VerticalContentAlignment}"> + + + @@ -1146,7 +1194,12 @@ + VerticalAlignment="{TemplateBinding VerticalContentAlignment}"> + + + + @@ -1191,16 +1192,19 @@ + diff --git a/Darling/PerformanceMonitor.Darling.Viewer/Themes/CoolBreezeTheme.xaml b/Darling/PerformanceMonitor.Darling.Viewer/Themes/CoolBreezeTheme.xaml index 654786044..62becce46 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/Themes/CoolBreezeTheme.xaml +++ b/Darling/PerformanceMonitor.Darling.Viewer/Themes/CoolBreezeTheme.xaml @@ -43,7 +43,13 @@ #2E7D32 - #F57F17 + + #9E4A0B #C62828 #1E6FA8 @@ -95,6 +101,13 @@ + + + #2E7D32 - #F57F17 + + #AE4F08 #C62828 - #2eaef1 + + #0369A1 2 @@ -91,6 +102,14 @@ + + + diff --git a/Lite/Themes/CoolBreezeTheme.xaml b/Lite/Themes/CoolBreezeTheme.xaml index f667df613..93044213b 100644 --- a/Lite/Themes/CoolBreezeTheme.xaml +++ b/Lite/Themes/CoolBreezeTheme.xaml @@ -43,7 +43,13 @@ #2E7D32 - #F57F17 + + #9E4A0B #C62828 #1E6FA8 @@ -95,6 +101,13 @@ + + + #2E7D32 - #F57F17 + + #AE4F08 #C62828 - #2eaef1 + + #0369A1 2 @@ -91,6 +102,14 @@ + + #2E7D32 - #F57F17 + + #9E4A0B #C62828 #1E6FA8 @@ -90,6 +91,9 @@ + + diff --git a/deprecated/Dashboard/Themes/DarkTheme.xaml b/deprecated/Dashboard/Themes/DarkTheme.xaml index d58f0ea65..e3d894995 100644 --- a/deprecated/Dashboard/Themes/DarkTheme.xaml +++ b/deprecated/Dashboard/Themes/DarkTheme.xaml @@ -87,6 +87,9 @@ + + diff --git a/deprecated/Dashboard/Themes/LightTheme.xaml b/deprecated/Dashboard/Themes/LightTheme.xaml index 0f76b5eb7..1b997ec5f 100644 --- a/deprecated/Dashboard/Themes/LightTheme.xaml +++ b/deprecated/Dashboard/Themes/LightTheme.xaml @@ -36,9 +36,11 @@ #2E7D32 - #F57F17 + + #AE4F08 #C62828 - #2eaef1 + + #0369A1 2 @@ -86,6 +88,10 @@ + + From c0d34528c1b41902ea809afbb50bd5636fb84219 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 14:42:35 -0400 Subject: [PATCH 49/69] =?UTF-8?q?The=20compression=20dead-job=20alert=20sa?= =?UTF-8?q?ys=20what=20a=20-infinity=20row=20IS=20on=20the=20store's=20Tim?= =?UTF-8?q?escaleDB:=20permanent=20below=202.26.4,=20a=20self-clearing=20c?= =?UTF-8?q?rash=20backoff=20from=202.26.4=20on=20=E2=80=94=20where=20the?= =?UTF-8?q?=20re-arm=20is=20measured=20to=20reset=20that=20backoff=20and?= =?UTF-8?q?=20is=20no=20longer=20sent=20(#3591)=20(#3629)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .../CompressionStuckConfirmReadTests.cs | 116 +++++++ .../Darling.Tests/DarlingSelfAlertTests.cs | 116 +++++++ .../Darling.Tests/TimescaleSupportTests.cs | 125 ++++++++ .../DarlingSelfAlertEvaluator.cs | 84 +++++- .../TimescaleSupport.cs | 284 ++++++++++++++++-- 5 files changed, 699 insertions(+), 26 deletions(-) diff --git a/Darling/Darling.Tests/CompressionStuckConfirmReadTests.cs b/Darling/Darling.Tests/CompressionStuckConfirmReadTests.cs index d60442c66..bf91691b5 100644 --- a/Darling/Darling.Tests/CompressionStuckConfirmReadTests.cs +++ b/Darling/Darling.Tests/CompressionStuckConfirmReadTests.cs @@ -378,6 +378,122 @@ passes would only ever disagree on in a test. */ Assert.Equal("confirm", job.HypertableName); } + /* ---------------- the version read: when it is taken and what it decides (#3591) ---------------- */ + + /// A scripted version read: hands back the planted version (or throws), counting calls. + private sealed class ScriptedVersion + { + private readonly Version? _version; + private readonly Exception? _throw; + public int Calls { get; private set; } + + public ScriptedVersion(Version? version) => _version = version; + public ScriptedVersion(Exception ex) => _throw = ex; + + public Task Read(CancellationToken ct) + { + Calls++; + return _throw is null ? Task.FromResult(_version) : Task.FromException(_throw); + } + } + + [Fact] + public async Task HealthyPass_NeverReadsTheVersion() + { + /* The version decides the -infinity arm's sentence and nothing else, so a pass with nothing on that arm + does not pay for it — the common hourly pass is still one read. */ + var reads = new ScriptedReads().Then(Healthy(1), HungRun(4)); + var version = new ScriptedVersion(new Version(2, 28, 1)); + + var result = await TimescaleSupport.ReadStuckCompressionJobsAsync( + reads.Read, new RecordedDelay().Wait, s_now, logger: null, TestContext.Current.CancellationToken, version.Read); + + Assert.Single(result); + Assert.Equal(0, version.Calls); + Assert.False(result[0].SchedulerRetries); + } + + [Fact] + public async Task PersistentNegInfinity_On_2_28_1_IsConfirmed_WithTheCrashBackoffSentence_AndSchedulerRetries() + { + /* The fleet's shape: a SIGKILLed worker's row on 2.28.1, reproduced on the rig. Still reported — the + arm is the arm — but the sentence is the true one for this version, and SchedulerRetries tells the + evaluator not to re-arm it (which, measured, resets the backoff rather than shortening it). */ + var reads = new ScriptedReads() + .Then(NegInfinityScheduled(7, "wait_stats")) + .Then(NegInfinityScheduled(7, "wait_stats")); + var version = new ScriptedVersion(new Version(2, 28, 1)); + + var result = await TimescaleSupport.ReadStuckCompressionJobsAsync( + reads.Read, new RecordedDelay().Wait, s_now, logger: null, TestContext.Current.CancellationToken, version.Read); + + var job = Assert.Single(result); + Assert.Equal(7L, job.JobId); + Assert.Equal(1, version.Calls); /* once per pass, like the confirm */ + Assert.Equal(TimescaleSupport.NextStartNegativeInfinityCrashBackoffReason, job.Reason); + Assert.True(job.SchedulerRetries); + } + + [Fact] + public async Task PersistentNegInfinity_BelowTheFix_KeepsTheOldSentence_AndTheReArm() + { + var reads = new ScriptedReads() + .Then(NegInfinityScheduled(7)) + .Then(NegInfinityScheduled(7)); + var version = new ScriptedVersion(new Version(2, 26, 3)); + + var result = await TimescaleSupport.ReadStuckCompressionJobsAsync( + reads.Read, new RecordedDelay().Wait, s_now, logger: null, TestContext.Current.CancellationToken, version.Read); + + var job = Assert.Single(result); + Assert.Equal(TimescaleSupport.NextStartNegativeInfinityPermanentReason, job.Reason); + Assert.False(job.SchedulerRetries); + } + + [Fact] + public async Task VersionReadFails_OrIsAbsent_IsTheOldSentence_AndNeverFailsThePass() + { + /* Words, not verdicts: a version read that throws is swallowed at Debug and the pass proceeds exactly as + if the version were unknown — the conservative sentence, the #1581 re-arm. The pre-#3591 seam (no + reader at all) is the same case. */ + foreach (var reader in new Func>?[] + { + new ScriptedVersion(new InvalidOperationException("permission denied for table pg_extension")).Read, + new ScriptedVersion((Version?)null).Read, + null, + }) + { + var reads = new ScriptedReads() + .Then(NegInfinityScheduled(7)) + .Then(NegInfinityScheduled(7)); + var log = new CapturingTestLogger(); + + var result = await TimescaleSupport.ReadStuckCompressionJobsAsync( + reads.Read, new RecordedDelay().Wait, s_now, log, TestContext.Current.CancellationToken, reader); + + var job = Assert.Single(result); + Assert.Equal(TimescaleSupport.NextStartNegativeInfinityPermanentReason, job.Reason); + Assert.False(job.SchedulerRetries); + Assert.DoesNotContain("Warning:", log.Joined, StringComparison.Ordinal); + } + } + + [Fact] + public void ConfirmStuckCompressionJobs_TheVersionPhrasesTheConfirmRow_HungRunUntouched() + { + /* The merge applies the version to the CONFIRM pass — the row that is carried — and the hung run, judged + from the first pass, never sees it. */ + var first = TimescaleSupport.ClassifyStuckCompressionJobs(new[] { HungRun(4), NegInfinityScheduled(1) }, s_now); + var merged = TimescaleSupport.ConfirmStuckCompressionJobs( + first, new[] { HungRun(4), NegInfinityScheduled(1) }, s_now, logger: null, new Version(2, 28, 1)); + + Assert.Equal(2, merged.Count); + Assert.Contains("Running", merged[0].Reason, StringComparison.Ordinal); + Assert.False(merged[0].SchedulerRetries); + Assert.Equal(TimescaleSupport.NextStartNegativeInfinityCrashBackoffReason, merged[1].Reason); + Assert.True(merged[1].SchedulerRetries); + } + /* ---------------- the delay constant, against what it has to clear and what it costs ---------------- */ [Fact] diff --git a/Darling/Darling.Tests/DarlingSelfAlertTests.cs b/Darling/Darling.Tests/DarlingSelfAlertTests.cs index 0777860b8..031d11e22 100644 --- a/Darling/Darling.Tests/DarlingSelfAlertTests.cs +++ b/Darling/Darling.Tests/DarlingSelfAlertTests.cs @@ -2736,6 +2736,122 @@ public async Task CompressionJobs_EvaluateWrapper_IsolatesAThrowingSeam_DoesNotP await e2.EvaluateCompressionJobsAsync(Stuck(1002), _ => throw new InvalidOperationException("boom"), Ct); } + /* ---------------- #3591: a crash-backoff row the scheduler recovers by itself ---------------- */ + + /// The 2.26.4+ shape of the -infinity row: the reader marks it SchedulerRetries with the version's sentence. + private static IReadOnlyList CrashBackoff(params long[] jobIds) => + jobIds.Select(id => new StuckCompressionJob( + id, "wait_stats", TimescaleSupport.NextStartNegativeInfinityCrashBackoffReason, SchedulerRetries: true)).ToList(); + + [Fact] + public async Task CompressionJobs_SchedulerRetries_FirstSight_NoRearm_NoPage_LoggedAtInformation() + { + /* Measured on a 2.28.1 rig: alter_job(next_start => now()) against a job in crash backoff RESETS the + backoff (the retry moved from crash+5:00 to re-arm+5:04, for the re-armed job and its un-re-armed + sibling) and overwrites the -infinity, so a re-arm here would be a later retry and two untrue + messages. First sight is remembered and written to the log; the scheduler gets one check cadence. */ + var h = new Harness(); + var e = h.Build(); + var rearm = new RearmRecorder(); + + await e.ApplyCompressionJobsStuckAsync(CrashBackoff(1001), rearm.Delegate, Ct); + + Assert.Empty(rearm.Calls); + Assert.Empty(h.Deliverer.Outcomes); + Assert.Empty(h.History.Records); + var info = Assert.Single(h.Log.Entries, x => x.Level == Microsoft.Extensions.Logging.LogLevel.Information); + Assert.Contains("1001", info.Message, StringComparison.Ordinal); + Assert.Contains("crash backoff", info.Message, StringComparison.Ordinal); + Assert.Contains("not re-armed, not alerted", info.Message, StringComparison.Ordinal); + Assert.Contains("#3591", info.Message, StringComparison.Ordinal); + } + + [Fact] + public async Task CompressionJobs_SchedulerRetries_StillThereAnHourOn_EscalatesOnce_NeverRearms() + { + /* The scheduler's own retry did not clear it within a check cadence — the jittered backoff fell past + the check, or the job crashed again and its backoff doubled. Either is a human's to read about in + the PostgreSQL log; neither is helped by alter_job. One Critical page, then the escalated state's + cooldown re-fires, and no re-arm at any point. */ + var h = new Harness(); + var e = h.Build(); + var rearm = new RearmRecorder(); + + await e.ApplyCompressionJobsStuckAsync(CrashBackoff(1001), rearm.Delegate, Ct); + h.Now = h.Now.AddHours(1); + await e.ApplyCompressionJobsStuckAsync(CrashBackoff(1001), rearm.Delegate, Ct); + + Assert.Empty(rearm.Calls); + var fired = Assert.Single(h.Deliverer.Outcomes); + Assert.Equal("Compression Job Stuck", fired.MetricName); + Assert.Equal(AlertSeverityLevel.Critical, fired.Severity); + Assert.Equal("compressjob:1001", fired.ServerKey); + Assert.Contains("still in crash backoff", fired.ShortMessage, StringComparison.Ordinal); + Assert.Contains("escalated", fired.ShortMessage, StringComparison.Ordinal); + Assert.DoesNotContain("re-armed", fired.ShortMessage, StringComparison.Ordinal); + + /* Escalated: an hour later, still there, still no re-arm; re-fires only on the cooldown. */ + h.Now = h.Now.AddHours(1); + await e.ApplyCompressionJobsStuckAsync(CrashBackoff(1001), rearm.Delegate, Ct); + Assert.Empty(rearm.Calls); + Assert.Equal(2, h.Deliverer.Outcomes.Count); + Assert.Contains("after escalation", h.Deliverer.Outcomes[1].ShortMessage, StringComparison.Ordinal); + } + + [Fact] + public async Task CompressionJobs_SchedulerRetries_ClearsBeforeTheSecondCheck_NoResolutionRow_StateDropped() + { + /* The expected path on the fleet: the scheduler re-ran the job inside the hour. Nothing was paged, so + nothing is "Recovered" — a lone resolution with no alert before it is the message shape #3575 + removed. The state is gone, so a later genuine sighting is a fresh first sight. */ + var h = new Harness(); + var e = h.Build(); + var rearm = new RearmRecorder(); + + await e.ApplyCompressionJobsStuckAsync(CrashBackoff(1001), rearm.Delegate, Ct); + h.Now = h.Now.AddHours(1); + await e.ApplyCompressionJobsStuckAsync(Stuck(), rearm.Delegate, Ct); + + Assert.Empty(rearm.Calls); + Assert.Empty(h.Deliverer.Outcomes); + Assert.Empty(h.History.Records); + Assert.Contains(h.Log.Entries, x => x.Level == Microsoft.Extensions.Logging.LogLevel.Information + && x.Message.Contains("running on schedule again", StringComparison.Ordinal) + && x.Message.Contains("#3591", StringComparison.Ordinal)); + + /* Fresh first sight afterwards: deferred again, not escalated — the state was dropped. */ + h.Now = h.Now.AddHours(1); + await e.ApplyCompressionJobsStuckAsync(CrashBackoff(1001), rearm.Delegate, Ct); + Assert.Empty(h.Deliverer.Outcomes); + Assert.Empty(rearm.Calls); + } + + [Fact] + public async Task CompressionJobs_PreFixRow_AndSchedulerRetriesRow_InOnePass_OnlyTheOldOneIsRearmedAndPaged() + { + /* The two kinds side by side, so the flag and not the position decides: the pre-2.26.4 row keeps every + #1581 semantic (re-armed once, paged Critical); the crash-backoff row is deferred. */ + var h = new Harness(); + var e = h.Build(); + var rearm = new RearmRecorder(); + + var both = new List(CrashBackoff(1001)); + both.AddRange(Stuck(1002)); + await e.ApplyCompressionJobsStuckAsync(both, rearm.Delegate, Ct); + + Assert.Equal(1002L, Assert.Single(rearm.Calls)); + var fired = Assert.Single(h.Deliverer.Outcomes); + Assert.Equal("compressjob:1002", fired.ServerKey); + Assert.Contains("auto-re-armed", fired.ShortMessage, StringComparison.Ordinal); + } + + [Fact] + public void CompressionJobs_TheOldRowShape_DefaultsToTheRearmPath() + { + /* The three-argument record every pre-#3591 caller and pin builds is the old semantics, by default. */ + Assert.False(new StuckCompressionJob(1L, "wait_stats", "next_start is -infinity — the scheduler will never run it again").SchedulerRetries); + } + /* ---------------- #991 Availability Groups: sync-behind decision (pure) ---------------- */ private static AgSyncJudgement Judge( diff --git a/Darling/Darling.Tests/TimescaleSupportTests.cs b/Darling/Darling.Tests/TimescaleSupportTests.cs index 58b3a0697..4488ca8e2 100644 --- a/Darling/Darling.Tests/TimescaleSupportTests.cs +++ b/Darling/Darling.Tests/TimescaleSupportTests.cs @@ -267,6 +267,131 @@ the point is that last_run_started_at is never read raw. */ TimescaleSupport.StuckCompressionJobsSql, StringComparison.Ordinal); } + /* ---------------- the -infinity arm's sentence, by TimescaleDB version (#3591) ---------------- */ + + [Theory] + [InlineData(null)] + [InlineData("2.14.2")] + [InlineData("2.26.0")] + [InlineData("2.26.3")] + public void NegInfinityArm_BelowTheFix_OrUnknown_SaysTheSchedulerWillNeverRunIt(string? extversion) + { + /* Below 2.26.4 a persisted -infinity is returned to the scheduler as the due time and the job is never + due again — #1581's sentence, byte for byte, because it is true there. An unknown version (null: the + read failed, or the pre-#3591 callers) gets the same text on purpose: it is the sentence that costs + nothing when wrong on a new store and everything when wrong on an old one. */ + var version = TimescaleSupport.ParseTimescaleVersion(extversion); + Assert.False(TimescaleSupport.SchedulerRecoversNegativeInfinity(version)); + + var arm = TimescaleSupport.ClassifyCompressionJob( + nextStartIsNegativeInfinity: true, jobStatus: "Scheduled", lastRunStartedAtUtc: s_now.AddHours(-1), + scheduleInterval: TimeSpan.FromHours(1), nowUtc: s_now, timescaleVersion: version, out var reason); + + Assert.Equal(StuckCompressionJobArm.NextStartNegativeInfinity, arm); + Assert.Equal("next_start is -infinity — the scheduler will never run it again", reason); + Assert.Equal(TimescaleSupport.NextStartNegativeInfinityPermanentReason, reason); + } + + [Theory] + [InlineData("2.26.4")] + [InlineData("2.27.0")] + [InlineData("2.28.1")] + [InlineData("2.29.0-dev")] + public void NegInfinityArm_FromTheFix_SaysCrashBackoff_AndStillFires(string extversion) + { + /* Upstream #9360 shipped in 2.26.4 (the CHANGELOG lists it there, not under 2.27.0), so from 2.26.4 the + only persistent -infinity is a crashed run in the scheduler's crash backoff. Same arm — the verdict + does not move — different sentence, naming the fix so the reader knows where the claim comes from. */ + var version = TimescaleSupport.ParseTimescaleVersion(extversion); + Assert.NotNull(version); + Assert.True(TimescaleSupport.SchedulerRecoversNegativeInfinity(version)); + + var arm = TimescaleSupport.ClassifyCompressionJob( + nextStartIsNegativeInfinity: true, jobStatus: "Scheduled", lastRunStartedAtUtc: s_now.AddHours(-1), + scheduleInterval: TimeSpan.FromHours(1), nowUtc: s_now, timescaleVersion: version, out var reason); + + Assert.Equal(StuckCompressionJobArm.NextStartNegativeInfinity, arm); + Assert.Equal(TimescaleSupport.NextStartNegativeInfinityCrashBackoffReason, reason); + Assert.Contains("crash backoff", reason, StringComparison.Ordinal); + Assert.Contains("#9360", reason, StringComparison.Ordinal); + Assert.Contains("2.26.4", reason, StringComparison.Ordinal); + Assert.DoesNotContain("never", reason, StringComparison.Ordinal); + + /* And the boolean projection with the version agrees with the classifier. */ + Assert.True(TimescaleSupport.IsCompressionJobStuck( + nextStartIsNegativeInfinity: true, jobStatus: "Scheduled", lastRunStartedAtUtc: s_now.AddHours(-1), + scheduleInterval: TimeSpan.FromHours(1), nowUtc: s_now, timescaleVersion: version, out var boolReason)); + Assert.Equal(reason, boolReason); + } + + [Fact] + public void NegInfinityArm_TheVersionChangesTheSentenceOnly_NeverTheVerdict() + { + /* Every row shape the predicate knows, on both sides of the fix: the arm is identical, and only the + -infinity arm's text differs. The stuck-Running arm did not change upstream and reads the version + for nothing. */ + var old = new Version(2, 26, 3); + var fixedVersion = new Version(2, 28, 1); + foreach (var (negInf, status, started) in new (bool, string, DateTime?)[] + { + (true, "Scheduled", null), /* the dead-job / crash-backoff arm */ + (true, "Running", s_now.AddMinutes(-3)), /* mid-run marker */ + (true, "Running", s_now.AddHours(-30)), /* hung run carrying the marker */ + (false, "Scheduled", s_now.AddMinutes(-5)), /* healthy */ + (false, "Running", s_now.AddHours(-8)), /* hung run */ + (false, "Running", DateTime.MinValue), /* #1760 sentinel */ + }) + { + var armOld = TimescaleSupport.ClassifyCompressionJob(negInf, status, started, TimeSpan.FromHours(1), s_now, old, out var reasonOld); + var armNew = TimescaleSupport.ClassifyCompressionJob(negInf, status, started, TimeSpan.FromHours(1), s_now, fixedVersion, out var reasonNew); + var armNone = TimescaleSupport.ClassifyCompressionJob(negInf, status, started, TimeSpan.FromHours(1), s_now, out var reasonNone); + + Assert.Equal(armOld, armNew); + Assert.Equal(armOld, armNone); + Assert.Equal(reasonOld, reasonNone); /* version-less IS the old text */ + if (armOld == StuckCompressionJobArm.NextStartNegativeInfinity) + { + Assert.NotEqual(reasonOld, reasonNew); + } + else + { + Assert.Equal(reasonOld, reasonNew); + } + } + } + + [Fact] + public void TimescaleNextStartSanitizedFrom_Is_2_26_4() + { + /* Pinned to the release the upstream CHANGELOG lists #9360 under. The issue and the first brief said + 2.27.0; a check keyed there would have told every 2.26.4–2.26.x store the old lie. */ + Assert.Equal(new Version(2, 26, 4), TimescaleSupport.TimescaleNextStartSanitizedFrom); + } + + [Theory] + [InlineData("2.28.1", "2.28.1")] + [InlineData(" 2.26.4 ", "2.26.4")] + [InlineData("2.29.0-dev", "2.29.0")] + [InlineData("2.28", "2.28")] + public void ParseTimescaleVersion_TakesTheNumericPrefix(string raw, string expected) + { + Assert.Equal(Version.Parse(expected), TimescaleSupport.ParseTimescaleVersion(raw)); + } + + [Theory] + [InlineData(null)] + [InlineData("")] + [InlineData(" ")] + [InlineData("2")] + [InlineData("dev")] + [InlineData("v2.28.1")] + public void ParseTimescaleVersion_AnythingElse_IsNull_WhichReadsAsOld(string? raw) + { + var parsed = TimescaleSupport.ParseTimescaleVersion(raw); + Assert.Null(parsed); + Assert.False(TimescaleSupport.SchedulerRecoversNegativeInfinity(parsed)); + } + [Fact] public void StuckRunningBound_UsesMaxOfTwiceIntervalAndFloor() { diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingSelfAlertEvaluator.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingSelfAlertEvaluator.cs index 1d9673a9f..d0943e066 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingSelfAlertEvaluator.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingSelfAlertEvaluator.cs @@ -628,7 +628,10 @@ is. The rule set is handed in by the worker from the live MuteRuleService cache when the job stops being stuck. Keyed by the CompressionKeyPrefix + job_id so the alert serverKey never collides with a real server_id (an int hash) — the deliverer's #1236 int.TryParse override no-ops on it, exactly like the non-numeric DiskKey. */ - private enum CompressionJobHealth { ReArmed, Escalated } + /* AwaitingSchedulerRetry (#3591): a -infinity row on a TimescaleDB whose scheduler recovers it by itself + (StuckCompressionJob.SchedulerRetries) — seen once, not re-armed, not paged; a second consecutive + sighting escalates. The other two states are #1581's. */ + private enum CompressionJobHealth { ReArmed, Escalated, AwaitingSchedulerRetry } private readonly ConcurrentDictionary _compressionJobState = new(StringComparer.Ordinal); private readonly ConcurrentDictionary _lastCompressionJobAlert = new(StringComparer.Ordinal); @@ -3955,6 +3958,31 @@ await RecordResolutionAsync(new AlertResolution( /// This method keeps its single-sample semantics — first sight IS first sight — because the input is now /// worth that trust, and adding hysteresis here instead would have bought the same protection for an /// hour of detection latency on a genuinely dead job. + /// + /// Which TimescaleDB the dead-job arm's sentence and its re-arm are true on (#3591). "The + /// scheduler will never run it again" was true of every TimescaleDB below 2.26.4: a persisted + /// next_start = -infinity was returned to the scheduler as the due time and the job was never due + /// again — the state #1581 was built against, and the one the re-arm genuinely rescues. Upstream #9360 + /// ("Sanitize DT_NOBEGIN next_start to recover jobs stuck after primary failover", in 2.26.4 and + /// every 2.27+ release; TimescaleSupport.TimescaleNextStartSanitizedFrom) removed that state. On a + /// fixed store the only PERSISTENT -infinity is a crashed run — a worker killed between its start + /// and end marks by a SIGKILL, a crash-restart or a failover — which the scheduler holds in a CRASH + /// BACKOFF of at least five minutes and, for a compression policy with the default one-hour + /// retry_period, about an hour (±13 % jitter), and then re-runs by itself. Rows of that kind arrive + /// here with set, and this machine treats them + /// differently on the evidence of a 2.28.1 rig: re-arming a job in crash backoff does not shorten the + /// wait, it RESETS it — alter_job refreshes the scheduler's job list and every crash row's backoff + /// is recomputed from that instant (the un-re-armed sibling crash row moved too), and the re-arm overwrites + /// the -infinity so the next hourly read would call the job healthy before it had run. A re-arm on + /// such a row is therefore three untrue messages and a later retry, which is why the first sighting of + /// one is NOT re-armed and NOT paged: it is logged at Information and remembered as + /// AwaitingSchedulerRetry. If the same job is still on the arm an hour later — the scheduler's own + /// retry has not cleared it, because the jittered backoff fell just past the check cadence or because the + /// job crashed AGAIN and its backoff doubled — it escalates: one Critical page naming a crash the operator + /// should read the PostgreSQL log for, no re-arm, and the cooldown re-fires of the escalated state. A job + /// that clears while awaiting the retry is dropped silently, since nothing was paged for it to recover + /// from. Stores below 2.26.4, and stores whose version could not be read, keep every #1581 semantic + /// exactly — an unknown version is treated as old, because on an old store the re-arm is the rescue. /// internal async Task ApplyCompressionJobsStuckAsync( IReadOnlyList stuckJobs, @@ -3980,6 +4008,19 @@ internal async Task ApplyCompressionJobsStuckAsync( if (!_compressionJobState.TryGetValue(key, out var state)) { + if (job.SchedulerRetries) + { + /* #3591: a crash-backoff row on a TimescaleDB that re-runs it by itself. Re-arming would + reset that backoff and erase the evidence (see the method doc); paging would narrate a + self-recovering condition as a rescue. Remember it, say so in the log, and give the + scheduler one check cadence to do what it does. */ + _compressionJobState[key] = CompressionJobHealth.AwaitingSchedulerRetry; + _logger?.LogInformation( + "TimescaleDB {Label} read {Reason}; the scheduler re-runs a crashed job by itself after its crash backoff (at least five minutes, about an hour for a compression policy's default retry period), and a re-arm would reset that backoff rather than shorten it — not re-armed, not alerted; escalates if still there next check (#3591)", + label, job.Reason); + continue; + } + /* First detection this episode: re-arm ONCE, then alert on the outcome. */ bool rearmed = await rearmAsync(job.JobId); _lastCompressionJobAlert[key] = now; @@ -3993,7 +4034,10 @@ await FireAsync( "(alter_job next_start => now). A stuck compression policy halts the store's archival tier, so " + "uncompressed data grows without bound until the disk fills and collection stops for the WHOLE " + "fleet, and a headless service has no dashboard to warn you. If it re-hangs the service will " + - "escalate and stop auto-re-arming — investigate the TimescaleDB background-worker health.", + "escalate and stop auto-re-arming — investigate the TimescaleDB background-worker health. " + + "(A next_start of -infinity is permanent only below TimescaleDB 2.26.4; from 2.26.4 on, upstream " + + "#9360, the scheduler recovers it by itself and the service leaves such rows to it — if this store " + + "is on 2.26.4 or later, its extension version could not be read on this pass.)", severity: AlertSeverityLevel.Critical, shortMessage: $"{label} was stuck — auto-re-armed", /* #1881: job.Reason is elapsed minutes when a run HUNG and a scheduler state with no @@ -4021,6 +4065,29 @@ await FireAsync( cancellationToken); } } + else if (state == CompressionJobHealth.AwaitingSchedulerRetry) + { + /* #3591: an hour on and the scheduler's own retry has not cleared it. Either the jittered backoff + landed just past this check, or the job crashed AGAIN and its backoff doubled — both are worth a + human reading the PostgreSQL log, and neither is helped by alter_job (which would reset the + backoff once more). Escalate: page once now, re-fire on the cooldown, never re-arm. */ + _compressionJobState[key] = CompressionJobHealth.Escalated; + _lastCompressionJobAlert[key] = now; + await FireAsync( + StoreKey(CompressionKeyPrefix + key), _storeLabel, CompressionJobMetric, + job.Reason, "running on schedule", + detail: $"TimescaleDB {label} has sat in the scheduler's crash backoff ({job.Reason}) since at least the previous " + + "hourly check, and the scheduler's own retry has not cleared it. A crashed background worker means the " + + "PostgreSQL cluster crash-restarted, failed over, or the worker was killed mid-run — read the PostgreSQL log " + + "around the job's last_run_started_at for the cause, and check whether it has crashed more than once (each " + + "consecutive crash doubles the backoff). The service did NOT re-arm it: on TimescaleDB 2.26.4+ (upstream " + + "#9360) alter_job(next_start => now()) against a job in crash backoff resets the backoff instead of " + + "shortening it. Compression stays paused for this hypertable until the scheduler's retry succeeds.", + severity: AlertSeverityLevel.Critical, + shortMessage: $"{label} still in crash backoff an hour on — escalated", + numericCurrentValue: StateOnlyValue, numericThresholdValue: StateOnlyValue, + cancellationToken); + } else if (state == CompressionJobHealth.ReArmed) { /* Still stuck after last check's self-heal = a RE-HANG. Escalate and STOP re-arming (looping @@ -4068,8 +4135,19 @@ collection under enumeration. */ continue; } - _compressionJobState.TryRemove(key, out _); + _compressionJobState.TryRemove(key, out var was); _lastCompressionJobAlert.TryRemove(key, out _); + if (was == CompressionJobHealth.AwaitingSchedulerRetry) + { + /* #3591: the scheduler's own retry cleared it and nothing was paged, so there is nothing to + resolve — a lone "Recovered" with no preceding alert would be the very message shape #3575 + removed. The log carries the outcome instead. */ + _logger?.LogInformation( + "TimescaleDB compression job {JobId} is running on schedule again — the scheduler's own retry cleared its crash backoff; nothing was re-armed or alerted (#3591)", + key); + continue; + } + await RecordResolutionAsync(new AlertResolution( StoreKey(CompressionKeyPrefix + key), _storeLabel, CompressionJobMetric, "Compression Job Recovered", diff --git a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs index 5bc895f6a..2e012f132 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs @@ -6690,10 +6690,119 @@ public static async Task EnsureCollectionLogHypertableAsync(NpgsqlConnecti /* ---------------- compression-job self-heal (#1581) ---------------- */ + /// + /// The TimescaleDB release from which a persisted next_start = -infinity stopped being a PERMANENT + /// state (#3591): upstream #9360, "Sanitize DT_NOBEGIN next_start to recover jobs stuck after + /// primary failover", shipped in the 2.26.4 patch release (2026-04-28 — the CHANGELOG lists it under + /// 2.26.4, not 2.27.0 as the issue first said) and is therefore in every 2.27+ release as well. + /// is the predicate over it. + /// + /// What the fix changed, from the 2.28.1 source (src/bgw/job_stat.c, + /// ts_bgw_job_stat_next_start). The scheduler computes each job's in-memory next start from + /// its stat row in three arms, in this order: a row with consecutive_crashes > 0 gets the CRASH + /// BACKOFF (below); otherwise, since #9360, a row whose persisted next_start is -infinity is + /// sanitized to "now" and runs at once; otherwise the persisted value stands. Before the fix the second arm + /// did not exist: the sentinel was returned as-is, the scheduler's due-time subtraction on INT64_MIN + /// wrapped, and the job was never due again — the permanent dead state #1581's arm was built against and + /// the field incident that filled a disk. A row reaches that state with consecutive_crashes = 0 through + /// on_failure_to_start_job's next_start != DT_NOBEGIN restore guard, or by inheriting a + /// mid-run row across a primary failover; both are named in the upstream fix's own comment. + /// + /// What a PERSISTENT -infinity is on a fixed store. Only the first arm's row: a worker + /// killed between mark_start and mark_end (a SIGKILL, a crash-restart, a failover — a SIGTERM + /// is caught and marked as a FAILURE with a finite next start) leaves next_start = -infinity, + /// last_finish = -infinity (the view's last_run_status IS NULL) and consecutive_crashes = 1, + /// and the scheduler holds it there, un-persisted, for + /// max(MIN_WAIT_AFTER_CRASH_MS, retry_period × crashes) capped at five schedule intervals and + /// jittered ±13 %: at least FIVE MINUTES, and for a compression policy — whose retry_period defaults + /// to one hour, measured on 2.28.1 — about an hour. Then it re-runs the job itself. (#3575's rig read + /// exactly five minutes because its 10-second probe policy capped the retry term at 50 s; the five-minute + /// floor is the whole backoff only when the retry term is smaller than it.) So on 2.26.4+ the row this + /// product's dead-job arm fires on is a self-recovering condition, not a permanent one, and the alert's + /// sentence has to say which — + /// does, by version. + /// + /// The version is read from pg_extension.extversion by + /// — the first place in this file to read it. Nothing else here declares a TimescaleDB floor (the stated + /// target is "2.x"), and this does not either: a version that cannot be read or parsed is treated as OLD, + /// because "the scheduler will never run it again" is the sentence that costs nothing when wrong on a new + /// store and everything when wrong on an old one. + /// + public static readonly Version TimescaleNextStartSanitizedFrom = new(2, 26, 4); + + /// + /// Whether has upstream #9360 (#3591): true from + /// up, false below it AND for null — an unknown + /// version is the old behaviour, deliberately (see the constant). Pure, so the version arms pin. + /// + public static bool SchedulerRecoversNegativeInfinity(Version? timescaleVersion) + => timescaleVersion is not null && timescaleVersion >= TimescaleNextStartSanitizedFrom; + + /// + /// pg_extension.extversion for timescaledb as a , or null when it cannot + /// be read (#3591). Failure-isolated for the same reason the job-stat read is: this decides a SENTENCE, + /// not whether to page, and a store hiccup on it must fall back to the conservative text rather than + /// fail the check. Debug on failure; the caller already gates on the extension being present. + /// + public static async Task ReadTimescaleVersionAsync( + NpgsqlConnection connection, ILogger? logger, CancellationToken cancellationToken = default) + { + if (connection is null) + { + throw new ArgumentNullException(nameof(connection)); + } + + try + { + using var command = new NpgsqlCommand( + "SELECT extversion FROM pg_extension WHERE extname = 'timescaledb'", connection) { CommandTimeout = JobCatalogReadTimeoutSeconds }; + var raw = await command.ExecuteScalarAsync(cancellationToken) as string; + var parsed = ParseTimescaleVersion(raw); + if (parsed is null) + { + logger?.LogDebug("TimescaleDB extversion '{Raw}' did not parse as a version — treating the store as pre-2.26.4 for the compression dead-job text (#3591)", raw ?? "(absent)"); + } + + return parsed; + } + catch (Exception ex) when (ex is not OperationCanceledException) + { + logger?.LogDebug("Could not read the TimescaleDB extension version — treating the store as pre-2.26.4 for the compression dead-job text (#3591): {Message}", ex.Message); + return null; + } + } + + /// + /// The numeric prefix of an extversion string as a (#3591): 2.28.1 + /// parses whole; a development suffix (2.29.0-dev) is dropped and the number kept, because the + /// scheduler code in a dev build past 2.26.4 has the fix; anything with fewer than two dotted numbers + /// (null, empty, a bare 2) is null, which the caller reads as OLD. Pure so it pins. + /// + public static Version? ParseTimescaleVersion(string? extversion) + { + if (string.IsNullOrWhiteSpace(extversion)) + { + return null; + } + + var span = extversion.AsSpan().Trim(); + int end = 0; + while (end < span.Length && (char.IsAsciiDigit(span[end]) || span[end] == '.')) + { + end++; + } + + var numeric = span[..end].TrimEnd('.'); + return numeric.Contains('.') && Version.TryParse(numeric, out var version) ? version : null; + } + /// /// The parameterized re-arm statement (#1581): reschedule a background job to run immediately, which /// un-sticks a job whose next_start has become -infinity (the scheduler will never re-fire - /// it otherwise — the field-incident root cause). The job_id is ALWAYS bound as $1, never + /// it otherwise — the field-incident root cause, on TimescaleDB below 2.26.4; see + /// for what the same row is above it, and + /// for why this statement must NOT be run against it + /// there — measured, it resets the scheduler's crash backoff rather than shortening it, #3591). The job_id is ALWAYS bound as $1, never /// interpolated (the discipline is uniform with DarlingRetention's parameterized paths); now() is /// SQL, not a value. It is cast $1::integer because TimescaleDB's alter_job takes /// job_id integer, but is a long that Npgsql sends as @@ -6858,9 +6967,11 @@ public static DateTime NextCompressionCheckUtc(DateTime nowUtc, TimeSpan interva /// /// The pure stuck-compression-job decision (#1581). A compression policy job is STUCK when either: /// - /// its next_start is -infinity while the job is NOT currently running — the scheduler - /// abandoned it and will NEVER re-fire it (the dead-job bug that let uncompressed data grow without bound - /// until the disk filled), or + /// its next_start is -infinity while the job is NOT currently running — on TimescaleDB + /// below 2.26.4 the scheduler abandoned it and will NEVER re-fire it (the dead-job bug that let uncompressed + /// data grow without bound until the disk filled); from 2.26.4 on (upstream #9360, + /// ) it is a crashed run in the scheduler's crash backoff, + /// which the scheduler clears by itself (#3591) — the same arm, a different sentence, and no re-arm; or /// it has been in the Running state since a last_run_started_at older than /// — a run that began long ago and never finished (a hung run). /// @@ -6895,6 +7006,31 @@ public static DateTime NextCompressionCheckUtc(DateTime nowUtc, TimeSpan interva /// -infinity never-ran sentinel, so this is the second line of defence: the sentinel maps to /// MinValue through Npgsql, and any future caller reading the column un-guarded would otherwise compute a /// ~739,000-day elapsed that clears every bound and flag a healthy job on its very first run. + /// + /// "Will never run it again" is true of TimescaleDB below 2.26.4 and false above it (#3591). + /// Upstream #9360 () made the scheduler sanitize a persisted + /// -infinity, so on a fixed store the only PERSISTENT -infinity is a crashed run sitting out + /// its crash backoff, which the scheduler clears by itself. The VERDICT is the same on every version — the + /// row is still reported, still re-armed once, still paged, because the arm is also the backstop for the + /// older 2.x stores the compatibility target admits and for whatever the next upstream regression is — but + /// the SENTENCE differs, and this overload without a version says the old one. Production goes through + /// + /// with the version read; the version-less form is the pre-#3591 + /// pins' entry point and the "unknown version" case, which are the same text. + /// + public static bool IsCompressionJobStuck( + bool nextStartIsNegativeInfinity, + string? jobStatus, + DateTime? lastRunStartedAtUtc, + TimeSpan? scheduleInterval, + DateTime nowUtc, + out string reason) + => ClassifyCompressionJob(nextStartIsNegativeInfinity, jobStatus, lastRunStartedAtUtc, scheduleInterval, nowUtc, timescaleVersion: null, out reason) + != StuckCompressionJobArm.None; + + /// + /// with the store's TimescaleDB version, so the -infinity arm's + /// reason tells the truth for that version (#3591). null is "unknown, assume old". /// public static bool IsCompressionJobStuck( bool nextStartIsNegativeInfinity, @@ -6902,8 +7038,9 @@ public static bool IsCompressionJobStuck( DateTime? lastRunStartedAtUtc, TimeSpan? scheduleInterval, DateTime nowUtc, + Version? timescaleVersion, out string reason) - => ClassifyCompressionJob(nextStartIsNegativeInfinity, jobStatus, lastRunStartedAtUtc, scheduleInterval, nowUtc, out reason) + => ClassifyCompressionJob(nextStartIsNegativeInfinity, jobStatus, lastRunStartedAtUtc, scheduleInterval, nowUtc, timescaleVersion, out reason) != StuckCompressionJobArm.None; /// @@ -6911,7 +7048,8 @@ public static bool IsCompressionJobStuck( /// , because that is the arm whose inputs /// race; is judged on six hours of elapsed time and /// a second read five seconds later could not change it. Same decision, same reason text — this is the - /// implementation and the boolean is its projection, so the two cannot drift. + /// implementation and the boolean is its projection, so the two cannot drift. Version-less: the + /// -infinity reason is the pre-2.26.4 text (#3591); see the overload below. /// public static StuckCompressionJobArm ClassifyCompressionJob( bool nextStartIsNegativeInfinity, @@ -6920,12 +7058,50 @@ public static StuckCompressionJobArm ClassifyCompressionJob( TimeSpan? scheduleInterval, DateTime nowUtc, out string reason) + => ClassifyCompressionJob(nextStartIsNegativeInfinity, jobStatus, lastRunStartedAtUtc, scheduleInterval, nowUtc, timescaleVersion: null, out reason); + + /// + /// The -infinity arm's reason on a TimescaleDB below , + /// or of unknown version: the #1581 sentence, byte for byte, because on those stores it is true — the + /// scheduler returns the sentinel as the due time and the job is never due again. + /// + public const string NextStartNegativeInfinityPermanentReason = + "next_start is -infinity — the scheduler will never run it again"; + + /// + /// The -infinity arm's reason on a TimescaleDB with upstream #9360 (#3591): the row is a crashed + /// run in the scheduler's crash backoff, and the scheduler clears it without help. Names the fix and the + /// floor so an operator reading the page knows which condition they are looking at and where the claim + /// comes from; how long the backoff is, and what the re-arm does to it, is the evaluator's detail text. + /// + public const string NextStartNegativeInfinityCrashBackoffReason = + "next_start is -infinity — a crashed run left the job in the scheduler's crash backoff, which the scheduler clears on its own (TimescaleDB 2.26.4+, upstream #9360)"; + + /// + /// with the + /// store's TimescaleDB version (#3591). The -infinity arm fires on exactly the same inputs whatever + /// the version — the confirm-read, the re-arm, the alert key and the severity all see one arm — and only + /// its changes: when + /// says the scheduler has the fix, + /// below it and for null. The stuck-Running + /// arm does not read the version; nothing about a hung run changed upstream. + /// + public static StuckCompressionJobArm ClassifyCompressionJob( + bool nextStartIsNegativeInfinity, + string? jobStatus, + DateTime? lastRunStartedAtUtc, + TimeSpan? scheduleInterval, + DateTime nowUtc, + Version? timescaleVersion, + out string reason) { var isRunning = string.Equals(jobStatus, "Running", StringComparison.OrdinalIgnoreCase); if (nextStartIsNegativeInfinity && !isRunning) { - reason = "next_start is -infinity — the scheduler will never run it again"; + reason = SchedulerRecoversNegativeInfinity(timescaleVersion) + ? NextStartNegativeInfinityCrashBackoffReason + : NextStartNegativeInfinityPermanentReason; return StuckCompressionJobArm.NextStartNegativeInfinity; } @@ -6977,6 +7153,17 @@ public static StuckCompressionJobArm ClassifyCompressionJob( /// which is why /// executes it TWICE, apart, when that arm trips. The text is /// unchanged from #1760; what changed is how many times it is asked. + /// + /// What a confirmed next_start_neg_infinity row IS depends on the store's TimescaleDB + /// (#3591), and this statement does not carry that. Below 2.26.4 it is the permanent dead state #1581 + /// was built against. From 2.26.4 (upstream #9360, ) the + /// scheduler sanitizes a persisted -infinity to "now", so the only row that stays -infinity + /// across the confirm delay is a crashed run in the scheduler's crash backoff, which the view shows as + /// -infinity + Scheduled + last_run_status IS NULL — and that last_run_status IS NULL is NOT a + /// discriminator between the two, because both are a mark_start whose mark_end never came. + /// The discriminator is pg_extension.extversion, read separately by + /// on the pass where the arm trips, and applied to the verdict's + /// text and re-arm by . /// public const string StuckCompressionJobsSql = @" SELECT @@ -7006,8 +7193,10 @@ WHERE j.proc_name LIKE '%compression%' /// once more, and reports the job only if the same arm trips /// again. A run-instant edge — the view pairing the scheduler's already-committed -infinity with a /// worker that is not yet, or no longer, visible as Running — is over in milliseconds and clears; - /// a row the scheduler has genuinely abandoned, or a crashed run sitting out its five-minute backoff, - /// reads the same on both passes and is reported with the latency it always had plus five seconds. One + /// a row the scheduler has genuinely abandoned (TimescaleDB below 2.26.4), or a crashed run sitting out its + /// crash backoff (the only persistent -infinity from 2.26.4 on, #3591 — reported with + /// set so the evaluator neither re-arms nor pages it on + /// first sight), reads the same on both passes and is reported with the latency it always had plus five seconds. One /// confirm per pass, not per job: the delay is taken once however many jobs tripped. The /// arm is reported from the first read as before — /// a six-hour elapsed bound has nothing to gain from a second look five seconds later. @@ -7038,7 +7227,8 @@ public static Task> ReadStuckCompressionJobsA Task.Delay, nowUtc, logger, - cancellationToken); + cancellationToken, + ct => ReadTimescaleVersionAsync(connection, logger, ct)); } /// @@ -7048,13 +7238,20 @@ public static Task> ReadStuckCompressionJobsA /// and the delay (so the pin can assert it is taken exactly when the -infinity arm tripped and /// never otherwise, without sleeping). Internal rather than private for that reason alone; production /// reaches it only through the connection overload. + /// + /// (#3591) is the store's TimescaleDB version, read ONLY on a pass + /// where the -infinity arm tripped — it decides that arm's sentence and nothing else, so the common + /// hourly pass still costs one read. Omitted (the pre-#3591 pins) or failing, the version is unknown and + /// the sentence is the conservative pre-2.26.4 one. Read before the confirm delay, on the same connection + /// the first read used, so the two reads that decide the page are not pushed further apart. /// internal static async Task> ReadStuckCompressionJobsAsync( Func>> readRows, Func delay, DateTime nowUtc, ILogger? logger, - CancellationToken cancellationToken) + CancellationToken cancellationToken, + Func>? readVersion = null) { if (readRows is null) { @@ -7071,11 +7268,28 @@ internal static async Task> ReadStuckCompress var first = ClassifyStuckCompressionJobs(await readRows(cancellationToken), nowUtc); if (!first.Any(f => f.Arm == StuckCompressionJobArm.NextStartNegativeInfinity)) { - /* Nothing on the racing arm: no delay, no second read. The common hourly pass costs exactly - what it did before #3575. */ + /* Nothing on the racing arm: no delay, no second read, no version read. The common hourly + pass costs exactly what it did before #3575. */ return first.Select(f => f.ToJob()).ToList(); } + /* #3591: the sentence the confirmed row will carry depends on whether this store's scheduler + recovers a -infinity row by itself. ReadTimescaleVersionAsync is failure-isolated to null, and + a scripted reader that throws is treated the same way here — the version decides words, never + the verdict, so it must not be able to fail the pass. */ + Version? timescaleVersion = null; + if (readVersion is not null) + { + try + { + timescaleVersion = await readVersion(cancellationToken); + } + catch (Exception ex) when (ex is not OperationCanceledException) + { + logger?.LogDebug("Could not read the TimescaleDB extension version — treating the store as pre-2.26.4 for the compression dead-job text (#3591): {Message}", ex.Message); + } + } + await delay(StuckCompressionConfirmDelay, cancellationToken); IReadOnlyList? confirm; @@ -7095,7 +7309,7 @@ what it did before #3575. */ ex.Message); } - return ConfirmStuckCompressionJobs(first, confirm, nowUtc, logger); + return ConfirmStuckCompressionJobs(first, confirm, nowUtc, logger, timescaleVersion); } catch (Exception ex) when (ex is not OperationCanceledException) { @@ -7124,12 +7338,18 @@ what it did before #3575. */ /// A job that appears on the confirm pass but not the first is not reported either: the confirm exists to /// ratify the first pass, not to widen it, and a job that only just went -infinity gets its own /// two reads next hour. + /// + /// (#3591) phrases the confirmed rows' reason for the store's + /// TimescaleDB; it is applied to the CONFIRM pass's classification because that is the row carried. The + /// first pass is classified without it on purpose — the arm is version-independent, and the first pass + /// only decides whether there is anything to confirm. /// internal static IReadOnlyList ConfirmStuckCompressionJobs( IReadOnlyList first, IReadOnlyList? confirm, DateTime nowUtc, - ILogger? logger) + ILogger? logger, + Version? timescaleVersion = null) { if (first is null) { @@ -7139,7 +7359,7 @@ internal static IReadOnlyList ConfirmStuckCompressionJobs( var result = new List(first.Count); var confirmed = confirm is null ? null - : ClassifyStuckCompressionJobs(confirm, nowUtc) + : ClassifyStuckCompressionJobs(confirm, nowUtc, timescaleVersion) .Where(c => c.Arm == StuckCompressionJobArm.NextStartNegativeInfinity) .ToDictionary(c => c.Row.JobId); @@ -7179,10 +7399,11 @@ internal static IReadOnlyList ConfirmStuckCompressionJobs( /// /// One pass of the pure predicate over a result set: every row it flags, with the arm that fired. Rows - /// the predicate clears are not returned. + /// the predicate clears are not returned. (#3591) phrases the + /// -infinity arm's reason; null is the pre-2.26.4 text. /// internal static List ClassifyStuckCompressionJobs( - IReadOnlyList rows, DateTime nowUtc) + IReadOnlyList rows, DateTime nowUtc, Version? timescaleVersion = null) { if (rows is null) { @@ -7193,10 +7414,12 @@ internal static List ClassifyStuckCompressionJobs( foreach (var row in rows) { var arm = ClassifyCompressionJob( - row.NextStartIsNegativeInfinity, row.JobStatus, row.LastRunStartedAtUtc, row.ScheduleInterval, nowUtc, out var reason); + row.NextStartIsNegativeInfinity, row.JobStatus, row.LastRunStartedAtUtc, row.ScheduleInterval, nowUtc, timescaleVersion, out var reason); if (arm != StuckCompressionJobArm.None) { - flagged.Add(new ClassifiedCompressionJob(row, arm, reason)); + flagged.Add(new ClassifiedCompressionJob( + row, arm, reason, + SchedulerRetries: arm == StuckCompressionJobArm.NextStartNegativeInfinity && SchedulerRecoversNegativeInfinity(timescaleVersion))); } } @@ -8125,8 +8348,21 @@ public static async Task TryRearmJobAsync( /// A COMPRESSION-policy background job that flagged /// as stuck (#1581): its immutable job_id, the hypertable it compresses (for a friendlier alert label — /// may be null on an odd catalog), and the human-readable reason the pure predicate produced. +/// +/// (#3591) is true for a -infinity row on a TimescaleDB at or +/// past : the row is a crashed run in the +/// scheduler's crash backoff, and the scheduler will re-run it by itself. The evaluator MUST NOT re-arm such a +/// row, measured on a 2.28.1 rig: alter_job(next_start => now()) against a job in crash backoff does +/// not shorten the wait — the scheduler's crash arm ignores the persisted next_start while +/// consecutive_crashes > 0 — it RESETS it. The re-arm refreshes the scheduler's job list, every crash +/// row's backoff is recomputed from that instant, and the retry moved from crash + 5:00 to re-arm + 5:04, for +/// the re-armed job AND for an un-re-armed sibling crash row in the same database. It also overwrites the +/// -infinity with a finite value, so the next hourly read would report the job healthy and post +/// "Recovered" before it had run. On the fleet's hourly policies (default retry_period one hour) that is a +/// retry pushed out by up to an hour and two untrue messages. false — the pre-2.26.4 row, an unknown +/// version, or the stuck-Running arm — keeps #1581's re-arm-once semantics exactly. /// -public sealed record StuckCompressionJob(long JobId, string? HypertableName, string Reason); +public sealed record StuckCompressionJob(long JobId, string? HypertableName, string Reason, bool SchedulerRetries = false); /// /// WHICH arm of fired (#3575), from @@ -8166,11 +8402,13 @@ internal sealed record CompressionJobStatRow( /// /// A the predicate flagged, with the arm that fired and its reason — /// the unit merges two passes of (#3575). +/// is , decided where the arm +/// was (#3591): the -infinity arm on a store whose scheduler has upstream #9360. /// -internal sealed record ClassifiedCompressionJob(CompressionJobStatRow Row, StuckCompressionJobArm Arm, string Reason) +internal sealed record ClassifiedCompressionJob(CompressionJobStatRow Row, StuckCompressionJobArm Arm, string Reason, bool SchedulerRetries = false) { /// The product-facing shape of this flag. - public StuckCompressionJob ToJob() => new(Row.JobId, Row.HypertableName, Reason); + public StuckCompressionJob ToJob() => new(Row.JobId, Row.HypertableName, Reason, SchedulerRetries); } /// From 665de2f19c1c57db7f4eb943bbcb4cd24f31a026 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 14:48:24 -0400 Subject: [PATCH 50/69] Slack text cuts land on whole characters, not UTF-16 indexes: an emoji astride a limit no longer reaches the reader as a replacement glyph, and a cut field's omitted count is in characters the reader can count (#3622) (#3625) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * Slack text cuts land on whole characters, not UTF-16 indexes: an emoji astride a limit no longer reaches the reader as a replacement glyph, and a cut field's omitted count is in characters the reader can count (#3622) Six cuts in WebhookAlertService sized themselves in UTF-16 units — the prose hard split and the omission line's quoted fragment (#3493); the detail heading, the field label, the field value and the omitted-headings list (#3612). A cut between the halves of a surrogate pair left an unpaired surrogate. Measured: System.Text.Json does NOT emit a lone escape and Slack does NOT reject the payload — the serializer substitutes U+FFFD (the relaxation EscapeForJson already relies on), so the payload delivered and lied: a � the value never held, a "leading stretch" that was no longer a prefix, a hard split that showed two glyphs and no emoji, and "N more characters" counted in code units. One helper, SlackCutLength, at all six sites: the cut lands on a text-element (grapheme) boundary, and falls back to a code-point boundary only when a single element is wider than the whole limit, so the hard-split loop always advances. SlackFieldText counts the omitted characters as text elements, so kept plus omitted is the value's own character count. Pins in both sibling suites place 🔥, e+U+0301 and 👍🏽 astride every boundary; 15 arms fail on the old code and pass on the new; the one-element-wider-than-a- section input pins that the splitter still terminates. * SlackCutLength hands the segmenter a window two units past the limit, so the cost claim is true by construction, and the field's tail count states its own cost (review notes on #3625) The review traced the "bounded by limit" claim and found it true of the iteration count but not of one call: text[cut..] ran to the end of the text, so a single long combining run made GetNextTextElementLength scan the whole remainder before the loop could conclude "crosses". The window ends two units past the limit — enough to hold any straddling scalar whole, and grapheme rules look left, so every boundary decision at or before the limit is the full text's. Four table rows pin it, including a window that ends mid-pair. The tail count in SlackFieldText is O(tail) by necessity — counting characters is a walk of them — and now says so at the site, with the upstream bound. --- Lite.Tests/SlackDetailsSizeTests.cs | 221 ++++++++++++++++++ Lite.Tests/SlackProseSplitTests.cs | 131 +++++++++++ .../WebhookAlertService.cs | 123 +++++++++- 3 files changed, 463 insertions(+), 12 deletions(-) diff --git a/Lite.Tests/SlackDetailsSizeTests.cs b/Lite.Tests/SlackDetailsSizeTests.cs index 400cdd3af..9454b25e3 100644 --- a/Lite.Tests/SlackDetailsSizeTests.cs +++ b/Lite.Tests/SlackDetailsSizeTests.cs @@ -47,6 +47,16 @@ namespace PerformanceMonitorLite.Tests; /// knows exactly where the message stopped. And a bounding pass that changed the SMALL case would repaint /// every engine alert and every analysis page that delivered before it, so the fitting shapes are pinned /// against the pre-#3612 rendering byte for byte — the oracle is the old loop, restated here verbatim. +/// +/// #3622: the per-text-object cuts land on whole characters, and the stated count is in +/// characters. The heading cap, the field cap (label and value) and the omitted-headings list were all +/// cut by UTF-16 index; an emoji or a combining sequence astride the boundary was cut in half, the +/// serializer relaxed the unpaired half to U+FFFD (measured — the payload delivered, contrary to the +/// issue's expectation of a rejection), and the reader was shown a replacement glyph in a value that never +/// held one, under a "leading stretch" that was no longer a prefix, beside a "more characters" count in +/// units the reader cannot see. The arms in the #3622 section place a character astride each boundary and +/// assert: no text a reader was sent carries U+FFFD, the cut landed one character earlier, and kept plus +/// omitted is the value's own character count. /// public class SlackDetailsSizeTests { @@ -283,6 +293,28 @@ private static List AllTexts(List blocks) => private static string OmissionText(List blocks) => Assert.Single(blocks.Select(SectionText), t => t is not null && t.StartsWith(OmissionHeading, StringComparison.Ordinal))!; + /* #3622 fixtures: one character each to a reader, two or four UTF-16 units to the index arithmetic. */ + private const string Fire = "\U0001F525"; // 🔥, a surrogate pair + private const string EAcute = "e\u0301"; // e + combining acute, two code points + private const string ThumbsUpMedium = "\U0001F44D\U0001F3FD"; // 👍🏽, two pairs in one grapheme + + private static void AssertNoReplacementGlyph(List blocks) => + Assert.All(AllTexts(blocks), t => Assert.DoesNotContain('\uFFFD', t)); + + /// The one cut field text of a payload, and its parts: the kept stretch of the value and the + /// omitted count the note states — read from the payload, never restated from the builder. + private static (string Kept, int Omitted) CutField(List blocks, string prefix) + { + var cut = Assert.Single(blocks.SelectMany(FieldTexts), f => f.StartsWith(prefix, StringComparison.Ordinal) && f.Contains("... (", StringComparison.Ordinal)); + Assert.True(cut.Length <= FieldTextLimit, $"the cut field is {cut.Length} chars"); + var noteAt = cut.IndexOf("... (", StringComparison.Ordinal); + var kept = cut[prefix.Length..noteAt]; + var omitted = int.Parse(cut[(noteAt + 5)..cut.IndexOf(" more characters", StringComparison.Ordinal)], NumberStyles.AllowThousands, CultureInfo.InvariantCulture); + return (kept, omitted); + } + + private static int Characters(string s) => new StringInfo(s).LengthInTextElements; + /// /// The pre-#3612 details loop, verbatim: divider, then a pointer section, a body section, or heading + /// fields in sections of ten. Serialized the way the builder serializes, this is the oracle every @@ -703,6 +735,195 @@ public void AFieldPastTheFieldCeiling_IsCutWithTheOmissionStatedInline() Assert.Contains("*#1 Database:*\nReportingDB", fields); } + /* ---------------- #3622: the cuts land on whole characters ---------------- */ + + /// + /// A field value with a character astride the field cut keeps everything before the character and + /// states the omission in characters. Self-calibrating: the control arm — an all-ASCII value of the + /// same length — reads the builder's own kept length K from the payload, and the value under test puts + /// the character at units K−1 and K, exactly one unit into the old cut. Before #3622 the kept stretch + /// ended in the character's first unit (a replacement glyph to the reader, and no longer a prefix of + /// the value) and the count was in UTF-16 units. Now the cut lands at K−1, the kept stretch is a + /// prefix, and kept + omitted is the value's own character count. + /// + [Theory] + [InlineData(Fire)] + [InlineData(EAcute)] + [InlineData(ThumbsUpMedium)] + public void AFieldValueWithACharacterAstrideTheCut_KeepsAPrefix_AndCountsInCharacters(string character) + { + const string prefix = "*#1 Query Text:*\n"; + const int length = 5000; + + static AlertContext ContextFor(string value) + { + var context = new AlertContext(); + var item = new AlertDetailItem { Heading = "Regressed Queries" }; + item.Fields.Add(("#1 Query Text", value)); + context.Details.Add(item); + return context; + } + + /* Control: where does the builder cut a plain value of this length? */ + using var controlDoc = JsonDocument.Parse(Payload(ContextFor(new string('x', length)))); + var (controlKept, controlOmitted) = CutField(Blocks(controlDoc), prefix); + var k = controlKept.Length; + Assert.Equal(length, k + controlOmitted); + Assert.InRange(k, 1000, FieldTextLimit); + + /* The character's first unit sits at K−1, so the old cut at K went through it. */ + var value = new string('x', k - 1) + character + new string('y', length - (k - 1) - character.Length); + Assert.Equal(length, value.Length); + Assert.True(char.IsHighSurrogate(value[k - 1]) || value[k] == '\u0301'); + + using var doc = JsonDocument.Parse(Payload(ContextFor(value))); + var blocks = Blocks(doc); + AssertInsideEveryCeiling(blocks); + AssertNoReplacementGlyph(blocks); + + var (kept, omitted) = CutField(blocks, prefix); + Assert.Equal(new string('x', k - 1), kept); + Assert.Equal(Characters(value) - Characters(kept), omitted); + Assert.Equal(Characters(value[(k - 1)..]), omitted); + } + + /// + /// The count is in the characters a reader would count for the whole value, not only at the cut: a + /// value that is nothing but multi-unit characters states an omitted count equal to how many of them + /// the reader is not shown, and the kept stretch is a whole number of them. + /// + [Theory] + [InlineData(Fire)] + [InlineData(EAcute)] + [InlineData(ThumbsUpMedium)] + public void AFieldValueOfMultiUnitCharacters_IsCutOnACharacter_AndCountedInCharacters(string character) + { + const string prefix = "*Advice:*\n"; + var value = string.Concat(Enumerable.Repeat(character, 1500)); + + var context = new AlertContext(); + var item = new AlertDetailItem { Heading = "Diagnosis" }; + item.Fields.Add(("Advice", value)); + context.Details.Add(item); + + using var doc = JsonDocument.Parse(Payload(context)); + var blocks = Blocks(doc); + AssertInsideEveryCeiling(blocks); + AssertNoReplacementGlyph(blocks); + + var (kept, omitted) = CutField(blocks, prefix); + Assert.Equal(0, kept.Length % character.Length); + Assert.Equal(value[..kept.Length], kept); + Assert.Equal(1500 - kept.Length / character.Length, omitted); + } + + /// + /// The label side of the same field: a label so long it alone crowds the field is bounded to half of + /// it, and that cut, too, lands on a whole character. Pathological (labels are producer literals), but + /// it is the same slice by the same arithmetic, and a fix that left it would be a fifth door. + /// + [Fact] + public void AFieldLabelWithACharacterAstrideItsBound_IsCutBeforeTheCharacter() + { + /* The prefix is "*" + label + ":*\n"; the bound is 1,000, so a character at label units 998–999 + sits at prefix units 999–1,000 — one unit into the old cut. */ + var label = new string('L', 998) + Fire + new string('L', 50); + var context = new AlertContext(); + var item = new AlertDetailItem { Heading = "Wide label" }; + item.Fields.Add((label, new string('v', 1500))); + context.Details.Add(item); + + using var doc = JsonDocument.Parse(Payload(context)); + var blocks = Blocks(doc); + AssertInsideEveryCeiling(blocks); + AssertNoReplacementGlyph(blocks); + + var cut = Assert.Single(blocks.SelectMany(FieldTexts), f => f.StartsWith("*LLLL", StringComparison.Ordinal)); + Assert.StartsWith("*" + new string('L', 998) + "vvv", cut, StringComparison.Ordinal); + Assert.Contains("more characters", cut, StringComparison.Ordinal); + } + + /// + /// A detail heading with a character astride the heading cap is cut before the character: the heading + /// is the detail's identity, and one ending in half an emoji names a different thing. Rendered as the + /// leading field of a field detail and as the header of a body detail alike. + /// + [Theory] + [InlineData(Fire)] + [InlineData(EAcute)] + public void ADetailHeadingWithACharacterAstrideTheCap_IsCutBeforeTheCharacter(string character) + { + var heading = new string('h', 499) + character + new string('h', 50); + var context = new AlertContext(); + var fieldItem = new AlertDetailItem { Heading = heading }; + fieldItem.Fields.Add(("Dedup Key", "k")); + context.Details.Add(fieldItem); + context.Details.Add(new AlertDetailItem { Heading = heading, Body = "Investigation: look." }); + + using var doc = JsonDocument.Parse(Payload(context)); + var blocks = Blocks(doc); + AssertInsideEveryCeiling(blocks); + AssertNoReplacementGlyph(blocks); + + var expected = "*" + new string('h', 499) + "...*"; + Assert.Contains(expected, blocks.SelectMany(FieldTexts)); + Assert.Contains(blocks.Select(SectionText), t => t?.StartsWith(expected + "\nInvestigation:", StringComparison.Ordinal) == true); + Assert.DoesNotContain(AllTexts(blocks), t => t.Contains(character, StringComparison.Ordinal)); + } + + /// + /// The omission item's heading list is quoted to 1,000 characters; a character astride that boundary + /// is left out rather than cut in half. The mixed page from + /// drops exactly "Wide 5" and "Wide 6"; "Wide 5; " is eight units, so a character at units 991–992 of + /// the sixth heading sits at list units 999–1,000. + /// + [Fact] + public void TheOmittedHeadingsList_WithACharacterAstrideItsBound_IsCutBeforeTheCharacter() + { + static AlertDetailItem Narrow(int i) + { + var item = new AlertDetailItem { Heading = string.Create(CultureInfo.InvariantCulture, $"Narrow {i}") }; + item.Fields.Add(("Dedup Key", "k")); + return item; + } + + static AlertDetailItem Wide(string heading) + { + var item = new AlertDetailItem { Heading = heading }; + for (var f = 0; f < 25; f++) + { + item.Fields.Add((string.Create(CultureInfo.InvariantCulture, $"f{f}"), "v")); + } + + return item; + } + + var context = new AlertContext(); + for (var i = 0; i < 11; i++) + { + context.Details.Add(Narrow(i)); + } + + for (var i = 0; i < 6; i++) + { + context.Details.Add(Wide(string.Create(CultureInfo.InvariantCulture, $"Wide {i}"))); + } + + var sixth = new string('w', 991) + Fire + new string('w', 100); + context.Details.Add(Wide(sixth)); + + using var doc = JsonDocument.Parse(Payload(context)); + var blocks = Blocks(doc); + Assert.Equal(47, blocks.Count); + AssertInsideEveryCeiling(blocks); + AssertNoReplacementGlyph(blocks); + + var omission = OmissionText(blocks); + Assert.Contains("2 more details did not fit", omission, StringComparison.Ordinal); + Assert.Contains("Wide 5; " + new string('w', 991) + "... - " + Pointer, omission, StringComparison.Ordinal); + Assert.DoesNotContain(Fire, omission, StringComparison.Ordinal); + } + /* ---------------- Teams ---------------- */ /// diff --git a/Lite.Tests/SlackProseSplitTests.cs b/Lite.Tests/SlackProseSplitTests.cs index a4844e2a0..119f33176 100644 --- a/Lite.Tests/SlackProseSplitTests.cs +++ b/Lite.Tests/SlackProseSplitTests.cs @@ -32,6 +32,16 @@ namespace PerformanceMonitorLite.Tests; /// because an omission note that is itself vague recreates the silent-truncation problem one level up. /// And a splitter that changed the SMALL case would repaint every ordinary alert's payload, so the /// one-section shape is pinned byte-for-byte. +/// +/// #3622: the cuts land on whole characters. Both of this splitter's cuts — the hard split +/// and the omission line's quoted fragment — were sized in UTF-16 units, so an emoji astride the boundary +/// was cut in half. JsonSerializer relaxes the unpaired half to U+FFFD rather than emitting a lone +/// escape (measured in the lane; the issue expected Slack to reject the payload), so the payload delivered +/// and the reader saw a replacement glyph that was never in the text — the hard split showed TWO of them +/// and no emoji. The arms below place a surrogate pair and a combining sequence exactly astride each +/// boundary and assert three things: no text a reader was sent contains U+FFFD, the cut landed one +/// character earlier, and reassembly still reproduces the original. The pathological single-element +/// input (a run of combining marks wider than a section) pins that the loop still advances. /// public class SlackProseSplitTests { @@ -141,6 +151,17 @@ private static string Line(int i, int length) private static string ManyLines(int count, int length) => string.Join('\n', Enumerable.Range(0, count).Select(i => Line(i, length))); + /* #3622 fixtures: one character each to a reader, two or four UTF-16 units to the index arithmetic. */ + private const string Fire = "\U0001F525"; // 🔥, a surrogate pair + private const string EAcute = "e\u0301"; // e + combining acute, two code points + private const string ThumbsUpMedium = "\U0001F44D\U0001F3FD"; // 👍🏽, two pairs in one grapheme + + /// Every mrkdwn text a reader was sent, parsed — an unpaired surrogate never survives into + /// the raw JSON (the serializer relaxes it to \uFFFD), so the only place it can be caught is the + /// decoded text. + private static void AssertNoReplacementGlyph(JsonDocument doc) => + Assert.All(AllTextStrings(doc.RootElement), t => Assert.DoesNotContain('\uFFFD', t)); + /* ---------------- the shape that must not change ---------------- */ /// @@ -250,6 +271,116 @@ public void ASingleLinePastTheCeiling_HardSplitsWithAContinuationMarker() Assert.Equal(line, reassembled); } + /* ---------------- #3622: the cuts land on whole characters ---------------- */ + + /// + /// The helper every Slack cut goes through, on its own: the whole text when it fits; the limit when the + /// limit is already a boundary; one unit earlier when the limit falls between the halves of a surrogate + /// pair; before the base letter when it falls between a letter and its combining accent; before a + /// four-unit emoji at every interior offset; and — the one case a whole-element cut cannot serve — a + /// single element wider than the limit falls back to the code-point boundary, so a caller that must + /// advance always can. The rows with a long combining run pin the segmenter's window: the helper hands + /// it two units past the limit rather than the whole text, and the answer must be the full text's + /// whether the element ends inside that window or runs through its end. + /// + [Theory] + [InlineData("abc", 5, 3)] // fits: the whole text + [InlineData("abcdef", 3, 3)] // ASCII: the limit is a boundary + [InlineData("ab\U0001F525cd", 3, 2)] // pair astride: one earlier + [InlineData("ab\U0001F525cd", 4, 4)] // pair inside: the limit + [InlineData("abe\u0301cd", 3, 2)] // letter + accent astride: before the letter + [InlineData("ab\U0001F44D\U0001F3FDcd", 3, 2)] // four-unit emoji, cut after its first unit + [InlineData("ab\U0001F44D\U0001F3FDcd", 4, 2)] // ...after its first pair + [InlineData("ab\U0001F44D\U0001F3FDcd", 5, 2)] // ...after its third unit + [InlineData("ab\U0001F44D\U0001F3FDcd", 6, 6)] // ...after the whole emoji: the limit + [InlineData("ab\u0301\u0301\u0301\u0301\u0301\u0301\u0301\u0301cd", 3, 1)] // b + eight accents runs through the window: before b + [InlineData("ab\u0301\u0301\u0301\u0301\u0301\u0301\u0301\u0301cd", 9, 1)] // ...one unit short of fitting: still before b + [InlineData("ab\u0301\u0301\u0301\u0301\u0301\u0301\u0301\u0301cd", 10, 10)] // ...fits exactly: after the last accent + [InlineData("a\U0001F44D\U0001F3FDcd", 2, 1)] // window ends mid-pair inside the emoji: before it + [InlineData("\u0301\u0301\u0301\u0301", 2, 2)] // one element wider than the limit: code point + [InlineData("\U0001F525\u0301\u0301\u0301", 3, 3)] // ...the limit itself when it is a code-point boundary + [InlineData("\U0001F44D\U0001F3FD\u0301\u0301", 3, 2)] // ...and one earlier when it is mid-pair + [InlineData("abc", 0, 0)] + public void SlackCutLength_LandsOnTheLastWholeCharacterInsideTheLimit(string text, int limit, int expected) + { + Assert.Equal(expected, WebhookAlertService.SlackCutLength(text, limit)); + } + + /// + /// A character astride the hard split's boundary is carried whole into the continuation piece rather + /// than cut in half. Before #3622 the first section ended with the pair's high half and the second + /// began with its low half after the marker: two replacement glyphs, no emoji, and a reassembly that + /// no longer matched the line. Each arm sizes its line so the boundary (the section's capacity, 2,990 + /// after the header) falls one unit into the character. + /// + [Theory] + [InlineData(Fire)] + [InlineData(EAcute)] + [InlineData(ThumbsUpMedium)] + public void ACharacterAstrideTheHardSplit_MovesWholeIntoTheContinuation(string character) + { + const string marker = "(cont.) "; + var line = new string('x', Capacity - 1) + character + new string('y', 100); + + using var doc = JsonDocument.Parse(Payload(line)); + AssertNoReplacementGlyph(doc); + var texts = ProseTexts(Blocks(doc)); + + Assert.Equal(2, texts.Count); + Assert.All(texts, t => Assert.True(t.Length <= 3000)); + Assert.Equal(Header + new string('x', Capacity - 1), texts[0]); + Assert.StartsWith(marker + character, texts[1], StringComparison.Ordinal); + + var reassembled = texts[0][Header.Length..] + string.Concat(texts.Skip(1).Select(t => t[marker.Length..])); + Assert.Equal(line, reassembled); + } + + /// + /// A single line that is ONE text element wider than a section — combining marks with no base — cannot + /// be cut on an element boundary at all. The cut falls back to the code point and the loop advances: + /// the line still splits, every piece fits, and the pieces reassemble to the line. Without the fallback + /// this input hangs the splitter, which is why it is pinned before any other property of it. + /// + [Fact] + public void ASingleElementWiderThanASection_StillHardSplits_AndReassembles() + { + const string marker = "(cont.) "; + var line = new string('\u0301', 7000); + + using var doc = JsonDocument.Parse(Payload(line)); + var texts = ProseTexts(Blocks(doc)); + + Assert.True(texts.Count >= 3); + Assert.All(texts, t => Assert.True(t.Length <= 3000)); + var reassembled = texts[0][Header.Length..] + string.Concat(texts.Skip(1).Select(t => t[marker.Length..])); + Assert.Equal(line, reassembled); + } + + /// + /// The omission line quotes the first dropped line's leading 120 characters; a character astride that + /// boundary is left out of the quote rather than cut in half. Every line carries the character at + /// units 119–120 (one unit into the 120 cut), so whichever line is the first dropped, the quote ends + /// one character earlier and the payload carries no replacement glyph. + /// + [Theory] + [InlineData(Fire)] + [InlineData(EAcute)] + public void ACharacterAstrideTheOmissionFragment_IsLeftOutOfTheQuote(string character) + { + string LineWith(int i) => Line(i, 119) + character + new string('z', 179); + var prose = string.Join('\n', Enumerable.Range(0, 600).Select(LineWith)); + + using var doc = JsonDocument.Parse(Payload(prose)); + AssertNoReplacementGlyph(doc); + var texts = ProseTexts(Blocks(doc)); + var lines = Reassemble(texts).Split('\n'); + var emitted = lines.Length - 1; + + Assert.Contains( + $"first omitted: \"{Line(emitted, 119)}...\"", lines[^1], StringComparison.Ordinal); + Assert.DoesNotContain(character, lines[^1], StringComparison.Ordinal); + } + /* ---------------- the block budget and the stated omission ---------------- */ /// diff --git a/PerformanceMonitor.Notifications/WebhookAlertService.cs b/PerformanceMonitor.Notifications/WebhookAlertService.cs index 1ad7f90e9..ef699011c 100644 --- a/PerformanceMonitor.Notifications/WebhookAlertService.cs +++ b/PerformanceMonitor.Notifications/WebhookAlertService.cs @@ -968,8 +968,8 @@ them is edited. */ /// /// The prose, as lines that each fit a section on their own. A single line longer than /// — pathological, but a delivery that fails over it would be this bug - /// again — hard-splits at character boundaries, every continuation piece marked with - /// . + /// again — hard-splits at character boundaries (whole characters, through , + /// since #3622), every continuation piece marked with . /// private static List SplitProseIntoSectionSafeLines(string prose, int capacity) { @@ -986,7 +986,12 @@ private static List SplitProseIntoSectionSafeLines(string prose, int cap while (start < raw.Length) { var prefix = start == 0 ? string.Empty : SlackProseContinuationMarker; - var take = Math.Min(capacity - prefix.Length, raw.Length - start); + /* #3622: the piece ends on a character boundary, never between the halves of a surrogate + pair or through a combining sequence — the index arithmetic alone put an emoji's two + halves in two sections, and the reader saw two replacement glyphs and no emoji. The + loop always advances: the capacity is in the thousands and the marker is eight, so the + cut can never back off to nothing. */ + var take = SlackCutLength(raw.AsSpan(start), capacity - prefix.Length); lines.Add(prefix + raw.Substring(start, take)); start += take; } @@ -1069,7 +1074,7 @@ fit MORE lines than the pass that already overflowed — the omission line is al var firstDropped = lines[placed]; var fragment = firstDropped.Length <= SlackOmissionFragmentLimit ? firstDropped - : firstDropped[..SlackOmissionFragmentLimit] + "..."; + : firstDropped[..SlackCutLength(firstDropped, SlackOmissionFragmentLimit)] + "..."; var noun = dropped == 1 ? "line" : "lines"; var omission = string.Create(CultureInfo.InvariantCulture, $"... and {dropped:N0} more {noun}, first omitted: \"{fragment}\"{SlackOmissionPointer}"); @@ -1220,9 +1225,10 @@ is for. */ list.Append(details[i].Heading); } - var headings = list.Length <= SlackOmittedHeadingsLimit - ? list.ToString() - : list.ToString(0, SlackOmittedHeadingsLimit) + "..."; + var listed = list.ToString(); + var headings = listed.Length <= SlackOmittedHeadingsLimit + ? listed + : listed[..SlackCutLength(listed, SlackOmittedHeadingsLimit)] + "..."; var noun = dropped == 1 ? "detail" : "details"; var omission = string.Create(CultureInfo.InvariantCulture, @@ -1279,9 +1285,11 @@ private static List RenderSlackDetail(AlertDetailItem detail, int bodyBl } /// A detail heading bounded to ; unchanged for every - /// heading a producer has ever emitted. + /// heading a producer has ever emitted. The cut lands on a whole character (#3622): a heading IS the + /// detail's identity, and one ending in half an emoji or a letter shorn of its accent names a + /// different thing. private static string SlackHeading(string heading) => - heading.Length <= SlackDetailHeadingLimit ? heading : heading[..SlackDetailHeadingLimit] + "..."; + heading.Length <= SlackDetailHeadingLimit ? heading : heading[..SlackCutLength(heading, SlackDetailHeadingLimit)] + "..."; /// /// One detail field's text object, *label:* over its value, inside @@ -1291,6 +1299,13 @@ private static string SlackHeading(string heading) => /// room for a closing line. Two passes size the kept stretch: the first against the longest count /// the note could carry, the second against the count it actually carries, so the result never /// exceeds the limit. Every field inside the limit renders the pre-#3612 bytes. + /// + /// #3622: the kept stretch ends on a whole character (), and the + /// omitted count is stated in the characters a reader would count — text elements, so one emoji is + /// one, a letter with its combining accent is one — not in UTF-16 units. Kept plus omitted is the + /// value's own character count, which is the only arithmetic a reader can check. The count can + /// only be at or below the UTF-16 count the first sizing pass allowed for, so the note can only be + /// narrower than the room held for it, and the field stays inside its limit. /// private static string SlackFieldText(string label, string value) { @@ -1307,14 +1322,98 @@ static string Note(int omitted) => string.Create(CultureInfo.InvariantCulture, the arithmetic below stays positive. */ if (prefix.Length > SlackFieldTextLimit / 2) { - prefix = prefix[..(SlackFieldTextLimit / 2)]; + prefix = prefix[..SlackCutLength(prefix, SlackFieldTextLimit / 2)]; } - var keep = Math.Max(0, SlackFieldTextLimit - prefix.Length - Note(value.Length).Length); - var note = Note(value.Length - keep); + var keep = SlackCutLength(value, Math.Max(0, SlackFieldTextLimit - prefix.Length - Note(value.Length).Length)); + /* Counting the omitted characters IS a walk of the omitted tail — the pre-#3622 subtraction counted + units, which is the thing that was wrong — so this costs the tail's length, once, on the + truncation path only. Every producer bounds its values upstream (the analysis formatter cuts + drill-down text at 300 characters; the longest measured field is 318), so the tail is short in + practice and the cost is the value's own length in the worst case (review note on #3625). */ + var note = Note(new StringInfo(value[keep..]).LengthInTextElements); return prefix + value[..keep] + note; } + /// + /// The length of the longest leading stretch of that fits inside + /// UTF-16 units without cutting through a character (#3622). Every Slack text + /// cut in this builder passes through here — the prose hard split and the omission line's quoted + /// fragment (#3493); the detail heading, the field label, the field value and the omitted-headings + /// list (#3612) — because each of them sized its cut in UTF-16 units, and a cut landing between the + /// two halves of a surrogate pair (any emoji, any supplementary-plane character, in a query text, a + /// database name or an advice string) left an unpaired surrogate at the boundary. JsonSerializer + /// does not emit that as a lone escape and Slack does not reject it — measured here, System.Text.Json + /// substitutes U+FFFD, the same relaxation already relies on — so the + /// payload delivered, and what it delivered was wrong: the reader saw a replacement glyph (\uFFFD) that + /// was never in the value, the "leading stretch" a cut field claims to show was no longer a prefix of + /// the value, the hard split lost the character outright (each half became its own glyph, one per + /// section), and the field's stated "N more characters" counted UTF-16 units the reader cannot see. + /// #3622 expected the invalid-payload door; the door it actually opened is a payload that lies. + /// + /// Grapheme, not code point, at every site. The cut lands on a text-element boundary — an + /// extended grapheme cluster, the unit a reader counts as one character: an emoji with its skin-tone + /// modifier, a base letter with its combining accent, a CR LF pair — so no site can leave half a + /// visible character behind, and the one omission that is stated in characters + /// () cuts and counts in the same unit. The recommendation on #3622 was a + /// code-point cut at the two long-prose sites on cost grounds, and that trade does not exist: the walk + /// is bounded by , not by the text — it stops at the first element that would + /// cross the limit, and the window it hands the segmenter ends two units past the limit rather than at + /// the end of the text (see the loop) — so it costs the same at a 120-character fragment as at a + /// 3,000-character section, and one rule at six sites is cheaper to keep true than two. The one input a whole-element cut cannot + /// serve is a single element wider than the whole limit — a run of combining marks with no base, the + /// "Zalgo text" a query comment can carry — where keeping whole elements would keep nothing and the + /// hard-split loop would never advance; there the cut falls back to the code-point boundary (one unit + /// before a surrogate pair, else the limit itself), because a valid payload showing a broken glyph + /// beats a delivery that hangs. Callers pass limits in the hundreds and thousands, so the fallback + /// always keeps at least one unit. + /// + internal static int SlackCutLength(ReadOnlySpan text, int limit) + { + if (text.Length <= limit) + { + return text.Length; + } + + if (limit <= 0) + { + return 0; + } + + /* text.Length > limit here, so the window is never empty while cut <= limit, and every element is + at least one unit wide, so the walk terminates at the first element that would cross. + + The window handed to the segmenter ends two units past the limit, not at the end of the text. + Two units hold any scalar that straddles the limit whole, and grapheme boundaries are decided + between one scalar and the next (every rule's context is to the LEFT), so every boundary + decision at or before the limit is the one the full text would make: an element that ends + inside the window ends where the full text ends it, and an element that reaches the window's + end has crossed the limit however the full text would segment the rest of it. That is what + makes the cost claim in the doc block true — without the window, one run of combining marks + makes this call scan to the end of the text before concluding that it crosses (review note on + #3625). */ + var cut = 0; + while (true) + { + var window = text.Slice(cut, Math.Min(text.Length - cut, limit - cut + 2)); + var element = StringInfo.GetNextTextElementLength(window); + if (cut + element > limit) + { + break; + } + + cut += element; + } + + if (cut > 0) + { + return cut; + } + + /* No whole element fits: a code-point boundary keeps the payload valid and the caller moving. */ + return char.IsHighSurrogate(text[limit - 1]) && char.IsLowSurrogate(text[limit]) ? limit - 1 : limit; + } + /// /// Builds a Slack incoming webhook payload with a colored attachment sidebar. /// Uses Slack Block Kit for rich formatting. From 14aed226b8f9f387d20c5251c238d38be859157f Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 14:50:33 -0400 Subject: [PATCH 51/69] Forced Plan Failing fires once per observation, not once per cooldown: the adapter returns the collection that produced the rise and the engine remembers which one it already reported (#3579) (#3628) * Forced Plan Failing fires once per observation, not once per cooldown: the adapter returns the collection that produced the rise and the engine remembers which one it already reported (#3579) * The live access-path seed is floored to whole microseconds so the #3579 stamp round-trips PostgreSQL's timestamp to the tick (#3579) --- Darling/Darling.Tests/AlertEngineTests.cs | 204 ++++++++++++++++++ .../ForcePlanFailuresAccessPathTests.cs | 37 +++- .../DarlingAlertReadAdapter.cs | 19 +- .../BlockingDeadlockContextBuilderTests.cs | 32 +++ Lite.Tests/ForcePlanFailuresReadTests.cs | 148 +++++++++++++ .../LocalDataService.ForcePlanFailures.cs | 15 +- PerformanceMonitor.Alerting/AlertEngine.cs | 98 ++++++++- .../ForcePlanFailureInfo.cs | 28 +++ 8 files changed, 568 insertions(+), 13 deletions(-) create mode 100644 Lite.Tests/ForcePlanFailuresReadTests.cs diff --git a/Darling/Darling.Tests/AlertEngineTests.cs b/Darling/Darling.Tests/AlertEngineTests.cs index b2712cf39..932ffc23d 100644 --- a/Darling/Darling.Tests/AlertEngineTests.cs +++ b/Darling/Darling.Tests/AlertEngineTests.cs @@ -2707,6 +2707,210 @@ the plan the way the firing message did — NOT the internal key. The first vers Assert.Contains("plan 22", resolution.Message, StringComparison.Ordinal); } + /* ---------------- #3579: one observation, one card ---------------- */ + + /// The production series' collection instants (#3579), so the pins below replay the shape that + /// was measured rather than an invented one: the 04:02→04:18 rise, six cooldowns of re-reads, the 04:50 + /// collection with the counter back at zero. + private static readonly DateTime Collection0402 = new(2026, 9, 18, 4, 2, 0, DateTimeKind.Utc); + private static readonly DateTime Collection0418 = new(2026, 9, 18, 4, 18, 0, DateTimeKind.Utc); + private static readonly DateTime Collection0450 = new(2026, 9, 18, 4, 50, 0, DateTimeKind.Utc); + + private static ForcePlanFailureInfo ForcePlanRow(DateTime? observedAt, long delta = 1, long total = 1) => new() + { + DatabaseName = "Sales", QueryId = 11, PlanId = 22, ForcingType = "AUTO", FailureReason = "NONE", + FailureDelta = delta, TotalFailures = total, ObservedAtUtc = observedAt + }; + + [Fact] + public async Task ForcePlanFailure_SameObservationAcrossSixCooldowns_FiresOnce_ThenResolvesOnTheNextCollection() + { + /* #3579's measured shape: one plan's counter went 0 → 1 at the 04:18 collection and back to 0 at + 04:50. Between those two collections the adapter returned the SAME row (newest = 04:18, previous = + 04:02) on every ~30 s pass, and the pre-#3579 engine fired at 04:18:53, 04:24:31, 04:30:04, + 04:35:33, 04:40:49 and 04:46:03 — six cards, every one reading New 1 / Total 1, for a force that + failed once. The cooldown elapsing is not proof a new observation exists. */ + var h = new Harness(); + h.Settings.ForcePlanFailureEnabled = true; + h.Settings.CooldownMinutes = 5; + h.Now = Collection0418.AddSeconds(53); + h.Adapter.ForcePlanFailures.Add(ForcePlanRow(Collection0418)); + var engine = h.Build(); + + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Single(h.Deliverer.Outcomes); + + /* The five re-reads that each fired before. Every one is past the cooldown; none is a new observation. */ + foreach (var refire in new[] { "04:24:31", "04:30:04", "04:35:33", "04:40:49", "04:46:03" }) + { + h.Now = DateTime.SpecifyKind(DateTime.Parse("2026-09-18 " + refire, System.Globalization.CultureInfo.InvariantCulture), DateTimeKind.Utc); + await engine.EvaluateServerAsync(Harness.Snapshot()); + } + + Assert.Single(h.Deliverer.Outcomes); + /* And the read still happened on every pass — the guard is on the FIRE, never on the fetch, so the + recovery arm keeps seeing fresh evidence. */ + Assert.Equal(6, h.Adapter.ForcePlanFetches); + + /* 04:50: APC released the forcing and the counter reset. The adapter's '>' filter drops the row, and + the recovery is announced exactly as before #3579. */ + h.Adapter.ForcePlanFailures.Clear(); + h.Now = Collection0450.AddSeconds(40); + await engine.EvaluateServerAsync(Harness.Snapshot()); + + Assert.Single(h.Deliverer.Outcomes); + var resolution = Assert.Single(h.Resolutions, r => r.MetricName == ForcePlanTokens.MetricName); + Assert.Contains("plan 22 no longer failing to force", resolution.Message, StringComparison.Ordinal); + } + + [Fact] + public async Task ForcePlanFailure_NewerObservationWithARise_FiresAgain_CarryingItsOwnNumbers() + { + /* A plan that keeps failing across successive collections is still a standing condition: each + collection is a NEW observation with a new rise, and it re-fires — the guard removes repeats of + one observation, not the second card for a second failure. The card carries the new observation's + delta and total, not the first one's. */ + var h = new Harness(); + h.Settings.ForcePlanFailureEnabled = true; + h.Settings.CooldownMinutes = 5; + h.Now = Collection0402.AddSeconds(30); + h.Adapter.ForcePlanFailures.Add(ForcePlanRow(Collection0402, delta: 1, total: 1)); + var engine = h.Build(); + + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Single(h.Deliverer.Outcomes); + + /* The next collection: the counter rose again (1 → 3). Same plan key, newer stamp. */ + h.Adapter.ForcePlanFailures.Clear(); + h.Adapter.ForcePlanFailures.Add(ForcePlanRow(Collection0418, delta: 2, total: 3)); + h.Now = Collection0418.AddSeconds(30); + await engine.EvaluateServerAsync(Harness.Snapshot()); + + Assert.Equal(2, h.Deliverer.Outcomes.Count); + var second = h.Deliverer.Outcomes[1]; + Assert.Equal(2, second.NumericCurrentValue); + Assert.Contains("failed to force 2x", second.ShortMessage, StringComparison.Ordinal); + Assert.Contains("Total Failures: 3", second.DetailText, StringComparison.Ordinal); + + /* And that observation, re-read past another cooldown, is one card too. */ + h.Now = h.Now.AddMinutes(6); + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Equal(2, h.Deliverer.Outcomes.Count); + Assert.Empty(h.Resolutions); + } + + [Fact] + public async Task ForcePlanFailure_NewObservationInsideTheCooldown_FiresOnceTheCooldownElapses() + { + /* The memory is 'last ALERTED observation', not 'last SEEN': a collection that lands while the + cooldown from the previous card is still running has not been reported, so when the cooldown + elapses and the row is still that observation, it fires. Folding the two memories into one + 'last seen' stamp would record it as seen on the quiet pass and then never fire it — the guard + would have been silencing a real second failure, which is worse than the repeat it replaces. */ + var h = new Harness(); + h.Settings.ForcePlanFailureEnabled = true; + h.Settings.CooldownMinutes = 5; + h.Now = Collection0402.AddSeconds(30); + h.Adapter.ForcePlanFailures.Add(ForcePlanRow(Collection0402)); + var engine = h.Build(); + + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Single(h.Deliverer.Outcomes); + + /* A fast collection: the next observation lands two minutes later, inside the five-minute cooldown. */ + var quickCollection = Collection0402.AddMinutes(2); + h.Adapter.ForcePlanFailures.Clear(); + h.Adapter.ForcePlanFailures.Add(ForcePlanRow(quickCollection, delta: 1, total: 2)); + h.Now = quickCollection.AddSeconds(30); + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Single(h.Deliverer.Outcomes); /* cooldown holds it — rate limiting is still the cooldown's job */ + + /* Cooldown elapsed, same not-yet-reported observation: it fires now. */ + h.Now = Collection0402.AddMinutes(5).AddSeconds(30); + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Equal(2, h.Deliverer.Outcomes.Count); + + /* And only once. */ + h.Now = h.Now.AddMinutes(6); + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Equal(2, h.Deliverer.Outcomes.Count); + } + + [Fact] + public async Task ForcePlanFailure_MutedFire_StampsTheObservation_SoUnmutingDoesNotReplayIt() + { + /* A muted fire is still a fire: it is delivered flagged Muted, it stamps the cooldown, and since + #3579 it stamps the observation. A mute rule lifted mid-interval must not turn the same 04:18 + rise into a fresh card — the operator muted the plan, not the engine's memory of it. */ + var h = new Harness(); + h.Settings.ForcePlanFailureEnabled = true; + h.Settings.CooldownMinutes = 5; + h.Muted = true; + h.Now = Collection0418.AddSeconds(53); + h.Adapter.ForcePlanFailures.Add(ForcePlanRow(Collection0418)); + var engine = h.Build(); + + await engine.EvaluateServerAsync(Harness.Snapshot()); + var muted = Assert.Single(h.Deliverer.Outcomes); + Assert.True(muted.Muted); + + h.Muted = false; + h.Now = h.Now.AddMinutes(6); + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Single(h.Deliverer.Outcomes); + } + + [Fact] + public async Task ForcePlanFailure_RecoveryForgetsTheObservation_SoANewEpisodeFires() + { + /* The falling edge clears the memory with the plan (the #2166 lesson, at plan grain): a plan that + recovers and later fails again is a new episode and its first rise fires, whatever the stamp. */ + var h = new Harness(); + h.Settings.ForcePlanFailureEnabled = true; + h.Settings.CooldownMinutes = 5; + h.Now = Collection0418.AddSeconds(53); + h.Adapter.ForcePlanFailures.Add(ForcePlanRow(Collection0418)); + var engine = h.Build(); + + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Single(h.Deliverer.Outcomes); + + h.Adapter.ForcePlanFailures.Clear(); + h.Now = Collection0450.AddSeconds(40); + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Single(h.Resolutions, r => r.MetricName == ForcePlanTokens.MetricName); + + /* Hours later, forced again and failing again. */ + var laterCollection = Collection0450.AddHours(3); + h.Adapter.ForcePlanFailures.Add(ForcePlanRow(laterCollection, delta: 1, total: 1)); + h.Now = laterCollection.AddSeconds(30); + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Equal(2, h.Deliverer.Outcomes.Count); + } + + [Fact] + public async Task ForcePlanFailure_StamplessRow_KeepsThePre3579CooldownRepeat() + { + /* The stated fallback for an adapter that supplies no observation stamp (the shipped two always do): + a null never matches a remembered stamp, so every read counts as new and the cooldown alone + rate-limits it — the pre-#3579 behaviour, degraded towards repetition rather than silence, the + same direction IAlertStateStore's no-op fallbacks degrade. Pinned so the fallback is a decision + and not an accident of null comparison. */ + var h = new Harness(); + h.Settings.ForcePlanFailureEnabled = true; + h.Settings.CooldownMinutes = 5; + h.Now = Collection0418.AddSeconds(53); + h.Adapter.ForcePlanFailures.Add(ForcePlanRow(observedAt: null)); + var engine = h.Build(); + + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Single(h.Deliverer.Outcomes); + + h.Now = h.Now.AddMinutes(6); + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Equal(2, h.Deliverer.Outcomes.Count); + } + [Fact] public async Task ForcePlanFailure_ExcludedDatabase_IsNeverAlerted() { diff --git a/Darling/Darling.Tests/ForcePlanFailuresAccessPathTests.cs b/Darling/Darling.Tests/ForcePlanFailuresAccessPathTests.cs index 01d407624..9062e2188 100644 --- a/Darling/Darling.Tests/ForcePlanFailuresAccessPathTests.cs +++ b/Darling/Darling.Tests/ForcePlanFailuresAccessPathTests.cs @@ -114,6 +114,31 @@ public void TheStatement_IsBuiltFromTheColumnList_AndTakesNeitherMeasuredTrap() Assert.DoesNotContain(" WHERE ", statement, StringComparison.Ordinal); /* not partial \u2014 see the statement's remarks */ } + /// + /// #3579: the observation stamp is the LAST column of the shipped read and is n.collection_time — the + /// newer sighting's collector clock — not a new qs. reference. Last, because the reader binds ordinals + /// 0–6 to the seven pre-#3579 columns and an inserted column would silently shift every one of them onto + /// its neighbour's type (a string read as a bigint fails; a bigint read as a bigint from the wrong column + /// does not). Not a qs. reference, because the covering pin above re-derives the index list from + /// exactly those references and a new one would demand a new INCLUDE column on the largest table in the + /// store; collection_time is already in the key. + /// + [Fact] + public void TheObservationStamp_IsTheLastColumn_AndIsTheNewerSightingsCollectionTime() + { + var sql = DarlingAlertReadAdapter.ForcePlanFailuresSql; + var selectList = sql[sql.LastIndexOf("SELECT", StringComparison.Ordinal)..sql.IndexOf("FROM ranked AS n", StringComparison.Ordinal)]; + var columns = selectList.Replace("SELECT", "", StringComparison.Ordinal) + .Split(',', StringSplitOptions.TrimEntries | StringSplitOptions.RemoveEmptyEntries); + + Assert.Equal(8, columns.Length); + Assert.Equal("n.failures AS total_failures", columns[6]); + Assert.Equal("n.collection_time AS observed_at", columns[7]); + + /* The set of scan columns did not grow — the same nine the index carried before #3579. */ + Assert.Equal(PgTableTuning.ForcePlanFailuresIndexColumns.Count, ColumnsTheReadReferences().Count); + } + /// /// The evidence no string pin can give: that the planner TAKES the index for the shipped statement. The /// production failure was a plan choice, not a missing object \u2014 the right composite was in the catalog and @@ -168,7 +193,12 @@ public async Task TheShippedRead_PlansAsAnIndexOnlyScanOnTheCoveringIndex_Agains servers in turn, each server's batch contiguous \u2014 the write pattern that makes server_id's correlation near zero, which is the condition the production plan was chosen under. All Kind-Unspecified: naive-UTC storage, see PgCollectorRowWriter. */ - var utcNow = DateTime.SpecifyKind(DateTime.UtcNow, DateTimeKind.Unspecified); + /* Floored to whole microseconds: PostgreSQL timestamp is microsecond-resolution and .NET ticks are + 100 ns, so a raw UtcNow does not survive the round trip and the #3579 stamp assertion below + (tick-equality against what was seeded) would fail on any clock that is not itself + microsecond-aligned — Windows' is not; the first CI run proved it by three ticks. */ + var rawNow = DateTime.UtcNow; + var utcNow = DateTime.SpecifyKind(new DateTime(rawNow.Ticks - (rawNow.Ticks % 10)), DateTimeKind.Unspecified); for (var pass = 7; pass >= 0; pass--) { var collectionTime = utcNow.AddMinutes(-2 - pass * 15); @@ -210,6 +240,11 @@ rather than inside the Index Cond. */ Assert.Equal(10L, failure.PlanId); Assert.Equal(1L, failure.FailureDelta); Assert.Equal(7L, failure.TotalFailures); + /* #3579: the observation's identity is the NEWEST sighting's collection_time (pass 0, two minutes + ago), read back through the real Npgsql path and stamped Utc. Ticks-equal to what was seeded: + the store holds naive UTC and the adapter only names the Kind, never shifts the value. */ + Assert.Equal((DateTime?)utcNow.AddMinutes(-2), failure.ObservedAtUtc); + Assert.Equal(DateTimeKind.Utc, failure.ObservedAtUtc!.Value.Kind); bodySucceeded = true; } diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingAlertReadAdapter.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingAlertReadAdapter.cs index 8abcb1503..f39429fc4 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingAlertReadAdapter.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingAlertReadAdapter.cs @@ -1136,6 +1136,15 @@ one carried over from a restart (#2166). A database cleared here is one that is /// The > comparison is what makes this a delta read: equal counters are silence, and a /// LOWER counter (unforce/re-force reset) is silence too rather than a negative delta. /// + /// The newer sighting's collection_time travels with the row (#3579) as + /// observed_at, the last column. The delta is a fact about two COLLECTIONS and stays byte-identical + /// on every alert pass until the next collection lands — "every row here is a live failure" is true at the + /// collection instant and stale for the rest of the interval — so the engine needs the observation's + /// identity to fire once per collection instead of once per cooldown. Appended rather than inserted so + /// the seven ordinals the reader already binds do not move. It is n.collection_time, already in + /// per_collection's GROUP BY and already carried by the covering index: no new qs. column, + /// so the access path below is untouched (the access-path pins re-derive the list from this text). + /// /// The access path is a covering index, and the column list here is what it covers (#3573). /// PgTableTuning.ForcePlanFailuresIndexName is (server_id, collection_time DESC) INCLUDE /// every other column this statement touches, so it runs as an Index Only Scan over one server's two @@ -1179,7 +1188,8 @@ FROM per_collection AS pc n.forcing_type, n.reason, n.failures - p.failures AS failure_delta, - n.failures AS total_failures + n.failures AS total_failures, + n.collection_time AS observed_at FROM ranked AS n JOIN ranked AS p ON p.database_name = n.database_name @@ -1213,7 +1223,12 @@ public async Task> GetForcePlanFailuresAsync( ForcingType = reader.IsDBNull(3) ? "" : reader.GetString(3), FailureReason = reader.IsDBNull(4) ? "" : reader.GetString(4), FailureDelta = reader.IsDBNull(5) ? 0 : reader.GetInt64(5), - TotalFailures = reader.IsDBNull(6) ? 0 : reader.GetInt64(6) + TotalFailures = reader.IsDBNull(6) ? 0 : reader.GetInt64(6), + /* #3579: the stored value is naive UTC (see the window-bound remarks above), read back with + Kind Unspecified; stamped Utc because that is what it IS and what the property's name says. + The engine only ever compares one plan's stamps with each other, so the Kind is honesty + rather than arithmetic. */ + ObservedAtUtc = reader.IsDBNull(7) ? null : DateTime.SpecifyKind(reader.GetDateTime(7), DateTimeKind.Utc) }); } diff --git a/Lite.Tests/BlockingDeadlockContextBuilderTests.cs b/Lite.Tests/BlockingDeadlockContextBuilderTests.cs index f8aa1a3f2..b2100cc38 100644 --- a/Lite.Tests/BlockingDeadlockContextBuilderTests.cs +++ b/Lite.Tests/BlockingDeadlockContextBuilderTests.cs @@ -564,6 +564,38 @@ public void PoisonWaitAccumulationSql_IsTheDarlingTextInDuckDbDialect() Assert.Contains("SUM(delta_wait_time_ms)::bigint", darlingSql, StringComparison.Ordinal); } + /// + /// #2157/#3579: Lite's forced-plan-failures read is Darling's text with ONE substitution — the dedup view + /// v_query_store_stats for the raw table — and nothing else. Both files say "shape-for-shape"; this + /// is the sentence as an assertion, held as whole-text equality rather than a clause list because the two + /// texts ARE equal today and any drift, in either direction, is the disagreement about "what counts as a + /// new failure" the files promise cannot happen. Read from source on the Darling side through + /// for the reason the poison pin above gives. + /// + /// Also pins the #3579 column: the newer sighting's collection_time travels as + /// observed_at, LAST, so the seven ordinals both readers already bind do not move. + /// + [Fact] + public void ForcePlanFailuresSql_IsTheDarlingText_ReadingTheDedupView() + { + var lite = LocalDataService.ForcePlanFailuresSql; + var darling = Lite.Tests.ParitySource.ReadFile("Darling/PerformanceMonitor.Darling.Service/DarlingAlertReadAdapter.cs"); + var darlingSql = darling[(darling.IndexOf("public const string ForcePlanFailuresSql = @\"", StringComparison.Ordinal) + "public const string ForcePlanFailuresSql = @\"".Length)..]; + darlingSql = darlingSql[..darlingSql.IndexOf("\";", StringComparison.Ordinal)]; + + Assert.Contains("FROM v_query_store_stats AS qs", lite, StringComparison.Ordinal); + Assert.Contains("FROM query_store_stats AS qs", darlingSql, StringComparison.Ordinal); + Assert.Equal( + darlingSql.ReplaceLineEndings("\n"), + lite.Replace("FROM v_query_store_stats AS qs", "FROM query_store_stats AS qs", StringComparison.Ordinal).ReplaceLineEndings("\n")); + + /* #3579: the observation stamp, last. */ + Assert.EndsWith( + "n.failures AS total_failures,\n n.collection_time AS observed_at\nFROM ranked AS n", + lite[..(lite.IndexOf("FROM ranked AS n", StringComparison.Ordinal) + "FROM ranked AS n".Length)].ReplaceLineEndings("\n"), + StringComparison.Ordinal); + } + [Fact] public void LiteAlertReadAdapter_ExposesAllSevenCollectedFeeds() { diff --git a/Lite.Tests/ForcePlanFailuresReadTests.cs b/Lite.Tests/ForcePlanFailuresReadTests.cs new file mode 100644 index 000000000..4b036c7e7 --- /dev/null +++ b/Lite.Tests/ForcePlanFailuresReadTests.cs @@ -0,0 +1,148 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor Lite. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Threading.Tasks; +using DuckDB.NET.Data; +using PerformanceMonitorLite.Database; +using PerformanceMonitorLite.Services; +using PerformanceMonitorLite.Tests; +using Xunit; + +namespace Lite.Tests; + +/// +/// Real-DuckDB round trip for Lite's forced-plan-failures read (#2157), replaying the #3579 production +/// series: a forced plan whose force_failure_count went 0 → 0 → 1 → 0 across four collections. The +/// read must return the plan exactly while the newest sighting is the rise, must carry that sighting's +/// collection_time as the observation stamp (the fact the engine keys its once-per-observation guard +/// on), and must return nothing once the next collection lands with the counter reset — the recovery. +/// +/// Two things a text pin cannot prove and this does: that DuckDB's TIMESTAMP comes back +/// through GetDateTime(7) at the ordinal the reader binds, and that the value round-trips to the +/// tick — the engine compares one plan's stamps for equality, so a stamp that came back shifted or +/// truncated would either never match (cooldown-repeat returns) or match a different collection. +/// +public sealed class ForcePlanFailuresReadTests : IClassFixture, IDisposable +{ + private const int ServerId = 3579; + private const string Db = "ForcedDb"; + + private readonly DuckDbInitializer _duckDb; + private DuckDBConnection? _seedConn; + private long _nextId = 1; + + public ForcePlanFailuresReadTests(SharedDuckDbFixture fixture) + { + fixture.ResetData(); + _duckDb = fixture.DuckDb; + } + + public void Dispose() => _seedConn?.Dispose(); + + /* Whole seconds by construction: DuckDB TIMESTAMP is microsecond-resolution, so raw DateTime ticks + would not survive the round trip and the tick-equality below would be testing the wrong thing. + Kind Unspecified, the store's naive-UTC convention. Inside the read's two-hour window. */ + private static readonly DateTime Base = MinuteFloor(DateTime.UtcNow.AddMinutes(-70)); + + private static DateTime MinuteFloor(DateTime t) => + DateTime.SpecifyKind(new DateTime(t.Ticks - (t.Ticks % TimeSpan.TicksPerMinute)), DateTimeKind.Unspecified); + + private async Task SeedConnectionAsync() + { + if (_seedConn is null) + { + _seedConn = _duckDb.CreateConnection(); + await _seedConn.OpenAsync(); + } + return _seedConn; + } + + /// One collection of one forced plan, two interval rows (the forcing columns are plan-level and + /// repeat across a plan's rows — the read's MAX collapse has to have something to collapse). + private async Task SeedCollectionAsync(DateTime collectionTime, long failures, bool forced, string forcingType, string? reason) + { + using var readLock = _duckDb.AcquireReadLock(); + var connection = await SeedConnectionAsync(); + for (var interval = 0; interval < 2; interval++) + { + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +INSERT INTO query_store_stats + (collection_id, collection_time, server_id, server_name, database_name, + query_id, plan_id, execution_type_desc, first_execution_time, last_execution_time, + query_text, query_hash, execution_count, avg_cpu_time_us, avg_duration_us, + avg_logical_io_reads, avg_logical_io_writes, avg_physical_io_reads, + query_plan_hash, is_forced_plan, force_failure_count, plan_forcing_type, last_force_failure_reason, + runtime_stats_interval_id, interval_start_time_utc) +VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11, $12, $13, $14, $15, $16, $17, $18, $19, $20, $21, $22, $23, $24, $25)"; + cmd.Parameters.Add(new DuckDBParameter { Value = _nextId++ }); + cmd.Parameters.Add(new DuckDBParameter { Value = collectionTime }); + cmd.Parameters.Add(new DuckDBParameter { Value = ServerId }); + cmd.Parameters.Add(new DuckDBParameter { Value = "ForcedSrv" }); + cmd.Parameters.Add(new DuckDBParameter { Value = Db }); + cmd.Parameters.Add(new DuckDBParameter { Value = 11L }); + cmd.Parameters.Add(new DuckDBParameter { Value = 22L }); + cmd.Parameters.Add(new DuckDBParameter { Value = "Regular" }); + cmd.Parameters.Add(new DuckDBParameter { Value = collectionTime.AddMinutes(-10 - interval) }); + cmd.Parameters.Add(new DuckDBParameter { Value = collectionTime }); + cmd.Parameters.Add(new DuckDBParameter { Value = "SELECT 11" }); + cmd.Parameters.Add(new DuckDBParameter { Value = "0xHASH11" }); + cmd.Parameters.Add(new DuckDBParameter { Value = 100L + interval }); + cmd.Parameters.Add(new DuckDBParameter { Value = 500L }); + cmd.Parameters.Add(new DuckDBParameter { Value = 900L }); + cmd.Parameters.Add(new DuckDBParameter { Value = 40L }); + cmd.Parameters.Add(new DuckDBParameter { Value = 0L }); + cmd.Parameters.Add(new DuckDBParameter { Value = 0L }); + cmd.Parameters.Add(new DuckDBParameter { Value = "0xPLAN22" }); + cmd.Parameters.Add(new DuckDBParameter { Value = forced }); + cmd.Parameters.Add(new DuckDBParameter { Value = failures }); + cmd.Parameters.Add(new DuckDBParameter { Value = forcingType }); + cmd.Parameters.Add(new DuckDBParameter { Value = (object?)reason ?? DBNull.Value }); + cmd.Parameters.Add(new DuckDBParameter { Value = 1000L + interval }); + cmd.Parameters.Add(new DuckDBParameter { Value = collectionTime.AddMinutes(-10 - interval) }); + await cmd.ExecuteNonQueryAsync(); + } + } + + [Fact] + public async Task TheRise_IsReturnedOnce_StampedWithTheNewestCollection_AndVanishesWhenTheCounterResets() + { + var service = new LocalDataService(_duckDb); + + /* 03:47 and 04:02 of the series: not forced, counter flat at zero. Nothing to report. */ + await SeedCollectionAsync(Base, failures: 0, forced: false, "NONE", null); + await SeedCollectionAsync(Base.AddMinutes(15), failures: 0, forced: false, "NONE", null); + Assert.Empty(await service.GetForcePlanFailuresAsync(ServerId)); + + /* 04:18: Automatic Plan Correction forced the plan and the force failed once. */ + var riseCollection = Base.AddMinutes(31); + await SeedCollectionAsync(riseCollection, failures: 1, forced: true, "AUTO", "GENERAL_FAILURE"); + + var row = Assert.Single(await service.GetForcePlanFailuresAsync(ServerId)); + Assert.Equal(Db, row.DatabaseName); + Assert.Equal(11L, row.QueryId); + Assert.Equal(22L, row.PlanId); + Assert.Equal("AUTO", row.ForcingType); + Assert.Equal("GENERAL_FAILURE", row.FailureReason); + Assert.Equal(1L, row.FailureDelta); + Assert.Equal(1L, row.TotalFailures); + /* #3579: the observation's identity is the newer sighting's collection_time, to the tick, Kind Utc. */ + Assert.Equal((DateTime?)riseCollection, row.ObservedAtUtc); + Assert.Equal(DateTimeKind.Utc, row.ObservedAtUtc!.Value.Kind); + + /* Re-reading between collections is the same observation: same row, same stamp. This is the read the + engine used to fire on six times; the stamp is what lets it recognise the repeat. */ + var reread = Assert.Single(await service.GetForcePlanFailuresAsync(ServerId)); + Assert.Equal(row.ObservedAtUtc, reread.ObservedAtUtc); + + /* 04:50: APC released the forcing; the counter reset to zero with it. A LOWER counter is silence. */ + await SeedCollectionAsync(Base.AddMinutes(63), failures: 0, forced: false, "NONE", null); + Assert.Empty(await service.GetForcePlanFailuresAsync(ServerId)); + } +} diff --git a/Lite/Services/LocalDataService.ForcePlanFailures.cs b/Lite/Services/LocalDataService.ForcePlanFailures.cs index b9a2da16f..52d20e4b3 100644 --- a/Lite/Services/LocalDataService.ForcePlanFailures.cs +++ b/Lite/Services/LocalDataService.ForcePlanFailures.cs @@ -44,6 +44,13 @@ public sealed partial class LocalDataService /// takes from the host — and the window silently widens west of UTC and narrows to nothing east of it. /// Darling's twin carries the same bound for the same reason; keeping the two shape-for-shape is what /// stops the apps disagreeing about what counts as a new failure. + /// + /// The newer sighting's collection_time rides along as observed_at, the last column + /// (#3579): the delta is a fact about two collections that reads identically on every alert pass until + /// the next collection lands, and the engine needs the observation's identity to fire once per + /// collection rather than once per cooldown. Appended so the seven ordinals already bound do not move; + /// Darling's twin carries the same column at the same position, and the Lite.Tests parity pin holds the + /// two texts equal but for the view name. /// public const string ForcePlanFailuresSql = @" WITH per_collection AS ( @@ -74,7 +81,8 @@ FROM per_collection AS pc n.forcing_type, n.reason, n.failures - p.failures AS failure_delta, - n.failures AS total_failures + n.failures AS total_failures, + n.collection_time AS observed_at FROM ranked AS n JOIN ranked AS p ON p.database_name = n.database_name @@ -107,7 +115,10 @@ public async Task> GetForcePlanFailuresAsync(int serv ForcingType = reader.IsDBNull(3) ? "" : reader.GetString(3), FailureReason = reader.IsDBNull(4) ? "" : reader.GetString(4), FailureDelta = reader.IsDBNull(5) ? 0 : reader.GetInt64(5), - TotalFailures = reader.IsDBNull(6) ? 0 : reader.GetInt64(6) + TotalFailures = reader.IsDBNull(6) ? 0 : reader.GetInt64(6), + /* #3579: a naive-UTC TIMESTAMP read back Kind-Unspecified, stamped Utc because that is what it + is. The engine compares one plan's stamps only with each other. */ + ObservedAtUtc = reader.IsDBNull(7) ? null : DateTime.SpecifyKind(reader.GetDateTime(7), DateTimeKind.Utc) }); } diff --git a/PerformanceMonitor.Alerting/AlertEngine.cs b/PerformanceMonitor.Alerting/AlertEngine.cs index aa2680971..2a00789b9 100644 --- a/PerformanceMonitor.Alerting/AlertEngine.cs +++ b/PerformanceMonitor.Alerting/AlertEngine.cs @@ -222,10 +222,45 @@ SUSPECT. Structural rather than concatenated because clearing a database's clock two plans failing on the same database are independent conditions that resolve independently. Keyed by the internal plan key but VALUED with the plan's identity, because the resolution has to name the plan in an operator-readable way: a bare key set left the recovery message reading - 'forceplan:Sales:11:22 no longer failing to force' in every email and webhook (review catch). */ - private readonly ConcurrentDictionary> _activeForcePlanAlerts = new(); + 'forceplan:Sales:11:22 no longer failing to force' in every email and webhook (review catch). + + #3579: the value also remembers the OBSERVATION each plan was last alerted on (the newer sighting's + collection_time), beside the identity rather than in a third dictionary, so the condition keeps one + per-plan state and one cooldown clock. The two are different memories: LastSeen is overwritten on + every pass the plan appears (it exists for the recovery message), LastAlertedObservedAtUtc only when + the plan actually fires (it exists so the same observation cannot fire twice). Folding them into one + stamp was the tempting shortcut and would have been wrong in the direction of silence: a new + collection landing INSIDE the cooldown would have been recorded as seen and then never fired. + + In-memory only, like the rest of this family: a restart empties it, so the first pass after one may + re-fire once for an observation still in the window — which is exactly what the cooldown clock, + also emptied, already did before #3579, so the guard never makes a restart noisier than it was. */ + private readonly ConcurrentDictionary> _activeForcePlanAlerts = new(); private readonly ConcurrentDictionary _lastForcePlanAlert = new(); + /// + /// One forced plan the engine currently holds active (#2157), with the #3579 observation memory. Mutable + /// by design: the pass updates it in place under the per-server evaluation gate, the same way the + /// database-state family mutates its per-server HashSet. + /// + private sealed class ForcePlanActivePlan + { + public ForcePlanActivePlan(ForcePlanFailureInfo lastSeen) + { + LastSeen = lastSeen; + } + + /// The row most recently returned for this plan — the identity the recovery message is + /// named from. Refreshed every pass the plan appears. + public ForcePlanFailureInfo LastSeen { get; set; } + + /// The of the row this plan last FIRED on + /// (#3579), or null when it has not fired since it entered the active set or the adapter supplied no + /// stamp. Set on fire only — including a muted fire, which stamps it exactly as it stamps the + /// cooldown — never on a pass that merely saw the plan. + public DateTime? LastAlertedObservedAtUtc { get; set; } + } + /// Live threshold surface — read every sweep, never cached. /// The collected alert feeds (slice B seam). /// Restart-surviving watermark persistence (#1145). @@ -2238,8 +2273,29 @@ await NotifyResolutionAsync(new AlertResolution( /// the optimizer picks. Nothing else in the product witnesses that — the operator's mitigation is /// silently not in effect, and the only trace is a counter climbing inside Query Store. /// - /// Standing condition with per-plan resolution, mirroring the database-state family: while a plan - /// keeps failing it re-fires on the cooldown, and when it stops appearing it announces a recovery. + /// Standing condition with per-plan resolution, firing once per OBSERVATION (#3579). The + /// original contract read "while a plan keeps failing it re-fires on the cooldown, mirroring the + /// database-state family", and that is the right contract for a condition whose input is re-measured + /// every pass. This condition's input is not: it is a stored delta between the two newest COLLECTIONS, + /// which changes only when the collector lands a row (the query_store cadence, 5 minutes by default and + /// ~15 on the store that found this), while the engine re-asks every ~30 s and the cooldown is 5 minutes. + /// Cooldown shorter than cadence means the cooldown expires several times against the SAME two rows, and + /// at pass granularity "still failing" and "no new data yet" are indistinguishable — the pre-#3579 loop + /// took the second for the first. Measured on one production store, one plan: the raw series was + /// 0, 0, 1 (forced AUTO), 0, 0 across five collections at 03:47 / 04:02 / 04:18 / 04:50 / 05:01, and the + /// engine fired SIX times between the 04:18 and 04:50 collections — 04:18:53, 04:24:31, 04:30:04, + /// 04:35:33, 04:40:49, 04:46:03, one per cooldown — every card honestly reading New 1 / Total 1, because + /// every pass re-read the same 04:02→04:18 rise. The arithmetic was right; the repetition was the defect. + /// The channel saw two of the six only because webhook-side throttling ate the rest. + /// + /// The contract now: a plan fires once per new observation that shows a rise — the engine + /// remembers, per plan, the it last fired on and + /// declines to fire the same stamp again regardless of cooldown; a newer stamp with a rise fires (a plan + /// failing across successive collections still re-fires, each collection being a new observation), the + /// cooldown still rate-limits those, and a plan that stops appearing announces a recovery exactly as + /// before. The poison-wait family took the same guard for the same shape under #2704; this is that + /// guard at plan grain. A row without a stamp falls back to the pre-#3579 cooldown-repeat rather than + /// to silence. /// private async Task CheckForcePlanFailuresAsync( string key, string serverName, DateTime now, TimeSpan alertCooldown, bool suppressed, CancellationToken ct) @@ -2289,13 +2345,34 @@ private async Task CheckForcePlanFailuresAsync( current[ForcePlanTokens.PlanKey(failure.DatabaseName, failure.QueryId, failure.PlanId)] = failure; } - var active = _activeForcePlanAlerts.GetOrAdd(key, _ => new Dictionary(StringComparer.Ordinal)); + var active = _activeForcePlanAlerts.GetOrAdd(key, _ => new Dictionary(StringComparer.Ordinal)); foreach (var (planKey, failure) in current) { - active[planKey] = failure; + if (active.TryGetValue(planKey, out var plan)) + { + plan.LastSeen = failure; + } + else + { + plan = new ForcePlanActivePlan(failure); + active[planKey] = plan; + } + + /* #3579: the observation guard. The row's stamp is the collection that produced the rise; if it + is not NEWER than the one this plan last fired on, this pass is re-reading an observation the + operator already has a card for, and no amount of elapsed cooldown makes it a second event. + "Not newer" rather than "equal" so the contract reads as stated — fire once per NEW + observation — and an older stamp (nothing produces one today; a deleted newest row or a clock + step would) is not mistaken for news. A null on either side never matches: a stampless row + keeps the cooldown-repeat, and a plan that has not fired yet is always eligible. */ + bool sameObservation = + failure.ObservedAtUtc is { } observedAt + && plan.LastAlertedObservedAtUtc is { } lastAlertedAt + && observedAt <= lastAlertedAt; + var cooldownKey = key + "|" + planKey; - if (!suppressed && CooldownElapsed(_lastForcePlanAlert, cooldownKey, now, alertCooldown)) + if (!suppressed && !sameObservation && CooldownElapsed(_lastForcePlanAlert, cooldownKey, now, alertCooldown)) { var reasonText = ForcePlanTokens.HumanizeReason(failure.FailureReason); var forcingText = string.IsNullOrWhiteSpace(failure.ForcingType) ? "unknown" : failure.ForcingType.Trim(); @@ -2307,6 +2384,7 @@ private async Task CheckForcePlanFailuresAsync( }; bool isMuted = _isAlertMuted(muteCtx); _lastForcePlanAlert[cooldownKey] = now; /* stamped even when muted, like the others */ + plan.LastAlertedObservedAtUtc = failure.ObservedAtUtc; /* #3579: and so is the observation */ /* #2109 discipline: the same facts the prose carries, as discrete fields, so a consumer never has to parse the title to learn which plan this is about. @@ -2350,8 +2428,12 @@ await FireAsync(new AlertOutcome( if (active.Count > 0) { var recovered = active.Where(p => !current.ContainsKey(p.Key)).ToList(); - foreach (var (planKey, lastSeen) in recovered) + foreach (var (planKey, recoveredPlan) in recovered) { + var lastSeen = recoveredPlan.LastSeen; + /* Removing the plan drops its #3579 observation memory with it, deliberately: a plan that + recovers and later fails again is a new episode and starts with no memory, exactly as the + cooldown clock beside it does. */ active.Remove(planKey); _lastForcePlanAlert.TryRemove(key + "|" + planKey, out _); if (!suppressed) diff --git a/PerformanceMonitor.Alerting/ForcePlanFailureInfo.cs b/PerformanceMonitor.Alerting/ForcePlanFailureInfo.cs index 7690a0d77..108df5eb0 100644 --- a/PerformanceMonitor.Alerting/ForcePlanFailureInfo.cs +++ b/PerformanceMonitor.Alerting/ForcePlanFailureInfo.cs @@ -6,6 +6,8 @@ * Licensed under the MIT License. See LICENSE file in the project root for full license information. */ +using System; + namespace PerformanceMonitor.Alerting; /// @@ -23,6 +25,16 @@ namespace PerformanceMonitor.Alerting; /// /// A counter that DROPS (an unforce/re-force cycle resets it) is not a failure and must not alert; /// the adapter treats that as a silent re-arm. +/// +/// A row is an observation, and it carries its own identity (#3579). "Rose between the two +/// most recent collections" becomes true the instant the newer collection lands and stays true, unchanged, +/// until the next one does — the adapter re-reads the SAME two rows on every alert pass in between. The +/// engine evaluates every ~30 s against a fact that changes once per collection (5–15 min) with a 5-minute +/// cooldown between them, so without it could not tell "still failing" from +/// "no new data yet" and treated the second as the first: one production store, one plan, a counter that +/// went 0 → 1 → 0 across three collections, and SIX alert rows in 28 minutes, each honestly reading +/// "New Failures: 1". The stamp is what lets the engine fire once per observation rather than once per +/// cooldown. /// public sealed class ForcePlanFailureInfo { @@ -54,4 +66,20 @@ public sealed class ForcePlanFailureInfo /// The cumulative count as of the newer sample, for context on whether this is chronic. public long TotalFailures { get; set; } + + /// + /// The collection_time of the NEWER of the two samples — the collector's clock, not the alert + /// sweep's — which is the identity of the observation this row reports (#3579). Two reads that return + /// the same stamp for the same plan are the same observation surfacing twice, not two failures; a newer + /// stamp is a new collection and a new rise. The engine keeps the stamp it last alerted on per plan and + /// declines to re-fire the same one regardless of cooldown (the poison-wait family's #2704 + /// unrefreshed-source-row guard, at plan grain). + /// + /// Nullable so an adapter that does not supply it degrades to the pre-#3579 cooldown-repeat rather + /// than to silence: a null never equals a remembered stamp, so every read counts as new and the + /// cooldown alone rate-limits it — the same stated fallback gives a host + /// that cannot persist. Both shipped adapters always supply it; collection_time is NOT NULL in both + /// stores. + /// + public DateTime? ObservedAtUtc { get; set; } } From b62a00eddb7d9a7dc83989eac76bc8841610e673 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 14:56:54 -0400 Subject: [PATCH 52/69] =?UTF-8?q?The=20File=20Growth=20threshold=20means?= =?UTF-8?q?=20one=20thing=20=E2=80=94=20megabytes=20per=20hour=20=E2=80=94?= =?UTF-8?q?=20at=20every=20surface=20that=20shows=20it,=20so=20a=2010=20GB?= =?UTF-8?q?=20bar=20stops=20meaning=2010=20GB=20per=20five=20minutes=20on?= =?UTF-8?q?=20one=20store=20and=20per=20day=20on=20another=20(#3539=20A8c)?= =?UTF-8?q?=20(#3631)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * The File Growth threshold means one thing — megabytes per hour — at every surface that shows it (#3539 A8c) FileGrowthRiseMb was compared against the raw growth inside the lookback window while the lookback was a second knob clamped 5..1440 minutes, so the same 10240 meant 10 GB per five minutes on one store and 10 GB per day on another. The knob is now a rate: GetBreachedFiles holds the in-window growth to FileGrowthRiseBarMb (rate × lookback/60), which is byte-identical on the shipped 60-minute lookback and the same sustained rate on every other. Scaled to the CONFIGURED window rather than read off the measured span, because the measured span on a freshly-collecting server is one collection interval and one autogrowth in it would extrapolate to twelve an hour. Both Settings rows, both preview lines, the alert's threshold line (rate + window + the bar in MB), the card's Growth field, IAlertEngineSettings, both MCP descriptions and the settings carriers now say MB/hr averaged over the lookback, in one shared phrase (AlertContextBuilders.FileGrowthRiseUnit) that FileGrowthRiseUnitCensusTests holds across nine surfaces. Key, column and integer are unchanged. * DarlingConfig.FileGrowthLookbackMinutes carries the same averaging-window note as its twins (#3539 A8c, review nit) --- Darling/Darling.Tests/AlertEngineTests.cs | 73 +++++++- .../DarlingMcpAlertToolsTests.cs | 31 ++++ .../FileGrowthRiseUnitCensusTests.cs | 174 ++++++++++++++++++ .../DarlingAlertSettings.cs | 3 +- .../DarlingConfig.cs | 3 + .../Mcp/DarlingMcpAlertTools.cs | 10 +- .../SettingsWindow.xaml | 4 +- .../SettingsWindow.xaml.cs | 5 +- .../ViewerDataService.AlertSettings.cs | 3 +- Lite.Tests/FileGrowthAlertTests.cs | 118 +++++++++++- Lite/App.xaml.cs | 4 +- Lite/Mcp/McpAlertTools.cs | 5 +- Lite/Services/AppAlertEngineSettings.cs | 2 +- Lite/Windows/SettingsWindow.xaml | 4 +- Lite/Windows/SettingsWindow.xaml.cs | 5 +- .../AlertContextBuilders.cs | 49 ++++- PerformanceMonitor.Alerting/AlertEngine.cs | 16 +- .../DatabaseFileGrowthInfo.cs | 5 +- .../IAlertEngineSettings.cs | 21 ++- 19 files changed, 496 insertions(+), 39 deletions(-) create mode 100644 Darling/Darling.Tests/FileGrowthRiseUnitCensusTests.cs diff --git a/Darling/Darling.Tests/AlertEngineTests.cs b/Darling/Darling.Tests/AlertEngineTests.cs index 932ffc23d..a2dfb366f 100644 --- a/Darling/Darling.Tests/AlertEngineTests.cs +++ b/Darling/Darling.Tests/AlertEngineTests.cs @@ -157,11 +157,18 @@ public Task> GetLongRunningQueriesAsync( public Task> GetVolumeFreeSpaceAsync(string serverKey, CancellationToken cancellationToken = default) => Task.FromResult(new List(Volumes)); - /* #2349: empty on purpose. These tests exercise other alerts, and a fabricated file would - make the file-growth gate fire inside an unrelated scenario. */ + /* #2349: EMPTY by default. These tests mostly exercise other alerts, and a fabricated file would + make the file-growth gate fire inside an unrelated scenario; the #3539 A8c pins below plant rows + and read back the lookback the engine asked for. */ + public List Files { get; } = new(); + public int? FileGrowthLookbackAsked { get; private set; } + public Task> GetDatabaseFileGrowthAsync( - string serverKey, int lookbackMinutes, CancellationToken cancellationToken = default) => - Task.FromResult(new List()); + string serverKey, int lookbackMinutes, CancellationToken cancellationToken = default) + { + FileGrowthLookbackAsked = lookbackMinutes; + return Task.FromResult(new List(Files)); + } public Task GetTempDbSpaceAsync(string serverKey, CancellationToken cancellationToken = default) => Task.FromResult(TempDb); @@ -2109,6 +2116,64 @@ public async Task LowDisk_GradesCriticallyLowBreaches_AndAStandingBreachDoesNotR Assert.Equal(3, h.Deliverer.Outcomes.Count); } + /* ---------------- database file growth (#2349; #3539 A8c: the rise is MB per HOUR) ---------------- */ + + private static DatabaseFileGrowthInfo GrowingFile(double growthMb, double windowMinutes) => new() + { + DatabaseName = "tempdb", FileName = "tempdev", PhysicalName = @"D:\data\tempdev.mdf", FileTypeDesc = "ROWS", + TotalSizeMb = 90_000, GrowthMb = growthMb, GrowthWindowMinutes = windowMinutes, + VolumeMountPoint = @"D:\", VolumeTotalMb = 4_000_000, VolumeFreeMb = 3_000_000, + }; + + /// + /// Through the ENGINE: the same growth rate pages on a five-minute lookback and on a one-day lookback, and + /// the threshold line on what fired names the rate, the window it was averaged over, and the megabytes + /// that rate amounts to inside the window — in the unit phrase both Settings windows use. The lookback the + /// engine hands the read is the configured one, so the store read and the bar cannot disagree about the + /// window. + /// + [Theory] + [InlineData(5, 853.34, "rise ≥ 10240 MB/hr averaged over 5 min (≥ 853 MB in the window) or file ≥ 60% of volume")] + [InlineData(1440, 245_760, "rise ≥ 10240 MB/hr averaged over 1440 min (≥ 245760 MB in the window) or file ≥ 60% of volume")] + public async Task FileGrowth_TheRiseIsPerHour_SoTheSameRatePagesOnAnyLookback_AndTheThresholdSaysSo( + int lookbackMinutes, double growthMb, string expectedThreshold) + { + var h = new Harness(); + h.Settings.FileGrowthEnabled = true; + h.Settings.FileGrowthLookbackMinutes = lookbackMinutes; + Assert.Equal(10_240, h.Settings.FileGrowthRiseMb); + var engine = h.Build(); + + h.Adapter.Files.Add(GrowingFile(growthMb, lookbackMinutes)); + await engine.EvaluateServerAsync(Harness.Snapshot()); + + Assert.Equal(lookbackMinutes, h.Adapter.FileGrowthLookbackAsked); + var fired = Assert.Single(h.Deliverer.Outcomes); + Assert.Equal("Database File Growth", fired.MetricName); + Assert.Equal(expectedThreshold, fired.ThresholdValue); + Assert.Contains(AlertContextBuilders.FileGrowthRiseUnit, fired.ThresholdValue, StringComparison.Ordinal); + } + + /// + /// The other half of the same property: a rate a tenth under the bar is silent on both lookbacks — including + /// the day-long one, where the per-window reading paged on 12 GB in a day because 12,288 is more than 10,240. + /// + [Theory] + [InlineData(5, 768)] + [InlineData(1440, 12_288)] + public async Task FileGrowth_ARateUnderTheBar_IsSilentOnAnyLookback(int lookbackMinutes, double growthMb) + { + var h = new Harness(); + h.Settings.FileGrowthEnabled = true; + h.Settings.FileGrowthLookbackMinutes = lookbackMinutes; + var engine = h.Build(); + + h.Adapter.Files.Add(GrowingFile(growthMb, lookbackMinutes)); + await engine.EvaluateServerAsync(Harness.Snapshot()); + + Assert.Empty(h.Deliverer.Outcomes); + } + /* ---------------- persistent version store (#1984) ---------------- */ [Fact] diff --git a/Darling/Darling.Tests/DarlingMcpAlertToolsTests.cs b/Darling/Darling.Tests/DarlingMcpAlertToolsTests.cs index dfc858075..a597533f6 100644 --- a/Darling/Darling.Tests/DarlingMcpAlertToolsTests.cs +++ b/Darling/Darling.Tests/DarlingMcpAlertToolsTests.cs @@ -294,6 +294,37 @@ public void FileGrowthWriteBounds_MatchTheEngineClamps() Assert.Contains("\"file_growth.lookback_minutes\", 5, 1440", tools, StringComparison.Ordinal); } + /// + /// #3539 A8c: file_growth.rise_mb became a RATE (megabytes per hour, averaged over + /// lookback_minutes) without changing its key, its column or its integer — so the wire contract is + /// pinned as UNCHANGED here: the read emits the row's value under the same key, the write accepts the same + /// key into the same column, and the meaning lives in both tool descriptions, which is where an agent reads + /// it. A rename would have broken every client that reads or writes the setting to say something the + /// description says just as well. + /// + [Fact] + public void FileGrowthRise_KeepsItsKeyAndColumn_AndBothDescriptionsSayItIsPerHour() + { + var payload = SerializedSettingsPayload(SampleSettingsRow()); + Assert.Equal(1024, payload["file_growth"]!["rise_mb"]!.GetValue()); + Assert.Equal(60, payload["file_growth"]!["lookback_minutes"]!.GetValue()); + + var (targets, error) = ParseAsPartialUpdate((JsonObject)JsonNode.Parse( + "{\"file_growth\":{\"rise_mb\":2048}}")!); + Assert.Null(error); + Assert.Equal(new[] { (DarlingMcpAlertTools.AlertSettingsTable, "file_growth_rise_mb") }, targets.ToArray()); + + /* The unit, on both descriptions — read off the attributes the MCP host serves, the way the + get_alert_history pin above reads its own. The census in FileGrowthRiseUnitCensusTests holds the + phrase across every surface; this is the MCP half stated where the MCP contract is pinned. */ + Assert.Contains("rise_mb is megabytes per HOUR", ToolDescription("get_alert_settings"), StringComparison.Ordinal); + Assert.Contains("file_growth.rise_mb is megabytes per HOUR", ToolDescription("update_alert_settings"), StringComparison.Ordinal); + } + + private static string ToolDescription(string toolName) => + ToolMethods().Single(m => m.GetCustomAttribute()!.Name == toolName) + .GetCustomAttribute()!.Description; + /// /// The SELECT's column count and the positional read must agree. Every field on /// AlertSettingsReadRow is read by ORDINAL, so a column inserted anywhere but the end re-maps diff --git a/Darling/Darling.Tests/FileGrowthRiseUnitCensusTests.cs b/Darling/Darling.Tests/FileGrowthRiseUnitCensusTests.cs new file mode 100644 index 000000000..c23236d28 --- /dev/null +++ b/Darling/Darling.Tests/FileGrowthRiseUnitCensusTests.cs @@ -0,0 +1,174 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Linq; +using PerformanceMonitor.Alerting; +using Xunit; +using static Darling.Tests.RepoFile; + +namespace Darling.Tests; + +/// +/// #3539 A8c: the File Growth rise threshold means ONE thing — megabytes per hour, averaged over the lookback +/// — and every surface that shows the number says so in the same words. +/// +/// The defect this holds shut. The knob shipped (#2349) as "a file grew at least this many MB +/// inside the lookback window", and the lookback shipped as a second knob clamped 5–1440 minutes. Nothing +/// related the two: the same 10,240 meant 10 GB per five minutes on a store whose operator had shortened the +/// window and 10 GB per day on one who had lengthened it, a 288× swing in what the threshold asked for, and +/// each surface described the number in its own words — "MB within N min" on the Settings rows, +/// "10240MB/60m" on the preview line, "rise ≥ 10240 MB" on the alert, a bare rise_mb on the wire — +/// none of which was wrong on its own and none of which could be compared with another. The fix made the +/// number a rate; this census is what keeps the surfaces from drifting back into private vocabularies, one +/// label at a time. +/// +/// What it asserts. Each surface either references +/// (code) or spells its literal value (XAML, which cannot +/// reference a C# constant without x:Static plumbing this row does not otherwise need), beside the +/// phrase that names the averaging window. And the phrases of the per-window reading are asserted ABSENT, so +/// a revert of one surface is a red build rather than a quiet regression — presence alone would pass with the +/// old label re-added beside the new one, which is the likelier accident. +/// +/// Both SKUs, one roster. The Darling viewer's Settings window is a copy of Lite's, so the two +/// XAML files and the two code-behinds are four rows here rather than two, and a fix that lands on one twin +/// without the other fails by name. The MCP descriptions are the fifth pair: Darling's are asserted by +/// reflection in DarlingMcpAlertToolsTests (where its wire contract is pinned) and by source here, so +/// the roster is complete on its own; Lite's has no reflection pin and is source-only. +/// +public class FileGrowthRiseUnitCensusTests +{ + /// The literal the XAML labels carry. Pinned to the constant so the two cannot drift apart: + /// XAML cannot reference the constant, so the constant's VALUE is the contract the XAML rows below are + /// held to. + [Fact] + public void TheUnitPhrase_IsMbPerHour() + { + Assert.Equal("MB/hr", AlertContextBuilders.FileGrowthRiseUnit); + } + + public static IEnumerable Surfaces() + { + /* Settings rows, both SKUs: the label between the rise box and the lookback box names the unit and + the averaging, and the old "MB within" (which read the box as a per-window total) is gone. */ + foreach (var xaml in new[] + { + "Lite/Windows/SettingsWindow.xaml", + "Darling/PerformanceMonitor.Darling.Viewer/SettingsWindow.xaml", + }) + { + yield return new object[] + { + xaml, + new[] { $"Text=\"{AlertContextBuilders.FileGrowthRiseUnit} averaged over\"" }, + new[] { "Text=\"MB within\"" }, + }; + } + + /* Preview lines, both SKUs: "Will alert when: file growth > 10240 MB/hr over 60m" — the constant, + and the old "10240MB/60m" shape gone. */ + foreach (var codeBehind in new[] + { + "Lite/Windows/SettingsWindow.xaml.cs", + "Darling/PerformanceMonitor.Darling.Viewer/SettingsWindow.xaml.cs", + }) + { + yield return new object[] + { + codeBehind, + new[] { "{AlertContextBuilders.FileGrowthRiseUnit} over {AlertFileGrowthLookbackMinutesBox.Text}m" }, + new[] { "}MB/{AlertFileGrowthLookbackMinutesBox.Text}m" }, + }; + } + + /* The alert's threshold line: rate, unit, the window it was averaged over, and the bar in megabytes + that rate amounts to inside it. The old "rise ≥ N MB or" is gone. */ + yield return new object[] + { + "PerformanceMonitor.Alerting/AlertEngine.cs", + new[] + { + "{AlertContextBuilders.FileGrowthRiseUnit} averaged over {_settings.FileGrowthLookbackMinutes} min", + "MB in the window)", + }, + new[] { "rise ≥ {_settings.FileGrowthRiseMb} MB or" }, + }; + + /* The card's Growth field — the rate the operator compares against the threshold line. */ + yield return new object[] + { + "PerformanceMonitor.Alerting/AlertContextBuilders.cs", + new[] { "({f.GrowthMbPerHour:F0} {FileGrowthRiseUnit})" }, + new[] { "MB/hr)\")," }, + }; + + /* The engine settings contract — the doc every implementation reads. */ + yield return new object[] + { + "PerformanceMonitor.Alerting/IAlertEngineSettings.cs", + new[] { "at least this many MB PER HOUR, averaged over" }, + new[] { "grew at least this many MB inside the lookback window" }, + }; + + /* The MCP descriptions, both SKUs: the key keeps its spelling, so the unit lives in the description. */ + yield return new object[] + { + "Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpAlertTools.cs", + new[] + { + "rise_mb is megabytes per HOUR, averaged over file_growth.lookback_minutes", + "file_growth.rise_mb is megabytes per HOUR averaged over file_growth.lookback_minutes", + }, + Array.Empty(), + }; + yield return new object[] + { + "Lite/Mcp/McpAlertTools.cs", + new[] { "rise_mb is megabytes per HOUR, averaged over file_growth.lookback_minutes" }, + Array.Empty(), + }; + } + + [Theory] + [MemberData(nameof(Surfaces))] + public void EverySurface_SpellsTheRiseAsARate_AndNotAsAPerWindowTotal( + string path, string[] mustContain, string[] mustNotContain) + { + var source = ReadRepoFile(path); + + foreach (var phrase in mustContain) + { + Assert.True( + source.Contains(phrase, StringComparison.Ordinal), + $"{path} no longer says «{phrase}» — the File Growth rise is MB per hour averaged over the lookback, and this surface must say so in the shared words"); + } + + foreach (var phrase in mustNotContain) + { + Assert.False( + source.Contains(phrase, StringComparison.Ordinal), + $"{path} still says «{phrase}», the per-window reading the rate replaced"); + } + } + + /// The roster above is the population; this is the floor under it, so a roster edited down to + /// nothing cannot read as clean. Nine rows: two XAML, two code-behinds, engine, builder, settings contract, + /// two MCP files — counted here rather than trusted. + [Fact] + public void TheRoster_CoversBothSkusAndTheSharedLibrary() + { + var paths = Surfaces().Select(row => (string)row[0]).ToList(); + + Assert.Equal(9, paths.Count); + Assert.Equal(paths.Count, paths.Distinct(StringComparer.Ordinal).Count()); + Assert.Contains(paths, p => p.StartsWith("Lite/", StringComparison.Ordinal)); + Assert.Contains(paths, p => p.StartsWith("Darling/", StringComparison.Ordinal)); + Assert.Contains(paths, p => p.StartsWith("PerformanceMonitor.Alerting/", StringComparison.Ordinal)); + } +} diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingAlertSettings.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingAlertSettings.cs index 54ac3ac26..ada1a5a06 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingAlertSettings.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingAlertSettings.cs @@ -183,7 +183,8 @@ the percent it has no meaningful upper bound. */ /* #2349: the file-growth gates. Clamped the same way the neighbours are -- a negative threshold would make the comparison always true, which for a gate whose whole job is to be quiet until something moves is the worst possible default. A ZERO is meaningful here rather than nonsense: it disables that one - gate, so an operator can run rise-only or level-only without a second switch. */ + gate, so an operator can run rise-only or level-only without a second switch. The rise is MB per HOUR + averaged over the lookback (#3539 A8c); the clamp does not care about the unit, the builder does. */ public bool FileGrowthEnabled => _config.Alerts.FileGrowthEnabled; public int FileGrowthRiseMb => Math.Max(0, _config.Alerts.FileGrowthRiseMb); public int FileGrowthVolumePercent => Math.Clamp(_config.Alerts.FileGrowthVolumePercent, 0, 100); diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingConfig.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingConfig.cs index 7975e1b9f..464248687 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingConfig.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingConfig.cs @@ -613,8 +613,11 @@ public sealed class AlertsConfig /* #2349: the database file-growth alert. Ships OFF -- a new alert that starts firing on upgrade is a bad citizen, and the right thresholds are a property of the fleet rather than of the product. */ public bool FileGrowthEnabled { get; set; } + /* #3539 A8c: MB per HOUR, averaged over FileGrowthLookbackMinutes. Same column (file_growth_rise_mb), same + integer, one meaning on every lookback -- the engine scales it to the window at comparison time. */ public int FileGrowthRiseMb { get; set; } = 10240; public int FileGrowthVolumePercent { get; set; } = 60; + /* The window the rise RATE is averaged over (#3539 A8c) -- it does not rescale the threshold. */ public int FileGrowthLookbackMinutes { get; set; } = 60; /// #2107: the store volume's self-alert warning percent (was a compile-time 10.0; diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpAlertTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpAlertTools.cs index ccbe75177..dd8c1f5bf 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpAlertTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpAlertTools.cs @@ -178,7 +178,7 @@ that shaped the population together with how much it removed. */ } } - [McpServerTool(Name = "get_alert_settings"), Description("Gets the current alert configuration the service is using: which alerts are enabled and their thresholds (CPU, blocking, deadlocks, poison waits, long-running queries/jobs, tempdb, low disk, failed jobs, database state, Availability Group health, connection loss), the cooldown, excluded databases, the deadlock/blocking delivery mode and cooldown, the scheduled-analysis cadence, and the fleet-sweep cadence. TWO different cooldowns are reported and they govern different stages: top-level cooldown_minutes gates whether the alert engine FIRES at all, while delivery.cooldown_minutes bounds the resulting Slack/Teams/PagerDuty/webhook/email post twice over: once per alert FINGERPRINT, and once per METRIC across the whole fleet for a RE-notification. The second bound is why one fault on forty servers does not cost forty posts an hour; the servers it holds back are named on the post that does go out, under an 'Other Servers Affected' section. A first notice is never held back by either bound, and PerEvent delivery mode opts out of the per-metric one. A channel going quiet with alerts still in get_alert_history is delivery.cooldown_minutes, not cooldown_minutes. The self_alerts group holds the thresholds for alerts about the MONITOR STORE itself rather than a monitored server — those arrive with Server: 'Monitor Store' by default, or with the store's own peers.storeName label when the operator set that file-only field (a multi-store estate names each store on its own self-alerts), so an alert naming either spelling is tuned here and nowhere else, including Retention Held's warn/critical ratios. Mute rules match the alert row's server spelling, so on a store with storeName set, scope self-alert mutes to that label, not to 'Monitor Store'. The health_bands group is NOT an alert: its two tiers decide what band a server's card, the worst-first ranking and get_fleet_overview's counts read, in deadlocks per HOUR normalised over whatever window was asked for — so the same pair means the same condition on a 1-hour read and a 24-hour one. Tuning deadlocks.count_threshold does not move the band and tuning health_bands does not move the alert. The fleet_sweep group is NOT an alert family either, and its cadence is a SECOND cadence, separate from the scheduled-analysis one: fleet_sweep.enabled turns the scheduled whole-fleet sweep report on or off, and fleet_sweep.interval_minutes (15–1440, default 60 — hourly) is how often it runs. The alerts_enabled master switch deliberately does not govern sweep production, only delivery: sweeps keep running under alerts_enabled: false — that is when they carry the would-have-paged ledger — so muting the fleet does not blind the report surface. Separately, deadlocks.pg_count_threshold and blocking.pg_count_threshold are the PostgreSQL versions of those two alerts' count gates, reported inside those same groups, and they are deliberately NOT the same numbers as the count_threshold beside them: a PostgreSQL server has no deadlock or blocking health band to calibrate against, and its blocking count is a periodic SAMPLE of pg_stat_activity rather than engine-recorded reports. The enabled switch in each group governs BOTH engines; the two thresholds do not move each other. On a store with no PostgreSQL targets both PostgreSQL keys are inert. SMTP/webhook delivery credentials are managed separately and are not reported here — configure them in the standalone Darling Viewer app's Settings window (Notifications section), which connects to this store (including remotely, not just localhost) rather than requiring desktop access to this specific box.")] + [McpServerTool(Name = "get_alert_settings"), Description("Gets the current alert configuration the service is using: which alerts are enabled and their thresholds (CPU, blocking, deadlocks, poison waits, long-running queries/jobs, tempdb, low disk, failed jobs, database state, Availability Group health, connection loss), the cooldown, excluded databases, the deadlock/blocking delivery mode and cooldown, the scheduled-analysis cadence, and the fleet-sweep cadence. TWO different cooldowns are reported and they govern different stages: top-level cooldown_minutes gates whether the alert engine FIRES at all, while delivery.cooldown_minutes bounds the resulting Slack/Teams/PagerDuty/webhook/email post twice over: once per alert FINGERPRINT, and once per METRIC across the whole fleet for a RE-notification. The second bound is why one fault on forty servers does not cost forty posts an hour; the servers it holds back are named on the post that does go out, under an 'Other Servers Affected' section. A first notice is never held back by either bound, and PerEvent delivery mode opts out of the per-metric one. A channel going quiet with alerts still in get_alert_history is delivery.cooldown_minutes, not cooldown_minutes. The self_alerts group holds the thresholds for alerts about the MONITOR STORE itself rather than a monitored server — those arrive with Server: 'Monitor Store' by default, or with the store's own peers.storeName label when the operator set that file-only field (a multi-store estate names each store on its own self-alerts), so an alert naming either spelling is tuned here and nowhere else, including Retention Held's warn/critical ratios. Mute rules match the alert row's server spelling, so on a store with storeName set, scope self-alert mutes to that label, not to 'Monitor Store'. The health_bands group is NOT an alert: its two tiers decide what band a server's card, the worst-first ranking and get_fleet_overview's counts read, in deadlocks per HOUR normalised over whatever window was asked for — so the same pair means the same condition on a 1-hour read and a 24-hour one. Tuning deadlocks.count_threshold does not move the band and tuning health_bands does not move the alert. The file_growth group's rise_mb is megabytes per HOUR, averaged over file_growth.lookback_minutes — a rate, not a total for the window: 10240 means 10 GB/hr whether the lookback is 5 minutes or 24 hours, and the engine scales it to the window (a 5-minute lookback asks for 853 MB inside it, a 24-hour one for 240 GB). The fleet_sweep group is NOT an alert family either, and its cadence is a SECOND cadence, separate from the scheduled-analysis one: fleet_sweep.enabled turns the scheduled whole-fleet sweep report on or off, and fleet_sweep.interval_minutes (15–1440, default 60 — hourly) is how often it runs. The alerts_enabled master switch deliberately does not govern sweep production, only delivery: sweeps keep running under alerts_enabled: false — that is when they carry the would-have-paged ledger — so muting the fleet does not blind the report surface. Separately, deadlocks.pg_count_threshold and blocking.pg_count_threshold are the PostgreSQL versions of those two alerts' count gates, reported inside those same groups, and they are deliberately NOT the same numbers as the count_threshold beside them: a PostgreSQL server has no deadlock or blocking health band to calibrate against, and its blocking count is a periodic SAMPLE of pg_stat_activity rather than engine-recorded reports. The enabled switch in each group governs BOTH engines; the two thresholds do not move each other. On a store with no PostgreSQL targets both PostgreSQL keys are inert. SMTP/webhook delivery credentials are managed separately and are not reported here — configure them in the standalone Darling Viewer app's Settings window (Notifications section), which connects to this store (including remotely, not just localhost) rather than requiring desktop access to this specific box.")] public static async Task GetAlertSettings( NpgsqlDataSource postgres) { @@ -324,6 +324,10 @@ be enabled only by UPDATEing config_alert_settings by hand. Reported by @gotqn. file_growth = new { enabled = s.FileGrowthEnabled, + /* #3539 A8c: MB per HOUR, averaged over lookback_minutes — a rate, not the in-window delta the key's + spelling suggests. The key keeps its name (a rename breaks every client that reads or writes it, + and Lite's McpAlertSettingsKeyTests derive its shape from this source); the unit is stated in + both tool descriptions, which is where an agent reads it. */ rise_mb = s.FileGrowthRiseMb, volume_percent = s.FileGrowthVolumePercent, lookback_minutes = s.FileGrowthLookbackMinutes @@ -475,6 +479,7 @@ and the second one is a mute somebody INTENDED that is no longer in force. "per hour is the tightest setting that is still a rate, so no value here can restore the 'any deadlock " + "is Critical' reading these tiers replaced. Setting critical BELOW warn is accepted and means every " + "banded rate is Critical. " + + "file_growth.rise_mb is megabytes per HOUR averaged over file_growth.lookback_minutes (a rate — the same 10240 is 10 GB/hr on any lookback; the engine scales it to the window), so shortening the lookback does not tighten the rise gate and lengthening it does not loosen it; only the rate does. " + "Two keys govern the PostgreSQL versions of the two count alerts and are NOT the same numbers as their SQL Server neighbours: deadlocks.pg_count_threshold and blocking.pg_count_threshold, both accepting 1 upward. They sit inside those groups rather than a section of their own so both engines' figures are visible together, but tuning deadlocks.count_threshold does NOT move the PostgreSQL gate and tuning deadlocks.pg_count_threshold does NOT move the SQL Server one. The enabled switch in each group DOES govern both engines. They are separate because the reason to move the SQL Server deadlock figure is agreement with health_bands.deadlock_warn_per_hour, and a PostgreSQL server has no deadlock band at all - its deadlocks are served by get_pg_deadlocks and are structurally absent from the fleet deadlock total - while on the blocking side the SQL Server count is engine-recorded blocked-process reports and the PostgreSQL one is distinct root blockers in a periodic SAMPLE of pg_stat_activity. Both PostgreSQL keys are ignored on a store with no PostgreSQL targets. " + "The fleet_sweep group is NOT an alert family and the alerts_enabled master switch does not govern it: " + "fleet_sweep.enabled turns the scheduled whole-fleet sweep report on or off, and " + @@ -1536,7 +1541,8 @@ including why the ceiling exists rather than leaving the knob open. */ [0,100] on the volume percent, [5,1440] on the lookback. If these drift apart the tool accepts a value the engine then silently rewrites, which reads as the setting not sticking. Zero on either gate disables that gate rather than being invalid (#2349), - which is why the rise floor is 0 and not 1. */ + which is why the rise floor is 0 and not 1. rise_mb is MB per HOUR (#3539 A8c); the + column takes the same integer it always did, and the engine scales it to the window. */ case "file_growth": Group(prop.Value, "file_growth", (k, n) => { diff --git a/Darling/PerformanceMonitor.Darling.Viewer/SettingsWindow.xaml b/Darling/PerformanceMonitor.Darling.Viewer/SettingsWindow.xaml index 99d292348..ddb579f0f 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/SettingsWindow.xaml +++ b/Darling/PerformanceMonitor.Darling.Viewer/SettingsWindow.xaml @@ -481,10 +481,10 @@ Foreground="{DynamicResource ForegroundBrush}"/> - - = {AlertPvsThresholdPercentBox.Text}% of database"); if (AlertFileGrowthCheckBox.IsChecked == true) - parts.Add($"file growth > {AlertFileGrowthRiseMbBox.Text}MB/{AlertFileGrowthLookbackMinutesBox.Text}m or volume > {AlertFileGrowthVolumePercentBox.Text}%"); + /* #3539 A8c: the rise is a RATE (MB per hour) averaged over the lookback, in the same unit phrase the + row's label, the alert's threshold line and the MCP payload description use. It used to read + "10240MB/60m", which was the per-window delta the engine then compared literally. */ + parts.Add($"file growth > {AlertFileGrowthRiseMbBox.Text} {AlertContextBuilders.FileGrowthRiseUnit} over {AlertFileGrowthLookbackMinutesBox.Text}m or volume > {AlertFileGrowthVolumePercentBox.Text}%"); if (AlertLongRunningJobCheckBox.IsChecked == true) parts.Add($"jobs > {AlertLongRunningJobMultiplierBox.Text}x avg"); if (AlertFailedJobCheckBox.IsChecked == true) diff --git a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.AlertSettings.cs b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.AlertSettings.cs index 6102bfde2..bb79e6ea0 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.AlertSettings.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.AlertSettings.cs @@ -446,7 +446,8 @@ public sealed class AlertSettingsRow public int FleetSweepIntervalMinutes { get; set; } = FleetSweepCadence.DefaultIntervalMinutes; /* #2391: defaults mirror the V79 column defaults, so a viewer prefilling against a store that has - not seeded the row shows what the store would have given it. Ships OFF, per #2349. */ + not seeded the row shows what the store would have given it. Ships OFF, per #2349. The rise column + is MB per HOUR averaged over the lookback (#3539 A8c) -- the Settings row's label says so. */ public bool FileGrowthEnabled { get; set; } public int FileGrowthRiseMb { get; set; } = 10240; public int FileGrowthVolumePercent { get; set; } = 60; diff --git a/Lite.Tests/FileGrowthAlertTests.cs b/Lite.Tests/FileGrowthAlertTests.cs index 888f9a176..b18ed4c2d 100644 --- a/Lite.Tests/FileGrowthAlertTests.cs +++ b/Lite.Tests/FileGrowthAlertTests.cs @@ -57,7 +57,7 @@ public void TheRiseGate_FiresOnGrowth_EvenWhenTheVolumeIsRoomy() /* 2% of a 4 TB volume — the level gate cannot see this, and it is exactly the case the issue is about. */ Assert.True(files[0].VolumePercent < 5); - var breached = AlertContextBuilders.GetBreachedFiles(files, riseMb: 10_240, volumePercent: 60); + var breached = AlertContextBuilders.GetBreachedFiles(files, riseMbPerHour: 10_240, volumePercent: 60, lookbackMinutes: 60); Assert.Single(breached); } @@ -72,7 +72,7 @@ public void TheLevelGate_FiresOnAFileThatIsLargeButNoLongerGrowing() { var files = new List { File(sizeMb: 400_000, growthMb: 0, volumeTotalMb: 500_000) }; - var breached = AlertContextBuilders.GetBreachedFiles(files, riseMb: 10_240, volumePercent: 60); + var breached = AlertContextBuilders.GetBreachedFiles(files, riseMbPerHour: 10_240, volumePercent: 60, lookbackMinutes: 60); Assert.Single(breached); Assert.Equal(80, breached[0].VolumePercent); @@ -84,7 +84,7 @@ public void AQuietFile_DoesNotFire() { var files = new List { File(sizeMb: 50_000, growthMb: 100, volumeTotalMb: 500_000) }; - Assert.Empty(AlertContextBuilders.GetBreachedFiles(files, riseMb: 10_240, volumePercent: 60)); + Assert.Empty(AlertContextBuilders.GetBreachedFiles(files, riseMbPerHour: 10_240, volumePercent: 60, lookbackMinutes: 60)); } /// @@ -98,12 +98,12 @@ public void ZeroDisablesOneGate_NotBoth() var large = new List { File(sizeMb: 400_000, growthMb: 0, volumeTotalMb: 500_000) }; /* level off: the rise still fires, the large-but-static file does not */ - Assert.Single(AlertContextBuilders.GetBreachedFiles(grew, riseMb: 10_240, volumePercent: 0)); - Assert.Empty(AlertContextBuilders.GetBreachedFiles(large, riseMb: 10_240, volumePercent: 0)); + Assert.Single(AlertContextBuilders.GetBreachedFiles(grew, riseMbPerHour: 10_240, volumePercent: 0, lookbackMinutes: 60)); + Assert.Empty(AlertContextBuilders.GetBreachedFiles(large, riseMbPerHour: 10_240, volumePercent: 0, lookbackMinutes: 60)); /* rise off: the level still fires, the growing-but-small-share file does not */ - Assert.Empty(AlertContextBuilders.GetBreachedFiles(grew, riseMb: 0, volumePercent: 60)); - Assert.Single(AlertContextBuilders.GetBreachedFiles(large, riseMb: 0, volumePercent: 60)); + Assert.Empty(AlertContextBuilders.GetBreachedFiles(grew, riseMbPerHour: 0, volumePercent: 60, lookbackMinutes: 60)); + Assert.Single(AlertContextBuilders.GetBreachedFiles(large, riseMbPerHour: 0, volumePercent: 60, lookbackMinutes: 60)); } /// @@ -116,7 +116,7 @@ public void AFileWithNoVolumeStats_IsNotLevelGated() var files = new List { File(sizeMb: 400_000, growthMb: 0, volumeTotalMb: 0) }; Assert.Equal(0, files[0].VolumePercent); - Assert.Empty(AlertContextBuilders.GetBreachedFiles(files, riseMb: 10_240, volumePercent: 60)); + Assert.Empty(AlertContextBuilders.GetBreachedFiles(files, riseMbPerHour: 10_240, volumePercent: 60, lookbackMinutes: 60)); } /// @@ -132,7 +132,7 @@ public void BreachedFiles_AreOrderedByHowCloseTheyAreToFillingTheirVolume() File(db: "tight", name: "f2", sizeMb: 400_000, growthMb: 20_000, volumeTotalMb: 500_000), }; - var breached = AlertContextBuilders.GetBreachedFiles(files, riseMb: 10_240, volumePercent: 60); + var breached = AlertContextBuilders.GetBreachedFiles(files, riseMbPerHour: 10_240, volumePercent: 60, lookbackMinutes: 60); Assert.Equal(2, breached.Count); Assert.Equal("tight", breached[0].DatabaseName); @@ -204,7 +204,7 @@ public void ASingleSampleWindow_ReportsNoRise() Assert.Equal(0, f.GrowthMb); Assert.Equal(0, f.GrowthMbPerHour); - Assert.Empty(AlertContextBuilders.GetBreachedFiles(new List { f }, riseMb: 10_240, volumePercent: 0)); + Assert.Empty(AlertContextBuilders.GetBreachedFiles(new List { f }, riseMbPerHour: 10_240, volumePercent: 0, lookbackMinutes: 60)); } /// The rate is derived from the MEASURED window, so a collection gap cannot make a slow rise @@ -219,4 +219,102 @@ public void TheRateUsesTheMeasuredWindow(double growthMb, double windowMinutes, Assert.Equal(expectedPerHour, f.GrowthMbPerHour, precision: 3); } + + /* ---------------- #3539 A8c: the rise threshold is a RATE ---------------- */ + + /// + /// The threshold means megabytes per HOUR on every lookback. Before #3539 A8c the stored number was compared + /// against the raw growth inside the window, so a 10 GB bar meant 10 GB per five minutes on a store with a + /// short lookback and 10 GB per day on one with a long one — the same file growing at the same rate paged + /// on one and not the other. Now a file growing at exactly the threshold rate for the whole window breaches + /// on every lookback the clamp allows, and one growing a tenth under it breaches on none. + /// + [Theory] + [InlineData(5)] + [InlineData(30)] + [InlineData(60)] + [InlineData(240)] + [InlineData(1440)] + public void TheSameGrowthRate_GivesTheSameVerdict_OnEveryLookback(int lookbackMinutes) + { + const int riseMbPerHour = 10_240; + var atTheRate = riseMbPerHour * lookbackMinutes / 60.0; + + var onTheBar = new List { File(growthMb: atTheRate, windowMinutes: lookbackMinutes, volumeTotalMb: 4_000_000) }; + var under = new List { File(growthMb: atTheRate * 0.9, windowMinutes: lookbackMinutes, volumeTotalMb: 4_000_000) }; + + Assert.Single(AlertContextBuilders.GetBreachedFiles(onTheBar, riseMbPerHour, volumePercent: 0, lookbackMinutes)); + Assert.Empty(AlertContextBuilders.GetBreachedFiles(under, riseMbPerHour, volumePercent: 0, lookbackMinutes)); + } + + /// + /// The lie, as arithmetic: 2 GB inside a five-minute window is 24 GB/hr. The per-window reading held it to + /// the full 10,240 and stayed silent; 24 GB/hr against a 10 GB/hr bar pages. And the reverse on a long + /// window: 12 GB over a day is 512 MB/hr, which the per-window reading paged on and the rate does not. + /// + [Fact] + public void ThePerWindowReading_GaveTheOppositeVerdict_AtBothEndsOfTheClamp() + { + var burstOnAShortWindow = new List { File(growthMb: 2_048, windowMinutes: 5, volumeTotalMb: 4_000_000) }; + var crawlOnALongWindow = new List { File(growthMb: 12_288, windowMinutes: 1440, volumeTotalMb: 4_000_000) }; + + /* per-window: 2048 < 10240 silent, 12288 >= 10240 pages */ + Assert.True(burstOnAShortWindow[0].GrowthMb < 10_240); + Assert.True(crawlOnALongWindow[0].GrowthMb >= 10_240); + + /* per hour: the burst pages, the crawl does not */ + Assert.Single(AlertContextBuilders.GetBreachedFiles(burstOnAShortWindow, riseMbPerHour: 10_240, volumePercent: 0, lookbackMinutes: 5)); + Assert.Empty(AlertContextBuilders.GetBreachedFiles(crawlOnALongWindow, riseMbPerHour: 10_240, volumePercent: 0, lookbackMinutes: 1440)); + } + + /// + /// The bar the growth is held to, in megabytes inside the window: the rate times the window in hours. The + /// shipped 10,240 on the shipped 60-minute lookback is 10,240 — the number every store on the defaults was + /// always compared against, which is what makes this change silent for them. 5 minutes asks for 853 MB, + /// a day for 240 GB. + /// + [Theory] + [InlineData(10_240, 60, 10_240.0)] + [InlineData(10_240, 5, 853.333)] + [InlineData(10_240, 1440, 245_760.0)] + [InlineData(1_024, 30, 512.0)] + [InlineData(0, 60, 0.0)] + public void TheBarIsTheRateTimesTheWindowInHours(int riseMbPerHour, int lookbackMinutes, double expectedBarMb) + { + Assert.Equal(expectedBarMb, AlertContextBuilders.FileGrowthRiseBarMb(riseMbPerHour, lookbackMinutes), precision: 3); + } + + /// + /// Growth observed over LESS than the window is held to the WHOLE window's bar — unobserved time counts + /// as no growth. The alternative, reading the rate off the measured span, extrapolates: a server that + /// started collecting five minutes ago with one 1 GB autogrowth in that span would read 12 GB/hr, page + /// the default bar, and resolve at the next sample when the span widened. Same conservative stance the + /// single-sample window already takes ("no rise observed", not "the whole file appeared"). + /// + [Fact] + public void GrowthObservedOverLessThanTheWindow_IsHeldToTheWholeWindowsBar() + { + var freshServer = File(growthMb: 1_024, windowMinutes: 5, volumeTotalMb: 4_000_000); + + /* The card's rate WOULD read as over the bar; the gate does not use it. */ + Assert.Equal(12_288, freshServer.GrowthMbPerHour, precision: 3); + Assert.Empty(AlertContextBuilders.GetBreachedFiles(new List { freshServer }, riseMbPerHour: 10_240, volumePercent: 0, lookbackMinutes: 60)); + + /* The same growth on a lookback as short as the span is a real 12 GB/hr and pages. */ + Assert.Single(AlertContextBuilders.GetBreachedFiles(new List { freshServer }, riseMbPerHour: 10_240, volumePercent: 0, lookbackMinutes: 5)); + } + + /// The card's rate carries the same unit phrase the threshold line and both Settings windows use, + /// so the two numbers read as comparable. + [Fact] + public void TheCardsRate_UsesTheSharedUnitPhrase() + { + var f = File(sizeMb: 400_000, growthMb: 40_000, windowMinutes: 60, volumeTotalMb: 500_000); + + var context = AlertContextBuilders.BuildFileGrowthContext(Server, new List { f }); + + var growth = Assert.Single(context!.Details.SelectMany(d => d.Fields), x => x.Item1 == "Growth"); + Assert.Equal($"39.1 GB in 60 min (40000 {AlertContextBuilders.FileGrowthRiseUnit})", growth.Item2); + Assert.Equal("MB/hr", AlertContextBuilders.FileGrowthRiseUnit); + } } diff --git a/Lite/App.xaml.cs b/Lite/App.xaml.cs index 9ec792d33..c250445ee 100644 --- a/Lite/App.xaml.cs +++ b/Lite/App.xaml.cs @@ -183,9 +183,9 @@ being handed back the stale in-memory version. The coordinator owns the mutex + public static int AlertPvsThresholdPercent { get; set; } = 40; // Alert when an ADR database's PVS >= X% of its data files (0 disables this check) public static int AlertPvsFloorGb { get; set; } = 1; // AND-qualifier: the PVS must also be >= X GB (0 removes the floor) public static bool AlertFileGrowthEnabled { get; set; } // #2349 database file growth -- OFF by default - public static int AlertFileGrowthRiseMb { get; set; } = 10240; // RISE gate: a file grew >= X MB in the window (0 disables this gate) + public static int AlertFileGrowthRiseMb { get; set; } = 10240; // RISE gate: a file growing >= X MB per HOUR, averaged over the lookback (#3539 A8c; 0 disables this gate) public static int AlertFileGrowthVolumePercent { get; set; } = 60; // LEVEL gate: a file is >= X% of its volume (0 disables this gate) - public static int AlertFileGrowthLookbackMinutes { get; set; } = 60;// how far back the rise is measured + public static int AlertFileGrowthLookbackMinutes { get; set; } = 60;// the window the rise rate is averaged over public static bool AlertLongRunningJobEnabled { get; set; } = true; public static int AlertLongRunningJobMultiplier { get; set; } = 3; public static bool AlertFailedJobEnabled { get; set; } = true; diff --git a/Lite/Mcp/McpAlertTools.cs b/Lite/Mcp/McpAlertTools.cs index 1e45b42ce..98a2fbe60 100644 --- a/Lite/Mcp/McpAlertTools.cs +++ b/Lite/Mcp/McpAlertTools.cs @@ -113,7 +113,7 @@ that shaped the population together with how much it removed. */ } } - [McpServerTool(Name = "get_alert_settings"), Description("Gets the current alert configuration this instance is running on: which alerts are enabled and their thresholds (CPU, blocking, deadlocks, poison waits, long-running queries and jobs, tempdb space, low disk, PVS, file growth, failed jobs, database state, Availability Group health, connection loss), the cooldown, the excluded databases, the deadlock/blocking delivery mode and cooldown, the scheduled-analysis cadence, and the SMTP email configuration. The two cooldowns govern different stages: top-level cooldown_minutes gates whether an alert FIRES, delivery.cooldown_minutes bounds the resulting email/Teams/Slack/PagerDuty/webhook send both per alert FINGERPRINT and, for a re-notification, per METRIC across every monitored server (the servers it holds back are named on the send that does go out, under an 'Other Servers Affected' section; a first notice is never held back, and delivery.mode PerEvent opts out of the per-metric bound). The same nested shape Darling's get_alert_settings returns, minus its self_alerts group (the headless service's own store-volume and collection-health thresholds, which a single-instance Lite install has no equivalent for) and plus smtp, which Lite delivers itself. Read-only: Lite has no update_alert_settings, so these change in the Settings window.")] + [McpServerTool(Name = "get_alert_settings"), Description("Gets the current alert configuration this instance is running on: which alerts are enabled and their thresholds (CPU, blocking, deadlocks, poison waits, long-running queries and jobs, tempdb space, low disk, PVS, file growth, failed jobs, database state, Availability Group health, connection loss), the cooldown, the excluded databases, the deadlock/blocking delivery mode and cooldown, the scheduled-analysis cadence, and the SMTP email configuration. The two cooldowns govern different stages: top-level cooldown_minutes gates whether an alert FIRES, delivery.cooldown_minutes bounds the resulting email/Teams/Slack/PagerDuty/webhook send both per alert FINGERPRINT and, for a re-notification, per METRIC across every monitored server (the servers it holds back are named on the send that does go out, under an 'Other Servers Affected' section; a first notice is never held back, and delivery.mode PerEvent opts out of the per-metric bound). The file_growth group's rise_mb is megabytes per HOUR, averaged over file_growth.lookback_minutes — a rate, not a total for the window: 10240 means 10 GB/hr whether the lookback is 5 minutes or 24 hours, and the engine scales it to the window. The same nested shape Darling's get_alert_settings returns, minus its self_alerts group (the headless service's own store-volume and collection-health thresholds, which a single-instance Lite install has no equivalent for) and plus smtp, which Lite delivers itself. Read-only: Lite has no update_alert_settings, so these change in the Settings window.")] public static Task GetAlertSettings() { try @@ -224,6 +224,9 @@ numeral here would be a frozen enumeration that the next Darling self-alert knob file_growth = new { enabled = App.AlertFileGrowthEnabled, + /* #3539 A8c: MB per HOUR averaged over lookback_minutes, like Darling's; the key keeps its + spelling (McpAlertSettingsKeyTests derives the shape from Darling's source) and the unit is + stated in the tool description above. */ rise_mb = App.AlertFileGrowthRiseMb, volume_percent = App.AlertFileGrowthVolumePercent, lookback_minutes = App.AlertFileGrowthLookbackMinutes diff --git a/Lite/Services/AppAlertEngineSettings.cs b/Lite/Services/AppAlertEngineSettings.cs index 66ea65f08..062475b27 100644 --- a/Lite/Services/AppAlertEngineSettings.cs +++ b/Lite/Services/AppAlertEngineSettings.cs @@ -89,7 +89,7 @@ would be silently reset on the first config_version bump. If this needs to be co /* #2349: the file-growth gates, same clamps as Darling's adapter so the two SKUs cannot disagree about what a threshold means. Zero disables one gate rather than being nonsense, so rise-only or level-only - needs no second switch. */ + needs no second switch. The rise is MB per HOUR averaged over the lookback (#3539 A8c). */ public bool FileGrowthEnabled => App.AlertFileGrowthEnabled; public int FileGrowthRiseMb => Math.Max(0, App.AlertFileGrowthRiseMb); public int FileGrowthVolumePercent => Math.Clamp(App.AlertFileGrowthVolumePercent, 0, 100); diff --git a/Lite/Windows/SettingsWindow.xaml b/Lite/Windows/SettingsWindow.xaml index 02e879e3c..83d5d386d 100644 --- a/Lite/Windows/SettingsWindow.xaml +++ b/Lite/Windows/SettingsWindow.xaml @@ -328,10 +328,10 @@ Foreground="{DynamicResource ForegroundBrush}"/> - - = {AlertPvsThresholdPercentBox.Text}% of database"); if (AlertFileGrowthCheckBox.IsChecked == true) - parts.Add($"file growth > {AlertFileGrowthRiseMbBox.Text}MB/{AlertFileGrowthLookbackMinutesBox.Text}m or volume > {AlertFileGrowthVolumePercentBox.Text}%"); + /* #3539 A8c: the rise is a RATE (MB per hour) averaged over the lookback, in the same unit phrase the + row's label, the alert's threshold line and the MCP payload description use. It used to read + "10240MB/60m", which was the per-window delta the engine then compared literally. */ + parts.Add($"file growth > {AlertFileGrowthRiseMbBox.Text} {AlertContextBuilders.FileGrowthRiseUnit} over {AlertFileGrowthLookbackMinutesBox.Text}m or volume > {AlertFileGrowthVolumePercentBox.Text}%"); if (AlertLongRunningJobCheckBox.IsChecked == true) parts.Add($"jobs > {AlertLongRunningJobMultiplierBox.Text}x avg"); if (AlertFailedJobCheckBox.IsChecked == true) diff --git a/PerformanceMonitor.Alerting/AlertContextBuilders.cs b/PerformanceMonitor.Alerting/AlertContextBuilders.cs index e8129fbdd..71805b5e5 100644 --- a/PerformanceMonitor.Alerting/AlertContextBuilders.cs +++ b/PerformanceMonitor.Alerting/AlertContextBuilders.cs @@ -337,23 +337,63 @@ public static IReadOnlyList FileGrowthIncidents( .Where(i => i is not null).Select(i => i!).ToList(); } + /// + /// The unit phrase every surface that shows the file-growth rise threshold uses (#3539 A8c). Both Settings + /// windows, the alert body's threshold line and the card spell the knob's unit with THIS string (the MCP tool + /// descriptions, which are prose, write it out as "megabytes per HOUR"), and a census test holds each of + /// them to it — the knob meant "per lookback" from the day it shipped because nothing held the surfaces to + /// one phrase. + /// + public const string FileGrowthRiseUnit = "MB/hr"; + + /// + /// The rise bar, in megabytes over the lookback window, for a threshold expressed in MB per HOUR (#3539 A8c). + /// + /// What the knob means. is a RATE — megabytes + /// per hour — and is the window that rate is + /// averaged over. Before this, the stored number was compared against the raw growth inside the window, so + /// the same 10,240 meant "10 GB in five minutes" on a store whose operator had shortened the lookback and + /// "10 GB in a day" on one who had lengthened it: two knobs, one of which silently rescaled the other by up + /// to 288×. Now 10,240 is 10 GB/hr everywhere; a 5-minute window asks for 853 MB inside it, a 24-hour + /// window for 240 GB, and both are the same sustained rate. + /// + /// Why the bar is scaled to the CONFIGURED window rather than the rate read off the MEASURED one. + /// divides by the width the samples actually span, + /// which on a server that started collecting five minutes ago — or just came out of a collection gap — is + /// five minutes: one 1 GB autogrowth in that span reads as 12 GB/hr, fires the default bar, and resolves at + /// the next sample when the span widens. Holding the growth to rate × configured window instead counts + /// unobserved time as no growth, the same conservative reading a single-sample window already gets ("no rise + /// observed", not "the whole file appeared"). It also makes the change byte-identical for every store on the + /// shipped 60-minute lookback: rate × 60 / 60 is the number that was always compared. + /// + public static double FileGrowthRiseBarMb(int riseMbPerHour, int lookbackMinutes) => + /* Product first, one division: the product of two ints is exact in a double for any value the knobs' + clamps allow (the write bound is int.MaxValue on the rate, so it is widened before multiplying), and + dividing once keeps "rate × 60 / 60" equal to the rate to the last bit on the shipped lookback. */ + (double)riseMbPerHour * Math.Max(1, lookbackMinutes) / 60.0; + /// /// #2349: the files breaching either gate, worst first. Both gates are applied HERE rather than in the /// engine so the render path, the observation path and the decision can never disagree about which files /// are involved. /// + /// The rise gate compares the growth inside the window against — the + /// MB-per-hour threshold scaled to the window it is averaged over (#3539 A8c), so the same rate gives the + /// same verdict whatever the lookback is set to. + /// /// Ordered by how much of its volume the file occupies, because that is the one number that says how /// close this is to becoming a Volume Free Space page — a 40 GB rise on a 4 TB volume is less urgent /// than a 10 GB file that is now 80% of a small one. /// public static List GetBreachedFiles( - IReadOnlyList? files, int riseMb, int volumePercent) + IReadOnlyList? files, int riseMbPerHour, int volumePercent, int lookbackMinutes) { if (files is null || files.Count == 0) return new List(); + var riseBarMb = FileGrowthRiseBarMb(riseMbPerHour, lookbackMinutes); var breached = files .Where(f => - (riseMb > 0 && f.GrowthMb >= riseMb) + (riseMbPerHour > 0 && f.GrowthMb >= riseBarMb) || (volumePercent > 0 && f.VolumeTotalMb > 0 && f.VolumePercent >= volumePercent)) .OrderByDescending(f => f.VolumePercent) .ThenByDescending(f => f.GrowthMb) @@ -382,7 +422,10 @@ public static List GetBreachedFiles( ("File", f.FileName), ("Physical Name", f.PhysicalName), ("Size", $"{f.TotalSizeGb:F1} GB"), - ("Growth", $"{f.GrowthGb:F1} GB in {f.GrowthWindowMinutes:F0} min ({f.GrowthMbPerHour:F0} MB/hr)"), + /* The rate here is over the MEASURED span (what the samples actually show); the threshold line + on the alert says what bar it was held to and over what window. Same unit phrase as the + threshold, so the two numbers read as comparable (#3539 A8c). */ + ("Growth", $"{f.GrowthGb:F1} GB in {f.GrowthWindowMinutes:F0} min ({f.GrowthMbPerHour:F0} {FileGrowthRiseUnit})"), ("Volume", string.IsNullOrEmpty(f.VolumeMountPoint) ? "(unknown)" : f.VolumeMountPoint), ("Volume Free", $"{f.VolumeFreeMb / 1024.0:F1} GB"), ("File % of Volume", $"{f.VolumePercent:F0}%"), diff --git a/PerformanceMonitor.Alerting/AlertEngine.cs b/PerformanceMonitor.Alerting/AlertEngine.cs index 2a00789b9..a2962d25e 100644 --- a/PerformanceMonitor.Alerting/AlertEngine.cs +++ b/PerformanceMonitor.Alerting/AlertEngine.cs @@ -1773,8 +1773,11 @@ private async Task CheckFileGrowthAsync( key, _settings.FileGrowthLookbackMinutes, ct); readClock.Restart(); + /* #3539 A8c: the rise knob is MB per HOUR and the lookback is the window that rate is averaged over, + so the builder scales the bar to the window rather than comparing the raw in-window delta against + a number whose meaning would otherwise change with the other knob. */ var breached = AlertContextBuilders.GetBreachedFiles( - files, _settings.FileGrowthRiseMb, _settings.FileGrowthVolumePercent); + files, _settings.FileGrowthRiseMb, _settings.FileGrowthVolumePercent, _settings.FileGrowthLookbackMinutes); var fileGrowthOccurrences = await ObserveOccurrencesAsync( key, FileGrowthWatermarkMetric, @@ -1803,10 +1806,19 @@ decide whether this is worth getting up for. The rise is in the card. */ + $"({worst.VolumePercent:F0}% of {worst.VolumeMountPoint}), " + $"grew {worst.GrowthGb:F1} GB in {worst.GrowthWindowMinutes:F0} min"; + /* The threshold line states the rate AND the window it was averaged over, in the same unit + phrase the Settings windows and the card use (#3539 A8c) — and the megabytes that rate + amounts to inside the window, which is the number the card's "Growth" figure was held to. + The card's own rate is over the MEASURED span, which can be narrower than the window on a + server that started collecting recently; naming the in-window bar is what lets the two be + compared without knowing that. */ + var riseBarMb = AlertContextBuilders.FileGrowthRiseBarMb( + _settings.FileGrowthRiseMb, _settings.FileGrowthLookbackMinutes); await FireAsync(new AlertOutcome( key, serverName, "Database File Growth", headline, - $"rise ≥ {_settings.FileGrowthRiseMb} MB or file ≥ {_settings.FileGrowthVolumePercent}% of volume", + $"rise ≥ {_settings.FileGrowthRiseMb} {AlertContextBuilders.FileGrowthRiseUnit} averaged over {_settings.FileGrowthLookbackMinutes} min " + + $"(≥ {riseBarMb:F0} MB in the window) or file ≥ {_settings.FileGrowthVolumePercent}% of volume", context, detailText, NumericCurrentValue: worst.VolumePercent, NumericThresholdValue: _settings.FileGrowthVolumePercent, diff --git a/PerformanceMonitor.Alerting/DatabaseFileGrowthInfo.cs b/PerformanceMonitor.Alerting/DatabaseFileGrowthInfo.cs index 4b1b2bdd8..4a1893fc6 100644 --- a/PerformanceMonitor.Alerting/DatabaseFileGrowthInfo.cs +++ b/PerformanceMonitor.Alerting/DatabaseFileGrowthInfo.cs @@ -62,7 +62,10 @@ public class DatabaseFileGrowthInfo public double TotalSizeGb => TotalSizeMb / 1024.0; public double GrowthGb => GrowthMb / 1024.0; - /// Growth per hour, for a message that distinguishes "80 GB in an hour" from "80 GB since Tuesday". + /// Growth per hour over the MEASURED span, for a message that distinguishes "80 GB in an hour" from + /// "80 GB since Tuesday". Display only: the rise gate holds to the threshold scaled + /// to the CONFIGURED window (), because this figure + /// extrapolates — one autogrowth inside a five-minute span reads as twelve an hour (#3539 A8c). public double GrowthMbPerHour => GrowthWindowMinutes > 0 ? GrowthMb / (GrowthWindowMinutes / 60.0) : 0; } diff --git a/PerformanceMonitor.Alerting/IAlertEngineSettings.cs b/PerformanceMonitor.Alerting/IAlertEngineSettings.cs index de5125516..9412c43e8 100644 --- a/PerformanceMonitor.Alerting/IAlertEngineSettings.cs +++ b/PerformanceMonitor.Alerting/IAlertEngineSettings.cs @@ -217,9 +217,17 @@ public interface IAlertEngineSettings int PvsFloorGb { get; } /// - /// The RISE gate: a file that grew at least this many MB inside the lookback window (#2349). Primary - /// rather than the level, for #2157's reason — a level alone re-pages every cooldown about a size that has - /// been true for a week, which trains people to mute it, while a rise is an event. + /// The RISE gate: a file growing at least this many MB PER HOUR, averaged over + /// (#2349, unit fixed by #3539 A8c). Primary rather than the level, + /// for #2157's reason — a level alone re-pages every cooldown about a size that has been true for a week, + /// which trains people to mute it, while a rise is an event. + /// + /// A rate, not an in-window delta. The name keeps its Mb suffix because the stored column + /// (config_alert_settings.file_growth_rise_mb) and the MCP key (file_growth.rise_mb) keep + /// theirs; the unit is stated where the number is shown — both Settings windows, the alert's threshold + /// line, the tool descriptions — with . Compared + /// through , which scales it to the window; on the + /// shipped 60-minute lookback that is the number itself. /// int FileGrowthRiseMb { get; } @@ -230,8 +238,11 @@ public interface IAlertEngineSettings /// int FileGrowthVolumePercent { get; } - /// How far back the rise is measured (#2349). The window is MEASURED from the samples rather than - /// assumed, so a gap in collection cannot make a slow rise look fast. + /// The window the rise RATE is averaged over (#2349; #3539 A8c). A short window catches a burst, a + /// long one asks the rate to be sustained — it does not rescale the threshold, which is per hour whatever + /// this is set to. The span the samples actually cover inside it is MEASURED rather than assumed, so a gap + /// in collection cannot make a slow rise look fast; growth observed over less than the window is held to + /// the whole window's bar, which reads unobserved time as no growth. int FileGrowthLookbackMinutes { get; } /// Fire when a running job exceeds this multiple of its historical average duration. From e609dbe480edc5287e92ce3f3f1a84a1de9f0826 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 15:09:58 -0400 Subject: [PATCH 53/69] Story confidence measures corroboration instead of path length, and the nightly rebuild's three cards become one incident that names the job (#3538 A6/A9) (#3632) * Story confidence measures corroboration instead of path length, and the nightly rebuild's three cards become one incident that names the job (#3538 A6/A9) * StoryConfidence in its own file; the exhaustive pins run to the engine's real 11-node maximum path (review) --- Darling/Darling.Tests/DarlingMcpToolsTests.cs | 5 +- .../Mcp/DarlingMcpInstructions.cs | 6 +- .../Mcp/DarlingMcpTools.cs | 27 +- Lite.Tests/McpAnalysisFindingsCommandTests.cs | 70 +++ Lite.Tests/McpMissMessageParityPinTests.cs | 13 + Lite.Tests/StoryConfidenceTests.cs | 407 ++++++++++++++++++ Lite/Mcp/McpAnalysisTools.cs | 27 +- Lite/Mcp/McpInstructions.cs | 6 +- PerformanceMonitor.Analysis/FactAdvice.cs | 79 +++- .../InferenceEngine.cs | 16 +- .../RelationshipGraph.cs | 44 ++ .../SameStatementPileupDetector.cs | 8 +- .../StoryConfidence.cs | 173 ++++++++ 13 files changed, 859 insertions(+), 22 deletions(-) create mode 100644 Lite.Tests/StoryConfidenceTests.cs create mode 100644 PerformanceMonitor.Analysis/StoryConfidence.cs diff --git a/Darling/Darling.Tests/DarlingMcpToolsTests.cs b/Darling/Darling.Tests/DarlingMcpToolsTests.cs index 2aa04fefe..87ea10ba0 100644 --- a/Darling/Darling.Tests/DarlingMcpToolsTests.cs +++ b/Darling/Darling.Tests/DarlingMcpToolsTests.cs @@ -350,7 +350,7 @@ nowhere near the window-covering cap carries a present-but-null truncation_note. including the #2000 occurrence stats. */ foreach (var field in new[] { - "finding_id", "analysis_time", "severity", "confidence", "category", + "finding_id", "analysis_time", "severity", "confidence", "confidence_basis", "category", "root_fact", "leaf_fact", "story_path", "story_path_hash", "fact_count", "incident_id", "occurrences", "first_seen", "last_seen", "peak_severity", "co_fired", "time_range", "advice", "remediation_command", "structured_remediation" @@ -368,6 +368,9 @@ including the #2000 occurrence stats. */ Assert.Equal(TestStoryHash, finding.GetProperty("story_path_hash").GetString()); Assert.Equal(2.5, finding.GetProperty("severity").GetDouble()); Assert.Equal(0.9, finding.GetProperty("confidence").GetDouble()); + /* #3538 A6: 0.9 on a two-node path is not the legacy (n-1)/n = 0.5, so the basis reads as + corroboration-derived; the legacy label is pinned on Lite's twin with a real 1.0/1 row. */ + Assert.StartsWith("corroboration (#3538)", finding.GetProperty("confidence_basis").GetString(), StringComparison.Ordinal); Assert.Equal("cpu", finding.GetProperty("category").GetString()); Assert.Equal("an4-incident-1", finding.GetProperty("incident_id").GetString()); Assert.Equal("SOS_SCHEDULER_YIELD", finding.GetProperty("root_fact").GetProperty("key").GetString()); diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs index d9250dfc8..da382529b 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs @@ -86,11 +86,11 @@ public static string Build(DarlingPeerDirectory.Snapshot peers) | Tool | Purpose | Key Parameters | |------|---------|----------------| - | `analyze_server` | Runs the inference engine: scores facts, traverses relationship graph, returns evidence-backed findings with severity and recommended next tools. A remediable finding also carries `remediation_command` — the full copy-paste T-SQL remediation (identical to the viewer card), with a two-sided risk-disclosure header on destructive changes; advisory only, never executed. Force-plan findings additionally carry `structured_remediation`: the verdict as machine-readable fields (eligible + named blockers), evidence, and split force/unforce/verify SQL. With `as_of` it analyzes a PAST window — anomaly baseline included — and is EXPLORATORY: the findings come back in full but are NOT persisted, which `persisted` / `persistence_note` state on every result | `server_name`, `hours_back` (default 4), `as_of` | - | `get_analysis_facts` | Exposes raw scored facts from the collect+score pipeline — every observation the engine sees with base severity, amplifiers, and metadata | `server_name`, `hours_back` (default 4), `source` (filter), `min_severity`, `as_of` | + | `analyze_server` | Runs the inference engine: scores facts, traverses relationship graph, returns evidence-backed findings with severity and recommended next tools. Each finding's `confidence` is an EVIDENCE score (0.20 for the symptom alone, more as the root fact's amplifier checks match and the chain deepens — a lone uncorroborated symptom is 0.20, never 1.0) and `confidence_basis` says in words what it rests on; rank by `severity` for impact and read `confidence` as how much of the engine's own corroboration showed up. A remediable finding also carries `remediation_command` — the full copy-paste T-SQL remediation (identical to the viewer card), with a two-sided risk-disclosure header on destructive changes; advisory only, never executed. Force-plan findings additionally carry `structured_remediation`: the verdict as machine-readable fields (eligible + named blockers), evidence, and split force/unforce/verify SQL. With `as_of` it analyzes a PAST window — anomaly baseline included — and is EXPLORATORY: the findings come back in full but are NOT persisted, which `persisted` / `persistence_note` state on every result | `server_name`, `hours_back` (default 4), `as_of` | + | `get_analysis_facts` | Exposes raw scored facts from the collect+score pipeline — every observation the engine sees with base severity, amplifiers, and metadata (an ANOMALY_* fact's `baseline_confidence` is the baseline's trustworthiness, not a finding's `confidence`) | `server_name`, `hours_back` (default 4), `source` (filter), `min_severity`, `as_of` | | `compare_analysis` | Compares two time periods (e.g., peak vs off-peak, before vs after a change) showing severity deltas for each fact. When NEITHER window produced facts the result is `unavailable` rather than an all-zero comparison, because "nothing to compare" is not "nothing changed"; when only ONE window is empty the payload carries a `caveat` saying so, since every fact then counts as new or resolved by default. `baseline_hours_back` is measured from the comparison window's END, so `as_of` moves BOTH windows together | `server_name`, `hours_back` (default 4), `baseline_hours_back` (default 28), `as_of` | | `audit_config` | Edition-aware configuration audit: evaluates CTFP, MAXDOP, max memory, and max worker threads against best practices | `server_name` | - | `get_analysis_findings` | Retrieves persisted findings from previous analysis runs (the service also analyzes on its own schedule, every 30 minutes per server), deduplicated to one entry per diagnostic chain (`story_path_hash` + `incident_id`): the latest occurrence plus `occurrences`/`first_seen`/`last_seen`/`peak_severity` spanning the window; each remediable finding carries `remediation_command` — the full copy-paste T-SQL remediation (identical to the viewer card), rendered from the persisted action, advisory only and never executed; force-plan findings additionally carry `structured_remediation` (verdict + evidence + split artifacts, machine-readable). Its window is on ANALYSIS TIME, so `as_of` asks what analysis was saying then rather than re-analyzing that window now | `server_name`, `hours_back` (default 24), `as_of` | + | `get_analysis_findings` | Retrieves persisted findings from previous analysis runs (each with `confidence_basis`: rows persisted before `confidence` measured corroboration are labelled `path-shape (pre-#3538)` — under that formula a lone symptom read 1.0, so do not read those as corroborated) (the service also analyzes on its own schedule, every 30 minutes per server), deduplicated to one entry per diagnostic chain (`story_path_hash` + `incident_id`): the latest occurrence plus `occurrences`/`first_seen`/`last_seen`/`peak_severity` spanning the window; each remediable finding carries `remediation_command` — the full copy-paste T-SQL remediation (identical to the viewer card), rendered from the persisted action, advisory only and never executed; force-plan findings additionally carry `structured_remediation` (verdict + evidence + split artifacts, machine-readable). Its window is on ANALYSIS TIME, so `as_of` asks what analysis was saying then rather than re-analyzing that window now | `server_name`, `hours_back` (default 24), `as_of` | | `mute_analysis_finding` | Mutes a finding pattern by story_path_hash so it won't appear in future runs. Reports what the write did: `registered`, and `matched_now` — how many stored findings in scope carry the hash (status `muted_unmatched` when 0: the mute is kept, but check the hash) | `story_path_hash` (required), `server_name`, `reason` | ### Plan-analysis tools diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTools.cs index 3e78b66b5..77ae09866 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTools.cs @@ -36,7 +36,7 @@ namespace PerformanceMonitor.Darling.Service.Mcp; [McpServerToolType] public sealed class DarlingMcpTools { - [McpServerTool(Name = "analyze_server"), Description("Runs the diagnostic inference engine against a server's collected data. Scores wait stats, blocking, memory, config, and other facts, then traverses a relationship graph to build evidence-backed stories about what's wrong and why. Anomaly detection compares the analysis window against 30-day time-bucketed baselines (hour-of-day x day-of-week) to identify deviations that are unusual for this specific time slot, not just unusual overall. Returns structured findings with severity scores, evidence chains, baseline context for anomalies, and recommended next tools to call. A remediable finding also carries remediation_command: the full copy-paste T-SQL remediation (identical to the viewer card), including a two-sided risk-disclosure comment header on destructive changes; it is advisory only and never executed. A force-plan remediation additionally carries structured_remediation: the same decision as machine-readable fields — eligible, named blockers (parameter_sensitivity_cofired, secondary_replica_evidence), evidence numbers, and split force_sql/unforce_sql/verify_sql artifacts — so agents consume the verdict as data instead of parsing comment prose. Set as_of to analyze a PAST window instead of the present — hours_back stays the window's LENGTH, and the anomaly baseline moves with it, so the findings are the ones that window deserves rather than today's findings over older rows. An anchored run is EXPLORATORY: its findings are returned in full but deliberately NOT written to the store, because a finding row is stamped with the time the analysis RAN and would then be read as this server's current state by get_analysis_findings and by the viewer. The result says so in persisted / persistence_note.")] + [McpServerTool(Name = "analyze_server"), Description("Runs the diagnostic inference engine against a server's collected data. Scores wait stats, blocking, memory, config, and other facts, then traverses a relationship graph to build evidence-backed stories about what's wrong and why. Anomaly detection compares the analysis window against 30-day time-bucketed baselines (hour-of-day x day-of-week) to identify deviations that are unusual for this specific time slot, not just unusual overall. Returns structured findings with severity scores, evidence chains, baseline context for anomalies, and recommended next tools to call. Each finding's confidence is an EVIDENCE score, not a probability: 0.20 for the fired symptom alone, plus up to 0.48 for the share of the root fact's amplifier checks (its expected companions) that matched and up to 0.32 for the depth of the evidence chain, so a lone uncorroborated symptom reads 0.20 and a fully corroborated deep chain approaches 1.0; confidence_basis says in words what each value rests on. Rank by severity for impact and by confidence for how much of the engine's own corroboration showed up; do not multiply them. A remediable finding also carries remediation_command: the full copy-paste T-SQL remediation (identical to the viewer card), including a two-sided risk-disclosure comment header on destructive changes; it is advisory only and never executed. A force-plan remediation additionally carries structured_remediation: the same decision as machine-readable fields — eligible, named blockers (parameter_sensitivity_cofired, secondary_replica_evidence), evidence numbers, and split force_sql/unforce_sql/verify_sql artifacts — so agents consume the verdict as data instead of parsing comment prose. Set as_of to analyze a PAST window instead of the present — hours_back stays the window's LENGTH, and the anomaly baseline moves with it, so the findings are the ones that window deserves rather than today's findings over older rows. An anchored run is EXPLORATORY: its findings are returned in full but deliberately NOT written to the store, because a finding row is stamped with the time the analysis RAN and would then be read as this server's current state by get_analysis_findings and by the viewer. The result says so in persisted / persistence_note.")] public static async Task AnalyzeServer( DarlingAnalysisService analysisService, NpgsqlDataSource postgres, @@ -163,6 +163,11 @@ one every caller already assumed. */ { severity = Math.Round(f.Severity, 2), confidence = Math.Round(f.Confidence, 2), + // #3538 A6: what the number rests on. Corroboration-derived since this change + // (matched amplifier share + path depth); a row persisted under the old path-shape + // formula is labelled as such, derived from the finding's own shape at read time + // because the store carries no version marker (no schema change). + confidence_basis = StoryConfidence.DescribeBasis(f.RootFactKey, f.Confidence, f.FactCount), category = f.Category, root_fact = new { key = f.RootFactKey, value = f.RootFactValue }, leaf_fact = f.LeafFactKey != null @@ -214,7 +219,7 @@ one every caller already assumed. */ } } - [McpServerTool(Name = "get_analysis_facts"), Description("Exposes the raw scored facts from the inference engine's collect+score pipeline WITHOUT graph traversal. Shows every observation the engine sees: wait stats as fraction-of-period, blocking rates, config settings, memory stats, plus base severity, final severity after amplifiers, and which amplifiers matched. Use this to understand exactly what the engine is working with, or to investigate facts that didn't reach the severity threshold for findings.")] + [McpServerTool(Name = "get_analysis_facts"), Description("Exposes the raw scored facts from the inference engine's collect+score pipeline WITHOUT graph traversal. Shows every observation the engine sees: wait stats as fraction-of-period, blocking rates, config settings, memory stats, plus base severity, final severity after amplifiers, and which amplifiers matched. For ANOMALY_* facts the metadata carries baseline_confidence — the baseline's own trustworthiness (tier x sample density), which the scorer multiplies into that fact's severity; it is a different quantity from a finding's confidence in analyze_server. Use this to understand exactly what the engine is working with, or to investigate facts that didn't reach the severity threshold for findings.")] public static async Task GetAnalysisFacts( DarlingAnalysisService analysisService, NpgsqlDataSource postgres, @@ -279,8 +284,17 @@ after configuration alone knows audit_config still has it. */ value = Math.Round(f.Value, 6), base_severity = Math.Round(f.BaseSeverity, 4), severity = Math.Round(f.Severity, 4), + // #3538 A6: an anomaly fact's metadata["confidence"] is the BASELINE's confidence (tier x + // density, BaselineBucket.Confidence) — the trustworthiness of the distribution the + // deviation was measured against, which the scorer multiplies into severity. It is not + // the story confidence analyze_server publishes, and one word for two quantities in one + // client session is the confusion this campaign item exists to remove, so the payload + // names it baseline_confidence. The fact's own metadata key is unchanged (the scorer + // reads it); this is a read-time projection only. metadata = f.Metadata.ToDictionary( - m => m.Key, + m => m.Key == "confidence" && string.Equals(f.Source, "anomaly", StringComparison.Ordinal) + ? "baseline_confidence" + : m.Key, m => Math.Round(m.Value, 2)), amplifiers = f.AmplifierResults.Count > 0 ? f.AmplifierResults.Select(a => new @@ -707,7 +721,7 @@ the collector missed changes nothing about what the server is configured to. */ } } - [McpServerTool(Name = "get_analysis_findings"), Description("Gets persisted findings from previous analysis runs without running a new analysis, deduplicated to one entry per diagnostic chain (story_path_hash + incident_id) - the engine re-persists the same stories every cycle, so each entry is the chain's LATEST occurrence plus occurrence stats (occurrences, first_seen, last_seen, peak_severity) spanning the window. Use this to review historical findings or check if anything has changed since the last analysis. A remediable finding carries remediation_command: the full copy-paste T-SQL remediation (identical to the viewer card), rendered from the finding's persisted action and including a two-sided risk-disclosure comment header on destructive changes; it is advisory only and never executed. A force-plan remediation additionally carries structured_remediation: the same decision as machine-readable fields — eligible, named blockers (parameter_sensitivity_cofired, secondary_replica_evidence), evidence numbers, and split force_sql/unforce_sql/verify_sql artifacts — so agents consume the verdict as data instead of parsing comment prose. Set include_drilldown to also return each chain's persisted evidence rows (the specific plans/queries behind the finding, capped at write time with an explicit _truncation_note; null on findings persisted before the column existed).")] + [McpServerTool(Name = "get_analysis_findings"), Description("Gets persisted findings from previous analysis runs without running a new analysis, deduplicated to one entry per diagnostic chain (story_path_hash + incident_id) - the engine re-persists the same stories every cycle, so each entry is the chain's LATEST occurrence plus occurrence stats (occurrences, first_seen, last_seen, peak_severity) spanning the window. Use this to review historical findings or check if anything has changed since the last analysis. Each finding's confidence is an EVIDENCE score (see analyze_server): 0.20 for the fired symptom alone, plus corroboration from matched amplifier checks and chain depth. Rows persisted before this definition carried a PATH-LENGTH statistic under the same name, with a lone symptom at 1.0 — confidence_basis labels those rows path-shape (pre-#3538) and they must not be read as corroborated. A remediable finding carries remediation_command: the full copy-paste T-SQL remediation (identical to the viewer card), rendered from the finding's persisted action and including a two-sided risk-disclosure comment header on destructive changes; it is advisory only and never executed. A force-plan remediation additionally carries structured_remediation: the same decision as machine-readable fields — eligible, named blockers (parameter_sensitivity_cofired, secondary_replica_evidence), evidence numbers, and split force_sql/unforce_sql/verify_sql artifacts — so agents consume the verdict as data instead of parsing comment prose. Set include_drilldown to also return each chain's persisted evidence rows (the specific plans/queries behind the finding, capped at write time with an explicit _truncation_note; null on findings persisted before the column existed).")] public static async Task GetAnalysisFindings( DarlingAnalysisService analysisService, NpgsqlDataSource postgres, @@ -802,6 +816,11 @@ stats. The store keeps every row — this shapes the read only. */ analysis_time = f.AnalysisTime.ToString("o"), severity = Math.Round(f.Severity, 2), confidence = Math.Round(f.Confidence, 2), + // #3538 A6: what the number rests on. Corroboration-derived since this change + // (matched amplifier share + path depth); a row persisted under the old path-shape + // formula is labelled as such, derived from the finding's own shape at read time + // because the store carries no version marker (no schema change). + confidence_basis = StoryConfidence.DescribeBasis(f.RootFactKey, f.Confidence, f.FactCount), category = f.Category, root_fact = new { key = f.RootFactKey, value = f.RootFactValue }, leaf_fact = f.LeafFactKey != null diff --git a/Lite.Tests/McpAnalysisFindingsCommandTests.cs b/Lite.Tests/McpAnalysisFindingsCommandTests.cs index 2c2a24dce..22c59c3c5 100644 --- a/Lite.Tests/McpAnalysisFindingsCommandTests.cs +++ b/Lite.Tests/McpAnalysisFindingsCommandTests.cs @@ -253,6 +253,76 @@ the window-covering cap carries a present-but-null truncation_note. */ single.GetProperty("last_seen").GetString()); } + /// + /// #3538 A6: every finding carries confidence_basis beside confidence, and the basis is + /// derived from the row's own shape at read time — the store has no version column, so a row + /// persisted under the pre-change path-shape formula (a lone symptom at 1.0) is recognised by its + /// VALUE and labelled as such, while a corroboration-derived value names the formula, and the pileup + /// detector's by-construction 1.0 is named by its root key rather than called legacy. Pinned through + /// Lite's OWN tool method and serializer (the Darling twin's equivalent lives in the gated e2e) so a + /// dropped projection field is caught in a default CI lane. + /// + [Fact] + public async Task GetAnalysisFindings_PublishesConfidenceBasis_LabellingLegacyRowsByTheirShape() + { + var store = new FindingStore(_duckDb); + var analysisTime = DateTime.UtcNow; + var context = new AnalysisContext + { + ServerId = _serverId, + ServerName = "TestServer", + TimeRangeStart = analysisTime.AddHours(-4), + TimeRangeEnd = analysisTime + }; + + /* A pre-#3538 row: lone symptom, confidence 1.0 — the legacy formula's signature. */ + var legacy = MakeFinding( + findingId: 930001, analysisTime, severity: 0.72, + rootFactKey: "SCH_M", storyPathHash: "basis_legacy_hash", remediation: null); + legacy.Confidence = 1.0; + legacy.FactCount = 1; + + /* A post-#3538 row: two-node chain, no catalogue — 0.20 + 0.80 × 0.5. */ + var corroborated = MakeFinding( + findingId: 930002, analysisTime, severity: 1.14, + rootFactKey: "IO_WRITE_LATENCY_MS", storyPathHash: "basis_corroborated_hash", remediation: null); + corroborated.StoryPath = "IO_WRITE_LATENCY_MS → RUNNING_JOBS"; + corroborated.FactCount = 2; + corroborated.Confidence = StoryConfidence.Compute(0, 0, 2); + + /* The pileup detector's story: 1.0 by construction, one node — the shape a legacy row has, told + apart by its root key. */ + var pileup = MakeFinding( + findingId: 930003, analysisTime, severity: 1.8, + rootFactKey: SameStatementPileupDetector.RootFactKey, storyPathHash: "basis_pileup_hash", remediation: null); + pileup.Confidence = 1.0; + pileup.FactCount = 1; + + await store.InsertFindingsAsync(new List { legacy, corroborated, pileup }, context); + + var json = await McpAnalysisTools.GetAnalysisFindings( + new AnalysisService(_duckDb), _serverManager, "TestServer", 24); + + using var doc = JsonDocument.Parse(json); + var findings = doc.RootElement.GetProperty("findings").EnumerateArray().ToList(); + Assert.Equal(3, findings.Count); + Assert.All(findings, f => Assert.True(f.TryGetProperty("confidence_basis", out _), + "every finding must expose confidence_basis beside confidence")); + + var legacyOut = findings.Single(f => f.GetProperty("story_path_hash").GetString() == "basis_legacy_hash"); + Assert.Equal(1.0, legacyOut.GetProperty("confidence").GetDouble()); + Assert.StartsWith("path-shape (pre-#3538)", legacyOut.GetProperty("confidence_basis").GetString(), StringComparison.Ordinal); + + var corroboratedOut = findings.Single(f => f.GetProperty("story_path_hash").GetString() == "basis_corroborated_hash"); + Assert.Equal(0.6, corroboratedOut.GetProperty("confidence").GetDouble()); + var basis = corroboratedOut.GetProperty("confidence_basis").GetString()!; + Assert.StartsWith("corroboration (#3538)", basis, StringComparison.Ordinal); + Assert.Contains("2-node story path", basis, StringComparison.Ordinal); + + var pileupOut = findings.Single(f => f.GetProperty("story_path_hash").GetString() == "basis_pileup_hash"); + Assert.StartsWith("detector-measured", pileupOut.GetProperty("confidence_basis").GetString(), StringComparison.Ordinal); + } + private AnalysisFinding MakeFinding( long findingId, DateTime analysisTime, double severity, string rootFactKey, string storyPathHash, RemediationAction? remediation) => diff --git a/Lite.Tests/McpMissMessageParityPinTests.cs b/Lite.Tests/McpMissMessageParityPinTests.cs index 1c92562a6..b83afae06 100644 --- a/Lite.Tests/McpMissMessageParityPinTests.cs +++ b/Lite.Tests/McpMissMessageParityPinTests.cs @@ -159,6 +159,19 @@ the user round a loop that never terminates. The correction has to be identical but this sentence is supplied by each tool body at its own call site, lives twice, and is exactly what drifts. A tree missing it is a tree whose get_running_jobs never grew the arm at all. */ "tables are not reachable to a monitoring login at all and no grant changes that.", + + /* The confidence DEFINITION (#3538 A6). The confidence_basis STRING itself is built by the shared + StoryConfidence.DescribeBasis and is byte-identical by construction, so it does not belong here; + what lives twice is the tool-description sentence teaching a caller what the number is — and + that is exactly the text that would drift into one SKU saying "evidence score" while the other + still implied a probability. analyze_server, get_analysis_findings (the legacy-row warning) and + get_analysis_facts (the baseline_confidence disambiguation), plus the instructions rows. */ + "confidence is an EVIDENCE score, not a probability: 0.20 for the fired symptom alone, plus up to 0.48 for the share of the root fact's amplifier checks (its expected companions) that matched and up to 0.32 for the depth of the evidence chain", + "Rank by severity for impact and by confidence for how much of the engine's own corroboration showed up; do not multiply them.", + "Rows persisted before this definition carried a PATH-LENGTH statistic under the same name, with a lone symptom at 1.0 — confidence_basis labels those rows path-shape (pre-#3538) and they must not be read as corroborated.", + "For ANOMALY_* facts the metadata carries baseline_confidence — the baseline's own trustworthiness (tier x sample density), which the scorer multiplies into that fact's severity; it is a different quantity from a finding's confidence in analyze_server.", + "Each finding's `confidence` is an EVIDENCE score (0.20 for the symptom alone, more as the root fact's amplifier checks match and the chain deepens — a lone uncorroborated symptom is 0.20, never 1.0) and `confidence_basis` says in words what it rests on", + "(an ANOMALY_* fact's `baseline_confidence` is the baseline's trustworthiness, not a finding's `confidence`)", }; [Theory] diff --git a/Lite.Tests/StoryConfidenceTests.cs b/Lite.Tests/StoryConfidenceTests.cs new file mode 100644 index 000000000..6e7de7484 --- /dev/null +++ b/Lite.Tests/StoryConfidenceTests.cs @@ -0,0 +1,407 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Linq; +using PerformanceMonitor.Analysis; +using Xunit; + +namespace PerformanceMonitorLite.Tests; + +/// +/// #3538 A6 + A9 — the two engine changes that make a finding's confidence mean corroboration and +/// make the nightly maintenance window one incident. +/// +/// A6. Confidence was (n-1)/n over the story path with a lone symptom at 1.0 — the +/// LEAST evidenced finding carried the HIGHEST confidence, exported under an evidence name. It is now an +/// evidence statistic (): a floor for the fired symptom, the share of the root +/// fact's amplifier checks that matched, and the depth of the traversed chain. These pins hold the +/// properties the formula was chosen for — monotonic in corroboration, lone symptom below any corroborated +/// chain, never 1.0, the worked examples in the class remarks — and the one property the read-time basis +/// label DEPENDS on: the new formula never lands on a legacy value for the same path length, which is +/// what lets confidence_basis call a pre-change row "path-shape" without a schema marker. +/// +/// A9. Nothing in the relationship graph touched SCH_M or RUNNING_JOBS, so the rebuild's +/// schema-lock waits, its long-running job and its write-latency / log-flush pair were two or three +/// unlinked single-node cards every night. Three symptom → job edges, gated on the job having FIRED, +/// let fold them into one incident; the same +/// predicate drives the advice sentence that names the job on each card. +/// +/// Pure logic over hand-built facts — no store, no fixture — so every value here was also executed +/// on the development machine before CI (the brief's execute-your-pins rule), not just compiled. +/// +public sealed class StoryConfidenceTests +{ + /* ── A6: the formula's properties ── */ + + /// + /// One more matched amplifier, or one more node on the path, never LOWERS confidence — the property + /// that makes "more corroboration" and "higher confidence" the same direction, which the legacy + /// formula inverted at the lone-symptom step. Enumerated over every catalogue size a real key has + /// (the largest amplifier list is six entries; twelve is headroom) and every path length the traversal + /// can build — the root plus MaxPathDepth hops, i.e. (11), + /// not a round ten: the engine's depth constant counts HOPS, and the first review of this lane caught + /// the off-by-one in the pins' upper bound. + /// + [Fact] + public void Compute_IsMonotonic_InMatchedAmplifiers_AndInPathDepth() + { + for (var defined = 0; defined <= 12; defined++) + for (var n = 1; n <= StoryConfidence.MaxPathNodes; n++) + for (var matched = 0; matched <= defined; matched++) + { + var here = StoryConfidence.Compute(matched, defined, n); + if (matched < defined) + Assert.True(StoryConfidence.Compute(matched + 1, defined, n) >= here, + $"matching one more amplifier lowered confidence at matched={matched} defined={defined} n={n}"); + Assert.True(StoryConfidence.Compute(matched, defined, n + 1) >= here, + $"one more path node lowered confidence at matched={matched} defined={defined} n={n}"); + } + } + + /// + /// The A6 defect, stated as the ordering it violated: a lone symptom with no corroboration sits at + /// the floor (0.20) whatever its catalogue size, and BELOW every corroborated shape — a two-node + /// chain with nothing else, a lone symptom with one match, a two-node chain on a catalogued key + /// with nothing matched. Under the old formula the lone symptom was 1.0 and outranked all of them. + /// + [Fact] + public void Compute_LoneUncorroboratedSymptom_SitsAtTheFloor_BelowEveryCorroboratedShape() + { + for (var defined = 0; defined <= 12; defined++) + Assert.Equal(StoryConfidence.Floor, StoryConfidence.Compute(0, defined, 1), precision: 9); + + var lone = StoryConfidence.Compute(0, 3, 1); + Assert.True(StoryConfidence.Compute(0, 0, 2) > lone, "an uncatalogued two-node chain must beat a lone symptom"); + Assert.True(StoryConfidence.Compute(1, 3, 1) > lone, "one matched amplifier must beat none"); + Assert.True(StoryConfidence.Compute(0, 3, 2) > lone, "a catalogued two-node chain with no matches must still beat a lone symptom"); + } + + /// + /// Nothing the formula builds is 1.0: path depth (n-1)/n never reaches 1, so even a fully matched + /// catalogue on the deepest path the engine allows stays under it. 1.0 is reserved for the two + /// by-construction stories (absolution, the pileup detector) and for legacy rows — which is what + /// lets a lone-symptom 1.0 be recognised as legacy at read time. + /// + [Fact] + public void Compute_NeverReachesOne() + { + var max = 0.0; + for (var defined = 0; defined <= 12; defined++) + for (var n = 1; n <= StoryConfidence.MaxPathNodes; n++) + max = Math.Max(max, StoryConfidence.Compute(defined, defined, n)); + Assert.True(max < 1.0, $"the formula reached {max}"); + /* The two ceilings at the deepest path the traversal builds (11 nodes = root + 10 hops): + uncatalogued 0.20 + 0.80 × 10/11, catalogued 0.20 + 0.48 + 0.32 × 10/11. */ + Assert.Equal(11, StoryConfidence.MaxPathNodes); + Assert.Equal(0.2 + 0.8 * 10.0 / 11.0, StoryConfidence.Compute(0, 0, StoryConfidence.MaxPathNodes), precision: 9); + Assert.Equal(0.2 + 0.48 + 0.32 * 10.0 / 11.0, StoryConfidence.Compute(5, 5, StoryConfidence.MaxPathNodes), precision: 9); + } + + /// + /// The worked examples from the remarks, as the class states them. + /// A comment that quotes numbers is a claim; this is the claim executed. + /// + [Theory] + [InlineData(0, 0, 1, 0.20)] // lone SCH_M: no catalogue, one node — the floor + [InlineData(0, 3, 1, 0.20)] // lone PAGEIOLATCH_SH, 0 of 3 matched — the floor + [InlineData(3, 3, 1, 0.68)] // lone PAGEIOLATCH_SH, 3 of 3 matched: 0.20 + 0.48 + [InlineData(2, 3, 3, 0.7333333333)] // PAGEIOLATCH_SH → RESOURCE_SEMAPHORE → MEMORY_GRANT_PENDING, 2 of 3: 0.20 + 0.32 + 0.32 × 2/3 + [InlineData(0, 0, 2, 0.60)] // SCH_M → RUNNING_JOBS: no catalogue, two nodes: 0.20 + 0.80 × 0.5 + [InlineData(0, 0, 10, 0.92)] // ten-node uncatalogued chain: 0.20 + 0.80 × 0.9 + [InlineData(0, 5, 10, 0.488)] // ten-node chain whose five checks all came back false: 0.20 + 0.32 × 0.9 + public void Compute_WorkedExamples(int matched, int defined, int pathLength, double expected) + { + Assert.Equal(expected, StoryConfidence.Compute(matched, defined, pathLength), precision: 9); + } + + /// + /// The property the read-time basis label rests on: for every path length the engine can build and + /// every catalogue size a key can have, the new value is never the legacy path-shape value for that + /// same length. Without this, confidence_basis could call a fresh corroboration score + /// "path-shape (pre-#3538)" — or a legacy row corroborated — and a schema marker would be the only + /// honest alternative (a rung this campaign is not permitted). The legacy shape is exactly 1.0 or + /// (n-1)/n; the formula's 0.20 floor keeps every lone value away from 1.0, and no small-integer + /// amplifier share puts a multi-node value on (n-1)/n. Enumerated to the engine's REAL maximum path + /// ( = root + MaxPathDepth hops = 11), not to ten. + /// + [Fact] + public void Compute_NeverCollidesWithTheLegacyPathShape_SoTheBasisCanBeReDerivedAtReadTime() + { + for (var n = 1; n <= StoryConfidence.MaxPathNodes; n++) + for (var defined = 0; defined <= 12; defined++) + for (var matched = 0; matched <= defined; matched++) + { + var value = StoryConfidence.Compute(matched, defined, n); + Assert.False(StoryConfidence.IsLegacyPathShape(value, n), + $"Compute({matched},{defined},{n}) = {value} equals the legacy path-shape value for n={n}"); + } + } + + /* ── A6: the basis string's arms ── */ + + [Fact] + public void DescribeBasis_LegacyRows_AreLabelledPathShape_AndNeverReadAsCorroborated() + { + var lone = StoryConfidence.DescribeBasis("SCH_M", 1.0, 1); + Assert.Contains("path-shape (pre-#3538)", lone, StringComparison.Ordinal); + Assert.Contains("UNCORROBORATED", lone, StringComparison.Ordinal); + Assert.DoesNotContain("corroboration (#3538)", lone, StringComparison.Ordinal); + + /* The multi-node legacy values too: (n-1)/n for n = 2, 3, 4. */ + Assert.Contains("path-shape (pre-#3538)", StoryConfidence.DescribeBasis("CXPACKET", 0.5, 2), StringComparison.Ordinal); + Assert.Contains("path-shape (pre-#3538)", StoryConfidence.DescribeBasis("CXPACKET", 2.0 / 3.0, 3), StringComparison.Ordinal); + Assert.Contains("path-shape (pre-#3538)", StoryConfidence.DescribeBasis("CXPACKET", 0.75, 4), StringComparison.Ordinal); + } + + [Fact] + public void DescribeBasis_CorroborationRows_NameTheFormula_AndThePathLength() + { + var basis = StoryConfidence.DescribeBasis("PAGEIOLATCH_SH", StoryConfidence.Compute(2, 3, 3), 3); + Assert.Contains("corroboration (#3538)", basis, StringComparison.Ordinal); + Assert.Contains("3-node story path", basis, StringComparison.Ordinal); + Assert.Contains("0.20 for the fired symptom", basis, StringComparison.Ordinal); + Assert.DoesNotContain("pre-#3538", basis, StringComparison.Ordinal); + + /* A fresh lone symptom (0.20) is corroboration-derived, not legacy — the floor is not 1.0. */ + Assert.Contains("corroboration (#3538)", StoryConfidence.DescribeBasis("SCH_M", StoryConfidence.Compute(0, 0, 1), 1), StringComparison.Ordinal); + } + + /// + /// The two 1.0s that are NOT legacy: the absolution story and the same-statement pileup detector's + /// story both carry 1.0 by construction and are named by root key, so a reader is never told a + /// directly measured convoy is an uncorroborated pre-change row. + /// + [Fact] + public void DescribeBasis_ByConstructionOnes_AreNamed_NotCalledLegacy() + { + var absolution = StoryConfidence.DescribeBasis(StoryConfidence.AbsolutionRootKey, 1.0, 1); + Assert.StartsWith("absolution:", absolution, StringComparison.Ordinal); + Assert.DoesNotContain("pre-#3538", absolution, StringComparison.Ordinal); + + var pileup = StoryConfidence.DescribeBasis(SameStatementPileupDetector.RootFactKey, 1.0, 1); + Assert.StartsWith("detector-measured:", pileup, StringComparison.Ordinal); + Assert.DoesNotContain("pre-#3538", pileup, StringComparison.Ordinal); + } + + /* ── A6: through the engine ── */ + + /// + /// End to end through FactScorer + InferenceEngine: a lone SCH_M that cleared its threshold roots a + /// story at the floor, and a PAGEIOLATCH_SH whose companions showed up — read latency over 20 ms and + /// memory-grant waiters (2 of its 3 amplifiers) — roots a story that traverses its highest-severity + /// active edge (executed here: PAGEIOLATCH_SH → IO_READ_LATENCY_MS, two nodes, 0.20 + 0.32 + 0.16 = + /// 0.68) and whose confidence is exactly the formula over its own root fact and path. The story's + /// confidence must equal over the facts the engine + /// actually saw, so the pin cannot drift from the code it describes. + /// + [Fact] + public void BuildStories_LoneSymptomScoresTheFloor_CorroboratedChainScoresHigher_AndBothMatchTheFormula() + { + var engine = new InferenceEngine(new RelationshipGraph()); + + var lone = new List { Wait("SCH_M", 0.05) }; + new FactScorer().ScoreAll(lone); + var loneStory = Assert.Single(engine.BuildStories(lone)); + Assert.Equal("SCH_M", loneStory.StoryPath); + Assert.Equal(StoryConfidence.Floor, loneStory.Confidence, precision: 9); + Assert.Empty(lone[0].AmplifierResults); // SCH_M has no catalogue: the uncatalogued arm + + var chain = new List + { + Wait("PAGEIOLATCH_SH", 0.30), + Wait("RESOURCE_SEMAPHORE", 0.05), + new() { Source = "io", Key = "IO_READ_LATENCY_MS", Value = 25 }, + new() { Source = "memory", Key = "MEMORY_GRANT_PENDING", Value = 3 }, + }; + new FactScorer().ScoreAll(chain); + var stories = engine.BuildStories(chain); + var root = stories.First(s => s.RootFactKey == "PAGEIOLATCH_SH"); + var rootFact = chain.Single(f => f.Key == "PAGEIOLATCH_SH"); + + Assert.Equal(3, rootFact.AmplifierResults.Count); + Assert.Equal(2, rootFact.AmplifierResults.Count(a => a.Matched)); // read latency ≥ 20 ms, grant waiters ≥ 1; SOS absent + Assert.True(root.Path.Count >= 2, $"expected a traversed chain, got {root.StoryPath}"); + Assert.Equal(StoryConfidence.Compute(rootFact, root.Path.Count), root.Confidence, precision: 9); + Assert.Equal(0.68, root.Confidence, precision: 9); // executed: PAGEIOLATCH_SH → IO_READ_LATENCY_MS + Assert.True(root.Confidence > loneStory.Confidence, + $"corroborated {root.StoryPath} ({root.Confidence:F3}) must outrank a lone symptom ({loneStory.Confidence:F3})"); + Assert.True(root.Confidence < 1.0); + } + + /* ── A9: the maintenance edges ── */ + + /// + /// Each symptom → job edge fires on RUNNING_JOBS having FIRED (base severity above zero — at least one + /// job running past its own history) and not on the fact merely being present: the collector emits + /// RUNNING_JOBS whenever any job is running, and an edge on presence would pin every schema-lock or + /// write-latency finding to whatever routine job happened to be executing. + /// + [Theory] + [InlineData("SCH_M")] + [InlineData("IO_WRITE_LATENCY_MS")] + [InlineData("WRITELOG")] + public void MaintenanceEdges_FireOnlyWhenRunningJobsFired_NotOnPresence(string symptom) + { + var graph = new RelationshipGraph(); + + var fired = new Dictionary + { + ["RUNNING_JOBS"] = new() { Key = "RUNNING_JOBS", Source = "jobs", Value = 1, BaseSeverity = 0.5, Severity = 0.5 } + }; + Assert.Contains(graph.GetActiveEdges(symptom, fired), e => e.Destination == "RUNNING_JOBS"); + + var presentButQuiet = new Dictionary + { + ["RUNNING_JOBS"] = new() { Key = "RUNNING_JOBS", Source = "jobs", Value = 0, BaseSeverity = 0, Severity = 0 } + }; + Assert.DoesNotContain(graph.GetActiveEdges(symptom, presentButQuiet), e => e.Destination == "RUNNING_JOBS"); + + Assert.DoesNotContain(graph.GetActiveEdges(symptom, new Dictionary()), e => e.Destination == "RUNNING_JOBS"); + } + + /// + /// The nightly rebuild, as the engine now reads it: SCH_M, IO_WRITE_LATENCY_MS, WRITELOG and a + /// RUNNING_JOBS that fired all score, and returns + /// ONE incident holding every story — the job is on the write-latency story's path (the highest root + /// follows its highest-severity active edge, and the job at 1.0 outranks WRITELOG), and the schema-lock + /// and log-flush stories union across their own edges onto it. Before the edges this was three + /// incidents. The same facts with the job NOT running long stay apart: SCH_M has no active edge to + /// anything present and is its own incident, while the write pair still unions on the pre-existing + /// IO_WRITE_LATENCY_MS ↔ WRITELOG edges — so the fold is the job's doing, not a side effect. + /// + [Fact] + public void ClusterIntoIncidents_FoldsTheRebuildIntoOneIncident_WhenTheJobFired_AndKeepsSchMApartWhenItDidNot() + { + var engine = new InferenceEngine(new RelationshipGraph()); + + var withJob = RebuildFacts(runningLong: 3); + new FactScorer().ScoreAll(withJob); + var stories = engine.BuildStories(withJob); + Assert.Equal(3, stories.Count); + Assert.Contains(stories, s => s.StoryPath == "IO_WRITE_LATENCY_MS → RUNNING_JOBS"); + Assert.Contains(stories, s => s.StoryPath == "SCH_M"); + Assert.Contains(stories, s => s.StoryPath == "WRITELOG"); + + var incidents = engine.ClusterIntoIncidents(stories, withJob); + var incident = Assert.Single(incidents); + Assert.Equal(3, incident.Count); + + var withoutJob = RebuildFacts(runningLong: 0); + new FactScorer().ScoreAll(withoutJob); + stories = engine.BuildStories(withoutJob); + Assert.Equal(2, stories.Count); + Assert.Contains(stories, s => s.StoryPath == "IO_WRITE_LATENCY_MS → WRITELOG"); + Assert.Contains(stories, s => s.StoryPath == "SCH_M"); + + incidents = engine.ClusterIntoIncidents(stories, withoutJob); + Assert.Equal(2, incidents.Count); + var schM = Assert.Single(incidents, i => i.Any(s => s.RootFactKey == "SCH_M")); + Assert.Single(schM); + } + + /* ── A9: the prose ── */ + + /// + /// The SCH_M card names the job when RUNNING_JOBS fired — the collected counts and overrun, and the + /// statement that the two cards are one incident — and says plainly that no job was running long + /// when it was not, instead of the static "open the Running Jobs tab to see whether". + /// + [Fact] + public void SchMAdvice_NamesTheLinkedJob_WhenRunningJobsFired_AndSaysNoJobOtherwise() + { + var withJob = RebuildFacts(runningLong: 3); + new FactScorer().ScoreAll(withJob); + var linked = FactAdvice.Compose("SCH_M", withJob.ToFactLookup()); + Assert.NotNull(linked); + Assert.Contains("RUNNING_JOBS fired in the same window", linked!.Investigation, StringComparison.Ordinal); + Assert.Contains("3 Agent jobs running well past normal duration", linked.Investigation, StringComparison.Ordinal); + Assert.Contains("400% of its historical average", linked.Investigation, StringComparison.Ordinal); + Assert.Contains("ONE incident, not two", linked.Investigation, StringComparison.Ordinal); + Assert.Contains("Agent job runs long", linked.Headline, StringComparison.Ordinal); + + var withoutJob = RebuildFacts(runningLong: 0); + new FactScorer().ScoreAll(withoutJob); + var alone = FactAdvice.Compose("SCH_M", withoutJob.ToFactLookup()); + Assert.NotNull(alone); + Assert.DoesNotContain("RUNNING_JOBS fired", alone!.Investigation, StringComparison.Ordinal); + Assert.Contains("No Agent job was running past its normal duration this window", alone.Investigation, StringComparison.Ordinal); + } + + /// + /// The other three cards carry the same link: the write-latency and log-flush remediations name the + /// job beside their storage advice, and the job card names the symptoms whose edges point at it. + /// + [Fact] + public void WriteLatency_Writelog_AndRunningJobs_AdviceCarryTheLink_BothWays() + { + var facts = RebuildFacts(runningLong: 3); + new FactScorer().ScoreAll(facts); + var lookup = facts.ToFactLookup(); + + var io = FactAdvice.Compose("IO_WRITE_LATENCY_MS", lookup); + Assert.Contains("RUNNING_JOBS fired in the same window", io!.Remediation, StringComparison.Ordinal); + + var writelog = FactAdvice.Compose("WRITELOG", lookup); + Assert.Contains("RUNNING_JOBS fired in the same window", writelog!.Remediation, StringComparison.Ordinal); + + var jobs = FactAdvice.Compose("RUNNING_JOBS", lookup); + Assert.Contains("SCH_M (schema-modification lock waits", jobs!.Investigation, StringComparison.Ordinal); + Assert.Contains("IO_WRITE_LATENCY_MS (", jobs.Investigation, StringComparison.Ordinal); + Assert.Contains("WRITELOG (", jobs.Investigation, StringComparison.Ordinal); + + /* And none of it when the job did not fire. */ + var quiet = RebuildFacts(runningLong: 0); + new FactScorer().ScoreAll(quiet); + Assert.DoesNotContain("RUNNING_JOBS fired", FactAdvice.Compose("IO_WRITE_LATENCY_MS", quiet.ToFactLookup())!.Remediation, StringComparison.Ordinal); + Assert.DoesNotContain("RUNNING_JOBS fired", FactAdvice.Compose("WRITELOG", quiet.ToFactLookup())!.Remediation, StringComparison.Ordinal); + } + + /* ── Helpers ── */ + + /// + /// The rebuild window: SCH_M at 5% of period (0.72 on its (0.01, 0.10) ramp), write latency 25 ms + /// (0.875 on (10, 30), then the WRITELOG amplifier), WRITELOG at 30% (0.60 on (0.25, 0.50)), and a + /// RUNNING_JOBS fact whose value is the running-long count — 3 saturates its (1, 3) ramp at 1.0; 0 + /// scores nothing and drops out of the working set, which is the "job present but not long" case. + /// Wait facts carry the metadata the advice composers read (wait_time_ms), the I/O fact the average + /// its composer reads, and the job fact the counts its sentence states. + /// + private static List RebuildFacts(int runningLong) => + [ + Wait("SCH_M", 0.05), + Wait("WRITELOG", 0.30), + new() + { + Source = "io", Key = "IO_WRITE_LATENCY_MS", Value = 25, + Metadata = new() { ["avg_write_latency_ms"] = 25, ["total_writes"] = 500_000, ["total_stall_write_ms"] = 12_500_000 } + }, + new() + { + Source = "jobs", Key = "RUNNING_JOBS", Value = runningLong, + Metadata = new() + { + ["running_count"] = 5, ["running_long_count"] = runningLong, + ["max_percent_of_average"] = runningLong > 0 ? 400 : 0, + ["max_duration_seconds"] = runningLong > 0 ? 10_800 : 0 + } + }, + ]; + + private static Fact Wait(string key, double fraction) => + new() + { + Source = "waits", Key = key, Value = fraction, + Metadata = new() + { + ["wait_time_ms"] = fraction * 4 * 3_600_000, ["waiting_tasks_count"] = 1_000, + ["avg_ms_per_wait"] = fraction * 4 * 3_600, ["period_duration_ms"] = 4 * 3_600_000 + } + }; +} diff --git a/Lite/Mcp/McpAnalysisTools.cs b/Lite/Mcp/McpAnalysisTools.cs index 413ee08bd..9dfd619b7 100644 --- a/Lite/Mcp/McpAnalysisTools.cs +++ b/Lite/Mcp/McpAnalysisTools.cs @@ -11,7 +11,7 @@ namespace PerformanceMonitorLite.Mcp; [McpServerToolType] public sealed class McpAnalysisTools { - [McpServerTool(Name = "analyze_server"), Description("Runs the diagnostic inference engine against a server's collected data. Scores wait stats, blocking, memory, config, and other facts, then traverses a relationship graph to build evidence-backed stories about what's wrong and why. Anomaly detection compares the analysis window against 30-day time-bucketed baselines (hour-of-day x day-of-week) to identify deviations that are unusual for this specific time slot, not just unusual overall. Returns structured findings with severity scores, evidence chains, baseline context for anomalies, and recommended next tools to call. A remediable finding also carries remediation_command: the full copy-paste T-SQL remediation (identical to the viewer card), including a two-sided risk-disclosure comment header on destructive changes; it is advisory only and never executed. A force-plan remediation additionally carries structured_remediation: the same decision as machine-readable fields — eligible, named blockers (parameter_sensitivity_cofired, secondary_replica_evidence), evidence numbers, and split force_sql/unforce_sql/verify_sql artifacts — so agents consume the verdict as data instead of parsing comment prose. Set as_of to analyze a PAST window instead of the present — hours_back stays the window's LENGTH, and the anomaly baseline moves with it, so the findings are the ones that window deserves rather than today's findings over older rows. An anchored run is EXPLORATORY: its findings are returned in full but deliberately NOT written to the store, because a finding row is stamped with the time the analysis RAN and would then be read as this server's current state by get_analysis_findings and by the viewer. The result says so in persisted / persistence_note.")] + [McpServerTool(Name = "analyze_server"), Description("Runs the diagnostic inference engine against a server's collected data. Scores wait stats, blocking, memory, config, and other facts, then traverses a relationship graph to build evidence-backed stories about what's wrong and why. Anomaly detection compares the analysis window against 30-day time-bucketed baselines (hour-of-day x day-of-week) to identify deviations that are unusual for this specific time slot, not just unusual overall. Returns structured findings with severity scores, evidence chains, baseline context for anomalies, and recommended next tools to call. Each finding's confidence is an EVIDENCE score, not a probability: 0.20 for the fired symptom alone, plus up to 0.48 for the share of the root fact's amplifier checks (its expected companions) that matched and up to 0.32 for the depth of the evidence chain, so a lone uncorroborated symptom reads 0.20 and a fully corroborated deep chain approaches 1.0; confidence_basis says in words what each value rests on. Rank by severity for impact and by confidence for how much of the engine's own corroboration showed up; do not multiply them. A remediable finding also carries remediation_command: the full copy-paste T-SQL remediation (identical to the viewer card), including a two-sided risk-disclosure comment header on destructive changes; it is advisory only and never executed. A force-plan remediation additionally carries structured_remediation: the same decision as machine-readable fields — eligible, named blockers (parameter_sensitivity_cofired, secondary_replica_evidence), evidence numbers, and split force_sql/unforce_sql/verify_sql artifacts — so agents consume the verdict as data instead of parsing comment prose. Set as_of to analyze a PAST window instead of the present — hours_back stays the window's LENGTH, and the anomaly baseline moves with it, so the findings are the ones that window deserves rather than today's findings over older rows. An anchored run is EXPLORATORY: its findings are returned in full but deliberately NOT written to the store, because a finding row is stamped with the time the analysis RAN and would then be read as this server's current state by get_analysis_findings and by the viewer. The result says so in persisted / persistence_note.")] public static async Task AnalyzeServer( AnalysisService analysisService, ServerManager serverManager, @@ -138,6 +138,11 @@ one every caller already assumed. */ { severity = Math.Round(f.Severity, 2), confidence = Math.Round(f.Confidence, 2), + // #3538 A6: what the number rests on. Corroboration-derived since this change + // (matched amplifier share + path depth); a row persisted under the old path-shape + // formula is labelled as such, derived from the finding's own shape at read time + // because the store carries no version marker (no schema change). + confidence_basis = StoryConfidence.DescribeBasis(f.RootFactKey, f.Confidence, f.FactCount), category = f.Category, root_fact = new { key = f.RootFactKey, value = f.RootFactValue }, leaf_fact = f.LeafFactKey != null @@ -189,7 +194,7 @@ one every caller already assumed. */ } } - [McpServerTool(Name = "get_analysis_facts"), Description("Exposes the raw scored facts from the inference engine's collect+score pipeline WITHOUT graph traversal. Shows every observation the engine sees: wait stats as fraction-of-period, blocking rates, config settings, memory stats, plus base severity, final severity after amplifiers, and which amplifiers matched. Use this to understand exactly what the engine is working with, or to investigate facts that didn't reach the severity threshold for findings.")] + [McpServerTool(Name = "get_analysis_facts"), Description("Exposes the raw scored facts from the inference engine's collect+score pipeline WITHOUT graph traversal. Shows every observation the engine sees: wait stats as fraction-of-period, blocking rates, config settings, memory stats, plus base severity, final severity after amplifiers, and which amplifiers matched. For ANOMALY_* facts the metadata carries baseline_confidence — the baseline's own trustworthiness (tier x sample density), which the scorer multiplies into that fact's severity; it is a different quantity from a finding's confidence in analyze_server. Use this to understand exactly what the engine is working with, or to investigate facts that didn't reach the severity threshold for findings.")] public static async Task GetAnalysisFacts( AnalysisService analysisService, ServerManager serverManager, @@ -254,8 +259,17 @@ after configuration alone knows audit_config still has it. */ value = Math.Round(f.Value, 6), base_severity = Math.Round(f.BaseSeverity, 4), severity = Math.Round(f.Severity, 4), + // #3538 A6: an anomaly fact's metadata["confidence"] is the BASELINE's confidence (tier x + // density, BaselineBucket.Confidence) — the trustworthiness of the distribution the + // deviation was measured against, which the scorer multiplies into severity. It is not + // the story confidence analyze_server publishes, and one word for two quantities in one + // client session is the confusion this campaign item exists to remove, so the payload + // names it baseline_confidence. The fact's own metadata key is unchanged (the scorer + // reads it); this is a read-time projection only. metadata = f.Metadata.ToDictionary( - m => m.Key, + m => m.Key == "confidence" && string.Equals(f.Source, "anomaly", StringComparison.Ordinal) + ? "baseline_confidence" + : m.Key, m => Math.Round(m.Value, 2)), amplifiers = f.AmplifierResults.Count > 0 ? f.AmplifierResults.Select(a => new @@ -682,7 +696,7 @@ the collector missed changes nothing about what the server is configured to. */ } } - [McpServerTool(Name = "get_analysis_findings"), Description("Gets persisted findings from previous analysis runs without running a new analysis, deduplicated to one entry per diagnostic chain (story_path_hash + incident_id) - the engine re-persists the same stories every cycle, so each entry is the chain's LATEST occurrence plus occurrence stats (occurrences, first_seen, last_seen, peak_severity) spanning the window. Use this to review historical findings or check if anything has changed since the last analysis. A remediable finding carries remediation_command: the full copy-paste T-SQL remediation (identical to the viewer card), rendered from the finding's persisted action and including a two-sided risk-disclosure comment header on destructive changes; it is advisory only and never executed. A force-plan remediation additionally carries structured_remediation: the same decision as machine-readable fields — eligible, named blockers (parameter_sensitivity_cofired, secondary_replica_evidence), evidence numbers, and split force_sql/unforce_sql/verify_sql artifacts — so agents consume the verdict as data instead of parsing comment prose. Set include_drilldown to also return each chain's persisted evidence rows (the specific plans/queries behind the finding, capped at write time with an explicit _truncation_note; null on findings persisted before the column existed).")] + [McpServerTool(Name = "get_analysis_findings"), Description("Gets persisted findings from previous analysis runs without running a new analysis, deduplicated to one entry per diagnostic chain (story_path_hash + incident_id) - the engine re-persists the same stories every cycle, so each entry is the chain's LATEST occurrence plus occurrence stats (occurrences, first_seen, last_seen, peak_severity) spanning the window. Use this to review historical findings or check if anything has changed since the last analysis. Each finding's confidence is an EVIDENCE score (see analyze_server): 0.20 for the fired symptom alone, plus corroboration from matched amplifier checks and chain depth. Rows persisted before this definition carried a PATH-LENGTH statistic under the same name, with a lone symptom at 1.0 — confidence_basis labels those rows path-shape (pre-#3538) and they must not be read as corroborated. A remediable finding carries remediation_command: the full copy-paste T-SQL remediation (identical to the viewer card), rendered from the finding's persisted action and including a two-sided risk-disclosure comment header on destructive changes; it is advisory only and never executed. A force-plan remediation additionally carries structured_remediation: the same decision as machine-readable fields — eligible, named blockers (parameter_sensitivity_cofired, secondary_replica_evidence), evidence numbers, and split force_sql/unforce_sql/verify_sql artifacts — so agents consume the verdict as data instead of parsing comment prose. Set include_drilldown to also return each chain's persisted evidence rows (the specific plans/queries behind the finding, capped at write time with an explicit _truncation_note; null on findings persisted before the column existed).")] public static async Task GetAnalysisFindings( AnalysisService analysisService, ServerManager serverManager, @@ -775,6 +789,11 @@ stats. The store keeps every row — this shapes the read only. */ analysis_time = f.AnalysisTime.ToString("o"), severity = Math.Round(f.Severity, 2), confidence = Math.Round(f.Confidence, 2), + // #3538 A6: what the number rests on. Corroboration-derived since this change + // (matched amplifier share + path depth); a row persisted under the old path-shape + // formula is labelled as such, derived from the finding's own shape at read time + // because the store carries no version marker (no schema change). + confidence_basis = StoryConfidence.DescribeBasis(f.RootFactKey, f.Confidence, f.FactCount), category = f.Category, root_fact = new { key = f.RootFactKey, value = f.RootFactValue }, leaf_fact = f.LeafFactKey != null diff --git a/Lite/Mcp/McpInstructions.cs b/Lite/Mcp/McpInstructions.cs index 3af1cb37f..5071344a6 100644 --- a/Lite/Mcp/McpInstructions.cs +++ b/Lite/Mcp/McpInstructions.cs @@ -218,11 +218,11 @@ You are connected to a SQL Server performance monitoring tool via Performance Mo ### Diagnostic Analysis Tools | Tool | Purpose | Key Parameters | |------|---------|----------------| - | `analyze_server` | Runs the inference engine: scores facts, traverses relationship graph, returns evidence-backed findings with severity and recommended next tools. A remediable finding also carries `remediation_command` — the full copy-paste T-SQL remediation (identical to the viewer card), with a two-sided risk-disclosure header on destructive changes; advisory only, never executed. Force-plan findings additionally carry `structured_remediation`: the verdict as machine-readable fields (eligible + named blockers), evidence, and split force/unforce/verify SQL. With `as_of` it analyzes a PAST window — anomaly baseline included — and is EXPLORATORY: the findings come back in full but are NOT persisted, which `persisted` / `persistence_note` state on every result | `server_name`, `hours_back` (default 4), `as_of` | - | `get_analysis_facts` | Exposes raw scored facts from the collect+score pipeline — every observation the engine sees with base severity, amplifiers, and metadata | `server_name`, `hours_back` (default 4), `source` (filter), `min_severity`, `as_of` | + | `analyze_server` | Runs the inference engine: scores facts, traverses relationship graph, returns evidence-backed findings with severity and recommended next tools. Each finding's `confidence` is an EVIDENCE score (0.20 for the symptom alone, more as the root fact's amplifier checks match and the chain deepens — a lone uncorroborated symptom is 0.20, never 1.0) and `confidence_basis` says in words what it rests on; rank by `severity` for impact and read `confidence` as how much of the engine's own corroboration showed up. A remediable finding also carries `remediation_command` — the full copy-paste T-SQL remediation (identical to the viewer card), with a two-sided risk-disclosure header on destructive changes; advisory only, never executed. Force-plan findings additionally carry `structured_remediation`: the verdict as machine-readable fields (eligible + named blockers), evidence, and split force/unforce/verify SQL. With `as_of` it analyzes a PAST window — anomaly baseline included — and is EXPLORATORY: the findings come back in full but are NOT persisted, which `persisted` / `persistence_note` state on every result | `server_name`, `hours_back` (default 4), `as_of` | + | `get_analysis_facts` | Exposes raw scored facts from the collect+score pipeline — every observation the engine sees with base severity, amplifiers, and metadata (an ANOMALY_* fact's `baseline_confidence` is the baseline's trustworthiness, not a finding's `confidence`) | `server_name`, `hours_back` (default 4), `source` (filter), `min_severity`, `as_of` | | `compare_analysis` | Compares two time periods (e.g., peak vs off-peak, before vs after a change) showing severity deltas for each fact. When NEITHER window produced facts the result is `unavailable` rather than an all-zero comparison, because "nothing to compare" is not "nothing changed"; when only ONE window is empty the payload carries a `caveat` saying so, since every fact then counts as new or resolved by default. `baseline_hours_back` is measured from the comparison window's END, so `as_of` moves BOTH windows together | `server_name`, `hours_back` (default 4), `baseline_hours_back` (default 28), `as_of` | | `audit_config` | Edition-aware configuration audit: evaluates CTFP, MAXDOP, max memory, and max worker threads against best practices | `server_name` | - | `get_analysis_findings` | Retrieves persisted findings from previous analysis runs, deduplicated to one entry per diagnostic chain (`story_path_hash` + `incident_id`): the latest occurrence plus `occurrences`/`first_seen`/`last_seen`/`peak_severity` spanning the window; each remediable finding carries `remediation_command` — the full copy-paste T-SQL remediation (identical to the viewer card), rendered from the persisted action, advisory only and never executed; force-plan findings additionally carry `structured_remediation` (verdict + evidence + split artifacts, machine-readable). Its window is on ANALYSIS TIME, so `as_of` asks what analysis was saying then rather than re-analyzing that window now | `server_name`, `hours_back` (default 24), `as_of` | + | `get_analysis_findings` | Retrieves persisted findings from previous analysis runs (each with `confidence_basis`: rows persisted before `confidence` measured corroboration are labelled `path-shape (pre-#3538)` — under that formula a lone symptom read 1.0, so do not read those as corroborated), deduplicated to one entry per diagnostic chain (`story_path_hash` + `incident_id`): the latest occurrence plus `occurrences`/`first_seen`/`last_seen`/`peak_severity` spanning the window; each remediable finding carries `remediation_command` — the full copy-paste T-SQL remediation (identical to the viewer card), rendered from the persisted action, advisory only and never executed; force-plan findings additionally carry `structured_remediation` (verdict + evidence + split artifacts, machine-readable). Its window is on ANALYSIS TIME, so `as_of` asks what analysis was saying then rather than re-analyzing that window now | `server_name`, `hours_back` (default 24), `as_of` | | `mute_analysis_finding` | Mutes a finding pattern by story_path_hash so it won't appear in future runs. Reports what the write did: `registered`, and `matched_now` — how many stored findings in scope carry the hash (status `muted_unmatched` when 0: the mute is kept, but check the hash) | `story_path_hash` (required), `server_name`, `reason` | ## Recommended Workflow diff --git a/PerformanceMonitor.Analysis/FactAdvice.cs b/PerformanceMonitor.Analysis/FactAdvice.cs index 036a6d9cc..f3c91c4ed 100644 --- a/PerformanceMonitor.Analysis/FactAdvice.cs +++ b/PerformanceMonitor.Analysis/FactAdvice.cs @@ -175,6 +175,9 @@ public static class FactAdvice "RESOURCE_SEMAPHORE" or "RESOURCE_SEMAPHORE_QUERY_COMPILE" or "WRITELOG" or "HADR_SYNC_COMMIT" or "LATCH_EX" or "LATCH_SH" or "PAGELATCH_UP" or "LCK_M_S" or "LCK_M_IS" => ComposeWaitByKey(rootFactKey, factsByKey), + // #3538 A9: schema-modification waits name the long-running job when RUNNING_JOBS fired — the + // same window the SCH_M → RUNNING_JOBS graph edge folds the two cards into one incident. + "SCH_M" => ComposeSchM(factsByKey), // Query-level: state the cost-swing ratio / regression factor / tempdb driver the engine // measured, and permute on the discriminating flag (forced-plan-failing, dominant consumer). "PARAMETER_SENSITIVITY" => ComposeParameterSensitivity(factsByKey), @@ -909,10 +912,12 @@ private static AdviceBlock ComposeSimpleWait(IReadOnlyDictionary f "Queries are queuing for compile memory — RESOURCE_SEMAPHORE_QUERY_COMPILE", "This is memory pressure during query COMPILATION, not execution — too many concurrent compilations competing for compile memory, typically a flood of unparameterized ad-hoc queries each compiling its own plan.", "Parameterize the ad-hoc queries so they reuse cached plans instead of compiling fresh, enable 'optimize for ad hoc workloads' so one-off plans stop bloating the cache, and address whatever generates the compile storm (an ORM emitting literal values, or unnecessary recompiles)."), + // #3538 A9: the WRITELOG → RUNNING_JOBS edge's prose — a fully logged rebuild or reload while a job + // runs long is the log-flush driver, so the job is named beside the storage advice. "WRITELOG" => ComposeSimpleWait(facts, key, "Log-flush waits", "Transaction log flushes are slow — WRITELOG", "WRITELOG is time spent waiting for the transaction log to harden to disk on commit. Sustained WRITELOG points at slow log storage or a commit-heavy workload doing many tiny transactions, each forcing its own flush.", - "Put the transaction log on the fastest storage available — write latency matters far more than throughput here — and batch tiny transactions where the application allows, so fewer, larger commits flush less often. Confirm the log is not autogrowing in small increments under load."), + "Put the transaction log on the fastest storage available — write latency matters far more than throughput here — and batch tiny transactions where the application allows, so fewer, larger commits flush less often. Confirm the log is not autogrowing in small increments under load." + LinkedJobClause(facts)), "HADR_SYNC_COMMIT" => ComposeSimpleWait(facts, key, "Synchronous-commit waits", "Commits are waiting on a synchronous availability-group secondary — HADR_SYNC_COMMIT", "HADR_SYNC_COMMIT is the time a committing transaction on the primary spends waiting for a synchronous-commit secondary to acknowledge that it has hardened the log block. It is the availability-group half of commit latency — the local log flush is WRITELOG — so on a synchronous AG every commit pays both. Sustained HADR_SYNC_COMMIT means the round trip to the secondary is slow: the secondary's own log write, the network between the replicas, or a secondary that is busy with redo or with readable-secondary queries and is slow to harden.", @@ -1040,6 +1045,9 @@ private static AdviceBlock ComposeIoLatency(IReadOnlyDictionary fa rem.Append("Write latency is usually the storage or the log path: confirm the data and log files are on adequately fast storage, watch for autogrowth events stalling writes, and check whether a backup, CHECKDB, or index maintenance overlapped the window."); if (Fired(facts, "WRITELOG")) rem.Append(" WRITELOG co-fired, pointing specifically at the log."); + // #3538 A9: when the IO_WRITE_LATENCY_MS → RUNNING_JOBS edge fired, the "check whether index + // maintenance overlapped the window" above is no longer a guess — say which job. + rem.Append(LinkedJobClause(facts)); } return fallback with @@ -1618,6 +1626,65 @@ private static AdviceBlock ComposePlanWarning(IReadOnlyDictionary }; } + /// + /// #3538 A9: the sentence that ties a symptom card to the long-running job when RUNNING_JOBS FIRED + /// this window (running-long count above zero — the same predicate the maintenance edges in + /// use, so the prose and the incident clustering agree on when the + /// link exists). States the job facts the engine collected (count, worst overrun, longest runtime) + /// rather than re-describing the symptom. Empty when the fact is absent or no job is running long, + /// so a caller can append it unconditionally. Leading space included. The RUNNING_JOBS fact carries + /// no job NAME (its metadata is counts and durations), so the sentence points at the Running Jobs + /// view for the name instead of pretending to know it. + /// + private static string LinkedJobClause(IReadOnlyDictionary facts) + { + if (!Fired(facts, "RUNNING_JOBS")) + return string.Empty; + var longCount = FactMeta(facts, "RUNNING_JOBS", "running_long_count"); + if (longCount is not > 0) + return string.Empty; + + var pct = FactMeta(facts, "RUNNING_JOBS", "max_percent_of_average"); + var dur = FactMeta(facts, "RUNNING_JOBS", "max_duration_seconds"); + var sb = new StringBuilder($" RUNNING_JOBS fired in the same window — {Plural(longCount.Value, "Agent job")} running well past normal duration"); + if (pct is > 0) + sb.Append($", the worst at {pct.Value:N0}% of its historical average"); + if (dur is > 0) + sb.Append($" (about {Duration(dur.Value * 1000)} so far)"); + sb.Append(" — so this finding and that job are ONE incident, not two: the Running Jobs view names the job, and moving or shortening it is the fix for both cards."); + return sb.ToString(); + } + + /// + /// SCH_M composed (#3538 A9): states the schema-lock wait totals and, when RUNNING_JOBS fired, names + /// the long-running job as the likely holder instead of telling the operator to go and look for one. + /// Falls back to the static block when the wait metadata is absent AND no job is running long. + /// + private static AdviceBlock ComposeSchM(IReadOnlyDictionary facts) + { + var fallback = _byKey["SCH_M"]; + var numbers = WaitNumbers(facts, "SCH_M", "Schema-modification lock waits"); + var linked = LinkedJobClause(facts); + if (numbers.Length == 0 && linked.Length == 0) + return fallback; + + var concept = + "SCH-M is the most exclusive lock SQL Server takes — it is incompatible with everything, including the IS lock a SELECT requires, so a session holding SCH-M on a hot table blocks the entire workload against that table. Sources: `ALTER TABLE`, `CREATE/DROP INDEX`, partition operations, and certain statistics updates."; + var inv = new StringBuilder(numbers).Append(concept); + if (linked.Length > 0) + inv.Append(linked); + else + inv.Append(" No Agent job was running past its normal duration this window, so look for ad-hoc DDL or an index operation issued outside the job schedule."); + + return fallback with + { + Headline = linked.Length > 0 + ? "Schema-modification lock waits while an Agent job runs long — its index maintenance is blocking the workload" + : fallback.Headline, + Investigation = inv.ToString() + }; + } + /// RUNNING_JOBS composed: states how many jobs overran and by how much. private static AdviceBlock ComposeRunningJobs(IReadOnlyDictionary facts) { @@ -1634,6 +1701,16 @@ private static AdviceBlock ComposeRunningJobs(IReadOnlyDictionary if (dur is > 0) inv.Append($" (longest current runtime about {Duration(dur.Value * 1000)})"); inv.Append(". A job running far past its average is usually stuck — blocked, waiting on a resource, or hung — rather than simply busy."); + // #3538 A9: the reverse link — the job card names the symptoms whose graph edges point at it, so + // whichever card the operator opens first says the incident is one thing. "Fired" here is the + // scorer's >0 working set (the same set the edge predicates evaluate against); a symptom that also + // ROOTED its own finding is clustered with this one, a trace-level one is merely linked. + var symptoms = new List(); + if (longCount.Value > 0 && Fired(facts, "SCH_M")) symptoms.Add("SCH_M (schema-modification lock waits — the job is probably index maintenance holding schema locks)"); + if (longCount.Value > 0 && Fired(facts, "IO_WRITE_LATENCY_MS")) symptoms.Add("IO_WRITE_LATENCY_MS (its write volume is a candidate for the latency)"); + if (longCount.Value > 0 && Fired(facts, "WRITELOG")) symptoms.Add("WRITELOG (a fully logged rebuild or reload drives log flushes)"); + if (symptoms.Count > 0) + inv.Append($" Also elevated this window and linked to this job by the engine: {string.Join("; ", symptoms)}. Where one of them rooted its own finding, it and this card are ONE incident, not two."); var rem = "Check the long-running jobs in Agent history: one at several times its normal runtime is typically " + diff --git a/PerformanceMonitor.Analysis/InferenceEngine.cs b/PerformanceMonitor.Analysis/InferenceEngine.cs index f25105990..bc6ffc819 100644 --- a/PerformanceMonitor.Analysis/InferenceEngine.cs +++ b/PerformanceMonitor.Analysis/InferenceEngine.cs @@ -22,7 +22,9 @@ namespace PerformanceMonitor.Analysis; public class InferenceEngine { private const double MinimumSeverityThreshold = 0.5; - private const int MaxPathDepth = 10; // Safety limit + /// Hops the greedy traversal will follow from a root before it stops (a safety limit); the + /// longest story path is therefore this plus the root — . + internal const int MaxPathDepth = 10; /// /// Config-advisory fact keys that root a finding at ANY positive severity, bypassing the @@ -122,6 +124,10 @@ public List BuildStories(List facts) RootFactKey = "server_health", RootFactValue = 0, Severity = 0, + // 1.0 by construction, not by evidence arithmetic: an absolution is the statement that + // every fact was scored and none rooted, which is exactly as certain as the scoring + // pass itself. StoryConfidence.DescribeBasis names this arm so a reader never mistakes + // it for the legacy path-shape 1.0 (#3538 A6). Confidence = 1.0, Category = "absolution", Path = ["server_health"], @@ -193,9 +199,11 @@ private static AnalysisStory BuildStory(List path, Dictionary HasFact(facts, "CPU_SPIKE") && facts["CPU_SPIKE"].BaseSeverity > 0); } + /* ── Maintenance ── */ + + /// + /// #3538 A9: the edges that let a maintenance window be ONE incident. Before these, nothing in the + /// graph touched SCH_M or RUNNING_JOBS, so a nightly index rebuild surfaced as two or three unlinked + /// single-node cards every night — the schema-lock waits, the long-running job, and the write-latency / + /// log-flush pair — each re-litigating the same event with nobody saying they were the same event. + /// unions stories across an ACTIVE edge, so one + /// fired edge from each symptom onto the job is what folds them into one incident that names it. + /// + /// The predicate on every edge is the same: RUNNING_JOBS FIRED — its base severity is above + /// zero, which FactScorer.ScoreJobFact grants only when at least one Agent job is running past + /// its own historical duration (the fact's value is the running-long count; the concerning bar is 1). + /// Mere presence of the fact would not do: the collector emits RUNNING_JOBS whenever ANY job is + /// running, and an edge on presence would pin every SCH_M or write-latency finding to whatever + /// routine job happened to be executing. The edges run symptom → job (the direction the operator + /// reasons in: "what is holding schema locks / hammering the log? — that job"), and only that way: + /// a job → symptom edge would let a RUNNING_JOBS root consume SCH_M into its own story and change + /// which key the persisted finding is rooted on, and the clustering needs one direction only. + /// + /// What this does NOT do: it does not model the maintenance SCHEDULE (a window that MOVES is + /// still undetectable), does not score recurrence (the same incident every night reads as new every + /// night), and does not name the job — the RUNNING_JOBS fact carries counts and durations, not a job + /// name; get_running_jobs names it. Those are the structural half of A9, deferred. + /// + private void BuildMaintenanceEdges() + { + // SCH_M → RUNNING_JOBS (schema-modification waits while a job runs long — index maintenance) + AddEdge("SCH_M", "RUNNING_JOBS", "maintenance", + "Agent job running past its normal duration — the schema locks are probably its index maintenance", + facts => HasFact(facts, "RUNNING_JOBS") && facts["RUNNING_JOBS"].BaseSeverity > 0); + + // IO_WRITE_LATENCY_MS → RUNNING_JOBS (write latency while a job runs long — rebuild/CHECKDB write volume) + AddEdge("IO_WRITE_LATENCY_MS", "RUNNING_JOBS", "maintenance", + "Agent job running past its normal duration — its write volume is a candidate for the latency", + facts => HasFact(facts, "RUNNING_JOBS") && facts["RUNNING_JOBS"].BaseSeverity > 0); + + // WRITELOG → RUNNING_JOBS (log-flush waits while a job runs long — a rebuild is fully logged) + AddEdge("WRITELOG", "RUNNING_JOBS", "maintenance", + "Agent job running past its normal duration — a fully logged rebuild or reload drives log flushes", + facts => HasFact(facts, "RUNNING_JOBS") && facts["RUNNING_JOBS"].BaseSeverity > 0); + } + private static bool HasFact(IReadOnlyDictionary facts, string key) { return facts.ContainsKey(key); diff --git a/PerformanceMonitor.Analysis/SameStatementPileupDetector.cs b/PerformanceMonitor.Analysis/SameStatementPileupDetector.cs index 0befd8156..9dd79d89c 100644 --- a/PerformanceMonitor.Analysis/SameStatementPileupDetector.cs +++ b/PerformanceMonitor.Analysis/SameStatementPileupDetector.cs @@ -496,8 +496,12 @@ notification cooldown must answer this statement's question with this statement' concurrent sessions in the pack. The pack's minimum elapsed rides as the leaf value. */ RootFactValue = sessions, Severity = severity, - /* Single-observation story — the symptom is directly measured, no traversal to dilute - (the InferenceEngine convention: single-node paths carry confidence 1.0). */ + /* 1.0 BY CONSTRUCTION, not by the engine's formula: the convoy is observed directly — N + concurrent sessions on one statement, each past that statement's own duration baseline — + so there is no corroboration left to look for. Since #3538 A6 the InferenceEngine scores a + lone symptom LOW (StoryConfidence: 0.20 uncorroborated), and StoryConfidence.DescribeBasis + names this root key so the MCP basis string says "detector-measured" rather than reading + this 1.0 as a legacy path-shape value. */ Confidence = 1.0, Category = "queries", Path = [RootFactKey], diff --git a/PerformanceMonitor.Analysis/StoryConfidence.cs b/PerformanceMonitor.Analysis/StoryConfidence.cs new file mode 100644 index 000000000..0b501325c --- /dev/null +++ b/PerformanceMonitor.Analysis/StoryConfidence.cs @@ -0,0 +1,173 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Linq; + +namespace PerformanceMonitor.Analysis; + +/// +/// Story confidence as an EVIDENCE statistic (#3538 A6), and the read-time description of what a +/// given confidence value rests on. +/// +/// The defect this replaces. Confidence used to be a path-shape statistic: +/// path.Count == 1 ? 1.0 : (path.Count - 1) / path.Count. A lone symptom with NO corroboration +/// carried the HIGHEST confidence the engine could express (1.0), a two-node chain 0.5, a ten-node chain +/// 0.9 — inverted at exactly the point that matters, and exported to LLM consumers under an evidence +/// name beside severity. An agent ranking findings by severity × confidence preferred the least +/// evidenced one. Meanwhile the engine already computed real corroboration for every root and threw +/// it away at this step: the FactScorer evaluates a per-key AMPLIFIER catalogue ("what else should be +/// true if this symptom is real?") and records each match on . +/// +/// The formula. Two corroboration terms, each in [0, 1], over a floor for the symptom +/// itself: +/// +/// amplifier share = matched ÷ defined amplifiers for the root fact — the +/// share of the engine's own "if this is real, X should also be true" checks that came back true. +/// path depth = (n − 1) ÷ n for an n-node path — each traversed edge is a +/// fired predicate onto a fact that itself scored, so a deeper chain is more corroborated, with +/// diminishing weight per extra node (0 for a lone symptom, 0.5 at two nodes, 0.9 at ten). +/// +/// confidence = 0.20 + 0.48 × amplifier share + 0.32 × path depth when the root has an +/// amplifier catalogue; 0.20 + 0.80 × path depth when it has none. The 0.20 floor is the fired +/// symptom itself: a threshold was crossed by a measured value, which is evidence of something, and a +/// finding the engine chose to root should never read as 0. The 60/40 split between the two terms +/// (0.48/0.32 of the 0.80 that corroboration can earn) puts the larger weight on the amplifier +/// catalogue because those checks are hand-written per key against the things that DISTINGUISH a +/// real instance of the symptom, while a graph edge is the next question to ask and fires on +/// presence more often than on distinguishing evidence. When the root has no catalogue the path +/// term carries the whole 0.80 rather than capping the story at 0.52: no catalogue is the ABSENCE of +/// evidence and is scored neutrally, whereas a catalogue whose checks all came back false is evidence +/// AGAINST — the engine looked for this symptom's usual companions and found none — and is scored down. +/// These weights are design constants, not fleet measurements; the property they were chosen to hold +/// is the ordering below, which the tests pin. +/// +/// Worked examples. A lone SCH_M (no catalogue, one node): 0.20. A lone PAGEIOLATCH_SH +/// with 0 of 3 amplifiers matched: 0.20. A lone PAGEIOLATCH_SH with 3 of 3 matched: 0.68. +/// A three-node chain such as PAGEIOLATCH_SH → RESOURCE_SEMAPHORE → MEMORY_GRANT_PENDING with 2 of 3 +/// matched: 0.20 + 0.32 + 0.21 = 0.73. SCH_M → RUNNING_JOBS (no catalogue, two nodes): 0.60. A ten-node +/// chain with no catalogue: 0.92; with a fully matched catalogue: 0.968. The deepest path the traversal +/// builds is eleven nodes (InferenceEngine.MaxPathDepth = 10 hops plus the root), where the two +/// ceilings are 0.927 and 0.971 — under 1.0, because path depth never reaches 1. Nothing built by this +/// formula is ever exactly 1.0, +/// and it is monotonic by construction: one more matched amplifier or one more path node never lowers +/// it. +/// +/// Legacy rows without a schema change. The persisted confidence column keeps its +/// name and pre-change rows keep their values; a migration rung to stamp a version marker was ruled +/// out (one un-landed rung at a time, repo-wide). Instead the basis is RE-DERIVED at read time from the +/// value and the path length, which is sound because the two formulas' ranges are disjoint at every +/// path length: the legacy value for n nodes is exactly 1.0 (n = 1) or (n − 1)/n, and this formula +/// never lands on that number for the same n with any small-integer amplifier catalogue (pinned +/// exhaustively over every path length the traversal can build — n ≤ — and +/// catalogue ≤ 12). therefore labels a +/// value that equals its path-shape number as path-shape (pre-#3538) and everything else as +/// corroboration-derived, and the two built-by-construction 1.0s (absolution, the same-statement +/// pileup detector) are named by their root key so they are never mislabelled as legacy. +/// +public static class StoryConfidence +{ + /// + /// The longest story path can produce: the root plus + /// hops. The exhaustive pins enumerate to this bound so the + /// no-collision guarantee covers every path the engine can actually build, not a round number. + /// + public const int MaxPathNodes = InferenceEngine.MaxPathDepth + 1; + + /// The fired symptom itself — a measured value crossed a threshold. Never 0. + public const double Floor = 0.20; + + /// Weight of the matched-amplifier share when the root has a catalogue. + public const double AmplifierWeight = 0.48; + + /// Weight of the path-depth term when the root has a catalogue. + public const double PathWeight = 0.32; + + /// Weight of the path-depth term when the root has NO amplifier catalogue (the amplifier + /// term's share is folded in rather than forfeited — see the class remarks). + public const double UncataloguedPathWeight = AmplifierWeight + PathWeight; + + /// The same-statement pileup detector's root key — its stories carry 1.0 by construction + /// (a directly observed convoy against its own baseline), not by this formula and not by the + /// legacy one. Named here so can say so instead of calling it legacy. + public const string PileupRootKey = "SAME_STATEMENT_PILEUP"; + + /// The absolution story's root key (see ). + public const string AbsolutionRootKey = "server_health"; + + /// + /// Computes a story's confidence from its root fact's amplifier results and its path length. The + /// FactScorer must have run first — is populated there and is + /// empty for a key with no catalogue (which is the uncatalogued arm, deliberately). A null root + /// (a path whose first key is not in the working set, which BuildStories never produces) is scored + /// as uncatalogued. + /// + public static double Compute(Fact? rootFact, int pathLength) + { + var results = rootFact?.AmplifierResults; + var defined = results?.Count ?? 0; + var matched = results?.Count(r => r.Matched) ?? 0; + return Compute(matched, defined, pathLength); + } + + /// + /// The pure arithmetic, exposed so the pins can enumerate it. below + /// 1 is treated as 1 (a story has at least its root). + /// + public static double Compute(int matchedAmplifiers, int definedAmplifiers, int pathLength) + { + var n = Math.Max(1, pathLength); + var pathDepth = (n - 1.0) / n; + if (definedAmplifiers <= 0) + return Floor + UncataloguedPathWeight * pathDepth; + + var share = Math.Clamp((double)matchedAmplifiers / definedAmplifiers, 0.0, 1.0); + return Floor + AmplifierWeight * share + PathWeight * pathDepth; + } + + /// + /// The value the PRE-#3538 formula produced for a path of nodes: + /// 1.0 for a lone symptom, (n − 1)/n otherwise. Kept only so a persisted row can be recognised as + /// legacy at read time; never used to score anything. + /// + public static double LegacyPathShape(int factCount) + { + var n = Math.Max(1, factCount); + return n == 1 ? 1.0 : (n - 1.0) / n; + } + + /// + /// True when is exactly the legacy path-shape number for a path of + /// nodes — a row written before confidence measured corroboration. + /// Exact to 1e-9: the two formulas' ranges are disjoint (class remarks), so equality IS the test. + /// + public static bool IsLegacyPathShape(double confidence, int factCount) => + Math.Abs(confidence - LegacyPathShape(factCount)) < 1e-9; + + /// + /// The confidence_basis string both MCP SKUs publish beside confidence — what the + /// number rests on, in words an agent can act on, derived from the finding's own shape so that a + /// row persisted before this change is labelled as such rather than read as a corroboration score. + /// One sentence per arm; the arms are: the two by-construction 1.0s (named by root key), a legacy + /// path-shape row, and a corroboration-derived value. + /// + public static string DescribeBasis(string? rootFactKey, double confidence, int factCount) + { + if (string.Equals(rootFactKey, AbsolutionRootKey, StringComparison.Ordinal)) + return "absolution: every fact was scored and none rooted a finding; 1.0 by construction, not an evidence score."; + + if (string.Equals(rootFactKey, PileupRootKey, StringComparison.Ordinal)) + return "detector-measured: the same-statement pileup is observed directly (concurrent sessions on one statement against that statement's own duration baseline), so it carries 1.0 by construction rather than by the corroboration formula."; + + if (IsLegacyPathShape(confidence, factCount)) + return "path-shape (pre-#3538): this row was persisted when confidence was (n-1)/n over the story path with a lone symptom at 1.0 — a path-length statistic, NOT an evidence score; it is not comparable to corroboration-derived values and a lone-symptom 1.0 here means UNCORROBORATED."; + + var n = Math.Max(1, factCount); + return $"corroboration (#3538): 0.20 for the fired symptom + up to 0.48 for the share of the root fact's amplifier checks that matched + up to 0.32 for path depth ((n-1)/n over the {n}-node story path; a root with no amplifier catalogue earns the full 0.80 from path depth). A lone uncorroborated symptom scores 0.20; a fully corroborated deep chain approaches but never reaches 1.0."; + } +} From f9797a8e9ed6d121e3fda66f952d14c45ff9a331 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 15:13:47 -0400 Subject: [PATCH 54/69] The daily digest and sweep rollup asked 'did I send' when the channel asks 'was it delivered': the once-a-day gate now stamps delivered-today in the store, so a restart re-announces nothing and a failed delivery still retries (#3580) (#3633) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * The daily digest and sweep rollup gate on delivered-today in the store, not fired-today in process memory: a restart no longer re-announces them, and a failed delivery still retries (#3580) Both once-a-day gates were a ConcurrentDictionary under one fixed key, so a fresh process delivered both documents again whatever the previous process had delivered an hour earlier — six re-announcements among ~23 channel posts on the v3.8.0 install night. The gate now reads a delivery stamp from collect.collector_state (V44, no rung; server_id 0 = the fleet sentinel both existing fleet-scope writers use, collector_name 'self_alert') and writes it only when the deliverer reports a disposition other than `failed`, so the install night's recovery case — a failed pair re-attempted after the restart and landing — is preserved by construction. - IAlertDeliverer.DeliverAndReportAsync: a default-implemented reporting twin of DeliverAsync (null = unreported); DarlingAlertDeliverer reports the AlertDelivery its history row was written with. - DarlingSelfAlertEvaluator: FireAsync returns the report; the two documents stamp AFTER the fire; the store seeds memory after a start (#981's cooldown seed shape), memory serves the process; a stamp read fault falls open to memory, warns and is counted (#3013); a stamp write fault warns, uncounted. - Census pins re-justified: AlertMasterSwitchSurfaceTests (new delivery seam name, generic FireAsync return), AlertReadFailureSurfaceTests (9th counted evaluator site, 12th exempt), Lite.Tests inventory (8th fleet-scoped read), AlertReadFailureCounter.FleetScopedReads names the new read. - Tests: SelfAlertDeliveryStampTests (both documents, simulated restarts, failed/muted/unreported dispositions, read/write faults) and a DARLING_TEST_PG-gated live round-trip through the real table. * The stamp store is asked once per document per process: an empty or failed answer is cached as a sentinel, so the apply half's gate does not re-read, re-warn or re-count a fault the evaluate half already met on the same tick (#3580 review) Review catch: the "one store read between the two halves" cost model only held when the read succeeded — a throw (or no row) returned without seeding memory, so the apply half re-asked, doubling the warning and the #3013 count per tick for one logical fault. Exact under the single-writer fact: a store with no row on this process's first tick cannot gain one except through this process, which seeds memory directly. New pin drives two consults on one tick with both the read and the delivery failing — one read, one warning, one count. --- .../AlertMasterSwitchSurfaceTests.cs | 12 +- .../AlertReadFailureSurfaceTests.cs | 18 +- ...lertDeliveryStampStoreLivePostgresTests.cs | 125 ++++ .../SelfAlertDeliveryStampTests.cs | 574 ++++++++++++++++++ .../DarlingAlertDeliverer.cs | 38 +- .../DarlingSelfAlertEvaluator.cs | 295 +++++++-- .../DarlingWorker.cs | 6 +- .../SelfAlertDeliveryStampStore.cs | 193 ++++++ Lite.Tests/AlertReadFailureSurfaceTests.cs | 8 +- .../AlertReadFailureCounter.cs | 10 +- .../IAlertDeliverer.cs | 29 + 11 files changed, 1253 insertions(+), 55 deletions(-) create mode 100644 Darling/Darling.Tests/SelfAlertDeliveryStampStoreLivePostgresTests.cs create mode 100644 Darling/Darling.Tests/SelfAlertDeliveryStampTests.cs create mode 100644 Darling/PerformanceMonitor.Darling.Service/SelfAlertDeliveryStampStore.cs diff --git a/Darling/Darling.Tests/AlertMasterSwitchSurfaceTests.cs b/Darling/Darling.Tests/AlertMasterSwitchSurfaceTests.cs index 3b8075980..d30f1d430 100644 --- a/Darling/Darling.Tests/AlertMasterSwitchSurfaceTests.cs +++ b/Darling/Darling.Tests/AlertMasterSwitchSurfaceTests.cs @@ -47,7 +47,9 @@ public sealed class AlertMasterSwitchSurfaceTests /* ---------------- the delivery-call census ---------------- */ /// - /// A call that can put an alert on a channel: the shared deliverer seam (DeliverAsync), the + /// A call that can put an alert on a channel: the shared deliverer seam (DeliverAsync, and its + /// #3580 reporting twin DeliverAndReportAsync — the same send, answering what the channels did, + /// which the self-alert funnel now calls so the two daily documents can stamp delivered-today), the /// analysis notify seam (NotifyAsync / SendFindingAlertAsync), Lite's direct send seam /// (TrySendAlertEmailAsync), the shared send core (TrySendAsync), and the deliberate /// channel-probe statics (SendTest*). Dot-qualified on purpose: a DECLARATION has no receiver, @@ -55,7 +57,7 @@ public sealed class AlertMasterSwitchSurfaceTests /// comment-and-string-stripped source, so prose mentioning a seam is not a site. /// private static readonly Regex s_deliveryCall = new( - @"\??\.\s*(?:DeliverAsync|NotifyAsync|TrySendAlertEmailAsync|TrySendAsync|SendFindingAlertAsync|SendTestPagerDutyAsync|SendTestTeamsAsync|SendTestSlackAsync|SendTestGenericAsync)\s*\(", + @"\??\.\s*(?:DeliverAsync|DeliverAndReportAsync|NotifyAsync|TrySendAlertEmailAsync|TrySendAsync|SendFindingAlertAsync|SendTestPagerDutyAsync|SendTestTeamsAsync|SendTestSlackAsync|SendTestGenericAsync)\s*\(", RegexOptions.Compiled | RegexOptions.CultureInvariant); /// How a censused site pays for its place on a delivery path. @@ -282,8 +284,10 @@ public void EverySelfAlertFamily_ConsultsTheMasterSwitch() for (var i = 0; i < lines.Length; i++) { - /* A firing call, not the funnel's own declaration: FireAsync is called bare (same class). */ - if (!Regex.IsMatch(lines[i], @"(? — the funnel reports + what the channels did), so the exclusion allows a type-argument list on the Task. */ + if (!Regex.IsMatch(lines[i], @"(?]*>)?\s+FireAsync\s*\(")) { continue; } diff --git a/Darling/Darling.Tests/AlertReadFailureSurfaceTests.cs b/Darling/Darling.Tests/AlertReadFailureSurfaceTests.cs index c67f45d54..da4223c1e 100644 --- a/Darling/Darling.Tests/AlertReadFailureSurfaceTests.cs +++ b/Darling/Darling.Tests/AlertReadFailureSurfaceTests.cs @@ -884,7 +884,7 @@ real failure rather than a matcher that never matches anything. */ private static readonly (string Path, int Counted, int Exempt)[] s_wholeFileScopes = { (Path.Combine("PerformanceMonitor.Alerting", "AlertEngine.cs"), 14, 6), - (Path.Combine("Darling", "PerformanceMonitor.Darling.Service", "DarlingSelfAlertEvaluator.cs"), 8, 11), + (Path.Combine("Darling", "PerformanceMonitor.Darling.Service", "DarlingSelfAlertEvaluator.cs"), 9, 12), }; /// @@ -945,9 +945,14 @@ private static readonly (string Path, int Counted, int Exempt)[] s_wholeFileScop /// delivery end. And AlertEngine.cs carries a FOURTEENTH since #3495: the maintenance-annotation /// probe on the High CPU fire path — counted because its swallowed failure silently costs the card the /// one line that closes the triage, and an operator chasing a mystery backup deserves to see that the - /// probe went blind rather than that no maintenance ran. + /// probe went blind rather than that no maintenance ran. And DarlingSelfAlertEvaluator.cs a + /// NINTH since #3580: the daily documents' delivered-today stamp read, ONE site serving both the digest + /// and the rollup (so one literal name; the warning beside it names the document) — counted because a + /// swallowed stamp read is the gate falling back to process memory, which is the pre-#3580 + /// re-announce-per-restart posture returning for that tick, and a population of those under store + /// contention is exactly what this census exists to make visible. Its sibling WRITE is exempt. /// - private const int CountedSites = 32; + private const int CountedSites = 33; /// /// Log-message fragments that identify a catch block DELIBERATELY not counted, each paired with the @@ -980,6 +985,7 @@ private static readonly (string Path, int Counted, int Exempt)[] s_wholeFileScop ["Failed to check failed jobs"] = "the fetcher reads the monitored server's msdb; the block's only store op is a write both stores swallow", ["CONVERTS the fault into the unreadable count"] = "a parse arm, not a read: the fleet-sweep rollup's store read is counted above it, and a document that does not parse becomes the rollup's own reportable unreadable count - the fault is evidence, not a swallow", ["Could not resolve Agent job names"] = "reads the monitored server's msdb through the host resolver, not the store - the Recently-failed-job precedent one seam over; the card degrades to the unresolved form whose raw marker keeps the gap visible, and the page still delivers", + ["delivery stamp could not be written"] = "a write (#3580): the daily document was already delivered and process memory already gates it; the dropped stamp costs one re-announcement at the next restart and never a delivery - the stamp READ beside it is the read, and it is counted", }; /// @@ -1070,8 +1076,10 @@ a person rather than netting out silently. */ rollup's own unreadable count rather than a read failure. 23rd since #3497: the Agent-job resolver's catch, an msdb read on the monitored server degrading to the unresolved form. 24th since #3514: the web-dashboard TLS certificate self-alert's catch, whose evidence is the in-memory - WebTlsCertificateState report the web host publishes - there is no store read to swallow. */ - Assert.Equal(24, totalExempt); + WebTlsCertificateState report the web host publishes - there is no store read to swallow. 25th + since #3580: the daily documents' delivery-stamp WRITE, a write whose loss costs one + re-announcement at the next restart and never a delivery. */ + Assert.Equal(25, totalExempt); /* Every exemption in the table is actually used. An exemption for a message that no longer exists is a hole this pin would otherwise keep open indefinitely — the shape that lets a real new catch diff --git a/Darling/Darling.Tests/SelfAlertDeliveryStampStoreLivePostgresTests.cs b/Darling/Darling.Tests/SelfAlertDeliveryStampStoreLivePostgresTests.cs new file mode 100644 index 000000000..1b283524b --- /dev/null +++ b/Darling/Darling.Tests/SelfAlertDeliveryStampStoreLivePostgresTests.cs @@ -0,0 +1,125 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Threading.Tasks; +using Npgsql; +using PerformanceMonitor.Darling.Service; +using PerformanceMonitor.Darling.Storage; +using Xunit; + +namespace Darling.Tests; + +/* #1776 own-store: every test here mints its own scratch database through ScratchPostgres and touches + nothing on the shared one, so serializing it against the live-postgres collection would cost suite time + and buy no isolation. */ + +/// +/// The half of #3580's stamp store no unit pin reaches: that the two statements PARSE against a migrated +/// store, that a stamp round-trips to the tick with its Kind, that the upsert replaces rather than +/// duplicates, that the row sits where the class remarks say it sits (server_id = 0, +/// collector_name = 'self_alert', updated_at a naive UTC write time), and that a value the +/// store cannot parse reads as no stamp rather than as a throw — the +/// shape, over a table that has existed since V44 and so needs no rung of its own. +/// +public sealed class SelfAlertDeliveryStampStoreLivePostgresTests +{ + [Fact] + public async Task AStampRoundTrips_Replaces_AndSitsUnderTheFleetSentinel() + { + var baseConnectionString = Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + Assert.SkipWhen(string.IsNullOrEmpty(baseConnectionString), + "Set DARLING_TEST_PG to a Postgres connection string to run the live delivery-stamp round-trip (it mints its own scratch database)."); + + var ct = TestContext.Current.CancellationToken; + await using var scratch = await ScratchPostgres.CreateAsync(baseConnectionString!, ct); + await using (var migrate = new NpgsqlConnection(scratch.ConnectionString)) + { + await migrate.OpenAsync(ct); + await PgMigrations.MigrateAsync(migrate, null, ct); + } + + await using var postgres = NpgsqlDataSource.Create(scratch.ConnectionString); + var log = new CapturingTestLogger(); + var store = new PgSelfAlertDeliveryStampStore(postgres, log); + + /* A fresh store answers null for both keys — no throw, no phantom row. */ + Assert.Null(await store.GetDeliveredAtUtcAsync(PgSelfAlertDeliveryStampStore.CostDigestStateKey, ct)); + Assert.Null(await store.GetDeliveredAtUtcAsync(PgSelfAlertDeliveryStampStore.FleetSweepRollupStateKey, ct)); + + /* Round-trip to the tick, Kind Utc coming back — and the two keys are independent rows. */ + var digestAt = new DateTime(2026, 9, 17, 23, 30, 12, DateTimeKind.Utc).AddTicks(1234567); + await store.RecordDeliveredAtUtcAsync(PgSelfAlertDeliveryStampStore.CostDigestStateKey, digestAt, ct); + var readBack = await store.GetDeliveredAtUtcAsync(PgSelfAlertDeliveryStampStore.CostDigestStateKey, ct); + Assert.Equal(digestAt, readBack); + Assert.Equal(DateTimeKind.Utc, readBack!.Value.Kind); + Assert.Null(await store.GetDeliveredAtUtcAsync(PgSelfAlertDeliveryStampStore.FleetSweepRollupStateKey, ct)); + + /* A Kind-Unspecified caller value (the evaluator's clock seam can hand one in) is stamped Utc on the + way in, so it reads back Utc and equal. */ + var rollupAt = new DateTime(2026, 9, 18, 0, 15, 0, DateTimeKind.Unspecified); + await store.RecordDeliveredAtUtcAsync(PgSelfAlertDeliveryStampStore.FleetSweepRollupStateKey, rollupAt, ct); + var rollupBack = await store.GetDeliveredAtUtcAsync(PgSelfAlertDeliveryStampStore.FleetSweepRollupStateKey, ct); + Assert.Equal(rollupAt, rollupBack); + Assert.Equal(DateTimeKind.Utc, rollupBack!.Value.Kind); + + /* The upsert REPLACES: a second delivery a day later overwrites the digest's row, and the table holds + exactly two rows under this owner — one per document — not three. */ + var laterDigestAt = digestAt.AddDays(1); + await store.RecordDeliveredAtUtcAsync(PgSelfAlertDeliveryStampStore.CostDigestStateKey, laterDigestAt, ct); + Assert.Equal(laterDigestAt, await store.GetDeliveredAtUtcAsync(PgSelfAlertDeliveryStampStore.CostDigestStateKey, ct)); + + await using (var connection = await postgres.OpenConnectionAsync(ct)) + { + await using var rows = new NpgsqlCommand(@" +SELECT server_id, collector_name, state_key, state_value, updated_at +FROM collect.collector_state +WHERE collector_name = $1 +ORDER BY state_key", connection); + rows.Parameters.AddWithValue(PgSelfAlertDeliveryStampStore.StateCollectorName); + await using var reader = await rows.ExecuteReaderAsync(ct); + + Assert.True(await reader.ReadAsync(ct)); + Assert.Equal(PgSelfAlertDeliveryStampStore.FleetServerId, reader.GetInt32(0)); + Assert.Equal("self_alert", reader.GetString(1)); + Assert.Equal(PgSelfAlertDeliveryStampStore.CostDigestStateKey, reader.GetString(2)); + /* The value is the round-trip text with its Z — what makes the read side's Kind honest. */ + Assert.EndsWith("Z", reader.GetString(3), StringComparison.Ordinal); + /* updated_at is the WRITE time, naive UTC: within a minute of now, read as Unspecified from a + `timestamp` column, and never the server's local rendering of a timestamptz cast (the trap the + runner's own comment measured at exactly one zone offset). */ + var updatedAt = reader.GetDateTime(4); + Assert.Equal(DateTimeKind.Unspecified, updatedAt.Kind); + Assert.InRange(DateTime.UtcNow - DateTime.SpecifyKind(updatedAt, DateTimeKind.Utc), TimeSpan.FromMinutes(-1), TimeSpan.FromMinutes(1)); + + Assert.True(await reader.ReadAsync(ct)); + Assert.Equal(PgSelfAlertDeliveryStampStore.FleetSweepRollupStateKey, reader.GetString(2)); + + Assert.False(await reader.ReadAsync(ct), "the upsert duplicated a stamp row instead of replacing it"); + } + + /* A value this build cannot parse — a hand edit, or a writer this build does not know — reads as NO + stamp with a warning, so the gate falls back to memory for the tick rather than the pass dying on a + FormatException; the next delivery overwrites it. */ + await using (var connection = await postgres.OpenConnectionAsync(ct)) + { + await using var poison = new NpgsqlCommand(@" +UPDATE collect.collector_state +SET state_value = 'yesterday-ish' +WHERE server_id = $1 AND collector_name = $2 AND state_key = $3", connection); + poison.Parameters.AddWithValue(PgSelfAlertDeliveryStampStore.FleetServerId); + poison.Parameters.AddWithValue(PgSelfAlertDeliveryStampStore.StateCollectorName); + poison.Parameters.AddWithValue(PgSelfAlertDeliveryStampStore.CostDigestStateKey); + Assert.Equal(1, await poison.ExecuteNonQueryAsync(ct)); + } + + Assert.Null(await store.GetDeliveredAtUtcAsync(PgSelfAlertDeliveryStampStore.CostDigestStateKey, ct)); + Assert.Contains("not a round-trip UTC instant", log.Joined, StringComparison.Ordinal); + Assert.Contains("yesterday-ish", log.Joined, StringComparison.Ordinal); + } +} diff --git a/Darling/Darling.Tests/SelfAlertDeliveryStampTests.cs b/Darling/Darling.Tests/SelfAlertDeliveryStampTests.cs new file mode 100644 index 000000000..67c062249 --- /dev/null +++ b/Darling/Darling.Tests/SelfAlertDeliveryStampTests.cs @@ -0,0 +1,574 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Linq; +using System.Text.Json; +using System.Text.RegularExpressions; +using System.Threading; +using System.Threading.Tasks; +using Microsoft.Extensions.Logging; +using PerformanceMonitor.Alerting; +using PerformanceMonitor.Collectors; +using PerformanceMonitor.Common; +using PerformanceMonitor.Darling.Service; +using PerformanceMonitor.Darling.Service.Mcp; +using PerformanceMonitor.Darling.Storage; +using PerformanceMonitor.Notifications; +using Xunit; + +namespace Darling.Tests; + +/// +/// #3580: the two daily documents — the collector-cost digest and the fleet-sweep rollup — gate on +/// DELIVERED-TODAY in the store rather than fired-today in process memory, so a service restart does not +/// re-announce a document the previous process delivered an hour ago, and DOES re-attempt one whose +/// delivery failed. +/// +/// Every pin here runs over BOTH documents (): the issue names both, +/// the fix is one shared gate, and a pin over one document would let the other drift back to memory-only +/// unnoticed. Each case builds the evaluator TWICE over one stamp store — the "simulated restart" is a new +/// evaluator instance with empty dictionaries and the same store, which is exactly what a fresh process is. +/// The deliverer is the product's own derivation fed a fabricated +/// fan-out result, not a hand-written disposition, so "failed" here is the value the shipped deliverer +/// would actually report for a webhook that came back 500. +/// +/// The install-night pair, as pins. Six re-announcements among ~23 posts is +/// ; the failed pair that +/// the restart correctly recovered is . +/// The two are asserted in the same file so neither can be "fixed" at the other's expense: a gate that +/// never re-delivers passes the first and fails the second; a gate that always re-delivers, the reverse. +/// +public class SelfAlertDeliveryStampTests +{ + private static CancellationToken Ct => TestContext.Current.CancellationToken; + + private static readonly DateTime Day = new(2026, 9, 17, 23, 30, 0, DateTimeKind.Utc); + + /* ---------------- fakes ---------------- */ + + /// A deliverer that records what it was handed and REPORTS a disposition — the product's own + /// derivation over a fabricated fan-out, so the values are the ones DarlingAlertDeliverer would + /// return, not literals a test author believes it returns. + private sealed class ReportingDeliverer : IAlertDeliverer + { + public List Outcomes { get; } = new(); + + /// What the webhook channel does on the next delivery. Delivered by default — the + /// steady state; a test flips it to Failed to stage the install-night fault. + public AlertChannelOutcome Webhook { get; set; } = AlertChannelOutcome.Delivered; + + public Task DeliverAsync(AlertOutcome outcome, CancellationToken cancellationToken = default) + { + Outcomes.Add(outcome); + return Task.CompletedTask; + } + + public Task DeliverAndReportAsync(AlertOutcome outcome, CancellationToken cancellationToken = default) + { + Outcomes.Add(outcome); + var attempted = !outcome.Muted; + var fanout = new EmailFanoutResult( + EmailOutcome: AlertChannelOutcome.NotAttempted, + SendError: null, + WebhookOutcome: attempted ? Webhook : AlertChannelOutcome.NotAttempted, + WebhookSendError: attempted && Webhook == AlertChannelOutcome.Failed ? "Slack: 500 Internal Server Error" : null, + AnyChannelConfigured: true); + return Task.FromResult(AlertDelivery.FromFanout(fanout, outcome.Muted, trayChannelPresent: false)); + } + } + + /// The pre-#3580 shape: records, never reports. What every other suite's fake is, and what + /// Lite's deliverer still is — the default interface method answers null. + private sealed class SilentDeliverer : IAlertDeliverer + { + public List Outcomes { get; } = new(); + + public Task DeliverAsync(AlertOutcome outcome, CancellationToken cancellationToken = default) + { + Outcomes.Add(outcome); + return Task.CompletedTask; + } + } + + private sealed class RecordingHistoryStore : IAlertHistoryStore + { + public Task RecordAlertAsync(AlertHistoryRecord record) => Task.CompletedTask; + + public Task GetLastEmailSentUtcAsync(string serverId, string metricName, string? dedupKey = null) => + Task.FromResult(null); + + public Task GetLastWebhookSentUtcAsync(string serverId, string metricName, string? dedupKey = null) => + Task.FromResult(null); + + public Task GetLastAlertTimeAsync(string serverId, string metricName, string? dedupKey = null) => + Task.FromResult(null); + } + + /// The store between two "processes": a dictionary keyed the way the table is, with switches + /// to fault either half so the fallback path is the REAL one and not a mirror of it. + private sealed class MemoryStampStore : ISelfAlertDeliveryStampStore + { + public Dictionary Stamps { get; } = new(StringComparer.Ordinal); + public bool ThrowOnRead { get; set; } + public bool ThrowOnWrite { get; set; } + public int Reads { get; private set; } + public int Writes { get; private set; } + + public Task GetDeliveredAtUtcAsync(string stateKey, CancellationToken cancellationToken) + { + Reads++; + if (ThrowOnRead) + { + throw new InvalidOperationException("stamp read: store unreachable"); + } + + return Task.FromResult(Stamps.TryGetValue(stateKey, out var at) ? at : (DateTime?)null); + } + + public Task RecordDeliveredAtUtcAsync(string stateKey, DateTime deliveredAtUtc, CancellationToken cancellationToken) + { + Writes++; + if (ThrowOnWrite) + { + throw new InvalidOperationException("stamp write: store unreachable"); + } + + Stamps[stateKey] = deliveredAtUtc; + return Task.CompletedTask; + } + } + + /// One "process": the product's own settings over a default config, a deliverer, a clock, a + /// logger, a counter — and the SHARED stamp store handed in, so two harnesses over one store are two + /// processes over one database. + private sealed class Harness + { + public DarlingConfig Config { get; } = new(); + public IAlertDeliverer Deliverer { get; } + public MemoryStampStore? Stamps { get; } + public CapturingTestLogger Log { get; } = new(); + public AlertReadFailureCounter ReadFailures { get; } = new(); + public bool Muted { get; set; } + public DateTime Now { get; set; } = Day; + + public Harness(MemoryStampStore? stamps, IAlertDeliverer? deliverer = null) + { + Stamps = stamps; + Deliverer = deliverer ?? new ReportingDeliverer(); + } + + public List Outcomes => Deliverer switch + { + ReportingDeliverer r => r.Outcomes, + SilentDeliverer s => s.Outcomes, + _ => throw new InvalidOperationException("unknown deliverer"), + }; + + public DarlingSelfAlertEvaluator Build() => new( + new DarlingAlertSettings(Config), Deliverer, new RecordingHistoryStore(), _ => Muted, + logger: Log, utcNow: () => Now, readFailures: ReadFailures, deliveryStamps: Stamps); + } + + /* ---------------- the two documents, driven through their apply seams ---------------- */ + + /// One daily document as this suite drives it: its stamp key, its interval, and an apply + /// that always has something to say (a reportable fixture), so the only thing deciding whether a + /// delivery happens is the gate under test. + private sealed record Document(string Name, string StampKey, TimeSpan Interval, Func ApplyAsync); + + /* The digest suite's measured #3440 fixture, named-argument for named-argument, so the row is one the + shipped read could actually return and a member reorder is a compile error here rather than a + silently transposed figure. */ + private static readonly DarlingCollectorCostReader.CostMover[] OneMover = + { + new(ServerId: 1, ServerName: "pm-server-1", CollectorName: "query_store", LatestRuns: 2, + LatestWorstMs: 17_548, LatestMsPerRun: 17_548.0, BaselineMsPerRun: 6_477.0, + BaselineP95MsPerRun: 17_935.0, BaselineWorstDayMsPerRun: 17_935.0, BaselineDays: 13, + EligiblePairs: 177), + }; + + private static readonly DarlingCollectorCostReader.CollectorCostSummaryRow[] OneCensusRow = + { + new(CollectorName: "query_stats", RunCount: 43_891, TotalSqlMs: 61_724_459, MaxSqlMs: 88_561, + TotalStorageMs: 0, TotalRows: 0, ServerCount: 43), + }; + + private static FleetSweepRun[] OneTransitionDay(DateTime now) => new[] + { + new FleetSweepRun( + 1, now.AddHours(-2), now.AddHours(-3), now.AddHours(-2), null, true, 3, 3, true, + "{\"alive\":true}", + JsonSerializer.Serialize(new + { + changes = new { band_transitions = new[] { new { server = "pm-server-1", from = "Healthy", to = "Critical", reason = "deadlocks in span" } } }, + watch = new { opened = Array.Empty(), closed = Array.Empty() }, + fleet = new { bands = new Dictionary { ["Critical"] = 1, ["Healthy"] = 2 } }, + })), + }; + + private static readonly Dictionary Names = new() { [1] = "pm-server-1", [2] = "pm-server-2", [3] = "pm-server-3" }; + + private const string Digest = "collector-cost digest"; + private const string Rollup = "fleet-sweep rollup"; + + /// The theory rows are the documents' NAMES (serializable, so each is its own test case in + /// the runner) and resolve to a driver here. The names double as the {Document} the evaluator's + /// warnings name, which the fault pins assert. + private static Document For(string name) => name switch + { + Digest => new Document( + Digest, + PgSelfAlertDeliveryStampStore.CostDigestStateKey, + DarlingSelfAlertEvaluator.CollectorCostDigestInterval, + (e, _) => e.ApplyCollectorCostDigestAsync(OneMover, OneCensusRow, Ct)), + Rollup => new Document( + Rollup, + PgSelfAlertDeliveryStampStore.FleetSweepRollupStateKey, + DarlingSelfAlertEvaluator.FleetSweepRollupInterval, + (e, now) => e.ApplyFleetSweepRollupAsync( + OneTransitionDay(now), Array.Empty(), Names, + now - DarlingSelfAlertEvaluator.FleetSweepRollupInterval, now, Ct)), + _ => throw new ArgumentOutOfRangeException(nameof(name), name, "not a daily document this suite knows"), + }; + + /* ---------------- the install-night pair ---------------- */ + + /// + /// The six re-announcements, retired: a document delivered by one process is NOT delivered again by a + /// fresh process an hour later. The stamp the first process wrote is the first process's clock at the + /// fire, Kind Utc; the second process — empty dictionaries, same store — reads it, gates, and never + /// reaches its deliverer. Past the interval the fresh process delivers, because the stamp is old, not + /// because it is fresh. + /// + [Theory] + [InlineData(Digest)] + [InlineData(Rollup)] + public async Task ADeliveredDocument_IsNotReDeliveredByANewProcess_InsideItsInterval(string document) + { + var doc = For(document); + var store = new MemoryStampStore(); + + var first = new Harness(store); + await doc.ApplyAsync(first.Build(), first.Now); + Assert.Single(first.Outcomes); + Assert.True(store.Stamps.TryGetValue(doc.StampKey, out var stamped), "no delivery stamp was written"); + Assert.Equal(first.Now, stamped); + Assert.Equal(DateTimeKind.Utc, stamped.Kind); + + /* The restart: a new evaluator, one hour on, same store. */ + var second = new Harness(store) { Now = first.Now.AddHours(1) }; + var e2 = second.Build(); + await doc.ApplyAsync(e2, second.Now); + Assert.Empty(second.Outcomes); + + /* And the same instance keeps gating for the rest of the day without asking the store again — + the fast path: one read to learn the stamp, then memory. */ + var readsAfterFirstAsk = store.Reads; + second.Now = first.Now.Add(doc.Interval).AddMinutes(-1); + await doc.ApplyAsync(e2, second.Now); + Assert.Empty(second.Outcomes); + Assert.Equal(readsAfterFirstAsk, store.Reads); + + /* Past the interval the fresh process delivers — and re-stamps at ITS clock, without asking the + store again: one writer per store, so once memory is seeded the store cannot know more than it. */ + second.Now = first.Now.Add(doc.Interval).AddMinutes(1); + await doc.ApplyAsync(e2, second.Now); + Assert.Single(second.Outcomes); + Assert.Equal(second.Now, store.Stamps[doc.StampKey]); + Assert.Equal(2, store.Reads); + } + + /// + /// The recovery the install night showed working, kept by construction: a delivery the deliverer + /// reports FAILED writes no stamp, so the next tick retries — in the same process (the memory gate was + /// not set either) and in a fresh one. When the retry lands, the stamp is written at the retry's + /// clock, and the document is quiet from there. + /// + [Theory] + [InlineData(Digest)] + [InlineData(Rollup)] + public async Task AFailedDelivery_WritesNoStamp_SoTheNextTick_RestartOrNot_Retries(string document) + { + var doc = For(document); + var store = new MemoryStampStore(); + var deliverer = new ReportingDeliverer { Webhook = AlertChannelOutcome.Failed }; + + var first = new Harness(store, deliverer); + var e1 = first.Build(); + await doc.ApplyAsync(e1, first.Now); + Assert.Single(first.Outcomes); + Assert.Empty(store.Stamps); + Assert.Contains("delivery failed", first.Log.Joined, StringComparison.Ordinal); + Assert.Contains("Slack: 500", first.Log.Joined, StringComparison.Ordinal); + + /* Same process, next hourly tick: retried, still failing, still no stamp — and the store is not + asked again: "no row" was cached on the first tick, and a store this process alone writes cannot + have gained a row since. */ + first.Now = first.Now.AddHours(1); + await doc.ApplyAsync(e1, first.Now); + Assert.Equal(2, first.Outcomes.Count); + Assert.Empty(store.Stamps); + Assert.Equal(1, store.Reads); + + /* The restart, channel repaired: a fresh process asks once, retries and lands, and NOW the stamp exists. */ + deliverer.Webhook = AlertChannelOutcome.Delivered; + var second = new Harness(store, deliverer) { Now = first.Now.AddHours(1) }; + var e2 = second.Build(); + await doc.ApplyAsync(e2, second.Now); + Assert.Equal(3, deliverer.Outcomes.Count); + Assert.Equal(second.Now, store.Stamps[doc.StampKey]); + Assert.Equal(2, store.Reads); + + /* And from there, quiet — in this process and in the next. */ + second.Now = second.Now.AddHours(1); + await doc.ApplyAsync(e2, second.Now); + var third = new Harness(store, deliverer) { Now = second.Now.AddHours(1) }; + await doc.ApplyAsync(third.Build(), third.Now); + Assert.Equal(3, deliverer.Outcomes.Count); + Assert.Equal(3, store.Reads); + } + + /* ---------------- the stamp's age is the whole test ---------------- */ + + /// A stamp OLDER than the interval gates nothing — a fresh process delivers and overwrites it; + /// one a minute inside the interval gates. The gate reads the stamp's age, not its presence. + [Theory] + [InlineData(Digest)] + [InlineData(Rollup)] + public async Task AStampsAge_DecidesTheGate_NotItsPresence(string document) + { + var doc = For(document); + var stale = new MemoryStampStore(); + stale.Stamps[doc.StampKey] = Day - doc.Interval - TimeSpan.FromMinutes(1); + var h1 = new Harness(stale); + await doc.ApplyAsync(h1.Build(), h1.Now); + Assert.Single(h1.Outcomes); + Assert.Equal(Day, stale.Stamps[doc.StampKey]); + + var fresh = new MemoryStampStore(); + fresh.Stamps[doc.StampKey] = Day - doc.Interval + TimeSpan.FromMinutes(1); + var h2 = new Harness(fresh); + await doc.ApplyAsync(h2.Build(), h2.Now); + Assert.Empty(h2.Outcomes); + Assert.Equal(0, fresh.Writes); + } + + /* ---------------- store faults fall open to memory, once, loudly ---------------- */ + + /// + /// A stamp READ that throws does not silence the document and does not spam it: the tick falls back + /// to the process-memory gate (empty in a fresh process, so it delivers once), the next tick in the + /// same process is gated by memory, a warning names the document and the fallback, and the read is + /// counted into #3013's swallowed-read census under a name that says which document could not ask. + /// The write that follows the delivery is attempted regardless — a store whose read failed may well + /// take the write, and if it does the NEXT process is spared the re-announcement. + /// + [Theory] + [InlineData(Digest)] + [InlineData(Rollup)] + public async Task AStampReadFault_FallsBackToProcessMemory_DeliversOnce_AndWarns(string document) + { + var doc = For(document); + var store = new MemoryStampStore { ThrowOnRead = true }; + var h = new Harness(store); + var e = h.Build(); + + await doc.ApplyAsync(e, h.Now); + Assert.Single(h.Outcomes); + Assert.Matches(new Regex(@"Warning: .*delivery stamp could not be read"), h.Log.Joined); + Assert.Contains(doc.Name, h.Log.Joined, StringComparison.Ordinal); + Assert.Equal(1, h.ReadFailures.ReadInstance().ReadFailures); + /* One read name for both documents — the #3013 census keys a counted site on a literal, and the + gate is one site; the log line above is where the document is named. */ + Assert.Equal("daily-document delivery-stamp self-alert", h.ReadFailures.ReadInstance().LastFailureRead); + + /* The write still happened, so a store that only failed to READ has the stamp for the next process. */ + Assert.Equal(1, store.Writes); + Assert.Equal(h.Now, store.Stamps[doc.StampKey]); + + /* Same process, next tick: memory gates it; the store is not asked again (memory answered). */ + h.Now = h.Now.AddHours(1); + await doc.ApplyAsync(e, h.Now); + Assert.Single(h.Outcomes); + Assert.Equal(1, h.ReadFailures.ReadInstance().ReadFailures); + Assert.Equal(1, store.Reads); + } + + /// + /// The fault is met ONCE per process, not once per gate consult. The evaluate half's pre-check and the + /// apply half's own check both run on one tick, seconds apart; a gate that re-asked the store after a + /// fault would log the same failure twice and count it twice in #3013's census on every tick the fault + /// persisted. Staged with a delivery that ALSO fails, so nothing but the cached "asked, nothing known" + /// sentinel can be what stops the second consult from reaching the store — a successful delivery would + /// have populated memory on its own and hidden the difference. + /// + [Theory] + [InlineData(Digest)] + [InlineData(Rollup)] + public async Task AStampReadFault_IsMetOnce_PerProcess_EvenWhenNothingLands(string document) + { + var doc = For(document); + var store = new MemoryStampStore { ThrowOnRead = true, ThrowOnWrite = true }; + var deliverer = new ReportingDeliverer { Webhook = AlertChannelOutcome.Failed }; + var h = new Harness(store, deliverer); + var e = h.Build(); + + /* Two consults on "one tick" (the evaluate half, then the apply half): one store read, one + warning, one count — and the document is attempted both times, because nothing is known to have + been delivered and the memory gate is open. */ + await doc.ApplyAsync(e, h.Now); + await doc.ApplyAsync(e, h.Now); + Assert.Equal(2, deliverer.Outcomes.Count); + Assert.Equal(1, store.Reads); + Assert.Equal(1, h.ReadFailures.ReadInstance().ReadFailures); + Assert.Equal(1, Regex.Matches(h.Log.Joined, "delivery stamp could not be read").Count); + + /* The next hour: still nothing landed, still one read on record — the retry runs from memory. */ + h.Now = h.Now.AddHours(1); + await doc.ApplyAsync(e, h.Now); + Assert.Equal(3, deliverer.Outcomes.Count); + Assert.Equal(1, store.Reads); + Assert.Equal(1, h.ReadFailures.ReadInstance().ReadFailures); + + /* A fresh process meets the fault once more — per process is the unit. */ + var next = new Harness(store, deliverer) { Now = h.Now.AddHours(1) }; + await doc.ApplyAsync(next.Build(), next.Now); + Assert.Equal(2, store.Reads); + Assert.Equal(1, next.ReadFailures.ReadInstance().ReadFailures); + } + + /// A stamp WRITE that throws leaves the document delivered and the process gated — memory is + /// stamped before the store is asked — with a warning that says the next restart will re-announce. + /// Not counted: a write is not a condition read (the resolution-row precedent). + [Theory] + [InlineData(Digest)] + [InlineData(Rollup)] + public async Task AStampWriteFault_StillGatesThisProcess_AndWarns_Uncounted(string document) + { + var doc = For(document); + var store = new MemoryStampStore { ThrowOnWrite = true }; + var h = new Harness(store); + var e = h.Build(); + + await doc.ApplyAsync(e, h.Now); + Assert.Single(h.Outcomes); + Assert.Matches(new Regex(@"Warning: .*delivery stamp could not be written"), h.Log.Joined); + Assert.Equal(0, h.ReadFailures.ReadInstance().ReadFailures); + + h.Now = h.Now.AddHours(1); + await doc.ApplyAsync(e, h.Now); + Assert.Single(h.Outcomes); + + /* And, stated: a fresh process over this store WILL re-announce — the bounded cost of a store + that would not take the write, and the pre-#3580 posture exactly. */ + var next = new Harness(store) { Now = h.Now }; + await doc.ApplyAsync(next.Build(), next.Now); + Assert.Single(next.Outcomes); + } + + /* ---------------- what counts as delivered ---------------- */ + + /// A deliverer that does not REPORT — the default interface method, every pre-#3580 fake, + /// Lite's deliverer — stamps: null is "unreported", treated as every fire before #3580 was, and never + /// read as "failed". A deliverer that knows a send failed says so. + [Theory] + [InlineData(Digest)] + [InlineData(Rollup)] + public async Task AnUnreportedDelivery_Stamps(string document) + { + var doc = For(document); + var store = new MemoryStampStore(); + var h = new Harness(store, new SilentDeliverer()); + await doc.ApplyAsync(h.Build(), h.Now); + + Assert.Single(h.Outcomes); + Assert.Equal(h.Now, store.Stamps[doc.StampKey]); + } + + /// A MUTED delivery stamps: the mute rule chose the silence, the history row was written + /// flagged muted, and nothing is owed — retrying hourly would write 24 muted rows a day for a document + /// the operator asked not to see. + [Theory] + [InlineData(Digest)] + [InlineData(Rollup)] + public async Task AMutedDelivery_Stamps_BecauseNothingIsOwed(string document) + { + var doc = For(document); + var store = new MemoryStampStore(); + var h = new Harness(store) { Muted = true }; + await doc.ApplyAsync(h.Build(), h.Now); + + var fired = Assert.Single(h.Outcomes); + Assert.True(fired.Muted); + Assert.Equal(h.Now, store.Stamps[doc.StampKey]); + } + + /// No stamp store at all is the pre-#3580 gate exactly — process memory only, so a fresh + /// evaluator re-delivers. Stated so the seam's default is a documented posture and not an accident; + /// production always passes the store. + [Theory] + [InlineData(Digest)] + [InlineData(Rollup)] + public async Task WithoutAStampStore_TheGateIsProcessMemoryOnly(string document) + { + var doc = For(document); + var first = new Harness(stamps: null); + var e1 = first.Build(); + await doc.ApplyAsync(e1, first.Now); + first.Now = first.Now.AddHours(1); + await doc.ApplyAsync(e1, first.Now); + Assert.Single(first.Outcomes); + + var second = new Harness(stamps: null) { Now = first.Now }; + await doc.ApplyAsync(second.Build(), second.Now); + Assert.Single(second.Outcomes); + } + + /* ---------------- the store's row identity ---------------- */ + + /// The stamp rows sit under the fleet sentinel BOTH existing fleet-scope writers already use + /// (the retention purge's run-record and the sweep's fleet watch items), and under an owner name that + /// is not a collector definition's — so no declared-key read and no per-database prune can reach them. + /// The two keys are distinct, and the SQL names the table with its schema. + [Fact] + public void TheStampRows_UseTheFleetSentinel_AndAnOwnerNameNoCollectorClaims() + { + Assert.Equal(FleetSweepStore.FleetScopeServerId, PgSelfAlertDeliveryStampStore.FleetServerId); + Assert.Equal(0, PgSelfAlertDeliveryStampStore.FleetServerId); + Assert.Equal("self_alert", PgSelfAlertDeliveryStampStore.StateCollectorName); + Assert.NotEqual(PgSelfAlertDeliveryStampStore.CostDigestStateKey, PgSelfAlertDeliveryStampStore.FleetSweepRollupStateKey); + + /* Not a collector definition's name, and not either of the two other worker-owned owner names + already in the table (the backfill worker's and the open-interval state's) — three owners, three + names, no row can be read as another's. */ + Assert.DoesNotContain( + PgSelfAlertDeliveryStampStore.StateCollectorName, + CollectorCatalog.All.Select(c => c.Name), + StringComparer.Ordinal); + Assert.NotEqual(QueryStoreBackfillState.StateCollectorName, PgSelfAlertDeliveryStampStore.StateCollectorName); + Assert.NotEqual(QueryStoreOpenIntervalState.StateCollectorName, PgSelfAlertDeliveryStampStore.StateCollectorName); + + Assert.Contains("collect.collector_state", PgSelfAlertDeliveryStampStore.GetSql, StringComparison.Ordinal); + Assert.Contains("collect.collector_state", PgSelfAlertDeliveryStampStore.UpsertSql, StringComparison.Ordinal); + Assert.Contains("ON CONFLICT (server_id, collector_name, state_key)", PgSelfAlertDeliveryStampStore.UpsertSql, StringComparison.Ordinal); + } + + /// The worker hands the evaluator the store-backed stamps — the seam's null default is for + /// harnesses, and a production evaluator built without it would be the six re-announcements back. + [Fact] + public void TheWorker_HandsTheEvaluatorTheStoreBackedStamps() + { + var worker = RepoFile.ReadRepoFile("Darling", "PerformanceMonitor.Darling.Service", "DarlingWorker.cs"); + var evaluatorBuild = worker.IndexOf("_selfAlerts = new DarlingSelfAlertEvaluator(", StringComparison.Ordinal); + Assert.True(evaluatorBuild >= 0, "#3580 pin: the worker no longer constructs DarlingSelfAlertEvaluator where this pin looks"); + var end = worker.IndexOf(");", evaluatorBuild, StringComparison.Ordinal); + Assert.Contains("deliveryStamps: new PgSelfAlertDeliveryStampStore(", worker[evaluatorBuild..end], StringComparison.Ordinal); + } +} diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingAlertDeliverer.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingAlertDeliverer.cs index af0dddb03..38e874bbd 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingAlertDeliverer.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingAlertDeliverer.cs @@ -74,7 +74,24 @@ public DarlingAlertDeliverer( _core = new EmailSendCore(settings, historyStore, webhookAlertService, s_branding, logger); } - public async Task DeliverAsync(AlertOutcome outcome, CancellationToken cancellationToken = default) + public Task DeliverAsync(AlertOutcome outcome, CancellationToken cancellationToken = default) => + DeliverAndReportAsync(outcome, cancellationToken); + + /// + /// The delivery, reporting its disposition (#3580). Every send goes through here — + /// is this with the answer discarded — so there is one delivery path and + /// not a reporting one beside a silent one. + /// + /// What comes back. On the combined send (Summary mode, or any alert without incidents, + /// which is every self-alert) the exact the history row was written with. + /// On a Per-event split there are N sends and N rows and no single disposition describes them, so this + /// returns null — "unreported" — rather than electing one; the two callers that read the + /// answer (the digest and the rollup) fire with Context: null and never take that path. The + /// belt-and-suspenders catch below also answers null: both TrySendAsync and + /// RecordAlertAsync are failure-isolated themselves, so a throw here is something outside the + /// channels and says nothing about whether they delivered. + /// + public async Task DeliverAndReportAsync(AlertOutcome outcome, CancellationToken cancellationToken = default) { if (outcome is null) { @@ -102,11 +119,11 @@ await SendAndRecordAsync( deliveryMode: mode); } - return; + return null; } /* Summary mode, or an alert with no incidents (CPU/low-disk/jobs): one combined send+row, unchanged. */ - await SendAndRecordAsync( + return await SendAndRecordAsync( outcome, outcome.CurrentValue, outcome.Context, outcome.DetailText, outcome.NumericCurrentValue, outcome.NumericThresholdValue, deliveryMode: mode); @@ -119,6 +136,7 @@ await SendAndRecordAsync( { _logger.LogError("Alert delivery failed for {Metric} on {Server}: {Message}", outcome.MetricName, outcome.ServerName, ex.Message); + return null; } } @@ -130,12 +148,14 @@ await SendAndRecordAsync( /// still record (flagged muted). /// /// - /// The mode resolved for this server, forwarded to the shared send core for - /// #3430's per-metric repeat ceiling. Passed rather than re-resolved so one alert's two channels and its - /// history row all describe the same decision, and passed FAITHFULLY on the Per-event split — those - /// messages must not be aggregated, which is that mode's own contract. + /// The mode resolved for this server, forwarded to the shared send + /// core for #3430's per-metric repeat ceiling. Passed rather than re-resolved so one alert's two channels + /// and its history row all describe the same decision, and passed FAITHFULLY on the Per-event split — + /// those messages must not be aggregated, which is that mode's own contract. /// - private async Task SendAndRecordAsync( + /// The disposition the history row was written with — the same value, so what the caller is + /// told and what the operator later reads in the alert log cannot disagree (#3580). + private async Task SendAndRecordAsync( AlertOutcome outcome, string currentValue, AlertContext? context, string? detailText, double? numericCurrentValue, double? numericThresholdValue, AlertNotificationMode deliveryMode) { @@ -175,5 +195,7 @@ await _historyStore.RecordAlertAsync(new AlertHistoryRecord( numericCurrentValue, numericThresholdValue, delivery, outcome.Muted, detailText, contextJson)); + + return delivery; } } diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingSelfAlertEvaluator.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingSelfAlertEvaluator.cs index d0943e066..06ad5082c 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingSelfAlertEvaluator.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingSelfAlertEvaluator.cs @@ -335,11 +335,28 @@ reports so the threshold a reader falsifies against is the one that selected the /// and cannot be taken from a phone at 3am. A longer interval alone would have produced the same /// unactionable card less often. /// - /// In-memory, like every sibling's interval, and a restart costs one extra digest. That is - /// the failure class and #3430's RepeatDeliveryBudget both already - /// accept, and it is what keeps this change free of a migration rung. An extra copy of a report is the - /// cheapest possible failure; nothing is lost either way, because the digest is recomputed from the - /// store every time rather than accumulated in process. + /// Gated on DELIVERED-TODAY in the store, not fired-today in memory (#3580). This shipped + /// as "in-memory, like every sibling's interval, and a restart costs one extra digest" — the failure + /// class and #3430's RepeatDeliveryBudget both accept, on the + /// argument that an extra copy of a report is the cheapest possible failure. The v3.8.0 install night + /// priced it: three stores restarted once each, and the channel carried SIX re-announcements among ~23 + /// overnight posts — a quarter of the channel was this document and the rollup, twice — because a fresh + /// process has an empty and the gate was answering "has THIS PROCESS sent + /// one today" when the reader's question is "has one been DELIVERED today". The same night showed the + /// case the fix must keep: a pair whose delivery had FAILED (a transport fault, not a suppression) was + /// correctly re-attempted after the restart and landed; the restart was the recovery. + /// + /// So the gate now reads a delivery stamp from (the + /// store's existing key/value state table, no rung) and skips while now - stamp is inside this + /// interval, restart or not; and the stamp is written only when the deliverer reports a disposition + /// other than . A failed delivery writes nothing, so the next + /// tick — restart or not — retries; the in-memory dictionary remains as a same-process fast path (23 of + /// 24 ticks still cost one lookup and no store read) but is no longer the authority. A store fault on + /// the stamp falls back to that fast path and warns — fail-open toward delivering, the direction every + /// store-fault posture in this evaluator already takes, because the alternative (skip on an unreadable + /// stamp) would let a store hiccup silence a daily document, and the memory gate still bounds the + /// fallback at one copy per process. The nothing-is-lost half of the original argument still holds: + /// the digest is recomputed from the store every time rather than accumulated in process. /// internal static readonly TimeSpan CollectorCostDigestInterval = TimeSpan.FromDays(1); @@ -352,9 +369,11 @@ reports so the threshold a reader falsifies against is the one that selected the /// How many collectors the digest's heaviest-first census spells out. private const int MaxListedCostHeaviest = 10; - /// When the digest was last sent — the idiom, one fixed key. - /// A digest has no active flag and no resolution edge: it is a report of a measurement, not a condition - /// that can be entered and left, so there is nothing to clear. + /// When the digest was last known DELIVERED — the idiom, one + /// fixed key, but since #3580 a CACHE of the store's stamp rather than the authority: filled from the + /// stamp on the first tick that has to ask, and from the fire itself on a delivery the deliverer did + /// not report failed. A digest has no active flag and no resolution edge: it is a report of a + /// measurement, not a condition that can be entered and left, so there is nothing to clear. private readonly ConcurrentDictionary _lastCostDigest = new(); /// @@ -378,10 +397,15 @@ reports so the threshold a reader falsifies against is the one that selected the /// the trailing day and only the trailing day, so what a post claims to cover and how often one can /// arrive are the same number and cannot drift apart. /// - /// In-memory, like , and a restart costs one extra - /// rollup. The same accepted failure class: an extra copy of a report is the cheapest possible - /// failure, and the rollup is recomputed from the sweep store every time rather than accumulated in - /// process, so nothing is lost in either direction. + /// Gated on delivered-today in the store, like and + /// for the same night's reason (#3580). This shipped in-memory with "a restart costs one extra + /// rollup" accepted as the cheapest failure; the install night's census — three restarts, six + /// re-announcements, this document being three of them — is what that acceptance cost, and the digest's + /// remarks carry the arc. The one-post-per-day CEILING above is a ruling, and a gate that a restart + /// resets is a ceiling the deployment procedure breaches on every install. The rollup takes the same + /// stamp store, the same failed-writes-nothing rule and the same fail-open fallback, under its own key. + /// Still recomputed from the sweep store every time rather than accumulated in process, so nothing is + /// lost in either direction. /// internal static readonly TimeSpan FleetSweepRollupInterval = TimeSpan.FromDays(1); @@ -398,9 +422,9 @@ reports so the threshold a reader falsifies against is the one that selected the /// stay readable on the fleet-wide day it exists for. private const int MaxListedRollupLedgerServers = 10; - /// When the rollup was last sent — the idiom, one fixed key. A - /// rollup is a report of a period, not a condition: no active flag, no resolution edge, nothing to - /// clear. + /// When the rollup was last known DELIVERED — the idiom, one + /// fixed key, and since #3580 the same cache-of-the-stamp role rather than the authority. A rollup is a + /// report of a period, not a condition: no active flag, no resolution edge, nothing to clear. private readonly ConcurrentDictionary _lastSweepRollup = new(); /// The fixed key for the fleet-level Store Disk Pressure edge (not a real server). @@ -776,7 +800,8 @@ public DarlingSelfAlertEvaluator( Func? retentionHoldWarnRatio = null, Func? retentionHoldCriticalRatio = null, AlertReadFailureCounter? readFailures = null, - string? storeName = null) + string? storeName = null, + ISelfAlertDeliveryStampStore? deliveryStamps = null) { _settings = settings ?? throw new ArgumentNullException(nameof(settings)); _deliverer = deliverer ?? throw new ArgumentNullException(nameof(deliverer)); @@ -811,8 +836,18 @@ than restated (#3060): a literal here is a third copy of the same number that a seam behaves like a store that never opted in — the AG-seam discipline, and the byte-identical promise the opt-in stands on. */ _storeLabel = EffectiveStoreLabel(storeName); + /* #3580: unsupplied means the two daily documents gate on process memory alone — the pre-#3580 + behavior, and what every test harness that does not care about restarts gets. Production + passes the store-backed stamps. */ + _deliveryStamps = deliveryStamps; } + /// + /// Where the two daily documents' DELIVERED-TODAY stamps live across restarts (#3580), or null when the + /// process-memory gate is the only gate. See for the arc. + /// + private readonly ISelfAlertDeliveryStampStore? _deliveryStamps; + /// /// Where a SWALLOWED self-alert store read is counted (#3013), or null when nothing is counting. /// Only the conditions that READ the store here increment it; the fleet-scoped conditions are handed @@ -1195,11 +1230,13 @@ rather than guessing. Falling back to "route everything to the page" would reins var routing = RouteCostRegressions(regressions, census); await ApplyCostRegressionsAsync(routing.Paging, cancellationToken); - /* The digest's own interval, checked here so 23 of every 24 hourly ticks do no extra store work. - The master switch already returned above, before the first store read (#3464); AlertsEnabled is + /* The digest's own interval, checked here so 23 of every 24 hourly ticks do no extra store work — + the delivered-today gate (#3580): memory first, the stamp only when memory cannot answer. The + master switch already returned above, before the first store read (#3464); AlertsEnabled is ALSO checked inside the apply, like every sibling, so a direct caller cannot skip it. */ - if (_lastCostDigest.TryGetValue(CollectorCostDigestKey, out var lastDigest) - && _utcNow() - lastDigest < CollectorCostDigestInterval) + if (await DocumentDeliveredInsideIntervalAsync( + _lastCostDigest, CollectorCostDigestKey, PgSelfAlertDeliveryStampStore.CostDigestStateKey, + CollectorCostDigestInterval, _utcNow(), "collector-cost digest", cancellationToken)) { return; } @@ -1467,8 +1504,9 @@ internal async Task ApplyCollectorCostDigestAsync( } var now = _utcNow(); - if (_lastCostDigest.TryGetValue(CollectorCostDigestKey, out var lastSent) - && now - lastSent < CollectorCostDigestInterval) + if (await DocumentDeliveredInsideIntervalAsync( + _lastCostDigest, CollectorCostDigestKey, PgSelfAlertDeliveryStampStore.CostDigestStateKey, + CollectorCostDigestInterval, now, "collector-cost digest", cancellationToken)) { return; } @@ -1478,10 +1516,13 @@ internal async Task ApplyCollectorCostDigestAsync( return; } - _lastCostDigest[CollectorCostDigestKey] = now; var (shortMessage, detail) = RenderCollectorCostDigest(movers, census); - await FireAsync( + /* #3580: the stamp is written AFTER the fire and only on a delivery the deliverer did not report + failed — it used to be written before the fire, unconditionally, which is the "fired-today" gate + this issue retires. A fire that throws before the deliverer is reached (the mute seam) delivered + nothing either, and takes the same path: no stamp, the next tick retries. */ + var delivery = await FireAsync( StoreKey(CollectorCostDigestKey), _storeLabel, CollectorCostDigestMetric, currentValue: movers.Count.ToString(CultureInfo.InvariantCulture), /* There is no threshold. Saying so in the string is the point of the string: this surface @@ -1497,6 +1538,10 @@ numeric below is the 0 the NOT NULL column demands. */ numericCurrentValue: movers.Count, numericThresholdValue: 0, cancellationToken); + + await RecordDocumentDeliveredAsync( + _lastCostDigest, CollectorCostDigestKey, PgSelfAlertDeliveryStampStore.CostDigestStateKey, + delivery, now, "collector-cost digest", cancellationToken); } /// @@ -1636,8 +1681,10 @@ is spelled from the constant so the words cannot drift from the read. */ /// left to discover it. /// /// Called from the worker's hourly store-metrics tick beside the collector-cost evaluation; 23 of - /// every 24 ticks cost one dictionary lookup. Testable through - /// with a recording deliverer and a controllable clock. + /// every 24 ticks cost one dictionary lookup, and the first tick after a start costs one stamp read + /// instead of a re-announcement (#3580 — ). Testable + /// through with a recording deliverer, a controllable clock and + /// an in-memory stamp store. /// public async Task EvaluateFleetSweepRollupAsync(NpgsqlDataSource postgres, CancellationToken cancellationToken) { @@ -1649,8 +1696,10 @@ public async Task EvaluateFleetSweepRollupAsync(NpgsqlDataSource postgres, Cance } var now = _utcNow(); - if (_lastSweepRollup.TryGetValue(FleetSweepRollupKey, out var lastSent) - && now - lastSent < FleetSweepRollupInterval) + /* #3580: delivered-today, memory first and the stamp when memory cannot answer. */ + if (await DocumentDeliveredInsideIntervalAsync( + _lastSweepRollup, FleetSweepRollupKey, PgSelfAlertDeliveryStampStore.FleetSweepRollupStateKey, + FleetSweepRollupInterval, now, "fleet-sweep rollup", cancellationToken)) { return; } @@ -1730,8 +1779,9 @@ internal async Task ApplyFleetSweepRollupAsync( } var now = _utcNow(); - if (_lastSweepRollup.TryGetValue(FleetSweepRollupKey, out var lastSent) - && now - lastSent < FleetSweepRollupInterval) + if (await DocumentDeliveredInsideIntervalAsync( + _lastSweepRollup, FleetSweepRollupKey, PgSelfAlertDeliveryStampStore.FleetSweepRollupStateKey, + FleetSweepRollupInterval, now, "fleet-sweep rollup", cancellationToken)) { return; } @@ -1742,10 +1792,10 @@ internal async Task ApplyFleetSweepRollupAsync( return; } - _lastSweepRollup[FleetSweepRollupKey] = now; var (shortMessage, detail) = RenderFleetSweepRollup(facts, spanStartUtc, spanEndUtc); - await FireAsync( + /* #3580: stamped after the fire, and only on a delivery not reported failed — the digest's rule. */ + var delivery = await FireAsync( StoreKey(FleetSweepRollupKey), _storeLabel, FleetSweepRollupMetric, currentValue: facts.Sweeps.ToString(CultureInfo.InvariantCulture), /* There is no threshold - the digest's exact posture, stated in the string because the NOT NULL @@ -1760,6 +1810,179 @@ column demands a value and "report" is the honest one. */ numericCurrentValue: facts.Sweeps, numericThresholdValue: 0, cancellationToken); + + await RecordDocumentDeliveredAsync( + _lastSweepRollup, FleetSweepRollupKey, PgSelfAlertDeliveryStampStore.FleetSweepRollupStateKey, + delivery, now, "fleet-sweep rollup", cancellationToken); + } + + /* ------------------------- #3580: the daily documents' delivered-today gate ------------------------- */ + + /// + /// Whether a daily document was DELIVERED inside its interval as of — the gate + /// both documents' evaluate and apply halves consult (#3580). The store SEEDS process memory after a + /// start; memory serves the process; every delivery writes both — the shape #981 gave the email + /// cooldown (IAlertHistoryStore.GetLastEmailSentUtcAsync seeds it across restart), applied to + /// the one gate that was still memory-only. + /// + /// Memory answers whenever it holds anything. 23 of every 24 hourly ticks fall inside the + /// interval of a delivery this process already knows about, and those ticks cost one dictionary lookup + /// and no store round-trip — the cost promise the documents' callers make. Memory is also allowed to + /// answer "outside the interval, deliver" without re-asking the store, because there is ONE writer per + /// store and it is this process: once memory is seeded the store can never hold a newer stamp than + /// memory does, so a second read could only return what memory already knows. That is also why the + /// evaluate half's pre-check and the apply half's own check cost one store read between them and not + /// two — the first seeds, the second finds memory populated. + /// + /// The stamp store is asked ONCE per document per process — on the first tick, when memory is + /// empty — and whatever it answers is cached, including "nothing": a stamp inside the interval + /// gates; a stamp outside it lets the document proceed to its reads and stays cached, so a subsequent + /// failed delivery leaves memory pointing at the last REAL delivery rather than at nothing; no row, or + /// a read that failed, caches , which reads as "deliver" from then on. + /// Caching the empty answer is exact under the single-writer fact above — a store that had no row for + /// this process's first tick cannot gain one except through this process, which would populate memory + /// directly — and it is what keeps the cost model honest in the fault case as well as the happy one: the + /// evaluate half's pre-check asks, and the apply half's own check, seconds later on the same tick, + /// finds memory populated whether the store answered, was empty, or threw. Re-asking on a fault would + /// log the same failure twice and count it twice in the #3013 census on every tick the fault persisted, + /// for one logical failure. The retry the install night's recovery case needs is unaffected: a document + /// that has never delivered reads "deliver" from memory on every tick until a delivery lands. + /// + /// A stamp read that fails falls OPEN to memory, warns, and is counted. Fail-open toward + /// delivering is the direction every store-fault posture in this evaluator already takes for its + /// documents — the rollup's read fault "skips the tick WITHOUT consuming the interval" so an unreadable + /// store cannot become a permanently quiet channel; the digest's does the same — and it is bounded: the + /// memory gate still holds within the process once one delivery lands, so a store that cannot answer + /// costs at most one extra copy per process, which is exactly the pre-#3580 posture and not a spam + /// path. Failing CLOSED (skip when the stamp cannot be read) would let a store hiccup silence a daily + /// document, the worse failure. Counted into #3013's census because it is a store read the alert pass + /// performed, failed and swallowed, and the census exists so that population is not invisible; the + /// warning beside it names the document that could not ask. Counted ONCE per process, per the + /// paragraph above — the census measures faults the pass met, and this pass meets this one once. + /// + /// No store configured ( null) is the pre-#3580 gate exactly: memory + /// only. + /// + private async Task DocumentDeliveredInsideIntervalAsync( + ConcurrentDictionary lastDelivered, string memoryKey, string stampKey, + TimeSpan interval, DateTime now, string documentName, CancellationToken cancellationToken) + { + if (lastDelivered.TryGetValue(memoryKey, out var known)) + { + return known != NoDeliveryKnown && now - known < interval; + } + + if (_deliveryStamps is null) + { + return false; + } + + DateTime? stamped; + var stampClock = Stopwatch.StartNew(); + try + { + stamped = await _deliveryStamps.GetDeliveredAtUtcAsync(stampKey, cancellationToken); + } + catch (OperationCanceledException) when (cancellationToken.IsCancellationRequested) + { + throw; + } + catch (Exception ex) + { + /* One read name for both documents, because the #3013 census keys a counted site on a LITERAL + name with its own clock and this is one site serving two callers; the log line beside it + names the document, so the actionable half is not lost — it is a line away. The sentinel + is what makes this warning and this count fire once per process rather than once per check: + the apply half's own gate, seconds from now, finds memory populated and does not re-ask. */ + _logger?.LogWarning(ex, + "{Document} delivery stamp could not be read after {ElapsedMs} ms; gating on process memory from here, which re-announces once per restart until the store answers", + documentName, stampClock.ElapsedMilliseconds); + _readFailures?.RecordReadFailure(null, "daily-document delivery-stamp self-alert", stampClock.ElapsedMilliseconds); + lastDelivered[memoryKey] = NoDeliveryKnown; + return false; + } + + if (stamped is not DateTime deliveredAt) + { + lastDelivered[memoryKey] = NoDeliveryKnown; + return false; + } + + lastDelivered[memoryKey] = deliveredAt; + return now - deliveredAt < interval; + } + + /// + /// What caches when the store was asked and had no + /// answer — no row, or a read that threw — so the store is asked once per document per process and + /// never re-asked on the same tick by the apply half (#3580). Reads as "deliver": the gate compares it + /// by identity before the interval arithmetic, so it can never be mistaken for a real stamp, and the + /// first delivery that lands replaces it with a real one. rather than a + /// nullable value because the dictionaries are the pre-#3580 shape and every sibling gate in this file + /// keys on presence; a value that means "asked, nothing known" keeps presence meaning "asked". + /// + private static readonly DateTime NoDeliveryKnown = DateTime.MinValue; + + /// + /// Records that a daily document was DELIVERED at — into process memory and, + /// when a store is configured, into the delivery stamp — unless the deliverer reported the send + /// (#3580). + /// + /// "Failed" is the one disposition that writes nothing, and the rule is stated as that one + /// exclusion on purpose. It is the disposition the install night's recovery case wore: a channel + /// was attempted and came back unsuccessful, so nothing reached a reader, and the next tick — restart + /// or not — must retry. Every other answer is either a delivery () or + /// the product's own decision that nothing is owed: (a mute + /// rule chose the silence), (no channel exists to + /// deliver to — the history row IS the delivery, and retrying hourly would write 24 rows a day for + /// nothing), (a copy went out inside the cooldown, so this + /// one is not owed) and (reported on another delivery's + /// roster). A null report — a deliverer that does not report, or one whose outer isolation + /// caught something outside the channels — is "unreported", not "failed", and stamps: that is how + /// every fire before #3580 was treated, and a deliverer that KNOWS a send failed says so. Stating the + /// exclusion rather than an allow-list means a disposition added later defaults to the quiet side; one + /// that means "not delivered and owed" has to be added here by name. + /// + /// The stamp write is failure-isolated and NOT counted — a write, not a condition read, + /// the distinction. Memory is stamped first, so a store that will + /// not take the write still gates this process; the warning says the next restart will re-announce. + /// The instant recorded is the evaluator's , the controllable clock, so a test + /// can place the stamp and the interval compare is against the same clock it was written from. + /// + private async Task RecordDocumentDeliveredAsync( + ConcurrentDictionary lastDelivered, string memoryKey, string stampKey, + AlertDelivery? delivery, DateTime now, string documentName, CancellationToken cancellationToken) + { + if (delivery is { Channel: AlertDelivery.ChannelFailed }) + { + _logger?.LogWarning( + "{Document} delivery failed ({Error}); no delivery stamp written, so the next tick retries it", + documentName, delivery.SendError ?? "no error text"); + return; + } + + lastDelivered[memoryKey] = now; + + if (_deliveryStamps is null) + { + return; + } + + try + { + await _deliveryStamps.RecordDeliveredAtUtcAsync(stampKey, now, cancellationToken); + } + catch (OperationCanceledException) when (cancellationToken.IsCancellationRequested) + { + throw; + } + catch (Exception ex) + { + /* NOT counted by #3013's swallowed-read counter: a stamp WRITE, not a condition read. */ + _logger?.LogWarning(ex, + "{Document} was delivered but its delivery stamp could not be written; process memory gates it until the next restart, which will re-announce it", + documentName); + } } /// One band transition a covered sweep reported: the server by name (the documents carry names @@ -4524,10 +4747,14 @@ FROM ag_database_replica_states /// own subject. That condition scans the rules it already holds for one that NAMES it and passes the /// verdict in here instead (#3348). Nothing else may pass this: a caller that hands in a decision it did /// not derive from an explicit naming has re-introduced the self-suppression the seam cannot see. + /// What the deliverer reported the channels did (#3580), or null when it reported + /// nothing — read by the two daily documents' stamps and ignored by every condition-class caller, + /// whose lifecycle is edge-driven and owes nothing to a failed send. See + /// for why the report rides a second method. /* The optional context TRAILS the cancellation token so the dozens of existing positional call sites stay untouched — only the callers that have discrete facts to carry (#2109: the AG database alerts) name it. Same for muted, which defaults to asking the seam like its siblings. */ - private async Task FireAsync( + private async Task FireAsync( string serverKey, string serverName, string metricName, string currentValue, string thresholdValue, string detail, AlertSeverityLevel? severity, string shortMessage, double? numericCurrentValue, double? numericThresholdValue, CancellationToken cancellationToken, @@ -4552,7 +4779,7 @@ so the service log showed "… Recovered" with nothing before it — which reads AlertFiringLog.Fired( serverName, metricName, severity?.ToString() ?? "Warning", shortMessage, isMuted)); - await _deliverer.DeliverAsync(new AlertOutcome( + return await _deliverer.DeliverAndReportAsync(new AlertOutcome( serverKey, serverName, metricName, currentValue, thresholdValue, Context: context, DetailText: detail, NumericCurrentValue: numericCurrentValue, NumericThresholdValue: numericThresholdValue, diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs index 30be47a40..735139e00 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs @@ -1704,7 +1704,11 @@ halves of a server's alert work. */ read would claim a hot-reload the config cannot deliver. Null/blank means the evaluator fires under the shipped "Monitor Store" constant, byte-identical to every release before the field existed. */ - storeName: config.Peers?.StoreName); + storeName: config.Peers?.StoreName, + /* #3580: the two daily documents' delivered-today stamps, in the store's own key/value state + table, so a restart of this process does not re-announce a digest or rollup the previous + process delivered an hour ago — and does re-attempt one whose delivery failed. */ + deliveryStamps: new PgSelfAlertDeliveryStampStore(postgres, _logger)); /* #1706: report this start's store runtime upgrade, now that there IS an alert engine to report it through. Fired once, here, and never re-evaluated — the store is down while an upgrade runs, so diff --git a/Darling/PerformanceMonitor.Darling.Service/SelfAlertDeliveryStampStore.cs b/Darling/PerformanceMonitor.Darling.Service/SelfAlertDeliveryStampStore.cs new file mode 100644 index 000000000..99de27119 --- /dev/null +++ b/Darling/PerformanceMonitor.Darling.Service/SelfAlertDeliveryStampStore.cs @@ -0,0 +1,193 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Globalization; +using System.Threading; +using System.Threading.Tasks; +using Microsoft.Extensions.Logging; +using Npgsql; +using NpgsqlTypes; + +namespace PerformanceMonitor.Darling.Service; + +/// +/// Where the two DOCUMENT-class self-alerts — the collector-cost digest (#3443) and the fleet-sweep daily +/// rollup (#3466) — record that a copy was DELIVERED, so their once-a-day gate can outlive the process +/// that fired them (#3580). +/// +/// The question this answers is "was one delivered today", not "did I send one today". Both +/// gates lived in process memory — one ConcurrentDictionary each, one fixed key — and a fresh +/// process has empty dictionaries, so the first tick after ANY start delivered both documents again +/// whatever the previous process had delivered an hour earlier. On the v3.8.0 install night three stores +/// restarted once each and the channel carried six re-announcements among ~23 overnight posts: a quarter +/// of the channel was the same two documents, twice. The same night showed the case any fix must keep: a +/// digest/rollup pair whose delivery had FAILED (a transport fault, not a suppression) was correctly +/// re-attempted after the restart and landed. A failed delivery is not a delivery. So the evaluator writes +/// a stamp here only when the deliverer reports a disposition other than +/// , reads it back before firing, +/// and skips while the stamp is inside the document's interval — restart or not. A failed delivery writes +/// nothing, so the next tick retries, restart or not. +/// +/// The store may throw; the evaluator isolates. The FleetSweepStore shape rather than +/// PgAlertStateStore's catch-inside: the evaluator is the one place that knows what a missing +/// answer means for the gate (fall back to the process-memory gate, warn, count the read into the #3013 +/// swallowed-read census), and putting the catch there means a fake that throws exercises the real +/// fallback path rather than a mirror of it. +/// +public interface ISelfAlertDeliveryStampStore +{ + /// The UTC instant the document keyed was last DELIVERED, or + /// null when no delivery has ever been stamped (a fresh store, or a document that has only ever + /// failed to deliver). + Task GetDeliveredAtUtcAsync(string stateKey, CancellationToken cancellationToken); + + /// Records that the document keyed was delivered at + /// , replacing any earlier stamp. + Task RecordDeliveredAtUtcAsync(string stateKey, DateTime deliveredAtUtc, CancellationToken cancellationToken); +} + +/// +/// over collect.collector_state (V44) — the store's +/// general key/value state table, under a fleet-sentinel server_id and an owner name of its own. No +/// migration rung: the table already holds one short row per (server, collector, key), which is exactly +/// two rows of this shape, and #3580's fix shape names an existing key/value table as the natural home. +/// +/// server_id = 0 — the fleet sentinel the store already uses, in two places. +/// DarlingObservability.FleetServerId is 0 for the fleet-wide retention purge's +/// collection_log run-record, and FleetSweepStore.FleetScopeServerId is 0 for watch items +/// about the fleet rather than one member; both rest on the same fact, that server_ids are FNV-1a hashes +/// of a storage name and no monitored server is registered at 0. These two documents are fleet-level +/// self-alerts (they fire under the synthetic store label, not a server), so they take the same id rather +/// than mint a third convention. The table's server_id is integer NOT NULL and part of the +/// primary key, which is also why a sentinel and not a NULL. +/// +/// collector_name = 'self_alert' — an owner name that is not a collector definition's. +/// The QueryStoreBackfill precedent: a worker that keeps state in this table under its own name, +/// deliberately outside the definitions' StateKeys machinery, so no collector's declared-key read or +/// per-database prune can reach it. Every prune in DarlingCollectorRunner is scoped to a real +/// server_id AND a collector's own name, and the two migration-time deletes (V77, V114) name their +/// collector; nothing retires rows under this name, which is the point — this is state, not facts, and +/// the table carries no retention by design. +/// +/// The stamp is the value, as ISO 8601 round-trip text, and updated_at is the write +/// time. Two different instants: the stamp is the evaluator's clock at the fire (the controllable +/// utcNow seam, so a test can place it), and updated_at is the wall clock the row was written +/// at, the column's meaning on every other row of the table. The stamp travels as text because the column +/// is text; the "O" format round-trips to the tick and carries its Z, so the read side gets +/// back without a SpecifyKind that could lie. A value that does not +/// parse — hand-edited, or written by nothing this build knows — reads as no stamp, with a warning, rather +/// than as a throw: the gate then falls back to memory for that tick, and the next successful delivery +/// overwrites the row. +/// +/// Schema-qualified like the V44 DDL and FleetSweepStore, not bare like the runner's own +/// reads: this store is also handed a scratch database in the live test, where the connection string +/// carries no search path and only the migrator's best-effort ALTER DATABASE would resolve a bare +/// name. Naming the schema costs nothing and removes the dependency. +/// +public sealed class PgSelfAlertDeliveryStampStore : ISelfAlertDeliveryStampStore +{ + /// See the class remarks: the fleet-wide sentinel both existing fleet-scope writers use. + public const int FleetServerId = DarlingObservability.FleetServerId; + + /// See the class remarks: the owner name, deliberately not a collector definition's. + public const string StateCollectorName = "self_alert"; + + /// The digest's stamp key — one fixed row, the document's one fixed in-memory key made + /// durable. + public const string CostDigestStateKey = "digest_delivered_at"; + + /// The fleet-sweep rollup's stamp key. + public const string FleetSweepRollupStateKey = "sweep_rollup_delivered_at"; + + /// The alert pass's own deadline (DarlingAlertReadAdapter.AlertPassCommandTimeoutSeconds), + /// because this read runs inside it: a stamp read that outlives the pass's budget is a stamp read that + /// should have failed toward the memory gate. + internal const int CommandTimeoutSeconds = DarlingAlertReadAdapter.AlertPassCommandTimeoutSeconds; + + internal const string GetSql = @" +SELECT state_value +FROM collect.collector_state +WHERE server_id = $1 +AND collector_name = $2 +AND state_key = $3"; + + internal const string UpsertSql = @" +INSERT INTO collect.collector_state (server_id, collector_name, state_key, state_value, updated_at) +VALUES ($1, $2, $3, $4, $5) +ON CONFLICT (server_id, collector_name, state_key) +DO UPDATE SET state_value = EXCLUDED.state_value, updated_at = EXCLUDED.updated_at"; + + private readonly NpgsqlDataSource _postgres; + private readonly ILogger? _logger; + + public PgSelfAlertDeliveryStampStore(NpgsqlDataSource postgres, ILogger? logger = null) + { + _postgres = postgres ?? throw new ArgumentNullException(nameof(postgres)); + _logger = logger; + } + + public async Task GetDeliveredAtUtcAsync(string stateKey, CancellationToken cancellationToken) + { + ArgumentException.ThrowIfNullOrEmpty(stateKey); + + await using var connection = await _postgres.OpenConnectionAsync(cancellationToken); + using var command = new NpgsqlCommand(GetSql, connection) { CommandTimeout = CommandTimeoutSeconds }; + command.Parameters.Add(new NpgsqlParameter { NpgsqlDbType = NpgsqlDbType.Integer, Value = FleetServerId }); + command.Parameters.Add(new NpgsqlParameter { NpgsqlDbType = NpgsqlDbType.Text, Value = StateCollectorName }); + command.Parameters.Add(new NpgsqlParameter { NpgsqlDbType = NpgsqlDbType.Text, Value = stateKey }); + + var result = await command.ExecuteScalarAsync(cancellationToken); + if (result is not string text) + { + return null; + } + + if (!DateTime.TryParseExact(text, "O", CultureInfo.InvariantCulture, DateTimeStyles.RoundtripKind, out var stamp)) + { + _logger?.LogWarning( + "Self-alert delivery stamp {Key} holds '{Value}', which is not a round-trip UTC instant; treating it as no stamp", + stateKey, text); + return null; + } + + /* "O" with its Z parses straight to Kind=Utc; a hand-written value carrying an offset lands as Local + and is converted rather than trusted, so the caller's `now - stamp` is a UTC-to-UTC subtraction + either way. */ + return stamp.Kind == DateTimeKind.Utc ? stamp : stamp.ToUniversalTime(); + } + + public async Task RecordDeliveredAtUtcAsync(string stateKey, DateTime deliveredAtUtc, CancellationToken cancellationToken) + { + ArgumentException.ThrowIfNullOrEmpty(stateKey); + + await using var connection = await _postgres.OpenConnectionAsync(cancellationToken); + using var command = new NpgsqlCommand(UpsertSql, connection) { CommandTimeout = CommandTimeoutSeconds }; + command.Parameters.Add(new NpgsqlParameter { NpgsqlDbType = NpgsqlDbType.Integer, Value = FleetServerId }); + command.Parameters.Add(new NpgsqlParameter { NpgsqlDbType = NpgsqlDbType.Text, Value = StateCollectorName }); + command.Parameters.Add(new NpgsqlParameter { NpgsqlDbType = NpgsqlDbType.Text, Value = stateKey }); + command.Parameters.Add(new NpgsqlParameter + { + NpgsqlDbType = NpgsqlDbType.Text, + /* Stamped Utc before formatting so the text always carries Z — a Kind-Unspecified caller value + would otherwise format without a zone and read back as Unspecified. */ + Value = DateTime.SpecifyKind(deliveredAtUtc, DateTimeKind.Utc).ToString("O", CultureInfo.InvariantCulture), + }); + command.Parameters.Add(new NpgsqlParameter + { + NpgsqlDbType = NpgsqlDbType.Timestamp, + /* Naive UTC, Kind-Unspecified — the product-wide PG `timestamp` discipline (PgAlertStateStore. + NaiveUtcNow, the runner's own SaveCollectorStateAsync): binding Kind=Utc against a `timestamp` + column does not fail, Npgsql infers timestamptz and PostgreSQL casts it into the SERVER's zone, + so the row lands silently offset while every other timestamp in the store is UTC. */ + Value = DateTime.SpecifyKind(DateTime.UtcNow, DateTimeKind.Unspecified), + }); + + await command.ExecuteNonQueryAsync(cancellationToken); + } +} diff --git a/Lite.Tests/AlertReadFailureSurfaceTests.cs b/Lite.Tests/AlertReadFailureSurfaceTests.cs index 681968b63..b95f97c63 100644 --- a/Lite.Tests/AlertReadFailureSurfaceTests.cs +++ b/Lite.Tests/AlertReadFailureSurfaceTests.cs @@ -244,8 +244,10 @@ one of the two and would have "proved" a single fleet-scoped site. */ } } - /* Seventh since #3466: the fleet-sweep rollup read, Darling-only like the store self-alerts. */ - Assert.Equal(7, nullKeyReads.Count); + /* Seventh since #3466: the fleet-sweep rollup read, Darling-only like the store self-alerts. Eighth + since #3580: the daily documents' delivery-stamp read — ONE site gating both the digest and the + rollup on delivered-today, so one name — Darling-only for the same reason. */ + Assert.Equal(8, nullKeyReads.Count); var inventory = AlertReadFailureCounter.FleetScopedReads; @@ -261,6 +263,7 @@ exactly when one of the three is the one that went quiet. */ Assert.Contains(nullKeyReads, r => r.Contains("collector-cost digest", StringComparison.Ordinal)); Assert.Contains(nullKeyReads, r => r.Contains("background-job health", StringComparison.Ordinal)); Assert.Contains(nullKeyReads, r => r.Contains("fleet-sweep rollup", StringComparison.Ordinal)); + Assert.Contains(nullKeyReads, r => r.Contains("delivery-stamp", StringComparison.Ordinal)); /* #3354: config_mute_rules belongs to the store, not to any monitored server, so its failed read lands in the instance total and in no server's count — exactly the case a per-server-only surface would have given no home. Recorded TWICE across the tree, once per SKU, and that is the @@ -276,6 +279,7 @@ entries are Darling store self-alerts with no Lite equivalent. */ Assert.Contains("background-job health", inventory, StringComparison.Ordinal); Assert.Contains("mute-rule reload", inventory, StringComparison.Ordinal); Assert.Contains("fleet-sweep rollup", inventory, StringComparison.Ordinal); + Assert.Contains("delivery-stamp", inventory, StringComparison.Ordinal); /* And the phantom stays gone. Disk pressure's feed reads are exempt — a local filesystem read and a recorded-store-size lookup that is context for the alert text — so naming it here would send an diff --git a/PerformanceMonitor.Alerting/AlertReadFailureCounter.cs b/PerformanceMonitor.Alerting/AlertReadFailureCounter.cs index b46f9d2cf..c88e01bcf 100644 --- a/PerformanceMonitor.Alerting/AlertReadFailureCounter.cs +++ b/PerformanceMonitor.Alerting/AlertReadFailureCounter.cs @@ -704,12 +704,20 @@ private static string FleetSentence(Reading reading) /// performs that read, Lite swallows its failure, and Lite's call site therefore records it here too. /// A read this constant names on a SKU that cannot increment it would be this class's own defect — a /// confident zero — reproduced in its documentation. + /// + /// #3580 added one more Darling-only member: the daily documents' DELIVERY-STAMP read, the gate + /// that decides whether the digest or the rollup was already delivered today. It is one read site + /// serving both documents (so one name here), and it is counted rather than exempt because its + /// swallowed failure is the gate falling back to process memory — which re-announces the document once + /// per restart until the store answers, the very behaviour #3580 retired. A nonzero count naming it + /// says the store could not be asked "was one delivered today", not that a document was lost. /// public const string FleetScopedReads = "the collector-cost regression self-alert and its two #3443 companions (the collector-cost census " + "read that decides paging-versus-digest routing, and the collector-cost digest read behind the " + "daily report), the mute-rule reload, the fleet-sweep rollup read behind the daily sweep report " - + "(#3466), and the store background-job " + + "(#3466), the daily documents' delivery-stamp read that gates the digest and the rollup on " + + "delivered-today (#3580), and the store background-job " + "health reads behind compression-job health, store-job cadence and retention holds"; /// diff --git a/PerformanceMonitor.Alerting/IAlertDeliverer.cs b/PerformanceMonitor.Alerting/IAlertDeliverer.cs index 7539bd5ad..32efe7055 100644 --- a/PerformanceMonitor.Alerting/IAlertDeliverer.cs +++ b/PerformanceMonitor.Alerting/IAlertDeliverer.cs @@ -91,4 +91,33 @@ public interface IAlertDeliverer /// must not throw for channel failures — a dead SMTP server must not abort the engine's sweep. /// Task DeliverAsync(AlertOutcome outcome, CancellationToken cancellationToken = default); + + /// + /// , and then SAYS what the channels did (#3580): the + /// the deliverer recorded on the alert's history row, or null when it + /// has no single answer to give. + /// + /// Why a second method rather than a return value on the first. The never-throws contract + /// above is deliberate and stays: a channel fault is the deliverer's to record, not the engine's to + /// handle, and every condition-class alert wants exactly that. The two DOCUMENT-class self-alerts (the + /// collector-cost digest and the fleet-sweep rollup) are the exception, because their once-a-day gate + /// has to answer "was one delivered today" and not "did this process fire one today" — on the v3.8.0 + /// install night three service restarts re-announced both documents on every store, six re-posts among + /// ~23 channel posts, while the one pair whose delivery had genuinely FAILED was correctly re-attempted + /// after the restart. Distinguishing those two cases needs the disposition at the fire site, and + /// nothing else on the engine's side does; so the report rides a separate method that the two askers + /// call and every other caller ignores. + /// + /// The default reports nothing. A deliverer that has not been taught to report delivers + /// exactly as before and returns null, which the askers treat the way every fire before #3580 + /// was treated: as delivered. That keeps Lite's deliverer and every test fake compiling and behaving + /// unchanged, and it means null is "unreported", never "failed" — a deliverer that KNOWS a send + /// failed reports , which is the one disposition the askers + /// withhold their delivered-today stamp on. + /// + async Task DeliverAndReportAsync(AlertOutcome outcome, CancellationToken cancellationToken = default) + { + await DeliverAsync(outcome, cancellationToken); + return null; + } } From 88be048531701ac0b91002675d129ceb0b7ee802 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 15:14:26 -0400 Subject: [PATCH 55/69] A server nothing has banded yet is Unknown, not Healthy, and the alert-history grids show the severity the alert actually fired at instead of the colour its name implies (#3539 A6/A8e) (#3635) * A server nothing has banded yet is Unknown, not Healthy, and the alert-history grids show the severity the alert fired at instead of the colour its name implies (#3539 A6/A8e) * Keep dismissed the last selected column (the pin anchors on it), and paint the nothing-measured border with the cached Warning brush --- .../AlertHistoryRowSeverityTests.cs | 291 ++++++++++++++++++ .../AlertMetricClassifierTests.cs | 6 + .../DarlingMcpAlertToolsTests.cs | 34 +- ...dCollectionStaleNamesItsPopulationTests.cs | 17 +- .../ServerHealthClassifierTests.cs | 76 ++++- .../UnmeasuredMetricsAreNotHealthyTests.cs | 139 ++++++++- .../ViewerOverviewExplainsItselfTests.cs | 6 +- Darling/Darling.Tests/ViewerW2aTests.cs | 17 +- .../DarlingAlertDeliverer.cs | 6 +- .../Mcp/DarlingAlertReader.cs | 13 +- .../Mcp/DarlingFleetReader.cs | 23 +- .../Mcp/DarlingMcpAlertTools.cs | 43 ++- .../wwwroot/js/pages/alerts.js | 21 ++ .../wwwroot/js/pages/fleet.js | 15 +- .../ViewerDataService.AlertHistory.cs | 11 +- .../ViewerDataService.Fleet.cs | 20 +- .../ViewerDataService.Overview.cs | 29 +- .../AlertHistoryRowSeverityLiteTests.cs | 87 ++++++ Lite/Mcp/McpAlertTools.cs | 42 ++- .../Services/LocalDataService.AlertHistory.cs | 10 +- .../AlertHistoryRowSeverity.cs | 128 ++++++++ .../AlertMetricClassifier.cs | 14 + .../ServerHealthBands.cs | 81 ++++- .../AlertContext.cs | 75 ++++- 24 files changed, 1105 insertions(+), 99 deletions(-) create mode 100644 Darling/Darling.Tests/AlertHistoryRowSeverityTests.cs create mode 100644 Lite.Tests/AlertHistoryRowSeverityLiteTests.cs create mode 100644 PerformanceMonitor.Alerting/AlertHistoryRowSeverity.cs diff --git a/Darling/Darling.Tests/AlertHistoryRowSeverityTests.cs b/Darling/Darling.Tests/AlertHistoryRowSeverityTests.cs new file mode 100644 index 000000000..6ef3e1cb6 --- /dev/null +++ b/Darling/Darling.Tests/AlertHistoryRowSeverityTests.cs @@ -0,0 +1,291 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.IO; +using System.Linq; +using PerformanceMonitor.Alerting; +using PerformanceMonitor.Common; +using PerformanceMonitor.Darling.Viewer; +using PerformanceMonitor.Notifications; +using Xunit; + +namespace Darling.Tests; + +/// +/// #3539 A8e: an alert-history row is styled and reported at the severity the alert actually FIRED at, +/// not the colour its name implies. +/// +/// The defect. is by NAME — "Poison Wait" is +/// red — and it was the only severity the grids had. Poison Wait has been graded Warning/Critical at its +/// fire site since #2711 (PostgreSQL) and #3539 A4 (SQL Server), so a Warning-graded fire rendered red in +/// Lite's grid and the Viewer's alike; the same gap ran the other way for every metric the engine grades +/// ABOVE its name (a CRITICAL low-disk fire, a SUSPECT database) — amber rows for critical pages. The tier +/// was on at fire time and the JSON projection dropped it, so +/// the row had nothing else to offer. +/// +/// The fix, and what these pins hold. carries the tier +/// as a trailing nullable member (both SKUs write through it, so no store changed on either); +/// reads it and falls back to the by-name classifier ONLY for rows +/// that carry none. The pins: the member round-trips by name and reads back; a row written before it +/// existed deserializes exactly as it did (the additive-only contract); the fallback fires only where no +/// tier exists, and says the right thing there; both grids' row classes reach the shared decision and no +/// grid still calls the by-name predicates directly. +/// +public sealed class AlertHistoryRowSeverityTests +{ + /* ─────────────────────────── the member on the wire ─────────────────────────── */ + + private static string WithSeverity(AlertSeverityLevel? level) + { + var context = new AlertContext { SeverityOverride = level }; + context.Details.Add(new AlertDetailItem { Heading = "THREADPOOL", Fields = { ("Accumulated wait", "61 s") } }); + return AlertContextSerializer.Serialize(context); + } + + [Theory] + [InlineData(AlertSeverityLevel.Warning, "\"Severity\":\"Warning\"")] + [InlineData(AlertSeverityLevel.Critical, "\"Severity\":\"Critical\"")] + public void TheTierIsPersistedByName_AndReadsBack(AlertSeverityLevel level, string fragment) + { + var json = WithSeverity(level); + + /* By NAME, not ordinal: the column outlives any build, and a stored "1" would change meaning the + day a member is inserted ahead of Critical. */ + Assert.Contains(fragment, json, StringComparison.Ordinal); + Assert.DoesNotContain("\"Severity\":" + (int)level, json, StringComparison.Ordinal); + + Assert.Equal(level, AlertContextSerializer.TryReadSeverity(json)); + + /* The full rehydration agrees with the cheap read, and the Details it always carried are intact. */ + Assert.True(AlertContextSerializer.TryDeserialize(json, out var restored)); + Assert.Equal(level, restored.SeverityOverride); + Assert.Single(restored.Details); + } + + /// + /// Additive only: a row written BEFORE the member existed — the exact JSON the pre-#3539 serializer + /// emitted, hand-written here so it cannot drift with the serializer — rehydrates its Details and + /// Incidents as it always did and reads back NO tier. Null is the honest answer for that row, and it is + /// what sends the grids to the by-name fallback. + /// + [Fact] + public void ARowWrittenBeforeTheMemberExisted_DeserializesUnchanged_AndCarriesNoTier() + { + const string legacy = + "{\"Details\":[{\"Heading\":\"THREADPOOL\",\"Fields\":[{\"Label\":\"Avg wait\",\"Value\":\"600 ms\"}],\"Body\":null,\"IsCodeBlock\":false,\"Remediation\":null}]," + + "\"Incidents\":[{\"DedupKey\":\"abc\",\"InvolvedObjects\":[\"x\"],\"OccurrenceCount\":2,\"WaitRange\":null,\"TotalOccurrences\":null,\"IncidentStartedUtc\":null,\"Database\":null,\"LastEventUtc\":null}]}"; + + Assert.Null(AlertContextSerializer.TryReadSeverity(legacy)); + + Assert.True(AlertContextSerializer.TryDeserialize(legacy, out var context)); + Assert.Null(context.SeverityOverride); + var detail = Assert.Single(context.Details); + Assert.Equal("THREADPOOL", detail.Heading); + Assert.Equal(("Avg wait", "600 ms"), Assert.Single(detail.Fields)); + var incident = Assert.Single(context.Incidents!); + Assert.Equal("abc", incident.DedupKey); + Assert.Equal(2, incident.OccurrenceCount); + } + + /// A fire with no override persists a null member and reads back none — the same state a + /// legacy row is in, which is why one fallback serves both. + [Fact] + public void AFireWithNoOverride_CarriesNoTier() + { + var json = WithSeverity(null); + Assert.Null(AlertContextSerializer.TryReadSeverity(json)); + Assert.True(AlertContextSerializer.TryDeserialize(json, out var restored)); + Assert.Null(restored.SeverityOverride); + } + + /// The cheap read refuses everything that is not the writer's exact spelling — garbage, a bare + /// ordinal (the coupling the string form exists to avoid), a case variant, a non-object root — and a + /// resolution row's null context. It must never throw: the grids call it once per row. + [Theory] + [InlineData(null)] + [InlineData("")] + [InlineData(" ")] + [InlineData("not json")] + [InlineData("[]")] + [InlineData("{\"Details\":[]}")] + [InlineData("{\"Details\":[],\"Severity\":null}")] + [InlineData("{\"Details\":[],\"Severity\":1}")] + [InlineData("{\"Details\":[],\"Severity\":\"1\"}")] + [InlineData("{\"Details\":[],\"Severity\":\"critical\"}")] + [InlineData("{\"Details\":[],\"Severity\":\"Fatal\"}")] + public void TheCheapRead_AnswersNull_ForAnythingThatIsNotAPersistedTier(string? json) + { + Assert.Null(AlertContextSerializer.TryReadSeverity(json)); + Assert.Null(AlertHistoryRowSeverity.FiredAt(json)); + } + + /* ─────────────────────────── the row's severity ─────────────────────────── */ + + /// + /// The headline case: a Poison Wait row that fired WARNING is a warning row, whatever the name says; + /// one that fired CRITICAL is critical; one that carries no tier is critical BY NAME — every SQL Server + /// Poison Wait row written before #3539 A4 was a presence-flat critical fire, so the fallback is the + /// faithful replay for exactly the rows that reach it. + /// + [Fact] + public void APoisonWaitRow_TakesTheTierItFiredAt_AndTheNameOnlyWhenItHasNone() + { + var warning = WithSeverity(AlertSeverityLevel.Warning); + Assert.True(AlertHistoryRowSeverity.IsWarning("Poison Wait", warning)); + Assert.False(AlertHistoryRowSeverity.IsCritical("Poison Wait", warning)); + Assert.Equal(("warning", AlertHistoryRowSeverity.SourceFired), AlertHistoryRowSeverity.Describe("Poison Wait", warning)); + + var critical = WithSeverity(AlertSeverityLevel.Critical); + Assert.True(AlertHistoryRowSeverity.IsCritical("Poison Wait", critical)); + Assert.False(AlertHistoryRowSeverity.IsWarning("Poison Wait", critical)); + Assert.Equal(("critical", AlertHistoryRowSeverity.SourceFired), AlertHistoryRowSeverity.Describe("Poison Wait", critical)); + + /* No tier on the row: the by-name arm, and it says so. */ + Assert.True(AlertHistoryRowSeverity.IsCritical("Poison Wait", null)); + Assert.True(AlertMetricClassifier.IsCritical("Poison Wait")); // the premise the fallback rests on + Assert.Equal(("critical", AlertHistoryRowSeverity.SourceMetricName), AlertHistoryRowSeverity.Describe("Poison Wait", null)); + } + + /// The other direction of the same gap: metrics the engine grades ABOVE their name's colour. + /// A CRITICAL-graded low-disk fire (#1136) and a SUSPECT database (Database State's critical arm) were + /// amber rows; with the tier on the row they are red. Their override-less rows keep the name's amber, + /// which is what the channels rendered for them too. + [Theory] + [InlineData("Volume Free Space")] + [InlineData("Database State")] + public void AMetricGradedAboveItsName_RendersTheGradeItFiredAt(string metric) + { + Assert.True(AlertMetricClassifier.IsWarning(metric)); // the name alone says amber + + var critical = WithSeverity(AlertSeverityLevel.Critical); + Assert.True(AlertHistoryRowSeverity.IsCritical(metric, critical)); + Assert.False(AlertHistoryRowSeverity.IsWarning(metric, critical)); + + Assert.False(AlertHistoryRowSeverity.IsCritical(metric, null)); + Assert.True(AlertHistoryRowSeverity.IsWarning(metric, null)); + } + + /// The presence-flat metrics fire with no override, so their rows carry no tier and the name + /// decides exactly as before — this change moves nothing for them. (Grading them is the engine's half of + /// A8e, not the grid's.) + [Theory] + [InlineData("Deadlocks Detected", true)] + [InlineData("High CPU", false)] + [InlineData("tempdb Space", false)] + [InlineData("Blocking Detected", false)] + public void APresenceFlatMetric_KeepsItsByNameColour(string metric, bool criticalByName) + { + Assert.Equal(criticalByName, AlertHistoryRowSeverity.IsCritical(metric, null)); + Assert.Equal(!criticalByName, AlertHistoryRowSeverity.IsWarning(metric, null)); + Assert.Equal(criticalByName, AlertMetricClassifier.IsCritical(metric)); + Assert.Equal( + (criticalByName ? "critical" : "warning", AlertHistoryRowSeverity.SourceMetricName), + AlertHistoryRowSeverity.Describe(metric, null)); + } + + /// Resolution rows are the name's business and never consult a tier — they are persisted with + /// a null context, and even a stray tier on one must not turn "Poison Waits Cleared" red. The two + /// deliberate INFO reports fire with no override and stay unhighlighted, as + /// intends. + [Fact] + public void ResolutionAndInformationalRows_AreTheNamesBusiness() + { + foreach (var json in new[] { null, WithSeverity(AlertSeverityLevel.Critical) }) + { + Assert.False(AlertHistoryRowSeverity.IsCritical("Poison Waits Cleared", json)); + Assert.False(AlertHistoryRowSeverity.IsWarning("Poison Waits Cleared", json)); + Assert.Equal(("resolution", AlertHistoryRowSeverity.SourceMetricName), AlertHistoryRowSeverity.Describe("Poison Waits Cleared", json)); + } + + Assert.False(AlertHistoryRowSeverity.IsCritical("Collector Cost Digest", null)); + Assert.False(AlertHistoryRowSeverity.IsWarning("Collector Cost Digest", null)); + Assert.Equal(("info", AlertHistoryRowSeverity.SourceMetricName), AlertHistoryRowSeverity.Describe("Collector Cost Digest", null)); + Assert.Equal(("info", AlertHistoryRowSeverity.SourceMetricName), AlertHistoryRowSeverity.Describe("Fleet Sweep Rollup", null)); + } + + /* ─────────────────────────── the Viewer's row ─────────────────────────── */ + + private static ViewerAlertRow ViewerRow(string metric, string? contextJson) => new() + { + AlertTime = new DateTime(2026, 9, 18, 12, 0, 0), + MetricName = metric, + CurrentValue = 61_000, + ThresholdValue = 60_000, + AlertSent = true, + NotificationType = "webhook", + Muted = false, + ContextJson = contextJson, + }; + + /// The Viewer's grid row reaches the shared decision: a Warning-graded Poison Wait row is + /// amber, not red; a row with no tier keeps the name's red. Lite's AlertHistoryRow is pinned the + /// same way in Lite.Tests (AlertHistoryRowSeverityLiteTests). + [Fact] + public void TheViewerGridRow_RendersTheTierTheAlertFiredAt() + { + var graded = ViewerRow("Poison Wait", WithSeverity(AlertSeverityLevel.Warning)); + Assert.True(graded.IsWarning); + Assert.False(graded.IsCritical); + Assert.False(graded.IsResolved); + + var legacy = ViewerRow("Poison Wait", null); + Assert.True(legacy.IsCritical); + Assert.False(legacy.IsWarning); + + var cleared = ViewerRow("Poison Waits Cleared", null); + Assert.True(cleared.IsResolved); + Assert.False(cleared.IsCritical); + Assert.False(cleared.IsWarning); + } + + /* ─────────────────────────── one decision, every grid ─────────────────────────── */ + + /// + /// Both SKUs' grid rows reach , and NO shipping file under Lite/ or + /// Darling/ calls the by-name IsCritical / IsWarning predicates directly any more — a + /// grid that did would be back to colouring a Warning-graded row red. The deprecated Dashboard is + /// outside the census on purpose: its own engine still fires Poison Wait presence-flat CRITICAL, so its + /// by-name red is faithful to its own rows, and it is a frozen twin. + /// + [Fact] + public void BothGridRows_ReachTheSharedDecision_AndNoGridStillClassifiesByNameAlone() + { + var rows = new[] + { + Path.Combine("Lite", "Services", "LocalDataService.AlertHistory.cs"), + Path.Combine("Darling", "PerformanceMonitor.Darling.Viewer", "ViewerDataService.AlertHistory.cs"), + }; + foreach (var relative in rows) + { + var code = CSharpSourceWalker.StripCommentsAndStrings(RepoFile.ReadRepoFile(relative.Split(Path.DirectorySeparatorChar))); + Assert.Contains("AlertHistoryRowSeverity.IsCritical(MetricName, ContextJson)", code, StringComparison.Ordinal); + Assert.Contains("AlertHistoryRowSeverity.IsWarning(MetricName, ContextJson)", code, StringComparison.Ordinal); + } + + var byNameCallers = new[] { "Lite", "Darling" } + .SelectMany(root => Directory.EnumerateFiles(Path.Combine(RepoFile.Root, root), "*.cs", SearchOption.AllDirectories)) + .Where(f => + { + var segments = f.Split(Path.DirectorySeparatorChar, Path.AltDirectorySeparatorChar); + return !segments.Contains("bin") && !segments.Contains("obj") + && !segments.Contains("Darling.Tests") && !segments.Contains("Lite.Tests"); + }) + .Where(f => + { + var code = CSharpSourceWalker.StripCommentsAndStrings(File.ReadAllText(f)); + return code.Contains("AlertMetricClassifier.IsCritical(", StringComparison.Ordinal) + || code.Contains("AlertMetricClassifier.IsWarning(", StringComparison.Ordinal); + }) + .Select(f => Path.GetRelativePath(RepoFile.Root, f)) + .OrderBy(f => f, StringComparer.Ordinal) + .ToArray(); + + Assert.Empty(byNameCallers); + } +} diff --git a/Darling/Darling.Tests/AlertMetricClassifierTests.cs b/Darling/Darling.Tests/AlertMetricClassifierTests.cs index 71c562296..c9386d4e4 100644 --- a/Darling/Darling.Tests/AlertMetricClassifierTests.cs +++ b/Darling/Darling.Tests/AlertMetricClassifierTests.cs @@ -18,6 +18,12 @@ namespace Darling.Tests; /// the old duplicated inline copies missed, and the last four are the same drift caught one layer down in /// Darling's self-alert recoveries (#991) — plus the critical (Deadlock/Poison) and warning buckets, over /// the metric names the alert engines actually emit. +/// +/// #3539 A8e: the critical/warning buckets are the FALLBACK the two SKU grids use for a row that +/// carries no persisted tier; a row that does is styled by the tier it fired at, through +/// AlertHistoryRowSeverity (pinned in AlertHistoryRowSeverityTests). The by-name pins here +/// therefore hold for the rows that predate the member, which is what "Poison Wait" being critical by name +/// means now — every such SQL Server row was a presence-flat critical fire. /// public class AlertMetricClassifierTests { diff --git a/Darling/Darling.Tests/DarlingMcpAlertToolsTests.cs b/Darling/Darling.Tests/DarlingMcpAlertToolsTests.cs index a597533f6..51e0fbeff 100644 --- a/Darling/Darling.Tests/DarlingMcpAlertToolsTests.cs +++ b/Darling/Darling.Tests/DarlingMcpAlertToolsTests.cs @@ -1955,6 +1955,20 @@ await DarlingMcpTestData.ExecAsync(connection, ct, VALUES ($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,$11)", when, ServerId, ServerName, "High CPU", 92.5, 80.0, true, "email", null, false, "CPU sustained above threshold"); + /* #3539 A8e: a Poison Wait row that FIRED Warning, with the tier persisted the way both SKUs' + deliverers persist it (the serializer's Severity member), and a legacy Deadlocks row carrying + no context at all. */ + var gradedContext = new AlertContext { SeverityOverride = AlertSeverityLevel.Warning }; + gradedContext.Details.Add(new AlertDetailItem { Heading = "THREADPOOL" }); + await DarlingMcpTestData.ExecAsync(connection, ct, + @"INSERT INTO config_alert_log (alert_time, server_id, server_name, metric_name, current_value, threshold_value, alert_sent, notification_type, send_error, muted, detail_text, context_json) +VALUES ($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,$11,$12)", + when.AddMinutes(-1), ServerId, ServerName, "Poison Wait", 61000.0, 60000.0, true, "webhook", null, false, "THREADPOOL", AlertContextSerializer.Serialize(gradedContext)); + await DarlingMcpTestData.ExecAsync(connection, ct, + @"INSERT INTO config_alert_log (alert_time, server_id, server_name, metric_name, current_value, threshold_value, alert_sent, notification_type, send_error, muted, detail_text) +VALUES ($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,$11)", + when.AddMinutes(-2), ServerId, ServerName, "Deadlocks Detected", 1.0, 1.0, true, "email", null, false, null); + /* Seed the single global settings row — every column has a default, so id alone suffices. BOTH singletons, because #3314 made get_alert_settings read the delivery cooldown off config_notification: the service seeds the two in one pass, and the tool reports `unavailable` @@ -1973,14 +1987,30 @@ await DarlingMcpTestData.ExecAsync(connection, ct, var scoped = await DarlingMcpAlertTools.GetAlertHistory(postgres, ServerName); DarlingMcpTestData.AssertEnvelope(scoped, ServerName, "alerts"); Assert.Contains("High CPU", scoped, StringComparison.Ordinal); - /* #3541 A3: one planted, undismissed row — the page says so, says the filter applied and hid + /* #3541 A3: three planted, undismissed rows — the page says so, says the filter applied and hid nothing, and carries no `total_` key. */ - JsonAssert.Contains("\"alerts_returned\": 1", scoped); + JsonAssert.Contains("\"alerts_returned\": 3", scoped); JsonAssert.Contains("\"truncated\": false", scoped); JsonAssert.Contains("\"dismissed_excluded\": true", scoped); JsonAssert.Contains("\"dismissed_excluded_count\": 0", scoped); Assert.DoesNotContain("total_alerts", scoped, StringComparison.Ordinal); + /* #3539 A8e: the tier the alert FIRED at, per row, and where it came from. The Poison Wait row + reads the Warning it fired at off its context ("fired") — not the red its name implies — while + the two rows with no context are classified by name and say so. Asserted on the row objects + rather than by substring, so a "warning" from one row cannot satisfy a pin about another. */ + using (var page = JsonDocument.Parse(scoped)) + { + var byMetric = page.RootElement.GetProperty("alerts").EnumerateArray() + .ToDictionary(a => a.GetProperty("metric_name").GetString()!, a => a); + Assert.Equal("warning", byMetric["Poison Wait"].GetProperty("severity").GetString()); + Assert.Equal(AlertHistoryRowSeverity.SourceFired, byMetric["Poison Wait"].GetProperty("severity_source").GetString()); + Assert.Equal("critical", byMetric["Deadlocks Detected"].GetProperty("severity").GetString()); + Assert.Equal(AlertHistoryRowSeverity.SourceMetricName, byMetric["Deadlocks Detected"].GetProperty("severity_source").GetString()); + Assert.Equal("warning", byMetric["High CPU"].GetProperty("severity").GetString()); + Assert.Equal(AlertHistoryRowSeverity.SourceMetricName, byMetric["High CPU"].GetProperty("severity_source").GetString()); + } + var fleet = await DarlingMcpAlertTools.GetAlertHistory(postgres); Assert.False(fleet.StartsWith("Error during", StringComparison.Ordinal), fleet); Assert.Contains("(all servers)", fleet, StringComparison.Ordinal); diff --git a/Darling/Darling.Tests/FleetCardCollectionStaleNamesItsPopulationTests.cs b/Darling/Darling.Tests/FleetCardCollectionStaleNamesItsPopulationTests.cs index 8ddff4138..bc35e729e 100644 --- a/Darling/Darling.Tests/FleetCardCollectionStaleNamesItsPopulationTests.cs +++ b/Darling/Darling.Tests/FleetCardCollectionStaleNamesItsPopulationTests.cs @@ -157,13 +157,14 @@ public void TheReasonString_AgreesWithTheFlagsName() var failingButCurrent = Card(TimeSpan.FromSeconds(5), failing: 1); var reason = DarlingFleetReader.BuildReason(failingButCurrent); - Assert.Equal("1 collector failing", reason); + /* "of 40": the helper declares the denominator (#3539 A8d names it when there is one). */ + Assert.Equal("1 of 40 collectors failing", reason); Assert.DoesNotContain("stale", reason, StringComparison.Ordinal); /* A card carrying both reports both, in their own clauses — the axes are additive in the prose exactly as they are in the payload. */ var both = Card(ServerHealthThresholds.StaleThreshold + TimeSpan.FromMinutes(1), failing: 2); - Assert.Equal("2 collectors failing, collection stale", DarlingFleetReader.BuildReason(both)); + Assert.Equal("2 of 40 collectors failing, collection stale", DarlingFleetReader.BuildReason(both)); } /// @@ -310,7 +311,10 @@ private static FleetServerCard Card(TimeSpan sinceLastCollection, int failing) var flags = ServerCollectionStatusRules.FlagsFor( ServerHealthClassifier.ClassifyFreshness(lastCollection, Now)); - var metrics = new ServerHealthMetrics { CpuPercentForAlert = 4, FailedCollectorCount = failing }; + /* Forty collectors banded (#3539 A6): the collectors row is a MEASURED reading, so "nothing failing" + is Healthy here and the two collection axes stay the only variables. */ + const int bandedCollectors = 40; + var metrics = new ServerHealthMetrics { CpuPercentForAlert = 4, FailedCollectorCount = failing, CollectorCount = bandedCollectors }; var overall = ServerHealthClassifier.OverallMetricSeverity(metrics); return new FleetServerCard @@ -325,9 +329,10 @@ private static FleetServerCard Card(TimeSpan sinceLastCollection, int failing) Status = ServerCollectionStatusRules .Classify(flags.IsOnline, flags.CollectionStale, flags.AwaitingFirstCollection).Word(), FailedCollectorCount = failing, - /* No denominator declared: the share cannot form, so a failing count is Warning and never - Critical (#3539 A8d) — the two collection axes stay the only variables. */ - CollectorSeverity = ServerHealthClassifier.CollectorSeverity(failing, collectorCount: 0), + CollectorCount = bandedCollectors, + /* Every failing count this file uses is under the 20% bar of forty, so a failing count is + Warning and never Critical (#3539 A8d) — the two collection axes stay the only variables. */ + CollectorSeverity = ServerHealthClassifier.CollectorSeverity(failing, bandedCollectors), OverallMetricSeverity = overall, Band = ServerHealthClassifier.ClassifyBand( flags.IsOnline, flags.AwaitingFirstCollection, flags.CollectionStale, overall), diff --git a/Darling/Darling.Tests/ServerHealthClassifierTests.cs b/Darling/Darling.Tests/ServerHealthClassifierTests.cs index b64095e11..e1a6028fc 100644 --- a/Darling/Darling.Tests/ServerHealthClassifierTests.cs +++ b/Darling/Darling.Tests/ServerHealthClassifierTests.cs @@ -197,12 +197,15 @@ public void BlockingTiers_SitAtTheTopOfTheQuietMode_AndTheFootOfTheStormMode() /// /// One failing of forty and forty of forty no longer band alike: any FAILING collector is Warning, and a /// FAILING share past the collector-health classifier's own 20% bar is Critical. Nothing failing is - /// Healthy whatever the denominator; no denominator with something failing is Warning and never - /// Critical — a share nobody computed cannot escalate. + /// Healthy when collectors were banded, and Unknown when none were (#3539 A6 — no collection to call + /// clean); no denominator with something failing is Warning and never Critical — a share nobody + /// computed cannot escalate. /// [Theory] [InlineData(0, 40, HealthSeverity.Healthy)] - [InlineData(0, 0, HealthSeverity.Healthy)] + [InlineData(0, 1, HealthSeverity.Healthy)] + [InlineData(0, 0, HealthSeverity.Unknown)] // nothing banded: not a clean collection, an unmeasured one + [InlineData(0, -1, HealthSeverity.Unknown)] [InlineData(1, 40, HealthSeverity.Warning)] // 2.5% [InlineData(8, 40, HealthSeverity.Warning)] // exactly 20% — the bar is strict, as the classifier's is [InlineData(9, 40, HealthSeverity.Critical)] // 22.5% @@ -305,11 +308,45 @@ public void OverallMetricSeverity_WarningWhenNoCritical() [Fact] public void OverallMetricSeverity_AllCalm_IsHealthy_UnknownNeverEscalates() { - // No CPU snapshot (Unknown) and no threads snapshot (Unknown) must not escalate the card. - var m = new ServerHealthMetrics { CpuPercentForAlert = null, TotalThreads = null }; + // No CPU snapshot (Unknown) and no threads snapshot (Unknown) must not escalate the card. Memory + // is measured and calm, so the fold has one real reading to answer Healthy from (#3539 A6: with + // NO reading it answers Unknown, pinned separately below). + var m = new ServerHealthMetrics { CpuPercentForAlert = null, TotalThreads = null, HasMemoryPressure = false }; Assert.Equal(HealthSeverity.Healthy, ServerHealthClassifier.OverallMetricSeverity(m)); } + /// + /// #3539 A6: a bundle on which NOT ONE metric was measured folds to Unknown, not Healthy, and the fleet + /// band reads it as Warning — the never-collected server's band — so it leaves the healthy mass. The + /// pre-fix fold answered Healthy here ("0 of 6 measured" was the only tell), which put an online server + /// nothing had banded yet in healthy_count. One measured reading is enough to lift the fold off + /// Unknown, which is what keeps #3528's partially-measured Healthy exactly where it was. + /// + [Fact] + public void OverallMetricSeverity_NothingMeasured_IsUnknown_AndBandsWarning() + { + var nothing = new ServerHealthMetrics(); + Assert.Equal((0, 6), ServerHealthClassifier.MeasuredMetricCounts(nothing)); + + var overall = ServerHealthClassifier.OverallMetricSeverity(nothing); + Assert.Equal(HealthSeverity.Unknown, overall); + Assert.Equal( + FleetHealthBand.Warning, + ServerHealthClassifier.ClassifyBand(isOnline: true, awaitingFirstCollection: false, collectionStale: false, overall)); + + /* One measured, calm reading and the fold is Healthy again — the #3528 partial-coverage card. */ + var one = nothing with { CollectorCount = 40 }; + Assert.Equal((1, 6), ServerHealthClassifier.MeasuredMetricCounts(one)); + Assert.Equal(HealthSeverity.Healthy, ServerHealthClassifier.OverallMetricSeverity(one)); + + /* A Warning among Unknowns is still Warning — the nothing-measured arm never de-escalates. */ + Assert.Equal(HealthSeverity.Warning, ServerHealthClassifier.OverallMetricSeverity(nothing with { CpuPercentForAlert = 85 })); + /* And the order the readings arrive in cannot matter: Warning first then Healthy, Healthy first + then Warning, both Warning. */ + Assert.Equal(HealthSeverity.Warning, ServerHealthClassifier.OverallMetricSeverity(nothing with { CpuPercentForAlert = 85, CollectorCount = 40 })); + Assert.Equal(HealthSeverity.Warning, ServerHealthClassifier.OverallMetricSeverity(nothing with { HasMemoryPressure = false, FailedCollectorCount = 1, CollectorCount = 40 })); + } + /* ── measured-metric coverage (#3528) ── */ [Fact] @@ -325,6 +362,7 @@ public void MeasuredMetricCounts_FullyMeasuredBundle_CountsAllSix() BlockingWindow = TimeSpan.FromHours(1), // #3539 A3: a zero count is measured only over a window DeadlockCount = 0, DeadlockWindow = TimeSpan.FromHours(1), + CollectorCount = 40, // #3539 A6: zero failing is measured only with a denominator }; Assert.Equal((6, 6), ServerHealthClassifier.MeasuredMetricCounts(m)); @@ -337,8 +375,9 @@ public void MeasuredMetricCounts_UnknownHeavyBundle_SaysSo_WhileTheFoldStillRead (no CPU/threads snapshot, DMV-sourced memory/blocking/deadlocks nulled), only the collector row measured. The fold deliberately skips Unknown, so the band label is still Healthy — and the counts are what let a consumer render that label as "Healthy — 1 of 6 measured" instead of an - unqualified green. */ - var m = new ServerHealthMetrics(); + unqualified green. The collector row is measured only because a denominator was declared + (#3539 A6): forty banded, none failing. */ + var m = new ServerHealthMetrics { CollectorCount = 40 }; Assert.Equal((1, 6), ServerHealthClassifier.MeasuredMetricCounts(m)); @@ -364,8 +403,12 @@ public void MeasuredMetricCounts_AreRankNeutral() BlockingWindow = TimeSpan.FromHours(1), DeadlockCount = 0, DeadlockWindow = TimeSpan.FromHours(1), + CollectorCount = 40, }; - var unmeasured = new ServerHealthMetrics(); + /* One of six measured (the collectors row, #3539 A6's denominator declared) against six of six: + the partial-coverage neutrality #3528 promised. A bundle measuring NOTHING is the one case that + does move, and OverallMetricSeverity_NothingMeasured_IsUnknown_AndBandsWarning owns it. */ + var unmeasured = new ServerHealthMetrics { CollectorCount = 40 }; Assert.NotEqual( ServerHealthClassifier.MeasuredMetricCounts(measured), @@ -402,6 +445,23 @@ public void ClassifyBand_OnlineCalm_IsHealthy() => Assert.Equal(FleetHealthBand.Healthy, ServerHealthClassifier.ClassifyBand(isOnline: true, awaitingFirstCollection: false, collectionStale: false, HealthSeverity.Healthy)); + /// #3539 A6: an online card whose fold is Unknown (nothing measured) is Warning — the same band + /// the awaiting-first-collection server gets, for the same reason — and never Healthy, stale or not. + /// Offline still wins over it. + [Fact] + public void ClassifyBand_OnlineNothingMeasured_IsWarning_LikeAwaitingFirstCollection() + { + Assert.Equal(FleetHealthBand.Warning, + ServerHealthClassifier.ClassifyBand(isOnline: true, awaitingFirstCollection: false, collectionStale: false, HealthSeverity.Unknown)); + Assert.Equal(FleetHealthBand.Warning, + ServerHealthClassifier.ClassifyBand(isOnline: true, awaitingFirstCollection: false, collectionStale: true, HealthSeverity.Unknown)); + Assert.Equal(FleetHealthBand.Offline, + ServerHealthClassifier.ClassifyBand(isOnline: false, awaitingFirstCollection: false, collectionStale: false, HealthSeverity.Unknown)); + Assert.Equal( + ServerHealthClassifier.ClassifyBand(isOnline: null, awaitingFirstCollection: true, collectionStale: false, HealthSeverity.Unknown), + ServerHealthClassifier.ClassifyBand(isOnline: true, awaitingFirstCollection: false, collectionStale: false, HealthSeverity.Unknown)); + } + /* ── worst-first score ── */ [Fact] diff --git a/Darling/Darling.Tests/UnmeasuredMetricsAreNotHealthyTests.cs b/Darling/Darling.Tests/UnmeasuredMetricsAreNotHealthyTests.cs index fdc16d8c0..4111528c9 100644 --- a/Darling/Darling.Tests/UnmeasuredMetricsAreNotHealthyTests.cs +++ b/Darling/Darling.Tests/UnmeasuredMetricsAreNotHealthyTests.cs @@ -40,17 +40,29 @@ namespace Darling.Tests; /// existing code rather than a happy accident of this change, and /// pins it so an "improvement" that /// made Unknown escalate would fail here instead of silently reordering the fleet. +/// +/// #3539 A6, the two edges of the same family. The collectors row had a Healthy arm no other +/// metric here has: (failed 0, banded 0) — a server nothing had banded yet — read Healthy, a green +/// dot for a collection nobody had classified. And the fold over a card on which NOTHING was measured +/// answered Healthy, so an online server with six Unknowns counted in the fleet's healthy mass with a +/// "0 of 6 measured" qualifier as its only tell. Both now read Unknown, and the all-Unknown card bands +/// Warning — the never-collected server's band. Neutrality is unchanged wherever anything IS measured: +/// the pins below hold a one-of-six card exactly where #3528 left it. /// public sealed class UnmeasuredMetricsAreNotHealthyTests { private static readonly DateTime Now = new(2026, 9, 10, 12, 0, 0, DateTimeKind.Utc); + /// Collectors banded for this server, none failing — forty by default so + /// the card's collectors row is a MEASURED calm reading and the DMV metrics stay this file's only + /// variable. Zero is the #3539 A6 shape: nothing banded, nothing measured. private static FleetServerCard Card( string? engineKind, bool memoryPressure = false, int blocking = 0, long maxBlockingWaitMs = 0, - int deadlocks = 0) => + int deadlocks = 0, + int bandedCollectors = 40) => DarlingFleetReader.BuildCard( new DarlingFleetReader.FleetServerRow(1, "t", "t", null, engineKind, false), default, @@ -61,7 +73,7 @@ private static FleetServerCard Card( new DarlingFleetReader.BlockingRow(blocking, maxBlockingWaitMs, 0, 0), new DarlingFleetReader.DeadlockRow(deadlocks, deadlocks > 0 ? Now.AddMinutes(-5) : null), Now.AddSeconds(-30), - default, + new DarlingFleetReader.CollectorCounts(bandedCollectors, 0, bandedCollectors), null, Now, /* #3368: a real one-hour window and the shipped tiers. This file's subject is the @@ -76,12 +88,15 @@ private static ServerSummaryItem ViewerCard( bool memoryPressure = false, int blocking = 0, long maxBlockingWaitMs = 0, - int deadlocks = 0) + int deadlocks = 0, + int bandedCollectors = 40) { var card = new ServerSummaryItem { ServerName = "t", ServerId = 1, + HealthyCollectorCount = bandedCollectors, + CollectorCount = bandedCollectors, MemoryWaiterCount = memoryPressure ? 3 : 0, BlockingCount = blocking, MaxBlockingWaitMs = maxBlockingWaitMs, @@ -312,16 +327,124 @@ public void UnknownIsBandAndRankNeutral_SoTheFixCannotReorderTheFleet(int blocki /// /// default(ServerHealthMetrics) flipped from "all zero, therefore Healthy" to "all null, - /// therefore Unknown" when the fields became nullable. That is only safe BECAUSE Unknown is inert, so it - /// is pinned rather than assumed: a bundle nobody populated still bands and scores as it did. + /// therefore Unknown" when the fields became nullable, and #3539 A6 closed the last gap: its collectors + /// row (0, 0) is Unknown too, so a bundle nobody populated measures NOTHING — and the fold says + /// so rather than answering Healthy from six Unknowns. Its band is the never-collected server's Warning, + /// and the score is that band's rank alone: the magnitude terms still skip Unknown, so it cannot climb + /// within the band on readings it does not have. /// [Fact] - public void AnUnpopulatedMetricBundleBandsAndScoresAsItAlwaysDid() + public void AnUnpopulatedMetricBundleMeasuresNothing_AndIsNotHealthy() { var empty = default(ServerHealthMetrics); - Assert.Equal(HealthSeverity.Healthy, ServerHealthClassifier.OverallMetricSeverity(empty)); - Assert.Equal(0L, ServerHealthClassifier.FleetHealthScore(FleetHealthBand.Healthy, empty)); + Assert.Equal((0, 6), ServerHealthClassifier.MeasuredMetricCounts(empty)); + Assert.Equal(HealthSeverity.Unknown, ServerHealthClassifier.OverallMetricSeverity(empty)); + + var band = ServerHealthClassifier.ClassifyBand(true, false, false, HealthSeverity.Unknown); + Assert.Equal(FleetHealthBand.Warning, band); + /* The Warning rank step and nothing else: no Critical/Warning metric to add magnitude, no incident + count — so an all-Unknown card sorts at the foot of the Warning band, under any card with a real + amber reading. */ + Assert.Equal(2000L, ServerHealthClassifier.FleetHealthScore(band, empty)); + } + + /* ─────────────────────────── #3539 A6: nothing banded, nothing measured ─────────────────────────── */ + + /// + /// The collectors row with NO collector banded reads Unknown on both cards, not Healthy — the #3539 A6 + /// sibling. The viewer's offline arm and the web's is_online === false chip cover a KNOWN-dark + /// server; this is a reachable one whose collection nobody has classified, and "nothing failing" is not + /// a health claim when nothing could have failed. Held on a SQL Server card so the other five rows are + /// measured and the collectors row is the only thing that changed. + /// + [Fact] + public void ZeroBandedCollectors_ReadUnknownNotHealthy_OnBothCards() + { + var card = Card(MonitoredEngineKind.SqlServer, bandedCollectors: 0); + Assert.Equal(0, card.CollectorCount); + Assert.Equal(0, card.FailedCollectorCount); + Assert.Equal(HealthSeverity.Unknown, card.CollectorSeverity); + + var viewer = ViewerCard(MonitoredEngineKind.SqlServer, bandedCollectors: 0); + Assert.Equal(HealthSeverity.Unknown, viewer.CollectorSeverity); + /* The word beside the dot agrees with it: "--" is the card's spelling of "no reading", where "OK" + was a green word under what is now a grey dot. */ + Assert.Equal("--", viewer.CollectorDisplay); + + /* And with ONE collector banded the arm is Healthy again, on both — the fix is the zero, not the + count. */ + Assert.Equal(HealthSeverity.Healthy, Card(MonitoredEngineKind.SqlServer, bandedCollectors: 1).CollectorSeverity); + Assert.Equal(HealthSeverity.Healthy, ViewerCard(MonitoredEngineKind.SqlServer, bandedCollectors: 1).CollectorSeverity); + Assert.Equal("OK", ViewerCard(MonitoredEngineKind.SqlServer, bandedCollectors: 1).CollectorDisplay); + + /* The collectors row is one of the six the coverage counts fold over: this helper's SQL Server card + measures memory, blocking and deadlocks (no CPU or threads row is handed in), so it says "3 of 6 + measured" with nothing banded and "4 of 6" with one collector — the row moved, and only the row. */ + Assert.Equal(3, card.MeasuredMetricCount); + Assert.Equal(6, card.MetricCount); + Assert.Equal(4, Card(MonitoredEngineKind.SqlServer, bandedCollectors: 1).MeasuredMetricCount); + Assert.Equal(FleetHealthBand.Healthy, card.Band); + } + + /// + /// The A6 card itself: an ONLINE PostgreSQL target with nothing banded — no CPU source, no threads, the + /// three DMV rows structurally null, zero collectors — measures nothing, and is NOT Healthy. Before this + /// it banded Healthy with "0 of 6 measured" as its only tell and counted in healthy_count on the + /// web fleet page, get_fleet_overview and the viewer's rollup. Now it bands Warning like a server + /// awaiting its first collection, leaves the healthy mass on every one of those surfaces, and the + /// ranking's reason says why in words rather than falling to "Needs attention". + /// + [Theory] + [InlineData(MonitoredEngineKind.Postgres)] + [InlineData(MonitoredEngineKind.AuroraPostgres)] + public void AnOnlineCardMeasuringNothing_IsNotInTheHealthyMass_OnAnySurface(string engineKind) + { + var card = Card(engineKind, bandedCollectors: 0); + Assert.True(card.IsOnline); + Assert.False(card.AwaitingFirstCollection); + Assert.Equal(0, card.MeasuredMetricCount); + Assert.Equal(6, card.MetricCount); + Assert.Equal(HealthSeverity.Unknown, card.OverallMetricSeverity); + Assert.Equal(FleetHealthBand.Warning, card.Band); + + /* The service's rollup — /api/fleet and get_fleet_overview read this: zero healthy, one warning. */ + var rollup = DarlingFleetReader.BuildRollup(new[] { card }, Now, Now.AddHours(-1), Now); + Assert.Equal(0, rollup.HealthyCount); + Assert.Equal(1, rollup.WarningCount); + var ranked = Assert.Single(rollup.WorstServers); + Assert.Equal(DarlingFleetReader.NoMetricMeasuredReason, ranked.Reason); + + /* The viewer's card and rollup, same server, same answer (#2473). */ + var viewer = ViewerCard(engineKind, bandedCollectors: 0); + Assert.Equal(0, viewer.MeasuredMetricCount); + Assert.Equal(HealthSeverity.Unknown, viewer.OverallMetricSeverity); + Assert.Equal(FleetHealthBand.Warning, FleetRollup.ClassifyBand(viewer)); + Assert.Equal(FleetRollup.NoMetricMeasuredReason, FleetRollup.BuildReason(viewer)); + Assert.Equal(DarlingFleetReader.NoMetricMeasuredReason, FleetRollup.NoMetricMeasuredReason); + + var viewerRollup = FleetRollup.Build(new[] { viewer }, new FleetTotals()); + Assert.Equal(0, viewerRollup.HealthyCount); + Assert.Equal(1, viewerRollup.WarningCount); + Assert.Contains(viewer, FleetRollup.NeedsAttention(new[] { viewer })); + + /* The tooltip's headline names the band and the reason once — not "Warning — no metric measured yet + · 0 of 6 measured", which would say the same thing twice. */ + Assert.StartsWith("Warning — " + FleetRollup.NoMetricMeasuredReason, viewer.StatusTooltip, StringComparison.Ordinal); + Assert.DoesNotContain("0 of 6", viewer.StatusTooltip, StringComparison.Ordinal); + + /* The border agrees with the band: the awaiting-first-collection amber, not the calm dark. */ + Assert.Equal("#FFFFD54F", viewer.CardBorderBrush.Color.ToString()); + + /* ONE banded collector and the same card is #3528's "Healthy — 1 of 6 measured", exactly where + that issue left it: the healthy mass loses only the cards that measured nothing. */ + var one = Card(engineKind, bandedCollectors: 1); + Assert.Equal(1, one.MeasuredMetricCount); + Assert.Equal(FleetHealthBand.Healthy, one.Band); + Assert.Equal(1, DarlingFleetReader.BuildRollup(new[] { one }, Now, Now.AddHours(-1), Now).HealthyCount); + var oneViewer = ViewerCard(engineKind, bandedCollectors: 1); + Assert.Equal(FleetHealthBand.Healthy, FleetRollup.ClassifyBand(oneViewer)); + Assert.StartsWith("Healthy — 1 of 6 measured", oneViewer.StatusTooltip, StringComparison.Ordinal); } /* ─────────────────────────── the ordering the guards depend on ─────────────────────────── */ diff --git a/Darling/Darling.Tests/ViewerOverviewExplainsItselfTests.cs b/Darling/Darling.Tests/ViewerOverviewExplainsItselfTests.cs index 98624168a..5cc26e8ea 100644 --- a/Darling/Darling.Tests/ViewerOverviewExplainsItselfTests.cs +++ b/Darling/Darling.Tests/ViewerOverviewExplainsItselfTests.cs @@ -49,6 +49,7 @@ private static ServerSummaryItem Busy(string name = "b1", int id = 2) => CpuPercent = 96, BlockingCount = 6, MaxBlockingWaitMs = 70000, + CollectorCount = 40, // #3539 A6: the collectors row is measured only with a denominator declared }; private static ServerSummaryItem Stale(string name = "s1", int id = 3) => @@ -136,6 +137,7 @@ private static ServerSummaryItem FullyMeasured(string name = "f1", int id = 7) = CurrentWorkers = 100, DeadlockWindow = TimeSpan.FromHours(1), BlockingWindow = TimeSpan.FromHours(1), // #3539 A3: a zero count is measured only over a window + CollectorCount = 40, // #3539 A6: zero failing is measured only with a denominator }; /// @@ -149,9 +151,11 @@ private static ServerSummaryItem FullyMeasured(string name = "f1", int id = 7) = public void TheCardsTooltip_QualifiesAHealthyBand_ThatFoldedOverUnmeasuredMetrics() { /* No CPU/threads snapshot, DMV-sourced memory/blocking/deadlocks nulled by the engine — only the - collector row measured. The same shape DarlingFleetReader's card serializes as 1-of-6. */ + collector row measured, which since #3539 A6 means its denominator is declared: forty banded, + none failing. The same shape DarlingFleetReader's card serializes as 1-of-6. */ var pg = Healthy(); pg.IsPostgres = true; + pg.CollectorCount = 40; Assert.Equal(1, pg.MeasuredMetricCount); Assert.Equal(6, pg.MetricCount); diff --git a/Darling/Darling.Tests/ViewerW2aTests.cs b/Darling/Darling.Tests/ViewerW2aTests.cs index 09882f4ea..63cbccf91 100644 --- a/Darling/Darling.Tests/ViewerW2aTests.cs +++ b/Darling/Darling.Tests/ViewerW2aTests.cs @@ -564,7 +564,14 @@ public void CollectorSeverity_FailingIsWarning_HealthyOtherwise() Assert.Equal(HealthSeverity.Warning, failing.CollectorSeverity); Assert.Equal("2 failed", failing.CollectorDisplay); Assert.Equal("Healthy: 28, Failing: 2", failing.CollectorDetail); - Assert.Equal("OK", new ServerSummaryItem { HealthyCollectorCount = 30 }.CollectorDisplay); + Assert.Equal("OK", new ServerSummaryItem { HealthyCollectorCount = 30, CollectorCount = 30 }.CollectorDisplay); + + /* #3539 A6: with NO collector banded the dot is Unknown and the word is the card's "no reading" + spelling, not a green "OK" — a reachable server nothing has classified yet is not a clean one. */ + var nothingBanded = new ServerSummaryItem { IsOnline = true }; + Assert.Equal(HealthSeverity.Unknown, nothingBanded.CollectorSeverity); + Assert.Equal("--", nothingBanded.CollectorDisplay); + Assert.Equal("Healthy: 0, Failing: 0", nothingBanded.CollectorDetail); } /// #3539 A8d: the collector dot is graded on the FAILING share — one of forty is Warning, @@ -607,19 +614,19 @@ public void CollectorSeverity_OfflineServer_ReadsStaleNeutral_NotGreenOk() // A GENUINE collector failure on a reachable server still surfaces red / "N failed" — not swallowed // into Stale. - var failing = new ServerSummaryItem { IsOnline = true, HealthyCollectorCount = 28, FailedCollectorCount = 2 }; + var failing = new ServerSummaryItem { IsOnline = true, HealthyCollectorCount = 28, FailedCollectorCount = 2, CollectorCount = 30 }; Assert.Equal("2 failed", failing.CollectorDisplay); Assert.Equal(HealthSeverity.Warning, failing.CollectorSeverity); - // A healthy ONLINE server is unchanged — green "OK". - var healthy = new ServerSummaryItem { IsOnline = true, HealthyCollectorCount = 30, FailedCollectorCount = 0 }; + // A healthy ONLINE server is unchanged — green "OK" (its thirty banded collectors declared, #3539 A6). + var healthy = new ServerSummaryItem { IsOnline = true, HealthyCollectorCount = 30, FailedCollectorCount = 0, CollectorCount = 30 }; Assert.Equal("OK", healthy.CollectorDisplay); Assert.Equal("Healthy: 30, Failing: 0", healthy.CollectorDetail); Assert.Equal(HealthSeverity.Healthy, healthy.CollectorSeverity); // Not-yet-connection-classified (IsOnline null — awaiting first collection) keeps the normal reading: // "Stale" is for a KNOWN-offline server only, matching the web's strict `is_online === false`. - var notChecked = new ServerSummaryItem { HealthyCollectorCount = 30, FailedCollectorCount = 0 }; + var notChecked = new ServerSummaryItem { HealthyCollectorCount = 30, FailedCollectorCount = 0, CollectorCount = 30 }; Assert.False(notChecked.IsOffline); Assert.Equal("OK", notChecked.CollectorDisplay); Assert.Equal(HealthSeverity.Healthy, notChecked.CollectorSeverity); diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingAlertDeliverer.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingAlertDeliverer.cs index 38e874bbd..c01fff4d5 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingAlertDeliverer.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingAlertDeliverer.cs @@ -163,8 +163,10 @@ private async Task SendAndRecordAsync( only Context.SeverityOverride — so every self-alert (fired with Context: null) rendered INFO-blue in Teams/Slack/PagerDuty/webhooks while its log line said Critical. Fold the outcome's severity into the context here, once, upstream of every channel; ??= so an - explicit override set by a context builder still wins. The context also serializes into - alert history, so replays keep the severity too. */ + explicit override set by a context builder still wins. The context serializes into alert + history below, and since #3539 A8e the serializer carries this property as the row's Severity + member — so the history grids and get_alert_history read the tier the alert fired at (before + that the projection dropped it, and this comment's "replays keep the severity" was not true). */ if (outcome.Severity is not null) { context ??= new AlertContext(); diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingAlertReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingAlertReader.cs index 27c60cf0c..db44ddc28 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingAlertReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingAlertReader.cs @@ -41,11 +41,14 @@ internal static class DarlingAlertReader /// is the operator's Viewer acknowledgement (#3541 A3): a row the /// operator hid from the Alert History grid. Always false on the default read, which excludes those rows; - /// carried so a read that INCLUDES them can label each one. + /// carried so a read that INCLUDES them can label each one. + /// (#3539 A8e) is the persisted context, read so the tool can report + /// the tier the alert FIRED at through AlertHistoryRowSeverity rather than the colour its name + /// implies; null on resolution rows and rows written with no context. public sealed record AlertHistoryReadRow( DateTime AlertTime, int ServerId, string ServerName, string MetricName, double CurrentValue, double ThresholdValue, bool AlertSent, string NotificationType, - string? SendError, bool Muted, string? DetailText, bool Dismissed); + string? SendError, bool Muted, string? DetailText, bool Dismissed, string? ContextJson = null); private const string AlertHistorySelectColumns = @" alert_time, @@ -59,6 +62,7 @@ public sealed record AlertHistoryReadRow( send_error, muted, detail_text, + context_json, dismissed"; /// Per-server alert history — the viewer's AlertHistorySql. $1 window start, $2 window @@ -166,7 +170,10 @@ public static async Task> GetAlertHistoryPageAsync( reader.IsDBNull(8) ? null : reader.GetString(8), !reader.IsDBNull(9) && reader.GetBoolean(9), reader.IsDBNull(10) ? null : reader.GetString(10), - !reader.IsDBNull(11) && reader.GetBoolean(11))); + /* context_json sits at ordinal 11 and dismissed stays the LAST column at 12 — the viewer's + own column order, and the "dismissed is selected" pin anchors on it closing the list. */ + !reader.IsDBNull(12) && reader.GetBoolean(12), + reader.IsDBNull(11) ? null : reader.GetString(11))); } return rows; diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingFleetReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingFleetReader.cs index 45a3cbf1a..f51daa77c 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingFleetReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingFleetReader.cs @@ -831,9 +831,22 @@ 1 in an hour and 1 in a day are the same string. The count stays because it is t parts.Add("collection stale"); } + if (c.MetricCount > 0 && c.MeasuredMetricCount == 0) + { + /* #3539 A6: the card banded Warning because NOTHING on it was measured (OverallMetricSeverity's + nothing-measured arm), and no per-metric clause above can fire for a card whose every band is + Unknown — so without this the ranking would show "Needs attention" against a card that + cannot say why. The same words the viewer's reason uses. */ + parts.Add(NoMetricMeasuredReason); + } + return parts.Count > 0 ? string.Join(", ", parts) : "Needs attention"; } + /// The reason clause for a card on which no metric was measured (#3539 A6) — the viewer's + /// FleetRollup.BuildReason spells it identically, so the two surfaces read alike. + internal const string NoMetricMeasuredReason = "no metric measured yet"; + /// The card's status word. Delegates to the one ladder every Darling surface renders (#2473): /// this file's own copy agreed with the WPF card, but the WPF sidebar row's copy did not, and three /// agreeing copies plus one that does not is still four places where the answer is decided. @@ -1430,17 +1443,23 @@ public sealed class FleetServerCard /// Every collector banded for this server in the health window, on any band (#3539 A8d) — the /// denominator collector_severity grades failed_collector_count against. Not /// healthy + failed: STALE, WARNING, STOPPED and the permission bands are banded collectors that are - /// neither. + /// neither. Zero with nothing failing bands collector_severity Unknown, not Healthy (#3539 + /// A6): no collector has been banded for this server, so there is no collection to call clean. [JsonPropertyName("collector_count")] public int CollectorCount { get; init; } [JsonPropertyName("collector_severity")] public HealthSeverity CollectorSeverity { get; init; } + /// The worst per-metric band, or Unknown when NOT ONE metric on the card was measured (#3539 + /// A6) — which band then reads as Warning, the never-collected server's band, rather than + /// Healthy. [JsonPropertyName("overall_metric_severity")] public HealthSeverity OverallMetricSeverity { get; init; } /// How many of the card's per-metric severities carried a real reading when it banded (#3528) /// — the band's fold skips Unknown, so a card can read Healthy off one measured metric of six. When /// this is below , the band label deserves the qualifier ("Healthy — 1 of 6 /// measured"); the web fleet page renders exactly that. Purely descriptive: it feeds neither the band - /// nor the worst-first score, so rank-neutrality of Unknown is unchanged. + /// nor the worst-first score, so rank-neutrality of Unknown is unchanged — with the one exception + /// #3539 A6 draws at zero: a card measuring NOTHING is not Healthy (see overall_metric_severity), + /// and its reason says so. [JsonPropertyName("measured_metric_count")] public int MeasuredMetricCount { get; init; } /// The denominator for — how many per-metric severities the diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpAlertTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpAlertTools.cs index dd8c1f5bf..f5afef035 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpAlertTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpAlertTools.cs @@ -75,7 +75,7 @@ namespace PerformanceMonitor.Darling.Service.Mcp; [McpServerToolType] public sealed class DarlingMcpAlertTools { - [McpServerTool(Name = "get_alert_history"), Description("Gets recent alert history from the alert log, NEWEST FIRST: what alerts fired, when, for which server, the current vs threshold value, whether email/webhook delivery succeeded, and whether the alert was muted. Omit server_name to see the whole fleet (each row names its server); pass one to scope to a single server. THE PAGE IS BOUNDED BY limit, NOT BY hours_back: alerts_returned is how many rows you got, truncated says the window held more than limit, and oldest_returned_alert_time / newest_returned_alert_time bound the page — under newest-first ordering the oldest stamp IS how far back this read reached, so on a noisy fleet a 24-hour request at the default limit may cover minutes. Raise limit or narrow hours_back when truncated is true; widening hours_back cannot help. BY DEFAULT THIS READ EXCLUDES DISMISSED ALERTS — rows an operator acknowledged in the Viewer's Alert History grid. Dismissal says nothing about whether the alert fired or mattered, so an incident reconstruction that ignores it can miss the very critical someone already looked at: dismissed_excluded says whether the filter applied and dismissed_excluded_count is how many rows in the window it removed, and include_dismissed = true returns them, each labelled dismissed = true. notification_type is the delivery disposition and is the ONLY field that says why a row did not deliver: 'email'/'webhook'/'email+webhook' delivered on that channel; 'throttled' means the delivery cooldown was still inside this alert's window so nothing was attempted (the throttle working, not a fault); 'folded' means a repeat was rolled onto another server's post for the same metric and is named there under 'Other Servers Affected', so it WAS reported; 'failed' means a channel was attempted and came back unsuccessful, with send_error carrying the first failing channel's text; 'unconfigured' means no email or webhook channel is set up; 'muted' means a mute rule suppressed it; 'none' is a resolution row, which no channel applies to. Do NOT split the not-delivered rows on send_error: it is null on 'throttled' and 'folded' rows and on every row written before those values existed, so a null error is not evidence of a working cooldown. 'undelivered' is a retained legacy value that means throttled OR folded OR failed with nothing in the row to say which — count those rows separately rather than attributing them.")] + [McpServerTool(Name = "get_alert_history"), Description("Gets recent alert history from the alert log, NEWEST FIRST: what alerts fired, when, for which server, the current vs threshold value, whether email/webhook delivery succeeded, and whether the alert was muted. Omit server_name to see the whole fleet (each row names its server); pass one to scope to a single server. THE PAGE IS BOUNDED BY limit, NOT BY hours_back: alerts_returned is how many rows you got, truncated says the window held more than limit, and oldest_returned_alert_time / newest_returned_alert_time bound the page — under newest-first ordering the oldest stamp IS how far back this read reached, so on a noisy fleet a 24-hour request at the default limit may cover minutes. Raise limit or narrow hours_back when truncated is true; widening hours_back cannot help. BY DEFAULT THIS READ EXCLUDES DISMISSED ALERTS — rows an operator acknowledged in the Viewer's Alert History grid. Dismissal says nothing about whether the alert fired or mattered, so an incident reconstruction that ignores it can miss the very critical someone already looked at: dismissed_excluded says whether the filter applied and dismissed_excluded_count is how many rows in the window it removed, and include_dismissed = true returns them, each labelled dismissed = true. notification_type is the delivery disposition and is the ONLY field that says why a row did not deliver: 'email'/'webhook'/'email+webhook' delivered on that channel; 'throttled' means the delivery cooldown was still inside this alert's window so nothing was attempted (the throttle working, not a fault); 'folded' means a repeat was rolled onto another server's post for the same metric and is named there under 'Other Servers Affected', so it WAS reported; 'failed' means a channel was attempted and came back unsuccessful, with send_error carrying the first failing channel's text; 'unconfigured' means no email or webhook channel is set up; 'muted' means a mute rule suppressed it; 'none' is a resolution row, which no channel applies to. Do NOT split the not-delivered rows on send_error: it is null on 'throttled' and 'folded' rows and on every row written before those values existed, so a null error is not evidence of a working cooldown. 'undelivered' is a retained legacy value that means throttled OR folded OR failed with nothing in the row to say which — count those rows separately rather than attributing them. severity is the row's tier — 'critical', 'warning', 'info' or 'resolution' — and severity_source says where it came from: 'fired' when the row persisted the tier the alert actually fired at (graded alerts such as Poison Wait, Volume Free Space and Database State fire Warning OR Critical by measurement), 'metric_name' when the row carries no tier and the metric's name is the only evidence (rows written before the tier was persisted, alerts whose severity is fixed per metric, and every resolution row). Do not infer a graded alert's tier from its name: a 'Poison Wait' row with severity 'warning' fired as a warning.")] public static async Task GetAlertHistory( NpgsqlDataSource postgres, [Description("Server name or display name. Omit to return alerts across all servers (the fleet default).")] string? server_name = null, @@ -136,22 +136,33 @@ the caller off widening a window whose contents they were never shown. */ : McpHelpers.Status("empty", "No alerts found in the specified time range."); } - var alerts = page.Select(r => new + var alerts = page.Select(r => { - alert_time = r.AlertTime.ToString("o"), - server_id = r.ServerId, - server_name = r.ServerName, - metric_name = r.MetricName, - current_value = r.CurrentValue, - threshold_value = r.ThresholdValue, - alert_sent = r.AlertSent, - notification_type = r.NotificationType, - send_error = r.SendError, - muted = r.Muted, - /* Per row, so a page that mixes the two populations labels each one. Always false on the - default read, which is a true statement about every row on it. */ - dismissed = r.Dismissed, - detail_text = r.DetailText + var (severity, severitySource) = AlertHistoryRowSeverity.Describe(r.MetricName, r.ContextJson); + return new + { + alert_time = r.AlertTime.ToString("o"), + server_id = r.ServerId, + server_name = r.ServerName, + metric_name = r.MetricName, + current_value = r.CurrentValue, + threshold_value = r.ThresholdValue, + alert_sent = r.AlertSent, + notification_type = r.NotificationType, + send_error = r.SendError, + muted = r.Muted, + /* Per row, so a page that mixes the two populations labels each one. Always false on the + default read, which is a true statement about every row on it. */ + dismissed = r.Dismissed, + /* #3539 A8e: the tier the alert FIRED at where the row persisted one ("fired"), else what + the metric NAME implies ("metric_name") — the same two arms both Alert History grids + colour rows by, so a caller reading "Poison Wait" here sees the Warning it fired at + rather than the red the name used to earn every row. The source is published because + the two are not equal evidence; see AlertHistoryRowSeverity.Describe. */ + severity, + severity_source = severitySource, + detail_text = r.DetailText, + }; }); return JsonSerializer.Serialize(new diff --git a/Darling/PerformanceMonitor.Darling.Service/wwwroot/js/pages/alerts.js b/Darling/PerformanceMonitor.Darling.Service/wwwroot/js/pages/alerts.js index 70aefbe97..2e48b625f 100644 --- a/Darling/PerformanceMonitor.Darling.Service/wwwroot/js/pages/alerts.js +++ b/Darling/PerformanceMonitor.Darling.Service/wwwroot/js/pages/alerts.js @@ -28,6 +28,7 @@ const ALERT_COLUMNS = [ { key: "alert_time", label: "Time", format: "time" }, { key: "server_name", label: "Server" }, { key: "metric_name", label: "Metric" }, + { key: "severity", label: "Severity", render: (a) => severityCell(a) }, { key: "current_value", label: "Value", format: "num1" }, { key: "threshold_value", label: "Threshold", format: "num1" }, { key: "status", label: "Status", render: (a) => statusCell(a) }, @@ -35,6 +36,26 @@ const ALERT_COLUMNS = [ { key: "triage", label: "Triage", render: (a) => triageCell(a) }, ]; +/* #3539 A8e: the tier the alert FIRED at, as the tool reports it — never re-derived here from the metric name + * (R1). The service reads it off the row's persisted context and falls back to the name only for rows that + * carry none; severity_source says which arm answered, and it rides as the cell's title because the two are + * not equal evidence: "critical" from the row is what the operator was paged with, "critical" from the name + * is what the map says about the name. The tone classes are the status cell's own. */ +const SEVERITY_TONE = { critical: "Critical", warning: "Warning", resolution: "Healthy", info: "Unknown" }; +const SEVERITY_SOURCE_TITLE = { + fired: "The tier this alert fired at, read from the row", + metric_name: "Implied by the metric name; this row carries no fired tier", +}; + +function severityCell(a) { + if (a.severity == null) return el("span", { class: "muted", text: "—" }); + return el("span", { + class: "status-cell sev-" + (SEVERITY_TONE[a.severity] || "Unknown"), + text: String(a.severity), + title: SEVERITY_SOURCE_TITLE[a.severity_source] || null, + }); +} + /* Deep-link into the #2710 triage page for this row — the SAME route the alert webhooks link to, anchored at * this row's own firing instant, so the in-app path and the delivered link land on an identical page. */ function triageCell(a) { diff --git a/Darling/PerformanceMonitor.Darling.Service/wwwroot/js/pages/fleet.js b/Darling/PerformanceMonitor.Darling.Service/wwwroot/js/pages/fleet.js index 81e24dd02..2a7cc3481 100644 --- a/Darling/PerformanceMonitor.Darling.Service/wwwroot/js/pages/fleet.js +++ b/Darling/PerformanceMonitor.Darling.Service/wwwroot/js/pages/fleet.js @@ -633,16 +633,25 @@ export function metricBands(c) { the reachability signal the card already carries (is_online, the same one that bands the card Offline and titles the header "no recent collection"): when the server is offline the chip reads "Stale" in the neutral Unknown tone instead of a green "OK · N healthy" (the "no recent collection" detail carries the specifics, - and "Stale" is the word the Collection Health tab lands on for these rows once its own floor is crossed). */ + and "Stale" is the word the Collection Health tab lands on for these rows once its own floor is crossed). + + #3539 A6: a REACHABLE server with no collector banded at all (collector_count 0, nothing failing) arrives + with collector_severity "Unknown" from the shared band, and its value reads "n/a" — the chip's word for a + metric with no reading (the Threads chip's) — rather than "OK". R1: the severity is read off the card, not + re-derived here; only the WORD keys on the count, and only so a green word never sits under a grey chip. */ const collectorsStale = c.is_online === false; const collectorsValue = collectorsStale ? "Stale" : c.failed_collector_count > 0 ? fmtInt(c.failed_collector_count) + " failing" - : "OK"; + : c.collector_count > 0 + ? "OK" + : "n/a"; const collectorsDetail = collectorsStale ? "no recent collection" + (c.last_collection ? " · last " + relTime(c.last_collection) : "") - : fmtInt(c.healthy_collector_count) + " healthy · " + fmtInt(c.failed_collector_count) + " failing"; + : c.collector_count > 0 + ? fmtInt(c.healthy_collector_count) + " healthy · " + fmtInt(c.failed_collector_count) + " failing" + : "no collector banded yet"; const collectorsSeverity = collectorsStale ? "Unknown" : c.collector_severity; return el("div", { class: "metric-bands" }, [ diff --git a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.AlertHistory.cs b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.AlertHistory.cs index 517fa5cd1..7954a9863 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.AlertHistory.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.AlertHistory.cs @@ -11,6 +11,7 @@ using System.Threading; using System.Threading.Tasks; using Npgsql; +using PerformanceMonitor.Alerting; using PerformanceMonitor.Common; using PerformanceMonitor.Notifications; @@ -21,7 +22,8 @@ namespace PerformanceMonitor.Darling.Viewer; /// (Lite/Services/LocalDataService.AlertHistory.cs): the same metric-keyed value formatting /// (#1134), the same shared behind /// (differing only in the tray answer each SKU gives it), and the shared -/// for critical/warning/resolved row emphasis. Carries +/// / pair for critical/warning/resolved +/// row emphasis (the tier the alert fired at where the row carries it, the name otherwise). Carries /// + so the all-servers Alert History surface (W2a) /// can show a Server column and key the dismiss write on (alert_time, server_id, metric_name). /// Darling has no parquet archive tier, so there is no Source/IsArchived split — every @@ -79,9 +81,12 @@ public sealed class ViewerAlertRow public bool IsResolved => AlertMetricClassifier.IsResolution(MetricName); - public bool IsCritical => AlertMetricClassifier.IsCritical(MetricName); + /* #3539 A8e: the emphasis is the tier the alert FIRED at, read off ContextJson, with the by-name + classifier only for rows that carry none — Lite's row, the web alerts page and get_alert_history make + the same call through AlertHistoryRowSeverity, whose summary states why the fallback still exists. */ + public bool IsCritical => AlertHistoryRowSeverity.IsCritical(MetricName, ContextJson); - public bool IsWarning => AlertMetricClassifier.IsWarning(MetricName); + public bool IsWarning => AlertHistoryRowSeverity.IsWarning(MetricName, ContextJson); /// /// This row as the a is judged against — the SAME diff --git a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Fleet.cs b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Fleet.cs index ae54f3961..1709ed67e 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Fleet.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Fleet.cs @@ -691,9 +691,23 @@ its first collection — see ServerCollectionStatus. */ parts.Add("collection stale"); } + if (s.MetricCount > 0 && s.MeasuredMetricCount == 0) + { + /* #3539 A6: the card banded Warning through OverallMetricSeverity's nothing-measured arm, and no + per-metric clause above can fire when every band is Unknown — without this the ranking and the + tooltip would fall to UnspecifiedReason against a card that CAN say why. The service's + DarlingFleetReader.BuildReason spells it identically. */ + parts.Add(NoMetricMeasuredReason); + } + return parts.Count > 0 ? string.Join(", ", parts) : UnspecifiedReason; } + /// The reason clause for a card on which no metric was measured (#3539 A6). The same words as + /// the service's fleet-card reason, kept as a constant so can recognise the + /// case it already covers. + public const string NoMetricMeasuredReason = "no metric measured yet"; + /// /// What answers when it can name nothing — a card banded away from Healthy by a /// severity whose display the reason does not cover. It reads fine in the ranking, where every row is a @@ -780,10 +794,12 @@ private static string BandHeadline(FleetHealthBand band, ServerSummaryItem s) /* The guarded reason-append stays WithReason's (one copy — its own doc says why); the qualifier rides after whatever it produced: beside a named reason as a second " · " phrase (the web status - line's list separator), else straight after the bare label. */ + line's list separator), else straight after the bare label. At ZERO measured the reason already + says "no metric measured yet" (#3539 A6), and "· 0 of 6 measured" after it would restate the + same fact in different words, so the qualifier stands down there and only there. */ var label = ServerHealthClassifier.BandLabel(band); var headline = WithReason(label, " — ", s); - if (coverage.Length == 0) + if (coverage.Length == 0 || s.MeasuredMetricCount == 0) { return headline; } diff --git a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Overview.cs b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Overview.cs index e192995fe..97232faf3 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Overview.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Overview.cs @@ -860,9 +860,17 @@ public string ThreadsDisplay /// while the server is dark). Rather than invent a stale-count threshold, reuse the reachability signal the /// card already carries (, the same one that drives the offline overlay): an offline /// server's collectors read a neutral "Stale", never a green "OK". + /// + /// #3539 A6: a REACHABLE server with no collector banded at all reads "--" — the card's word for a + /// row with no reading ( uses it for a missing scheduler snapshot) — rather + /// than "OK". "OK" was the value beside a dot the shared band now paints Unknown for exactly this case, + /// and a green word under a grey dot is the card contradicting itself. /// public string CollectorDisplay => - IsOffline ? "Stale" : FailedCollectorCount > 0 ? $"{FailedCollectorCount} failed" : "OK"; + IsOffline ? "Stale" + : FailedCollectorCount > 0 ? $"{FailedCollectorCount} failed" + : CollectorCount > 0 ? "OK" + : "--"; /// Collectors detail — "No recent collection" when offline, else "Healthy: N, Failing: M" (Dashboard's CollectorDetailText). public string CollectorDetail => @@ -1017,7 +1025,9 @@ raw value would show 0.0/hr on a card whose severity says Unknown. */ /// Collectors band — neutral Unknown when the server is offline (its collectors are unmeasured, not /// healthy — #2784), else the shared graded band (#3539 A8d): Warning on any FAILING collector, /// Critical when the FAILING share of passes the collector-health - /// classifier's 20% bar. Offline is already painted by the card border / overlay, so this governs only + /// classifier's 20% bar, and Unknown again when is zero with nothing + /// failing — a reachable server nothing has banded yet (#3539 A6), which the offline arm here does not + /// cover. Offline is already painted by the card border / overlay, so this governs only /// the per-metric dot: it must not show a green "healthy" dot on a dark server. The overall metric band /// reads the counts straight from ToHealthMetrics(), not this property, so the neutral offline reading /// never leaks into the card's worst-band or fleet score. @@ -1085,9 +1095,17 @@ raw value would show 0.0/hr on a card whose severity says Unknown. */ /// /// The card border reflects the worst signal: offline (red) > a Critical metric (red) > a Warning - /// metric (amber-orange) > a stale collection (amber) > calm (dark). Enriches Lite's border (which - /// only knew CPU / blocking / deadlock) with the added Threads / Memory / Collectors bands via - /// . + /// metric (amber-orange) > a stale collection, a never-collected server, or a card on which NOTHING + /// was measured (amber) > calm (dark). Enriches Lite's border (which only knew CPU / blocking / + /// deadlock) with the added Threads / Memory / Collectors bands via . + /// + /// The nothing-measured arm (#3539 A6) paints the Warning brush, because that is the band it is: + /// folds an all-Unknown card to Unknown and + /// bands that Warning the way it bands a server awaiting its + /// first collection — and the two amber arms below are the SAME hex (#FFD54F), so the border says + /// what the band says in the colour the card already uses for "nothing to report yet". A dark border + /// here was the card claiming calm about readings it never took. The cached brush rather than + /// another MakeBrush: this getter runs per card per refresh. /// public SolidColorBrush CardBorderBrush { @@ -1098,6 +1116,7 @@ public SolidColorBrush CardBorderBrush { HealthSeverity.Critical => s_criticalBrush, HealthSeverity.Warning => s_warningBrush, + HealthSeverity.Unknown => s_warningBrush, _ => CollectionStale || AwaitingFirstCollection ? MakeBrush("#FFD54F") : MakeBrush("#2a2d35"), }; } diff --git a/Lite.Tests/AlertHistoryRowSeverityLiteTests.cs b/Lite.Tests/AlertHistoryRowSeverityLiteTests.cs new file mode 100644 index 000000000..5f01224cf --- /dev/null +++ b/Lite.Tests/AlertHistoryRowSeverityLiteTests.cs @@ -0,0 +1,87 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using PerformanceMonitor.Alerting; +using PerformanceMonitor.Notifications; +using PerformanceMonitorLite.Services; +using Xunit; + +namespace Lite.Tests; + +/// +/// #3539 A8e, Lite's half: the Alert History grid row () colours itself by the +/// tier the alert FIRED at, read off the row's persisted context_json, and by its metric NAME only +/// when the row carries no tier. Lite's rows have always carried context_json (the in-app detail +/// dialog rehydrates it), so there is no DuckDB schema step here — the member the serializer now writes +/// is what the row reads. The shared decision and its fallback reasons are +/// 's, pinned in full on the Darling side +/// (AlertHistoryRowSeverityTests); this file pins that Lite's row reaches it. +/// +public sealed class AlertHistoryRowSeverityLiteTests +{ + private static string WithSeverity(AlertSeverityLevel? level) + { + var context = new AlertContext { SeverityOverride = level }; + context.Details.Add(new AlertDetailItem { Heading = "THREADPOOL", Fields = { ("Accumulated wait", "61 s") } }); + return AlertContextSerializer.Serialize(context); + } + + private static AlertHistoryRow Row(string metric, string? contextJson) => new() + { + AlertTime = new DateTime(2026, 9, 18, 12, 0, 0, DateTimeKind.Utc), + MetricName = metric, + CurrentValue = 61_000, + ThresholdValue = 60_000, + AlertSent = true, + NotificationType = AlertDelivery.ChannelWebhook, + ContextJson = contextJson, + }; + + /// A Poison Wait that fired WARNING (SQL Server grades it since #3539 A4) is an amber row, not + /// the red its name earned every row before; one that fired CRITICAL is red; a row written before the + /// tier was persisted keeps the name's red, which is faithful — every such SQL Server row was a + /// presence-flat critical fire. + [Fact] + public void ThePoisonWaitRow_RendersTheTierItFiredAt() + { + var warning = Row("Poison Wait", WithSeverity(AlertSeverityLevel.Warning)); + Assert.True(warning.IsWarning); + Assert.False(warning.IsCritical); + Assert.False(warning.IsResolved); + + var critical = Row("Poison Wait", WithSeverity(AlertSeverityLevel.Critical)); + Assert.True(critical.IsCritical); + Assert.False(critical.IsWarning); + + var legacy = Row("Poison Wait", null); + Assert.True(legacy.IsCritical); + Assert.False(legacy.IsWarning); + } + + /// The other direction: a CRITICAL-graded low-disk fire (#1136) was an amber row by name and is + /// red by tier; the presence-flat metrics carry no tier and keep the name's colour exactly. + [Fact] + public void GradedAboveTheName_IsRed_AndPresenceFlatIsUnchanged() + { + var lowDisk = Row("Volume Free Space", WithSeverity(AlertSeverityLevel.Critical)); + Assert.True(lowDisk.IsCritical); + Assert.False(lowDisk.IsWarning); + + var deadlocks = Row("Deadlocks Detected", null); + Assert.True(deadlocks.IsCritical); + var cpu = Row("High CPU", null); + Assert.True(cpu.IsWarning); + Assert.False(cpu.IsCritical); + + var cleared = Row("Poison Waits Cleared", null); + Assert.True(cleared.IsResolved); + Assert.False(cleared.IsCritical); + Assert.False(cleared.IsWarning); + } +} diff --git a/Lite/Mcp/McpAlertTools.cs b/Lite/Mcp/McpAlertTools.cs index 98a2fbe60..44f4b6f48 100644 --- a/Lite/Mcp/McpAlertTools.cs +++ b/Lite/Mcp/McpAlertTools.cs @@ -1,6 +1,7 @@ using System.ComponentModel; using System.Text.Json; using ModelContextProtocol.Server; +using PerformanceMonitor.Alerting; using PerformanceMonitor.Notifications; using PerformanceMonitorLite.Services; using PerformanceMonitor.Common; @@ -27,7 +28,7 @@ public sealed class McpAlertTools internal static string CpuModeFor(CpuAlertMode mode) => mode == CpuAlertMode.SqlOnly ? CpuModeSql : CpuModeTotal; - [McpServerTool(Name = "get_alert_history"), Description("Gets recent alert history from the alert log, NEWEST FIRST. Shows what alerts fired, when, and whether email was sent successfully. THE PAGE IS BOUNDED BY limit, NOT BY hours_back: alerts_returned is how many rows you got, truncated says the window held more than limit, and oldest_returned_alert_time / newest_returned_alert_time bound the page — under newest-first ordering the oldest stamp IS how far back this read reached. Raise limit or narrow hours_back when truncated is true; widening hours_back cannot help. BY DEFAULT THIS READ EXCLUDES DISMISSED ALERTS — rows an operator acknowledged in the Alerts History tab. Dismissal says nothing about whether the alert fired or mattered, so an incident reconstruction that ignores it can miss the very critical someone already looked at: dismissed_excluded says whether the filter applied and dismissed_excluded_count is how many rows in the window it removed, and include_dismissed = true returns them, each labelled dismissed = true. On this edition an alert that was dismissed AFTER aging into the parquet archive is removed by the archive view itself and can be neither returned nor counted here. notification_type is the delivery disposition and is the ONLY field that says why a row did not deliver: 'email'/'webhook'/'email+webhook' delivered on that channel; 'tray' is this instance's own balloon notification, which every non-muted alert gets, so it is what most rows read here and it does NOT report the email or webhook outcome; 'failed' means a channel was attempted and came back unsuccessful, with send_error carrying the first failing channel's text; 'muted' means a mute rule suppressed it; 'none' is a resolution row, which no channel applies to. Do NOT split the not-delivered rows on send_error: it is null whenever a cooldown or the per-metric repeat budget suppressed a send, and on every row written before those dispositions existed. 'throttled' and 'folded' are recorded by the headless service; on this instance the tray channel answers first, so a cooldown-suppressed or folded send is stored as 'tray'. 'undelivered' is a retained legacy value that means throttled OR folded OR failed with nothing in the row to say which.")] + [McpServerTool(Name = "get_alert_history"), Description("Gets recent alert history from the alert log, NEWEST FIRST. Shows what alerts fired, when, and whether email was sent successfully. THE PAGE IS BOUNDED BY limit, NOT BY hours_back: alerts_returned is how many rows you got, truncated says the window held more than limit, and oldest_returned_alert_time / newest_returned_alert_time bound the page — under newest-first ordering the oldest stamp IS how far back this read reached. Raise limit or narrow hours_back when truncated is true; widening hours_back cannot help. BY DEFAULT THIS READ EXCLUDES DISMISSED ALERTS — rows an operator acknowledged in the Alerts History tab. Dismissal says nothing about whether the alert fired or mattered, so an incident reconstruction that ignores it can miss the very critical someone already looked at: dismissed_excluded says whether the filter applied and dismissed_excluded_count is how many rows in the window it removed, and include_dismissed = true returns them, each labelled dismissed = true. On this edition an alert that was dismissed AFTER aging into the parquet archive is removed by the archive view itself and can be neither returned nor counted here. notification_type is the delivery disposition and is the ONLY field that says why a row did not deliver: 'email'/'webhook'/'email+webhook' delivered on that channel; 'tray' is this instance's own balloon notification, which every non-muted alert gets, so it is what most rows read here and it does NOT report the email or webhook outcome; 'failed' means a channel was attempted and came back unsuccessful, with send_error carrying the first failing channel's text; 'muted' means a mute rule suppressed it; 'none' is a resolution row, which no channel applies to. Do NOT split the not-delivered rows on send_error: it is null whenever a cooldown or the per-metric repeat budget suppressed a send, and on every row written before those dispositions existed. 'throttled' and 'folded' are recorded by the headless service; on this instance the tray channel answers first, so a cooldown-suppressed or folded send is stored as 'tray'. 'undelivered' is a retained legacy value that means throttled OR folded OR failed with nothing in the row to say which. severity is the row's tier — 'critical', 'warning', 'info' or 'resolution' — and severity_source says where it came from: 'fired' when the row persisted the tier the alert actually fired at (graded alerts such as Poison Wait, Volume Free Space and Database State fire Warning OR Critical by measurement), 'metric_name' when the row carries no tier and the metric's name is the only evidence (rows written before the tier was persisted, alerts whose severity is fixed per metric, and every resolution row). Do not infer a graded alert's tier from its name: a 'Poison Wait' row with severity 'warning' fired as a warning.")] public static async Task GetAlertHistory( LocalDataService dataService, [Description("Hours of history. Default 24.")] int hours_back = 24, @@ -72,22 +73,31 @@ the caller off widening a window whose contents they were never shown. */ : McpHelpers.Status("empty", "No alerts found in the specified time range."); } - var alerts = page.Select(r => new + var alerts = page.Select(r => { - alert_time = r.AlertTime.ToString("o"), - server_id = r.ServerId, - server_name = r.ServerName, - metric_name = r.MetricName, - current_value = r.CurrentValue, - threshold_value = r.ThresholdValue, - alert_sent = r.AlertSent, - notification_type = r.NotificationType, - send_error = r.SendError, - muted = r.Muted, - /* Per row, so a page that mixes the two populations labels each one. Always false on the - default read, which is a true statement about every row on it. */ - dismissed = r.Dismissed, - detail_text = r.DetailText + var (severity, severitySource) = AlertHistoryRowSeverity.Describe(r.MetricName, r.ContextJson); + return new + { + alert_time = r.AlertTime.ToString("o"), + server_id = r.ServerId, + server_name = r.ServerName, + metric_name = r.MetricName, + current_value = r.CurrentValue, + threshold_value = r.ThresholdValue, + alert_sent = r.AlertSent, + notification_type = r.NotificationType, + send_error = r.SendError, + muted = r.Muted, + /* Per row, so a page that mixes the two populations labels each one. Always false on the + default read, which is a true statement about every row on it. */ + dismissed = r.Dismissed, + /* #3539 A8e: the tier the alert FIRED at where the row persisted one ("fired"), else what + the metric NAME implies ("metric_name") — the Darling tool's twin fields, from the same + shared decision the Alerts History tab colours its rows by. */ + severity, + severity_source = severitySource, + detail_text = r.DetailText, + }; }).ToList(); return JsonSerializer.Serialize(new diff --git a/Lite/Services/LocalDataService.AlertHistory.cs b/Lite/Services/LocalDataService.AlertHistory.cs index 9f0c7706e..08f165588 100644 --- a/Lite/Services/LocalDataService.AlertHistory.cs +++ b/Lite/Services/LocalDataService.AlertHistory.cs @@ -10,6 +10,7 @@ using System.Collections.Generic; using System.Threading.Tasks; using DuckDB.NET.Data; +using PerformanceMonitor.Alerting; using PerformanceMonitor.Common; using PerformanceMonitor.Notifications; using PerformanceMonitorLite.Database; @@ -522,7 +523,12 @@ public class AlertHistoryRow AlertDeliveryStatus.Describe(AlertSent, NotificationType, SendError, producerHadTrayChannel: true); public bool IsResolved => AlertMetricClassifier.IsResolution(MetricName); - public bool IsCritical => AlertMetricClassifier.IsCritical(MetricName); - public bool IsWarning => AlertMetricClassifier.IsWarning(MetricName); + + /* #3539 A8e: the row's emphasis is the tier the alert FIRED at, read off the persisted context, and the + by-name classifier only for rows that carry none (written before the member existed, or fired with no + grade). The two arms live in AlertHistoryRowSeverity — one decision for this grid, the Darling + Viewer's, the web page and get_alert_history — and its summary is where the fallback's reasons are. */ + public bool IsCritical => AlertHistoryRowSeverity.IsCritical(MetricName, ContextJson); + public bool IsWarning => AlertHistoryRowSeverity.IsWarning(MetricName, ContextJson); } diff --git a/PerformanceMonitor.Alerting/AlertHistoryRowSeverity.cs b/PerformanceMonitor.Alerting/AlertHistoryRowSeverity.cs new file mode 100644 index 000000000..c8508042c --- /dev/null +++ b/PerformanceMonitor.Alerting/AlertHistoryRowSeverity.cs @@ -0,0 +1,128 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using PerformanceMonitor.Common; +using PerformanceMonitor.Notifications; + +namespace PerformanceMonitor.Alerting; + +/// +/// The one place an alert-history ROW's severity is decided for the surfaces that style or report it — +/// Lite's Alert History grid, the Darling Viewer's, the web alerts page and get_alert_history +/// (#3539 A8e): the tier the alert actually FIRED at where the row carries one, and the by-name +/// only where it does not. +/// +/// The defect. classifies by metric NAME — +/// "Poison Wait" is red — and that was the whole of what the grids knew. Since #2711 (PostgreSQL) and #3539 +/// A4 (SQL Server) Poison Wait is GRADED at its fire site, Warning at one task continuously stuck across the +/// ten-minute window and Critical at ten, so a Warning-graded fire arrived in both grids wearing the red the +/// name implies. The same gap ran the other way for every metric the engine grades above its name's colour: +/// a CRITICAL-graded low-disk fire (#1136), a SUSPECT database (Database State's CRITICAL arm), a forced-plan +/// failure at its CRITICAL grade — each rendered the amber of an ordinary warning. The channels never had +/// this problem, because they read at render time; the grids +/// read a row weeks later, and the row did not carry it. +/// +/// The fix is a member on the row, not a column. Both SKUs persist the alert's context as JSON +/// (context_json) through one serializer, and both deliverers fold the fire site's severity into that +/// context before serializing (#2090). So now carries the tier as a +/// trailing nullable member, and reads it back — no +/// DuckDB schema step for Lite, no PostgreSQL migration rung for Darling, and one write path covers every +/// graded metric on both engines rather than one per metric. +/// +/// Why the by-name fallback still exists, and what it is for. A row carries no tier when it was +/// written before the member existed, when the alert fired with no override (the per-metric map decided, +/// and the name IS the tier — Deadlocks Detected, High CPU, tempdb Space are presence-flat at their fire +/// sites today), or when it is a resolution row (persisted with a null context). For all of those the +/// name is the only evidence the row has, and the classifier's reading of it is the reading the channels +/// gave at the time — AlertSeverity.ForMetric's override-less arm, which is why its "Poison Wait" +/// arm stays CRITICAL: every SQL Server poison row written before #3539 A4 WAS a critical fire. The +/// fallback is therefore faithful for the rows that reach it, and it must not be "improved" by grading +/// old rows on their stored value: the value's unit changed under #3539 A4 and the bar it crossed is not +/// in the row. +/// +/// Resolution and informational rows are the name's business, unchanged. +/// is a suffix convention, not a tier, and a resolution +/// row carries no context to read; names the two +/// reports that fire with no override on purpose so the map's INFO arm decides. Neither is consulted +/// against a persisted tier because neither ever has one. +/// +public static class AlertHistoryRowSeverity +{ + /// + /// The tier the row FIRED at, or null when the row carries none — see the class summary for the + /// three populations that leaves. Exposed so a surface that wants to SAY where its severity came from + /// (get_alert_history's severity_source) can ask the same question the styling asks. + /// + public static AlertSeverityLevel? FiredAt(string? contextJson) => + AlertContextSerializer.TryReadSeverity(contextJson); + + /// + /// True when the grids should give this row the critical emphasis: the row fired Critical, or it carries + /// no tier and its NAME is one the classifier calls critical. A row that fired Warning is never critical + /// here, whatever its name says — that is the Poison Wait case this class exists for. + /// + public static bool IsCritical(string? metricName, string? contextJson) + { + if (AlertMetricClassifier.IsResolution(metricName)) + return false; + + return FiredAt(contextJson) is { } fired + ? fired == AlertSeverityLevel.Critical + : AlertMetricClassifier.IsCritical(metricName); + } + + /// + /// True for the ordinary actionable emphasis: the row fired Warning, or it carries no tier and its NAME + /// is neither a resolution, nor critical, nor one of the deliberate INFO reports. Exactly one of + /// / / + /// is true for a row that carries a tier; for one that does not, the classifier's own partition holds + /// (which leaves an informational row with none of the three, on purpose). + /// + public static bool IsWarning(string? metricName, string? contextJson) + { + if (AlertMetricClassifier.IsResolution(metricName)) + return false; + + return FiredAt(contextJson) is { } fired + ? fired == AlertSeverityLevel.Warning + : AlertMetricClassifier.IsWarning(metricName); + } + + /// The source word for a tier read off the row itself. + public const string SourceFired = "fired"; + + /// The source word for a tier implied by the metric name alone. + public const string SourceMetricName = "metric_name"; + + /// + /// The row's severity as two words for a wire consumer (get_alert_history, and the web alerts + /// page through it): the tier — critical, warning, info or resolution — and + /// where it came from, when the row persisted the tier the alert fired at and + /// when the name was all there was. The source is published rather than + /// folded away because the two are not equally strong evidence: a "critical" read off the row is what + /// the operator was paged with; a "critical" read off the name is what the map says about the name. + /// + public static (string Severity, string Source) Describe(string? metricName, string? contextJson) + { + if (AlertMetricClassifier.IsResolution(metricName)) + return ("resolution", SourceMetricName); + + if (FiredAt(contextJson) is { } fired) + { + return (fired == AlertSeverityLevel.Critical ? "critical" : "warning", SourceFired); + } + + if (AlertMetricClassifier.IsCritical(metricName)) + return ("critical", SourceMetricName); + + if (AlertMetricClassifier.IsInformational(metricName)) + return ("info", SourceMetricName); + + return ("warning", SourceMetricName); + } +} diff --git a/PerformanceMonitor.Common/AlertMetricClassifier.cs b/PerformanceMonitor.Common/AlertMetricClassifier.cs index 065869691..0dc1bed5d 100644 --- a/PerformanceMonitor.Common/AlertMetricClassifier.cs +++ b/PerformanceMonitor.Common/AlertMetricClassifier.cs @@ -16,6 +16,14 @@ namespace PerformanceMonitor.Common /// ) are used by ALL THREE Alert History grids — Lite's, the Darling /// Viewer's and the deprecated Dashboard's — and by the Dashboard sidebar's Alert badge count. /// + /// Since #3539 A8e the two SKU grids reach / + /// only as a FALLBACK, through AlertHistoryRowSeverity (Alerting): a row that persisted + /// the tier the alert fired at (the serializer's Severity member) is styled by that tier, and + /// the name decides only for rows that carry none — written before the member existed, fired with no + /// runtime grade, or a resolution. stays the primary arm it always was: a + /// resolution is a naming convention, not a tier. The Dashboard still calls the name predicates + /// directly; its own engine fires presence-flat, so for its rows the name IS the tier. + /// /// Alert classification across this codebase is metric-name based — there is no structural "kind" /// field on a row — so this centralizes a string convention that was previously duplicated inline /// in Dashboard's AlertsHistoryContent and Lite's AlertHistoryRow, and had drifted: both copies @@ -79,6 +87,12 @@ public static bool IsResolution(string? metricName) /// /// True when the metric name denotes a critical-severity alert (deadlock or poison wait), /// used for row emphasis in the history grids. Mirrors the long-standing inline convention. + /// + /// "Poison" stays here although Poison Wait is GRADED at both engines' fire sites now (#2711, + /// #3539 A4), because this predicate is what a row with NO persisted tier is styled by, and every + /// SQL Server Poison Wait row written before #3539 A4 was a presence-flat CRITICAL fire — so red is + /// the faithful replay for exactly the rows that still reach this by name. A graded row never + /// does: AlertHistoryRowSeverity reads its tier first (#3539 A8e). /// public static bool IsCritical(string? metricName) { diff --git a/PerformanceMonitor.Common/ServerHealthBands.cs b/PerformanceMonitor.Common/ServerHealthBands.cs index 0ea1b927f..fa69c46af 100644 --- a/PerformanceMonitor.Common/ServerHealthBands.cs +++ b/PerformanceMonitor.Common/ServerHealthBands.cs @@ -600,7 +600,8 @@ public readonly record struct ServerHealthMetrics /// turns into a share. Zero means no denominator was declared; /// then bands a non-zero failing count /// Warning and never Critical, the fail-away-from-Healthy reading every undeclared denominator - /// here takes. + /// here takes — and a ZERO failing count Unknown rather than Healthy (#3539 A6), because with no + /// collector banded there was nothing that could have failed. /// public int CollectorCount { get; init; } } @@ -999,8 +1000,9 @@ public static HealthSeverity ThreadsSeverity(int? totalThreads, int? availableTh collectorCount > 0 ? failedCollectorCount * 100.0 / collectorCount : null; /// - /// Collectors band (#3539 A8d): Healthy with nothing FAILING; otherwise Warning, escalating to - /// Critical when the FAILING share of this server's banded collectors exceeds + /// Collectors band (#3539 A8d): Unknown when NO collector has been banded for this server at all; + /// Healthy with collectors banded and nothing FAILING; otherwise Warning, escalating to Critical + /// when the FAILING share of this server's banded collectors exceeds /// . /// /// Graded on a share, because a count of failing collectors was presence-flat. The arm @@ -1023,9 +1025,27 @@ public static HealthSeverity ThreadsSeverity(int? totalThreads, int? availableTh /// gap in what this server's other bands can see, and the count is disclosed beside the band, so /// the Healthy arm stays reserved for nothing failing. With no denominator declared the share cannot /// be formed and a non-zero count fails away from Healthy into Warning, never into Critical — the - /// unrateable-window discipline the rate bands above follow. Nothing failing bands Healthy whatever - /// the denominator, as before: the counts come off a seven-day aggregate that only lacks rows for a - /// server that has collected nothing in a week, which the freshness axis already paints. + /// unrateable-window discipline the rate bands above follow. + /// + /// Zero banded collectors is Unknown, not Healthy (#3539, A6's sibling). The arm this + /// replaces read failing == 0 → Healthy whatever the denominator, on the argument that the + /// counts come off a seven-day aggregate that only lacks rows for a server that has collected nothing + /// in a week, which the freshness axis already paints. That argument covers OFFLINE. It does not + /// cover a server registered and reachable whose first collection has not landed in the aggregate + /// yet, or a store whose collector-health read returned no rows for it: both handed this band + /// (0, 0) and got a green dot for a collection nobody had banded — the same positive claim of + /// health for an unmeasured metric that 's null arm exists to refuse, + /// one row down the card. "Nothing failing" is only a health claim when there was something that + /// could have failed; with no collector banded there was not, and Unknown is the honest reading. + /// The viewer's offline arm and the web's is_online === false chip stay where they are: they + /// paint a KNOWN-dark server, and this arm paints one nothing has banded yet, which are different + /// facts on different axes. + /// + /// Unknown here is rank-neutral like every other Unknown on the card: the worst-of fold and the + /// fleet score skip it, and stops counting the collectors row as + /// measured, so a card whose ONLY reading used to be this green dot now reads "0 of 6 measured" and + /// bands through 's nothing-measured arm rather than as Healthy. + /// A card with any collector banded is unchanged. /// /// Collectors whose seven-day band is FAILING. /// Collectors banded at all in the same window (every band). Required @@ -1035,7 +1055,10 @@ public static HealthSeverity CollectorSeverity(int failedCollectorCount, int col { if (failedCollectorCount <= 0) { - return HealthSeverity.Healthy; + /* Nothing failing is Healthy only when something was banded; (0, 0) is a collection nobody + has classified, not a clean one. A non-zero failing count with no denominator still falls + through to the Warning arm below: a failure was observed even if the population was not. */ + return collectorCount > 0 ? HealthSeverity.Healthy : HealthSeverity.Unknown; } var share = FailingCollectorSharePercent(failedCollectorCount, collectorCount); @@ -1060,11 +1083,28 @@ public static IEnumerable MetricSeverities(ServerHealthMetrics m /// /// The card's worst metric band (offline handled separately by the border / overlay). Unknown and Healthy - /// never escalate — matching ServerHealthStatus.OverallSeverity's reduce. + /// never escalate — matching ServerHealthStatus.OverallSeverity's reduce — and a card on which + /// NO metric was measured folds to , not Healthy. + /// + /// The nothing-measured arm (#3539 A6). The fold skips Unknown so that an unmeasured + /// metric can never escalate a card, and that is still right: a partially-measured card bands on + /// what WAS read and lets every label say how much that was + /// ("Healthy — 1 of 6 measured", #3528). But a fold over six Unknowns has nothing to fold, and + /// answering Healthy from it was a positive claim about a server on which not one reading had been + /// taken — an online server whose collectors had not yet been banded counted in the fleet's healthy + /// mass and drew a green card, with only a "0 of 6 measured" qualifier to say the green was empty. + /// Unknown is the reading the fold actually has, and gives it the band + /// the product already gives a server nothing has measured yet: the awaiting-first-collection + /// Warning. + /// + /// This does not move a partially-measured card. One measured Healthy metric among five + /// Unknowns still folds to Healthy, exactly as before, so the #3528 rank-neutrality of Unknown holds + /// wherever there is anything measured to be neutral against; the only cards that move are the ones + /// on which there was nothing. /// public static HealthSeverity OverallMetricSeverity(in ServerHealthMetrics m) { - var worst = HealthSeverity.Healthy; + var worst = HealthSeverity.Unknown; foreach (var s in MetricSeverities(m)) { if (s == HealthSeverity.Critical) @@ -1076,6 +1116,12 @@ public static HealthSeverity OverallMetricSeverity(in ServerHealthMetrics m) { worst = HealthSeverity.Warning; } + else if (s == HealthSeverity.Healthy && worst == HealthSeverity.Unknown) + { + /* The first measured reading lifts the fold off Unknown; a Warning already found keeps + its place, because Healthy never de-escalates. */ + worst = HealthSeverity.Healthy; + } } return worst; @@ -1108,7 +1154,21 @@ public static (int Measured, int Total) MeasuredMetricCounts(in ServerHealthMetr /// /// Collapses a server's health to one fleet band, mirroring the card border: offline collection -> Offline; /// a never-collected (queued-during-bootstrap) server -> Warning (attention-worthy but not the red overlay); - /// else the card's worst metric band, with a stale collection also Warning. + /// else the card's worst metric band, with a stale collection also Warning, and a card on which nothing + /// was measured ( overall) Warning for the same reason the + /// never-collected server is. + /// + /// Why Unknown overall is Warning and not a band of its own (#3539 A6). An online server + /// with no metric measured is in the same epistemic state as one awaiting its first collection — the + /// product knows nothing about its health — and that state already has a band here: Warning, + /// "attention-worthy but not the red overlay", chosen after a 24-server bootstrap incident put a + /// never-collected fleet under the red Offline overlay. A fifth member + /// would have been the alternative, and was rejected: it is a wire-visible enum on /api/fleet + /// and get_fleet_overview, every band consumer (the web tiles, the viewer's brushes, the + /// worst-first score's rank steps, the sweep's counts) would need an arm, and the reading it would + /// give — "this needs a look" — is the one Warning already gives. What this arm changes is that such + /// a server leaves the healthy mass: healthy_count no longer counts it, and it appears in the + /// worst-first ranking with a reason that says nothing was measured. /// public static FleetHealthBand ClassifyBand(bool? isOnline, bool awaitingFirstCollection, bool collectionStale, HealthSeverity overallMetricSeverity) { @@ -1126,6 +1186,7 @@ public static FleetHealthBand ClassifyBand(bool? isOnline, bool awaitingFirstCol { HealthSeverity.Critical => FleetHealthBand.Critical, HealthSeverity.Warning => FleetHealthBand.Warning, + HealthSeverity.Unknown => FleetHealthBand.Warning, _ => collectionStale ? FleetHealthBand.Warning : FleetHealthBand.Healthy, }; } diff --git a/PerformanceMonitor.Notifications/AlertContext.cs b/PerformanceMonitor.Notifications/AlertContext.cs index b95f292d6..345aed411 100644 --- a/PerformanceMonitor.Notifications/AlertContext.cs +++ b/PerformanceMonitor.Notifications/AlertContext.cs @@ -9,6 +9,7 @@ using System; using System.Collections.Generic; using System.Text.Json; +using System.Text.Json.Serialization; using PerformanceMonitor.Analysis; namespace PerformanceMonitor.Notifications; @@ -27,9 +28,18 @@ public class AlertContext /// Forces the rendered severity tier (email badge/color, Teams/Slack accent) regardless of /// metric name, for metrics graded at runtime — low-disk fires WARNING normally and CRITICAL /// when critically low (#1136). null = use the per-metric - /// map. Deliberately not persisted (like ): it drives the live - /// email/webhook render only, and the alert-history UI does not re-derive severity, so the - /// JSON projection () need not carry it. + /// map. + /// + /// Persisted since #3539 A8e, as the trailing Severity member of the JSON projection + /// (), so the alert-history row carries the tier the alert actually + /// FIRED at. It was deliberately not persisted before, on the argument that it drove the live + /// email/webhook render only and the alert-history UI did not re-derive severity — which was true, and + /// was the defect: the history grids styled a row by its metric NAME + /// (AlertMetricClassifier.IsCritical), so once Poison Wait became graded (#2711 on + /// PostgreSQL, #3539 A4 on SQL Server) a WARNING-graded fire rendered red in every grid, and a + /// CRITICAL-graded low-disk fire rendered amber. AlertHistoryRowSeverity (Alerting) is the read side: + /// the row's own tier where one was persisted, the by-name classifier for rows that predate the member. + /// stays unpersisted for its own reason (no dialog surface, and size). /// public AlertSeverityLevel? SeverityOverride { get; set; } @@ -210,8 +220,18 @@ public class AlertDetailItem /// persisted context survives the round-trip into the in-app dialog. /// / /// are deliberately not persisted (the dialog has no attachment surface). +/// +/// Severity (#3539 A8e) is the tier the alert fired at — +/// — trailing and nullable like Incidents, so a row written before it existed rehydrates to null, +/// which reads as "this row carries no tier" and sends the grids to the by-name fallback. Serialized as the +/// enum's NAME rather than its ordinal: the column outlives any build, and a persisted 1 would change +/// meaning the day a member is inserted ahead of it, where "Critical" cannot. +/// /// -public record AlertContextDto(List Details, List? Incidents = null); +public record AlertContextDto( + List Details, + List? Incidents = null, + [property: JsonConverter(typeof(JsonStringEnumConverter))] AlertSeverityLevel? Severity = null); public record AlertDetailItemDto(string Heading, List Fields, string? Body, bool IsCodeBlock, RemediationActionDto? Remediation = null); public record FieldDto(string Label, string Value); @@ -467,10 +487,51 @@ public static string Serialize(AlertContext context) d.Body, d.IsCodeBlock, ToDto(d.Remediation))), - ToDto(context.Incidents)); + ToDto(context.Incidents), + /* #3539 A8e: the tier the alert fired at rides the row. Both SKUs' deliverers fold + AlertOutcome.Severity into this property before serializing (#2090), so a graded fire on + either engine persists its grade here with no store change on either side. */ + context.SeverityOverride); return JsonSerializer.Serialize(dto); } + /// + /// The tier a persisted alert-history row FIRED at (#3539 A8e), or null when the row carries none + /// — written before the member existed, an alert that fired with no override (so the per-metric map + /// decided), a resolution row (persisted with a null context), or unparseable JSON. Reads the one + /// property rather than rehydrating the whole context: the grids call this once per row, and a Details + /// list with remediation actions is the expensive part of a row it does not need. + /// + public static AlertSeverityLevel? TryReadSeverity(string? contextJson) + { + if (string.IsNullOrWhiteSpace(contextJson)) + return null; + + try + { + using var document = JsonDocument.Parse(contextJson); + if (document.RootElement.ValueKind != JsonValueKind.Object + || !document.RootElement.TryGetProperty(nameof(AlertContextDto.Severity), out var severity) + || severity.ValueKind != JsonValueKind.String) + { + return null; + } + + /* Name match only, exact: the writer emits the enum's name, and Enum.TryParse on its own would + also accept a bare digit ("1"), which is the ordinal coupling the string form exists to avoid + — so the parsed member must spell itself back to the stored text. */ + var text = severity.GetString(); + return Enum.TryParse(text, ignoreCase: false, out var level) + && string.Equals(Enum.GetName(level), text, StringComparison.Ordinal) + ? level + : null; + } + catch (JsonException) + { + return null; + } + } + /// /// #2302: just the incidents array, for the generic webhook's {{incidents_json}} token — /// the SAME projection embeds in the @@ -544,6 +605,10 @@ public static bool TryDeserialize(string? json, out AlertContext context) if (dto?.Details is null) return false; + /* #3539 A8e: the tier the alert fired at. A row written before the member existed rehydrates it + null, which is the same state a fire with no override left it in. */ + context.SeverityOverride = dto.Severity; + foreach (var d in dto.Details) { var item = new AlertDetailItem From 2a1e367fecdf1bef0414cf11a57db36bfa19a787 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 15:19:57 -0400 Subject: [PATCH 56/69] Every delta family now stores the interval its deltas accrued over, and query_stats stores the statement offsets its delta key is made of, so no restart zero reads as a measurement anywhere and the query seed can finally find its keys (#3540, V128 / Lite v61) (#3630) * Every delta family stores the interval its deltas accrued over, and query_stats stores the statement offsets its delta key is made of (#3540, V128 / Lite v61) The completion of V127: sample_interval_seconds on procedure_stats, memory_grant_stats, pg_wait_stats and pg_statement_stats, written as the minimum over each row's delta groups; statement_start_offset / statement_end_offset on query_stats, stored raw (-1 included) so both hosts' restart seeds rebuild the collector's delta key from the store. The census's still-naked list is empty; the seeding census's pass-window-only set is empty. Readers of the four families that divide by time take the stored interval first, LAG only for pre-V128 rows, and never ELSE 0. get_resource_semaphore emits sample_interval_seconds again on both hosts. * CI round 1: the PostgreSQL pair's CREATE TABLE rungs carry the interval for the fresh population (the V101 two-places rule); the resolving-view definer literal moves to 128; the backlog fallback pin records why it stays hash-keyed; the pre-V128 first snapshot is absent in the trend read test; two of my own pins fixed (V4 passthrough skipped, CTE-first FROM) * CI round 2: the two Lite procedure-trend tool tests expect the pre-v61 first snapshot to be absent, the same correction v60 made for the wait trends --- .../Darling.Tests/DarlingDeltaSeederTests.cs | 97 ++++- .../DarlingMcpMemoryGrantToolsTests.cs | 53 ++- .../DarlingMcpTrendToolsTests.cs | 33 ++ .../DarlingPerformanceTrendsReadTests.cs | 18 +- .../DeltaFamilyIntervalColumnsRungTests.cs | 52 ++- ...milyIntervalCompletionLivePostgresTests.cs | 233 ++++++++++ .../DeltaFamilyIntervalCompletionRungTests.cs | 407 ++++++++++++++++++ .../Darling.Tests/OversizedPlanBacklogPins.cs | 52 ++- .../Darling.Tests/PayloadDimensionTests.cs | 10 +- Darling/Darling.Tests/PgTrendReaderTests.cs | 30 ++ .../StartupCommandTimeoutTests.cs | 2 +- .../Darling.Tests/ViewerHistoryWindowTests.cs | 9 +- .../Darling.Tests/ViewerQueriesRestTests.cs | 27 ++ .../DarlingDeltaCalculator.cs | 91 ++-- .../Mcp/DarlingMcpMemoryGrantTools.cs | 11 +- .../Mcp/DarlingMemoryGrantReader.cs | 31 +- .../Mcp/DarlingTrendReader.cs | 26 +- .../DarlingPgTrendReader.cs | 31 +- .../OversizedPlanBacklog.cs | 17 +- .../PgMigrations.cs | 115 ++++- .../StorageVersion.cs | 2 +- .../ViewerDataService.ItemHistory.cs | 8 +- .../ViewerDataService.QueryTrends.cs | 32 +- .../ViewerDataService.cs | 41 +- Darling/README.md | 4 +- Lite.Tests/DeltaFamilyIntervalColumnTests.cs | 119 +++-- Lite.Tests/DeltaFamilySeedingCensusTests.cs | 61 ++- .../DeltaFamilyUnknowableRowReadTests.cs | 126 ++++++ Lite.Tests/DuckDbSchemaEquivalenceTests.cs | 15 + Lite.Tests/LiteDeltaSeederTests.cs | 125 +++++- .../MemoryGrantsCollectorDefinitionTests.cs | 50 ++- Lite.Tests/PerformanceTrendsToolTests.cs | 21 +- Lite.Tests/PgStatementStatsDeltaSkipTests.cs | 10 +- Lite.Tests/PgStatementStatsFlavorTests.cs | 4 +- .../PgWaitStatsCollectorDefinitionTests.cs | 42 +- Lite.Tests/PgWaitStatsDeltaSkipTests.cs | 10 +- .../ProcedureStatsCollectorDefinitionTests.cs | 64 ++- .../QueryStatsCollectorDefinitionTests.cs | 29 +- Lite/Database/DuckDbInitializer.cs | 64 ++- Lite/Mcp/McpMemoryTools.cs | 11 +- Lite/Services/DeltaCalculator.cs | 91 ++-- .../Services/LocalDataService.MemoryGrants.cs | 17 +- Lite/Services/LocalDataService.QueryStats.cs | 31 +- .../CollectorDeltaCalculator.cs | 46 +- .../MemoryGrantsCollector.cs | 31 +- .../PgStatementStatsCollector.cs | 35 +- .../PgWaitStatsCollector.cs | 29 +- .../ProcedureStatsCollector.cs | 43 +- .../QueryStatsCollector.cs | 34 +- 49 files changed, 2224 insertions(+), 316 deletions(-) create mode 100644 Darling/Darling.Tests/DeltaFamilyIntervalCompletionLivePostgresTests.cs create mode 100644 Darling/Darling.Tests/DeltaFamilyIntervalCompletionRungTests.cs diff --git a/Darling/Darling.Tests/DarlingDeltaSeederTests.cs b/Darling/Darling.Tests/DarlingDeltaSeederTests.cs index 37e25731e..8343383c6 100644 --- a/Darling/Darling.Tests/DarlingDeltaSeederTests.cs +++ b/Darling/Darling.Tests/DarlingDeltaSeederTests.cs @@ -28,8 +28,9 @@ namespace Darling.Tests; /// the per-group pass window the #2235 series-age rescue reads. Two query shapes are pinned here: the /// original latest-collection-per-server row-value probe (now also latch_stats and spinlock_stats), and /// the latest-row-per-key DISTINCT ON form for the families whose collectors do not write every -/// key every pass (procedure_stats, pg_wait_stats, pg_statement_stats). query_stats has a pass-window -/// seed only; the reason is in the calculator's header and the census in Lite.Tests names it. +/// key every pass (procedure_stats, pg_wait_stats, pg_statement_stats — and query_stats since V128, #3540, +/// once the store persisted the two statement offsets its key is made of; a pre-V128 row seeds the pass +/// window and no key). The census in Lite.Tests holds the set. /// /* Live-fixture tests share one Postgres store; the collection serializes them so cross-test row churn (inserts/purges/deletes) cannot race another class's assertions. */ @@ -168,21 +169,27 @@ public void SeedSql_PostgresPair_PartitionByTheCollectorsKeyAndTakeTheLatestRowP } /// - /// query_stats seeds its PASS WINDOW only — the store persists neither statement offset the delta key - /// carries, so no row can reproduce the key. The read is the distinct collection times per server - /// inside the cutoff, and nothing else: a key seed here would be a claim the store cannot back. + /// query_stats is key-seeded since V128 (#3540): the store persists both statement offsets now, so + /// the read partitions by the collector's FULL key — sql_handle, both offsets, plan_handle — and + /// returns each key's latest row inside the window, with the eight counters the collector's eight + /// series difference. The offsets are selected RAW (no COALESCE, no arithmetic) because the seeder + /// rebuilds the key from them with the collector's own interpolation, and there is no offset filter + /// in the SQL: a pre-V128 row (NULL offsets) is read for the pass window and skipped for keys in C#, + /// so one read serves both halves. /// [Fact] - public void SeedSql_QueryStats_IsThePassWindowOnly() + public void SeedSql_QueryStats_PartitionsByTheFullDeltaKeyAndSelectsTheOffsetsRaw() { - var sql = DarlingDeltaCalculator.QueryStatsPassSeedSql; - Assert.Contains("SELECT server_id, collection_time", sql, StringComparison.Ordinal); + var sql = DarlingDeltaCalculator.QueryStatsSeedSql; + Assert.Contains("SELECT DISTINCT ON (server_id, sql_handle, statement_start_offset, statement_end_offset, plan_handle)", sql, StringComparison.Ordinal); + Assert.Contains("server_id, sql_handle, statement_start_offset, statement_end_offset, plan_handle,", sql, StringComparison.Ordinal); + Assert.Contains("execution_count, total_worker_time, total_elapsed_time,", sql, StringComparison.Ordinal); + Assert.Contains("total_logical_reads, total_logical_writes, total_physical_reads, total_rows, total_spills,", sql, StringComparison.Ordinal); Assert.Contains("FROM query_stats", sql, StringComparison.Ordinal); - Assert.Contains("GROUP BY server_id, collection_time", sql, StringComparison.Ordinal); - Assert.Equal(1, CountOccurrences(sql, "collection_time >= $1")); - Assert.DoesNotContain("sql_handle", sql, StringComparison.Ordinal); - Assert.DoesNotContain("plan_handle", sql, StringComparison.Ordinal); - Assert.DoesNotContain("execution_count", sql, StringComparison.Ordinal); + Assert.Contains("ORDER BY server_id, sql_handle, statement_start_offset, statement_end_offset, plan_handle, collection_time DESC", sql, StringComparison.Ordinal); + Assert.DoesNotContain("IS NOT NULL", sql, StringComparison.Ordinal); + Assert.DoesNotContain("COALESCE", sql, StringComparison.Ordinal); + Assert.DoesNotContain("GROUP BY", sql, StringComparison.Ordinal); } public static TheoryData SeedQueries() => new() @@ -195,10 +202,11 @@ public void SeedSql_QueryStats_IsThePassWindowOnly() { DarlingDeltaCalculator.SpinlockStatsSeedSql, "spinlock_stats" }, }; - /// The latest-row-per-key shape (#3540 A4): one table read, one bound, DISTINCT ON. + /// The latest-row-per-key shape (#3540 A4, query_stats since V128): one table read, one bound, DISTINCT ON. public static TheoryData PerKeySeedQueries() => new() { { DarlingDeltaCalculator.ProcedureStatsSeedSql, "procedure_stats" }, + { DarlingDeltaCalculator.QueryStatsSeedSql, "query_stats" }, { DarlingDeltaCalculator.PgWaitStatsSeedSql, "pg_wait_stats" }, { DarlingDeltaCalculator.PgStatementStatsSeedSql, "pg_statement_stats" }, }; @@ -435,7 +443,8 @@ public async Task EverySeedQuery_RunsAgainstTheRealSchema_AgainstDevPostgres() (DarlingDeltaCalculator.LatchStatsSeedSql, 6), (DarlingDeltaCalculator.SpinlockStatsSeedSql, 7), (DarlingDeltaCalculator.ProcedureStatsSeedSql, 10), - (DarlingDeltaCalculator.QueryStatsPassSeedSql, 2), + /* #3540 V128: 5 key parts + 8 counters + collection_time */ + (DarlingDeltaCalculator.QueryStatsSeedSql, 14), (DarlingDeltaCalculator.PgWaitStatsSeedSql, 5), (DarlingDeltaCalculator.PgStatementStatsSeedSql, 9), }; @@ -500,11 +509,18 @@ public async Task EndToEnd_EveryFamilyAndThePassWindow_SeedFromStore_AgainstDevP await ProcAsync(connection, latest, null, "db", "dbo", "proc3", 3); await ProcAsync(connection, stale, "0x02", "db", "dbo", "p2", 1); - /* query_stats: three passes, one stale — the pass window must come from the two inside. */ + /* query_stats: three PRE-V128 passes (no offsets), one stale — the pass window must come from the + two inside; plus V128 rows carrying the offsets (#3540): the whole-batch statement (0, -1) in + both passes, a second statement of the same batch/plan (100, 240) only in the older pass, and a + null-handle row. */ foreach (var t in new[] { stale, older, latest }) { await QueryStatsAsync(connection, t); } + await KeyedQueryStatsAsync(connection, older, "0xSH1", 0, -1, "0xPH1", 10); + await KeyedQueryStatsAsync(connection, latest, "0xSH1", 0, -1, "0xPH1", 15); + await KeyedQueryStatsAsync(connection, older, "0xSH1", 100, 240, "0xPH1", 7); + await KeyedQueryStatsAsync(connection, latest, null, 0, -1, null, 3); /* pg_wait_stats: 1001 both passes; 1002 idle at the latest pass, so its newest row is the older. */ await PgWaitAsync(connection, older, 1001, 5, 500); @@ -560,10 +576,25 @@ public async Task EndToEnd_EveryFamilyAndThePassWindow_SeedFromStore_AgainstDevP Assert.Equal(1, deltas.CalculateDelta(TestServerId, "pg_statement_stats_rows", "11|16384|10|1", 61, pass, Gap)); Assert.Equal(1, deltas.CalculateDelta(TestServerId, "pg_statement_stats_calls", "12|16384|10|0", 8, pass, Gap)); - /* The series-age rescue on the FIRST post-restart pass: query_stats has no key seed, but its pass - window is seeded from the table, so a plan compiled 30 s ago (inside the ~120 s since the last - pre-restart pass) is credited in full with a real interval, while a plan older than that gap - baselines honestly. Unseeded, both are (0, 0) — the defect this lane closes. */ + /* query_stats KEYS (#3540, V128): the latest pass is the baseline for the key both passes wrote, + spelled with the raw -1 as the collector spells it; the statement that fell out of the TOP (n) + is restored from its OLDER row over ~240 s; a null handle formats as empty on both sides; and + the pre-V128 rows seeded NOTHING — not even under a normalizing guess (gap policy off, so only a + first sighting reads 0; a seeded baseline of 1 would return 4). */ + Assert.Equal(3, deltas.CalculateDeltaWithInterval(TestServerId, "query_stats_exec", "0xSH1:0:-1:0xPH1", 18, out var keyedInterval, pass, Gap)); + Assert.InRange(keyedInterval, 118, 122); + Assert.Equal(210, deltas.CalculateDelta(TestServerId, "query_stats_spills", "0xSH1:0:-1:0xPH1", 1260, pass, Gap)); + Assert.Equal(2, deltas.CalculateDeltaWithInterval(TestServerId, "query_stats_exec", "0xSH1:100:240:0xPH1", 9, out var fellOutInterval, pass, Gap)); + Assert.InRange(fellOutInterval, 238, 242); + Assert.Equal(2, deltas.CalculateDelta(TestServerId, "query_stats_exec", ":0:-1:", 5, pass, Gap)); + Assert.Equal(0, deltas.CalculateDelta(TestServerId, "query_stats_exec", "sh:0:0:ph", 5, pass, 0)); + Assert.Equal(0, deltas.CalculateDelta(TestServerId, "query_stats_exec", "sh:0:-1:ph", 5, pass, 0)); + + /* The series-age rescue on the FIRST post-restart pass: the pass window is seeded from EVERY + query_stats row, the pre-V128 ones included, so a plan compiled 30 s ago (inside the ~120 s + since the last pre-restart pass) is credited in full with a real interval, while a plan older + than that gap baselines honestly. Unseeded, both are (0, 0) — the defect #3614 closed and V128 + must not reopen on the first restart after the upgrade, when the window holds only such rows. */ Assert.Equal(900, deltas.CalculateDeltaWithSeriesAge(TestServerId, "query_stats_worker", "sh:0:99:newplan", 900, 30, out var rescueInterval, pass, Gap)); Assert.InRange(rescueInterval, 118, 122); Assert.Equal(0, deltas.CalculateDeltaWithSeriesAge(TestServerId, "query_stats_exec", "sh:0:99:oldplan", 900, 3_000, out var oldInterval, pass, Gap)); @@ -710,6 +741,32 @@ private static async Task QueryStatsAsync(NpgsqlConnection connection, DateTime await cmd.ExecuteNonQueryAsync(TestContext.Current.CancellationToken); } + /// A V128 query_stats row (#3540): both offsets stored raw; the eight counters are multiples of the + /// execution count so every group's expected delta is derivable. + private static async Task KeyedQueryStatsAsync(NpgsqlConnection connection, DateTime t, string? sqlHandle, int start, int end, string? planHandle, long executions) + { + using var cmd = new NpgsqlCommand( + "INSERT INTO query_stats (collection_id, collection_time, server_id, server_name, query_hash, sql_handle, plan_handle, " + + "statement_start_offset, statement_end_offset, " + + "execution_count, total_worker_time, total_elapsed_time, total_logical_reads, total_logical_writes, total_physical_reads, total_rows, total_spills) " + + "VALUES (1, $1, $2, 'delta-seed-e2e', 'qh', $3, $4, $5, $6, $7, $8, $9, $10, $11, $12, $13, $14)", connection); + cmd.Parameters.AddWithValue(t); + cmd.Parameters.AddWithValue(TestServerId); + cmd.Parameters.AddWithValue((object?)sqlHandle ?? DBNull.Value); + cmd.Parameters.AddWithValue((object?)planHandle ?? DBNull.Value); + cmd.Parameters.AddWithValue(start); + cmd.Parameters.AddWithValue(end); + cmd.Parameters.AddWithValue(executions); + cmd.Parameters.AddWithValue(executions * 10); + cmd.Parameters.AddWithValue(executions * 20); + cmd.Parameters.AddWithValue(executions * 30); + cmd.Parameters.AddWithValue(executions * 40); + cmd.Parameters.AddWithValue(executions * 50); + cmd.Parameters.AddWithValue(executions * 60); + cmd.Parameters.AddWithValue(executions * 70); + await cmd.ExecuteNonQueryAsync(TestContext.Current.CancellationToken); + } + private static async Task PgWaitAsync(NpgsqlConnection connection, DateTime t, long eventId, long waits, long waitTimeUs) { using var cmd = new NpgsqlCommand( diff --git a/Darling/Darling.Tests/DarlingMcpMemoryGrantToolsTests.cs b/Darling/Darling.Tests/DarlingMcpMemoryGrantToolsTests.cs index a377a9995..f125184e9 100644 --- a/Darling/Darling.Tests/DarlingMcpMemoryGrantToolsTests.cs +++ b/Darling/Darling.Tests/DarlingMcpMemoryGrantToolsTests.cs @@ -11,6 +11,7 @@ using System.ComponentModel; using System.Linq; using System.Reflection; +using System.Text.Json; using System.Threading.Tasks; using Microsoft.Extensions.DependencyInjection; using ModelContextProtocol.Server; @@ -88,6 +89,31 @@ public void ResourceSemaphoreLatestSql_LatestSnapshot_CarriesCeilingColumns() Assert.Contains("resource_semaphore_id", sql, StringComparison.Ordinal); Assert.Contains("timeout_error_count_delta", sql, StringComparison.Ordinal); Assert.Contains("forced_grant_count_delta", sql, StringComparison.Ordinal); + /* #3540 (V128): the Dashboard's sample_interval_seconds is back — the collector stores it now — and + it rides LAST so every ordinal the reader indexes is unchanged. */ + var interval = sql.IndexOf("sample_interval_seconds", StringComparison.Ordinal); + Assert.True(interval > sql.IndexOf("forced_grant_count_delta,", StringComparison.Ordinal), "the interval must be selected after the last pre-V128 column"); + /* LastIndexOf: the CTE's MAX(collection_time) probe reads the view first; the select list sits ahead + of the SECOND FROM. */ + Assert.True(interval < sql.LastIndexOf("FROM v_memory_grant_stats", StringComparison.Ordinal), "the interval must be in the SELECT list"); + Assert.Equal(1, sql.Split("sample_interval_seconds").Length - 1); + } + + /// + /// #3540 (V128): the resource-semaphore row carries the stored interval and reports it the way the file-I/O + /// row does — IsUnknowable is true ONLY for a stored 0 (the calculator's marker), never for a + /// pre-V128 NULL, which is "never recorded" rather than "unknowable". The tool then hands the caller a + /// null interval for both, with interval_known saying so. + /// + [Fact] + public void ResourceSemaphoreRow_IsUnknowable_OnlyForAStoredZeroInterval() + { + static DarlingMemoryGrantReader.ResourceSemaphoreRow Row(int? interval) => new( + DateTime.UnixEpoch, 0, 2, 100, 200, 90, 80, 10, 8, 3, 1, 5, 2, 0, 0, interval); + + Assert.True(Row(0).IsUnknowable); + Assert.False(Row(120).IsUnknowable); + Assert.False(Row(null).IsUnknowable); } [Fact] @@ -203,18 +229,37 @@ public async Task MemoryGrantTools_ReadPlantedRows_AgainstDevPostgres() await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); var t = DarlingMcpTestData.TruncateToSeconds(DateTime.UtcNow).AddMinutes(-2); - foreach (var (semaphore, pool) in new[] { ((short)0, 1), ((short)0, 2) }) + /* #3540 (V128): three semaphores at one collection — pool 1 a restart marker (stored interval 0), + pool 2 measured (120 s), pool 3 a pre-V128 row (NULL). The tool reports the interval only for + the measured one and says interval_known for exactly that one. */ + foreach (var (semaphore, pool, interval) in new[] { ((short)0, 1, (object)0), ((short)0, 2, 120), ((short)0, 3, DBNull.Value) }) { await DarlingMcpTestData.ExecAsync(connection, ct, - @"INSERT INTO memory_grant_stats (collection_id, collection_time, server_id, server_name, resource_semaphore_id, pool_id, target_memory_mb, max_target_memory_mb, total_memory_mb, available_memory_mb, granted_memory_mb, used_memory_mb, grantee_count, waiter_count, timeout_error_count, forced_grant_count, timeout_error_count_delta, forced_grant_count_delta) -VALUES ($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,$11,$12,$13,$14,$15,$16,$17,$18)", - CollectionIdGenerator.Next(), t, ServerId, ServerName, semaphore, pool, 8000m, 12000m, 8000m, 6000m, 2000m, 1500m, 3, 1, 4L, 2L, 1L, 0L); + @"INSERT INTO memory_grant_stats (collection_id, collection_time, server_id, server_name, resource_semaphore_id, pool_id, target_memory_mb, max_target_memory_mb, total_memory_mb, available_memory_mb, granted_memory_mb, used_memory_mb, grantee_count, waiter_count, timeout_error_count, forced_grant_count, timeout_error_count_delta, forced_grant_count_delta, sample_interval_seconds) +VALUES ($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,$11,$12,$13,$14,$15,$16,$17,$18,$19)", + CollectionIdGenerator.Next(), t, ServerId, ServerName, semaphore, pool, 8000m, 12000m, 8000m, 6000m, 2000m, 1500m, 3, 1, 4L, 2L, 1L, 0L, interval); } var semaphoreJson = await DarlingMcpMemoryGrantTools.GetResourceSemaphore(postgres, ServerName); DarlingMcpTestData.AssertEnvelope(semaphoreJson, ServerName, "grants"); Assert.Contains("max_target_memory_mb", semaphoreJson, StringComparison.Ordinal); + using (var doc = JsonDocument.Parse(semaphoreJson)) + { + var byPool = doc.RootElement.GetProperty("grants").EnumerateArray() + .ToDictionary(g => g.GetProperty("pool_id").GetInt32()); + Assert.Equal(3, byPool.Count); + + Assert.Equal(JsonValueKind.Null, byPool[1].GetProperty("sample_interval_seconds").ValueKind); + Assert.False(byPool[1].GetProperty("interval_known").GetBoolean()); + + Assert.Equal(120, byPool[2].GetProperty("sample_interval_seconds").GetInt32()); + Assert.True(byPool[2].GetProperty("interval_known").GetBoolean()); + + Assert.Equal(JsonValueKind.Null, byPool[3].GetProperty("sample_interval_seconds").ValueKind); + Assert.False(byPool[3].GetProperty("interval_known").GetBoolean()); + } + var grantsJson = await DarlingMcpMemoryGrantTools.GetMemoryGrants(postgres, ServerName); DarlingMcpTestData.AssertEnvelope(grantsJson, ServerName, "grants"); Assert.Contains("granted_memory_mb", grantsJson, StringComparison.Ordinal); diff --git a/Darling/Darling.Tests/DarlingMcpTrendToolsTests.cs b/Darling/Darling.Tests/DarlingMcpTrendToolsTests.cs index 571e02f81..66a7314fa 100644 --- a/Darling/Darling.Tests/DarlingMcpTrendToolsTests.cs +++ b/Darling/Darling.Tests/DarlingMcpTrendToolsTests.cs @@ -204,6 +204,39 @@ public void ProcedureDurationTrendSql_SameRate_OverProcedureStats() Assert.Contains("executions_per_second", sql, StringComparison.Ordinal); } + /// + /// #3540 (V128): the procedure trend reads the collection's STORED interval — MAX over the collection's + /// rows, 0 → NULL through NULLIF so a restart's marker collection drops rather than plotting 0.00 — and + /// falls back to the LAG derivation only for a pre-V128 collection. No ELSE 0. Byte-identical to the + /// viewer's copy apart from the database filter, as the pair always were. + /// + [Fact] + public void ProcedureDurationTrendSql_PrefersTheStoredInterval_NeverFabricatesZero_AndMirrorsTheViewer() + { + var sql = DarlingTrendReader.ProcedureDurationTrendSql; + Assert.Contains("CASE WHEN MAX(sample_interval_seconds) IS NULL", sql, StringComparison.Ordinal); + Assert.Contains("ELSE NULLIF(MAX(sample_interval_seconds), 0)", sql, StringComparison.Ordinal); + Assert.DoesNotContain("ELSE 0", sql, StringComparison.Ordinal); + Assert.Contains("CASE WHEN interval_seconds > 0 THEN total_elapsed_ms / interval_seconds END AS elapsed_ms_per_second", sql, StringComparison.Ordinal); + + /* The viewer's copy minus its database-filter line is this string, whitespace aside. */ + var viewer = string.Join('\n', ViewerDataService.ProcedureDurationTrendSql + .Replace("\r\n", "\n", StringComparison.Ordinal) + .Split('\n') + .Where(l => !l.Contains("$4::text[]", StringComparison.Ordinal)) + .Select(l => l.Trim())); + var mcp = string.Join('\n', sql.Replace("\r\n", "\n", StringComparison.Ordinal).Split('\n').Select(l => l.Trim())); + Assert.Equal(viewer, mcp); + + /* And the shared reader DROPS a NULL-rate row rather than reading it as 0 — the C# half of the idiom. */ + var source = RepoFile.ReadRepoFile("Darling", "PerformanceMonitor.Darling.Service", "Mcp", "DarlingTrendReader.cs"); + var reader = source[source.IndexOf("private static async Task> ReadDurationPointsAsync(", StringComparison.Ordinal)..]; + reader = reader[..reader.IndexOf("return items;", StringComparison.Ordinal)]; + Assert.Contains("if (reader.IsDBNull(1))", reader, StringComparison.Ordinal); + Assert.Contains("continue;", reader, StringComparison.Ordinal); + Assert.DoesNotContain("reader.IsDBNull(1) ? 0", reader, StringComparison.Ordinal); + } + /// /// #2484: the Query Store trend carries the #1841 tier-2 interval placement, copied from the viewer's /// read rather than rewritten. Both arms are pinned because losing either one changes the numbers: drop diff --git a/Darling/Darling.Tests/DarlingPerformanceTrendsReadTests.cs b/Darling/Darling.Tests/DarlingPerformanceTrendsReadTests.cs index c426640ba..c2c62736f 100644 --- a/Darling/Darling.Tests/DarlingPerformanceTrendsReadTests.cs +++ b/Darling/Darling.Tests/DarlingPerformanceTrendsReadTests.cs @@ -123,6 +123,11 @@ an empty raw answer the requested start stands and the served span is the whole Two snapshots five minutes apart, two executions between them: 0.0067/sec. The shipped integer field truncates that to 0, which reads as an idle server; the double does not. This is the whole reason executions_per_second exists. + + These rows carry no sample_interval_seconds (the pre-V128 shape), so the read LAG-derives + the interval — and since V128 (#3540) the FIRST snapshot, which has nothing to LAG against, + is absent rather than a fabricated 0.0 point (the correction V127 made for the wait trends). + One point comes back: the second snapshot, whose rate is the thing under test. */ await SeedProcedureAsync(connection, ct, MinutesAgo(20), executions: 0, elapsedUs: 0); await SeedProcedureAsync(connection, ct, MinutesAgo(15), executions: 2, elapsedUs: 600_000); @@ -130,9 +135,9 @@ This is the whole reason executions_per_second exists. var procs = JsonDocument.Parse( await DarlingMcpTrendTools.GetProcedureDurationTrend(postgres, ServerName, 4)).RootElement; var procTrend = procs.GetProperty("trend"); - Assert.Equal(2, procTrend.GetArrayLength()); + Assert.Equal(1, procTrend.GetArrayLength()); - var second = procTrend[1]; + var second = procTrend[0]; Assert.True(second.GetProperty("value").GetDouble() > 0, "elapsed ms/sec must be a real rate"); Assert.Equal(0, second.GetProperty("execution_count").GetInt64()); Assert.True( @@ -146,10 +151,11 @@ execution_count precedent. */ /* #3541 A2: the disclosure block. A 4-hour window anchored at now sits inside the raw horizon (the shared fixture carries no continuous aggregates — every test that builds them mints a - ScratchPostgres — so raw is also the only tier here), and the series the store held begins - at the 20-minutes-ago seed: effective_start says so, and the head sits three-plus hours past - the requested start, which is what `truncated` means. The point is that the label matches - the data rather than the request. + ScratchPostgres — so raw is also the only tier here), and the series the read SERVED begins + at its first point — the 15-minutes-ago seed, since V128 dropped the prior-less first + snapshot: effective_start says so, and the head sits three-plus hours past the requested + start, which is what `truncated` means. The point is that the label matches the data rather + than the request. */ Assert.Equal("raw", procs.GetProperty("source").GetString()); Assert.Equal("per-collection", procs.GetProperty("bucket").GetString()); diff --git a/Darling/Darling.Tests/DeltaFamilyIntervalColumnsRungTests.cs b/Darling/Darling.Tests/DeltaFamilyIntervalColumnsRungTests.cs index ae56e40b6..9966215c8 100644 --- a/Darling/Darling.Tests/DeltaFamilyIntervalColumnsRungTests.cs +++ b/Darling/Darling.Tests/DeltaFamilyIntervalColumnsRungTests.cs @@ -29,9 +29,12 @@ namespace Darling.Tests; /// perfmon_stats and query_stats carried the column from the start; this rung gives the other four the same /// column in the same type. /// -/// This file carries the "I am the top rung" claims that moved off the previous top rung's test when this -/// rung landed — a fully-migrated store must map to EXACTLY this version, or the viewer's connect-time gate -/// refuses a store that is actually current. +/// The "I am the top rung" claims have moved ON to +/// (V128), the same handoff this file received from (V126) and +/// that file received from (V125). What stays here is +/// everything true of this rung wherever it sits in the ladder; what left is every claim that was really +/// about being NEWEST — keeping a copy of those would assert this rung is still the top, which is how the +/// NEXT rung's build goes red. /// public sealed class DeltaFamilyIntervalColumnsRungTests { @@ -39,7 +42,8 @@ public sealed class DeltaFamilyIntervalColumnsRungTests private const int PreviousVersion = 126; - /// This rung's sentinel ordinal in the viewer probe — the newest, so the last argument. + /// This rung's sentinel ordinal in the viewer probe. No longer the last argument — V128 + /// appended its own — so this is a position within the signature rather than its end. private const int ProbeOrdinal = 102; private const string IntervalColumn = "sample_interval_seconds"; @@ -49,7 +53,7 @@ public sealed class DeltaFamilyIntervalColumnsRungTests /* ---- the rung ------------------------------------------------------------------------------------ */ [Fact] - public void TheRungIsRegisteredAtTheTopOfADenseLadder() + public void TheRungIsRegisteredInADenseLadder() { var versions = PgMigrations.Scripts.Select(s => s.Version).ToList(); @@ -59,7 +63,11 @@ public void TheRungIsRegisteredAtTheTopOfADenseLadder() Assert.Equal(StorageVersion.SchemaVersion, PgMigrations.Scripts[^1].Version); Assert.Equal(StorageVersion.SchemaVersion, versions.Max()); - Assert.Equal(RungVersion, StorageVersion.SchemaVersion); + + /* Not `RungVersion == SchemaVersion` any more: that asserted this rung is the newest, which + stopped being true when V128 landed. The invariant that outlives the handoff is that the + LADDER's top and the declared version agree, which the two lines above already say. */ + Assert.True(RungVersion < StorageVersion.SchemaVersion); Assert.Equal(versions.Distinct().OrderBy(v => v), versions); } @@ -140,18 +148,18 @@ public void TheGeneratedCreateTable_CarriesTheColumnLast_OnAllFour() } } - /* ---- the probe (three sites, top arm) ------------------------------------------------------------- */ + /* ---- the probe (three sites) ---------------------------------------------------------------------- */ /// - /// The viewer probe's three sites carry this rung's sentinel, and the map treats it as the TOP arm. + /// The viewer probe's three sites carry this rung's sentinel, and a store that stopped here maps to it. /// /// The probe asks the question, the caller reads the answer, the map has the parameter — three /// sites, and a sentinel present at only some of them shifts every LATER ordinal onto the wrong column. - /// Miss all three and a fully-migrated store probes one rung short, so the connect-time gate refuses a - /// store that is in fact current — permanently, because no later upgrade changes the answer. + /// The top-arm claims (last argument, returns the build's version) moved to + /// with V128. /// [Fact] - public void TheProbeMapsAFullyMigratedStoreToThisTopRung() + public void TheProbeCarriesThisRungsSentinel_AndAFullyMigratedStoreMapsToTheLaddersTop() { Assert.Contains( $"table_name = 'wait_stats'\n AND column_name = '{IntervalColumn}'", @@ -167,8 +175,10 @@ public void TheProbeMapsAFullyMigratedStoreToThisTopRung() .GetMethod("MapProbedSchemaVersion", BindingFlags.NonPublic | BindingFlags.Static)!; var arity = method.GetParameters().Length; - /* The top rung's sentinel IS the last argument. */ - Assert.Equal(ProbeOrdinal, arity - 1); + /* A position within the signature, not its end: `ProbeOrdinal == arity - 1` asserted this rung is + the NEWEST sentinel, which stopped being true the moment V128 appended its own. Strictly-less is + the form every other non-top rung's test here uses. */ + Assert.True(ProbeOrdinal < arity - 1); /* Every sentinel true = a fully-migrated store, which must map to exactly this version. Built by reflection so the arity tracks the signature. */ @@ -187,14 +197,15 @@ test of that rung. */ Assert.Equal(PreviousVersion, (int)method.Invoke(null, behind)!); /* And in the source, the arm sits ABOVE the previous rung's — newest-first is the whole contract of - that method — and returns this build's version rather than a literal that could drift from it. */ + that method. It returns this rung's own literal now, not the build's version: the "returns + StorageVersion.SchemaVersion" half of the top-arm claim moved to V128's test with the top. */ var thisArm = viewer.IndexOf("if (hasDeltaFamilyIntervalColumns)", StringComparison.Ordinal); var previousArm = viewer.IndexOf("if (hasSelfDiskWarnGbFloor)", StringComparison.Ordinal); - Assert.True(thisArm >= 0, "the viewer has no V127 sentinel arm — a fully-migrated store would map one rung low"); + Assert.True(thisArm >= 0, "the viewer has no V127 sentinel arm — a store that stopped here would map to 126"); Assert.True(previousArm >= 0, "the previous rung's arm is gone, so this pin is comparing against nothing"); - Assert.True(thisArm < previousArm, "the V127 arm sits below the previous rung's, so a current store maps one rung low"); + Assert.True(thisArm < previousArm, "the V127 arm sits below the previous rung's, so a V127 store maps one rung low"); Assert.Contains( - "return " + StorageVersion.SchemaVersion.ToString(CultureInfo.InvariantCulture) + ";", + "return " + RungVersion.ToString(CultureInfo.InvariantCulture) + ";", viewer[thisArm..], StringComparison.Ordinal); } @@ -229,11 +240,12 @@ public void TheFourCollectors_WriteTheIntervalAsTheMinimumOverTheirDeltaGroups() /// /// The calculator's own doc claim is TRUE again. It said "every consumer already maps 0 to NULL via /// NULLIF(sample_interval_seconds, 0)" while four of six families discarded the interval at the write; - /// the sentence now names which families the claim holds for and which still do not persist one, so it - /// cannot silently become false again by a seventh family shipping naked. + /// the sentence now names which families this rung dressed and when, so the history of how the claim + /// went false and came back is in the file that made it. (The "and which still do not" half of this + /// pin moved to when V128 emptied that list.) /// [Fact] - public void TheCalculatorDoc_NamesTheFamiliesTheNullifClaimHoldsFor() + public void TheCalculatorDoc_NamesTheFamiliesThisRungDressed() { var source = RepoFile.ReadRepoFile("PerformanceMonitor.Collectors", "CollectorDeltaCalculator.cs"); diff --git a/Darling/Darling.Tests/DeltaFamilyIntervalCompletionLivePostgresTests.cs b/Darling/Darling.Tests/DeltaFamilyIntervalCompletionLivePostgresTests.cs new file mode 100644 index 000000000..d370d8b81 --- /dev/null +++ b/Darling/Darling.Tests/DeltaFamilyIntervalCompletionLivePostgresTests.cs @@ -0,0 +1,233 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading; +using System.Threading.Tasks; +using Npgsql; +using PerformanceMonitor.Darling.Service.Mcp; +using PerformanceMonitor.Darling.Storage; +using PerformanceMonitor.Darling.Viewer; +using Xunit; + +namespace Darling.Tests; + +/// +/// #3540 (V128), the read half of the completion, proven against LIVE Postgres (gated on +/// DARLING_TEST_PG) through the real readers: a collection every row of which the collector stored with +/// sample_interval_seconds = 0 — the calculator's "no delta knowable" marker, in practice a restart — +/// is NOT a point on the procedure duration trend or the per-statement PostgreSQL trend. It used to be a +/// confident 0.00 ms/sec and 0.00 calls/sec at exactly the moment nothing was knowable. +/// +/// Three states per collection, read distinctly: a MEASURED interval (MAX over the collection's rows) +/// divides the summed deltas and wins over the LAG derivation; the marker (every row 0) yields no point; and +/// NULL — every row collected before V128 — falls back to the LAG over collection_time those reads always +/// used. The procedure history grid shows the stored interval itself, the marker's 0 included, because a +/// displayed interval is not a rate. Lite's twin of these claims runs in-process on DuckDB +/// (DeltaFamilyUnknowableRowReadTests). +/// +/// Every expected value below was worked by hand from the rows and reproduced against PG18 + +/// TimescaleDB 2.28.1 before this was written. +/// +[Collection("live-postgres")] +public sealed class DeltaFamilyIntervalCompletionLivePostgresTests +{ + private const int ServerId = -128128; + private const string ServerName = "delta-interval-completion-e2e"; + + private static string? ConnectionString => Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + + /// + /// The procedure duration trend, both copies (the viewer's read and the MCP's SQL, which the pin in + /// DarlingMcpTrendToolsTests proves are one string): t1/t2 pre-V128 (NULL) — t1 no prior, no point; t2 the + /// LAG's 300 s. t3 a restart — every row 0 — absent. t4 a steady pass with a readmitted plan (its row 0) + /// beside a measured 120 s row: MAX 120 wins over the LAG's 300, and the readmitted plan adds 0. + /// + [Fact] + public async Task ProcedureDurationTrend_DropsTheUnknowableCollection_PrefersTheStoredInterval_AgainstDevPostgres() + { + var cs = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(cs), "Set DARLING_TEST_PG to a Postgres connection string to run the live procedure-trend test."); + + var ct = TestContext.Current.CancellationToken; + using var connection = new NpgsqlConnection(cs); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + await DeleteRowsAsync(connection, ct); + + await using var viewer = new ViewerDataService(cs!); + await using var postgres = NpgsqlDataSource.Create(cs!); + + var bodySucceeded = false; + try + { + var t1 = Naive(TruncateToSeconds(DateTime.UtcNow.AddHours(-2))); + var t2 = t1.AddMinutes(5); + var t3 = t2.AddMinutes(5); + var t4 = t3.AddMinutes(5); + + await ProcedureAsync(connection, t1, "usp_A", deltaExecutions: 5, deltaElapsedUs: 100_000, interval: null, ct); + await ProcedureAsync(connection, t2, "usp_A", 30, 600_000, null, ct); + await ProcedureAsync(connection, t3, "usp_A", 0, 0, 0, ct); + await ProcedureAsync(connection, t4, "usp_A", 24, 1_200_000, 120, ct); + await ProcedureAsync(connection, t4, "usp_New", 0, 0, 0, ct); + + var points = await viewer.GetProcedureDurationTrendAsync(ServerId, t1.AddMinutes(-1), t4.AddMinutes(1), cancellationToken: ct); + Assert.Equal(new[] { t2, t4 }, points.Select(p => p.CollectionTime).ToArray()); + Assert.Equal(2.0, points[0].Value, precision: 6); /* 600 ms / LAG 300 s */ + Assert.Equal(0, points[0].ExecutionCount); /* 30 / 300 = 0.1 executions/sec, truncated to long as always */ + Assert.Equal(10.0, points[1].Value, precision: 6); /* 1200 ms / STORED 120 s, not the LAG's 4.0 */ + + /* The MCP copy, run as the tool would run it on the raw tier. */ + await using (var command = postgres.CreateCommand(DarlingTrendReader.ProcedureDurationTrendSql)) + { + command.Parameters.AddWithValue(ServerId); + command.Parameters.AddWithValue(t1.AddMinutes(-1)); + command.Parameters.AddWithValue(t4.AddMinutes(1)); + var mcp = new List<(DateTime At, double? Rate, double? Executions)>(); + await using var reader = await command.ExecuteReaderAsync(ct); + while (await reader.ReadAsync(ct)) + { + mcp.Add((reader.GetDateTime(0), + reader.IsDBNull(1) ? null : Convert.ToDouble(reader.GetValue(1)), + reader.IsDBNull(2) ? null : Convert.ToDouble(reader.GetValue(2)))); + } + + /* The SQL returns all four collections; t1 and t3 carry NULL rates, which the shared reader drops. */ + Assert.Equal(new[] { t1, t2, t3, t4 }, mcp.Select(m => m.At).ToArray()); + Assert.Null(mcp[0].Rate); + Assert.Equal(2.0, mcp[1].Rate!.Value, precision: 6); + Assert.Null(mcp[2].Rate); + Assert.Null(mcp[2].Executions); + Assert.Equal(10.0, mcp[3].Rate!.Value, precision: 6); + Assert.Equal(0.2, mcp[3].Executions!.Value, precision: 6); + } + + /* The history grid: the stored interval as stored, the marker's 0 included; LAG only for NULL. */ + var history = await viewer.GetProcedureStatsHistoryAsync(ServerId, "AppDb", "dbo", "usp_A", t1.AddMinutes(-1), t4.AddMinutes(1), ct); + Assert.Equal(new[] { t1, t2, t3, t4 }, history.Select(h => h.CollectionTime).ToArray()); + Assert.Null(history[0].SampleIntervalSeconds); + Assert.Equal(300, history[1].SampleIntervalSeconds); + Assert.Equal(0, history[2].SampleIntervalSeconds); + Assert.Equal(120, history[3].SampleIntervalSeconds); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, async (cleanup, cleanupCt) => + await DeleteRowsAsync(cleanup, cleanupCt)); + } + } + + /// + /// The per-statement PostgreSQL trend (): the same + /// four collections for one queryid across two (dbid, userid, toplevel) entries. t3 is a + /// pg_stat_statements_reset() — both entries' rows store 0 — and is absent; at t4 one entry is a + /// first sighting (0) beside the other's measured 120 s, so MAX is 120 and calls_per_second is the measured + /// entry's 24 calls over 120 s, not over the LAG's 300. + /// + [Fact] + public async Task PgQueryDurationTrend_DropsTheUnknowableSnapshot_PrefersTheStoredInterval_AgainstDevPostgres() + { + var cs = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(cs), "Set DARLING_TEST_PG to a Postgres connection string to run the live pg-statement-trend test."); + + var ct = TestContext.Current.CancellationToken; + using var connection = new NpgsqlConnection(cs); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + await DeleteRowsAsync(connection, ct); + + await using var postgres = NpgsqlDataSource.Create(cs!); + + var bodySucceeded = false; + try + { + var t1 = Naive(TruncateToSeconds(DateTime.UtcNow.AddHours(-2))); + var t2 = t1.AddMinutes(5); + var t3 = t2.AddMinutes(5); + var t4 = t3.AddMinutes(5); + const long QueryId = 4242; + + await PgStatementAsync(connection, t1, QueryId, userId: 10, deltaCalls: 5, deltaMs: 100, interval: null, ct); + await PgStatementAsync(connection, t2, QueryId, 10, 30, 600, null, ct); + await PgStatementAsync(connection, t3, QueryId, 10, 0, 0, 0, ct); + await PgStatementAsync(connection, t3, QueryId, 11, 0, 0, 0, ct); + await PgStatementAsync(connection, t4, QueryId, 10, 24, 1200, 120, ct); + await PgStatementAsync(connection, t4, QueryId, 11, 0, 0, 0, ct); + + var points = await DarlingPgTrendReader.GetQueryDurationTrendAsync(postgres, ServerId, QueryId, t1.AddMinutes(-1), t4.AddMinutes(1), ct); + + Assert.Equal(new[] { t2, t4 }, points.Select(p => p.CollectionTimeUtc).ToArray()); + Assert.Equal(30, points[0].Calls); + Assert.Equal(0.1, points[0].CallsPerSecond, precision: 6); /* 30 / LAG 300 s */ + Assert.Equal(20.0, points[0].MeanExecMs, precision: 6); /* 600 / 30 */ + Assert.Equal(24, points[1].Calls); + Assert.Equal(0.2, points[1].CallsPerSecond, precision: 6); /* 24 / STORED 120 s, not the LAG's 0.08 */ + Assert.Equal(50.0, points[1].MeanExecMs, precision: 6); /* 1200 / 24 */ + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, async (cleanup, cleanupCt) => + await DeleteRowsAsync(cleanup, cleanupCt)); + } + } + + /* ---- seeding ---------------------------------------------------------------------------------------- */ + + private static DateTime Naive(DateTime t) => DateTime.SpecifyKind(t, DateTimeKind.Unspecified); + + private static DateTime TruncateToSeconds(DateTime value) => + new(value.Ticks - (value.Ticks % TimeSpan.TicksPerSecond), value.Kind); + + private static async Task ProcedureAsync(NpgsqlConnection connection, DateTime t, string objectName, long deltaExecutions, long deltaElapsedUs, int? interval, CancellationToken ct) + { + using var cmd = new NpgsqlCommand( + "INSERT INTO procedure_stats (collection_id, collection_time, server_id, server_name, database_name, schema_name, object_name, object_type, " + + "execution_count, total_worker_time, total_elapsed_time, total_logical_reads, total_physical_reads, total_logical_writes, " + + "delta_execution_count, delta_worker_time, delta_elapsed_time, sample_interval_seconds) " + + "VALUES (1, $1, $2, $3, 'AppDb', 'dbo', $4, 'PROCEDURE', 0, 0, 0, 0, 0, 0, $5, 0, $6, $7)", connection); + cmd.Parameters.AddWithValue(t); + cmd.Parameters.AddWithValue(ServerId); + cmd.Parameters.AddWithValue(ServerName); + cmd.Parameters.AddWithValue(objectName); + cmd.Parameters.AddWithValue(deltaExecutions); + cmd.Parameters.AddWithValue(deltaElapsedUs); + cmd.Parameters.AddWithValue(interval.HasValue ? interval.Value : DBNull.Value); + await cmd.ExecuteNonQueryAsync(ct); + } + + private static async Task PgStatementAsync(NpgsqlConnection connection, DateTime t, long queryId, long userId, long deltaCalls, long deltaMs, int? interval, CancellationToken ct) + { + using var cmd = new NpgsqlCommand( + "INSERT INTO pg_statement_stats (collection_id, collection_time, server_id, server_name, queryid, database_id, user_id, toplevel, " + + "calls, total_exec_time_ms, rows_returned, delta_calls, delta_total_exec_time_ms, delta_rows, sample_interval_seconds) " + + "VALUES (1, $1, $2, $3, $4, 16384, $5, true, 0, 0, 0, $6, $7, 0, $8)", connection); + cmd.Parameters.AddWithValue(t); + cmd.Parameters.AddWithValue(ServerId); + cmd.Parameters.AddWithValue(ServerName); + cmd.Parameters.AddWithValue(queryId); + cmd.Parameters.AddWithValue(userId); + cmd.Parameters.AddWithValue(deltaCalls); + cmd.Parameters.AddWithValue(deltaMs); + cmd.Parameters.AddWithValue(interval.HasValue ? interval.Value : DBNull.Value); + await cmd.ExecuteNonQueryAsync(ct); + } + + private static async Task DeleteRowsAsync(NpgsqlConnection connection, CancellationToken ct) + { + using var cleanup = new NpgsqlCommand( + $"DELETE FROM procedure_stats WHERE server_id = {ServerId}; DELETE FROM pg_statement_stats WHERE server_id = {ServerId};", connection); + await cleanup.ExecuteNonQueryAsync(ct); + } +} diff --git a/Darling/Darling.Tests/DeltaFamilyIntervalCompletionRungTests.cs b/Darling/Darling.Tests/DeltaFamilyIntervalCompletionRungTests.cs new file mode 100644 index 000000000..0ace3a7bc --- /dev/null +++ b/Darling/Darling.Tests/DeltaFamilyIntervalCompletionRungTests.cs @@ -0,0 +1,407 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Globalization; +using System.Linq; +using System.Reflection; +using PerformanceMonitor.Collectors; +using PerformanceMonitor.Darling.Storage; +using PerformanceMonitor.Darling.Viewer; +using Xunit; + +namespace Darling.Tests; + +/// +/// V128 / #3540: the completion of V127. sample_interval_seconds on the four delta families V127 left +/// naked — procedure_stats, memory_grant_stats, pg_wait_stats, pg_statement_stats +/// — and the two statement offsets on query_stats that its delta key is made of. After this rung every +/// member of CollectorDeltaCalculator.DeltaFamilyCollectors stores the interval its deltas accrued +/// over, so the calculator's (delta 0, interval 0) "no delta knowable" marker reaches the store from every +/// family, and the restart seed can rebuild the one key the store could not reproduce. +/// +/// This file carries the "I am the top rung" claims that moved off +/// (V127) when this rung landed, the same handoff that file +/// received from (V126) — a fully-migrated store must map to +/// EXACTLY this version, or the viewer's connect-time gate refuses a store that is actually current. +/// +public sealed class DeltaFamilyIntervalCompletionRungTests +{ + private const int RungVersion = 128; + + private const int PreviousVersion = 127; + + /// This rung's sentinel ordinal in the viewer probe — the newest, so the last argument. + private const int ProbeOrdinal = 103; + + private const string IntervalColumn = "sample_interval_seconds"; + + private static readonly string[] IntervalTables = { "procedure_stats", "memory_grant_stats", "pg_wait_stats", "pg_statement_stats" }; + + private static readonly string[] OffsetColumns = { "statement_start_offset", "statement_end_offset" }; + + /* ---- the rung ------------------------------------------------------------------------------------ */ + + [Fact] + public void TheRungIsRegisteredAtTheTopOfADenseLadder() + { + var versions = PgMigrations.Scripts.Select(s => s.Version).ToList(); + + Assert.Equal( + "delta-family-interval-completion", + PgMigrations.Scripts.Single(s => s.Version == RungVersion).Name); + + Assert.Equal(StorageVersion.SchemaVersion, PgMigrations.Scripts[^1].Version); + Assert.Equal(StorageVersion.SchemaVersion, versions.Max()); + Assert.Equal(RungVersion, StorageVersion.SchemaVersion); + + Assert.Equal(versions.Distinct().OrderBy(v => v), versions); + } + + /// + /// The hand-written head of the rung — everything before the V54 pre-add the ladder entry concatenates + /// — adds ONE nullable, default-less integer interval column to each of the four tables and the two + /// offset columns to query_stats, all schema-qualified and idempotent; refreshes the ONE + /// SELECT * passthrough among them; drops the payload-resolving v_query_stats for the + /// regenerated definition to follow; and does nothing else. + /// + /// integer is pinned against the type perfmon_stats already uses through the generator, not + /// as a literal alone, so the ten interval columns cannot drift apart and + /// NULLIF(sample_interval_seconds, 0) means the same thing on every one. No DEFAULT and no backfill + /// on any of the six: a historical row never recorded its interval or its offsets, a backfilled 0 + /// interval would stamp all of history "unknowable", and a backfilled 0/-1 offset pair would seed + /// baselines under a key nothing will ever present. + /// + [Fact] + public void TheRungAddsSixNullableIntegers_SchemaQualified_RefreshesTheGrantView_AndDropsTheResolvingView() + { + var full = PgMigrations.Scripts.Single(s => s.Version == RungVersion).Sql.Replace("\r\n", "\n", StringComparison.Ordinal); + + /* The V121 idiom: the hand-written head, then V54's gz pre-add, then every payload column's pre-add, + then the regenerated resolving view. The head is what this test inspects statement by statement; + MigrationLadderPins holds the ordering of the rest. */ + var v54 = full.IndexOf("ALTER TABLE query_plan_dim", StringComparison.Ordinal); + Assert.True(v54 > 0, "the rung does not carry V54's gz pre-add ahead of the regenerated resolving view"); + var head = full[..v54]; + + var perfmonType = PgSchemaGenerator.TypeFor( + PerfmonStatsCollector.Instance.PayloadColumns.Single(c => c.Name == IntervalColumn)); + Assert.Equal("integer", perfmonType); + + foreach (var table in IntervalTables) + { + Assert.Equal(1, CountOf(head, $"ALTER TABLE collect.{table}\n")); + Assert.DoesNotContain($"ALTER TABLE {table}\n", head, StringComparison.Ordinal); + Assert.Contains( + $"ALTER TABLE collect.{table}\n ADD COLUMN IF NOT EXISTS {IntervalColumn} {perfmonType};", + head, StringComparison.Ordinal); + } + + foreach (var column in OffsetColumns) + { + Assert.Contains( + $"ALTER TABLE collect.query_stats\n ADD COLUMN IF NOT EXISTS {column} integer;", + head, StringComparison.Ordinal); + } + + Assert.Equal(6, CountOf(head, "ADD COLUMN IF NOT EXISTS")); + + /* The one SELECT * passthrough among the five tables (PgSchemaGenerator.AllPassthroughViews pins the + set): Postgres freezes a view's column list at CREATE (V14, V80, V81, V127). */ + Assert.Equal(1, CountOf(head, "CREATE OR REPLACE VIEW collect.v_memory_grant_stats AS SELECT * FROM collect.memory_grant_stats;")); + Assert.Equal(1, CountOf(head, "CREATE OR REPLACE VIEW collect.v_")); + foreach (var table in new[] { "procedure_stats", "pg_wait_stats", "pg_statement_stats" }) + { + Assert.DoesNotContain($"v_{table}", head, StringComparison.Ordinal); + Assert.DoesNotContain("v_" + table, PgSchemaGenerator.AllPassthroughViews); + } + + /* v_query_stats is the payload-RESOLVING view (#1767) and the offsets land ahead of its digest columns, + an alteration CREATE OR REPLACE VIEW refuses: DROP here, regenerate in the tail (V51 / V121). */ + Assert.Equal(1, CountOf(head, "DROP VIEW IF EXISTS collect.v_query_stats;")); + Assert.DoesNotContain("CASCADE", head, StringComparison.Ordinal); + var tail = full[v54..]; + Assert.Equal(1, CountOf(tail, "CREATE OR REPLACE VIEW v_query_stats AS")); + Assert.True( + tail.LastIndexOf("CREATE OR REPLACE VIEW v_query_stats AS", StringComparison.Ordinal) + > tail.LastIndexOf("ADD COLUMN IF NOT EXISTS", StringComparison.Ordinal), + "the regenerated resolving view must be the LAST statement, after every pre-add"); + foreach (var column in OffsetColumns) + { + Assert.Contains($"f.{column}", tail, StringComparison.Ordinal); + } + + /* Nullable, no default, no backfill, no CHECK, no GRANT, and no touch of any continuous aggregate — + the wait_stats_baseline follow-up V127 documented is an aggregate-plus-retirement operation, not a + column, and not this rung. */ + Assert.DoesNotContain("DEFAULT", head, StringComparison.Ordinal); + Assert.DoesNotContain("NOT NULL", head, StringComparison.Ordinal); + Assert.DoesNotContain("UPDATE ", head, StringComparison.Ordinal); + Assert.DoesNotContain("CHECK", head, StringComparison.Ordinal); + Assert.DoesNotContain("GRANT", head, StringComparison.Ordinal); + Assert.DoesNotContain("MATERIALIZED", head, StringComparison.Ordinal); + Assert.DoesNotContain("wait_stats_baseline", head, StringComparison.Ordinal); + Assert.DoesNotContain("_hourly", head, StringComparison.Ordinal); + } + + /// + /// The offsets' semantics are stated in the rung's own doc, in the words a reader will search for: they + /// are BYTE offsets into the batch's nvarchar text (so a character slice divides by two), -1 as the + /// end offset means "to the end of the batch", and they are stored VERBATIM — never normalized — because + /// the delta key is built over the raw values. A future reader will second-guess exactly that pair, and + /// the rung doc is where they will look. + /// + [Fact] + public void TheRungDoc_StatesTheOffsetsSemantics_BytesEndOfBatchAndVerbatim() + { + var source = RepoFile.ReadRepoFile("Darling", "PerformanceMonitor.Darling.Storage", "PgMigrations.cs"); + var start = source.IndexOf("/// V128 —", StringComparison.Ordinal); + var end = source.IndexOf("private const string V128Sql", StringComparison.Ordinal); + Assert.True(start >= 0 && end > start, "the V128 rung has no XML doc block ahead of V128Sql"); + var doc = source[start..end]; + + Assert.Contains("BYTES", doc, StringComparison.Ordinal); + Assert.Contains("divides by two", doc, StringComparison.Ordinal); + Assert.Contains("statement_end_offset = -1", doc, StringComparison.Ordinal); + Assert.Contains("means \"to the end of the batch\"", doc, StringComparison.Ordinal); + Assert.Contains("verbatim as the DMV reports them", doc, StringComparison.Ordinal); + Assert.Contains("never normalized", doc, StringComparison.Ordinal); + /* And the SQL itself repeats the two facts beside the ALTERs. */ + var sql = PgMigrations.Scripts.Single(s => s.Version == RungVersion).Sql; + Assert.Contains("BYTE offsets", sql, StringComparison.Ordinal); + Assert.Contains("statement_end_offset = -1 means", sql, StringComparison.Ordinal); + } + + /// + /// A FRESH store gets the columns from the generated CREATE TABLE (the collector definitions carry them + /// now): the interval as the trailing column on all four, in the same type the rung's ALTER adds, and the + /// two offsets as the trailing pair on query_stats — so fresh-through-V128 and upgraded-to-V128 stores are + /// shaped identically and the rung's ALTERs no-op on the former. + /// + [Fact] + public void TheGeneratedCreateTable_CarriesTheColumnsLast_OnAllFive() + { + foreach (var table in IntervalTables) + { + var definition = CollectorCatalog.Find(table); + Assert.NotNull(definition); + + Assert.Equal(IntervalColumn, definition!.PayloadColumns[^1].Name); + Assert.Equal(CollectorColumnType.Integer, definition.PayloadColumns[^1].Type); + + var ddl = PgSchemaGenerator.CreateTable(definition); + Assert.EndsWith($" {IntervalColumn} integer\n);", ddl, StringComparison.Ordinal); + } + + var queryStats = CollectorCatalog.Find("query_stats")!; + Assert.Equal(OffsetColumns[0], queryStats.PayloadColumns[^2].Name); + Assert.Equal(OffsetColumns[1], queryStats.PayloadColumns[^1].Name); + Assert.All(OffsetColumns, c => Assert.Equal(CollectorColumnType.Integer, queryStats.PayloadColumns.Single(p => p.Name == c).Type)); + Assert.EndsWith(" statement_start_offset integer,\n statement_end_offset integer\n);", PgSchemaGenerator.CreateTable(queryStats), StringComparison.Ordinal); + + /* The COPY column list is the PayloadColumns order, so the new columns ride LAST there too. */ + Assert.EndsWith(", query_plan_xml_bytes, statement_start_offset, statement_end_offset) FROM STDIN (FORMAT BINARY)", + PgCollectorRowWriter.CopyCommandFor(QueryStatsCollector.Instance), StringComparison.Ordinal); + foreach (var table in IntervalTables) + { + Assert.EndsWith($", {IntervalColumn}) FROM STDIN (FORMAT BINARY)", + PgCollectorRowWriter.CopyCommandFor(CollectorCatalog.Find(table)!), StringComparison.Ordinal); + } + } + + /// + /// The regenerated resolving view names every query_stats payload column in order — the two offsets ahead + /// of the digests, which is WHY the rung drops and recreates rather than replacing — and the pre-add guard + /// covers them, so a store replaying V51, V54 or V121 on this build (all of which re-emit the view) has + /// the columns before the view names them. + /// + [Fact] + public void TheResolvingView_NamesTheOffsets_AndEveryRungReEmittingItPreAddsThem() + { + var view = PgSchemaGenerator.GenerateQueryStatsResolvingView(); + var startAt = view.IndexOf("f.statement_start_offset", StringComparison.Ordinal); + var endAt = view.IndexOf("f.statement_end_offset", StringComparison.Ordinal); + var digestAt = view.IndexOf("f.query_text_digest", StringComparison.Ordinal); + Assert.True(startAt > 0 && endAt > startAt && digestAt > endAt, + "the offsets must be projected in order and ahead of the digest columns"); + + var preAdds = PgSchemaGenerator.GenerateQueryStatsPayloadColumnPreAdds(); + foreach (var column in OffsetColumns) + { + Assert.Contains($"ALTER TABLE query_stats ADD COLUMN IF NOT EXISTS {column} integer;", preAdds, StringComparison.Ordinal); + } + + /* Every rung that emits the RESOLVING view (V38, V51, V54, V121, V128) names f.; V4 and V14 + define the passthrough and name neither, so they are skipped on `use < 0` the way + MigrationLadderPins skips them. The resolving definers must pre-add first: presence, then order. */ + var resolvingDefiners = 0; + foreach (var rung in PgMigrations.Scripts.Where(m => m.Sql.Contains("VIEW v_query_stats AS", StringComparison.Ordinal))) + { + foreach (var column in OffsetColumns) + { + var use = rung.Sql.IndexOf($"f.{column}", StringComparison.Ordinal); + if (use < 0) + { + continue; + } + + resolvingDefiners++; + var guard = rung.Sql.IndexOf($"ADD COLUMN IF NOT EXISTS {column} ", StringComparison.Ordinal); + Assert.True(guard >= 0, $"V{rung.Version} names f.{column} and never adds it"); + Assert.True(guard < use, $"V{rung.Version} adds {column} AFTER the view uses it"); + } + } + + /* Two offsets × the five resolving definers: a scan that skipped everything would pass while asserting + nothing. */ + Assert.Equal(10, resolvingDefiners); + } + + /* ---- the probe (three sites, top arm) ------------------------------------------------------------- */ + + /// + /// The viewer probe's three sites carry this rung's sentinel, and the map treats it as the TOP arm. + /// + /// The probe asks the question, the caller reads the answer, the map has the parameter — three + /// sites, and a sentinel present at only some of them shifts every LATER ordinal onto the wrong column. + /// Miss all three and a fully-migrated store probes one rung short, so the connect-time gate refuses a + /// store that is in fact current — permanently, because no later upgrade changes the answer. The + /// sentinel shares V127's column NAME on a different TABLE, which is exactly why the table is in the + /// predicate. + /// + [Fact] + public void TheProbeMapsAFullyMigratedStoreToThisTopRung() + { + Assert.Contains( + $"table_name = 'procedure_stats'\n AND column_name = '{IntervalColumn}'", + ViewerDataService.StoreSchemaProbeSql.Replace("\r\n", "\n", StringComparison.Ordinal), StringComparison.Ordinal); + + var viewer = RepoFile.ReadRepoFile("Darling", "PerformanceMonitor.Darling.Viewer", "ViewerDataService.cs"); + Assert.Contains($"reader.GetBoolean({ProbeOrdinal})", viewer, StringComparison.Ordinal); + Assert.DoesNotContain($"reader.GetBoolean({ProbeOrdinal + 1})", viewer, StringComparison.Ordinal); + Assert.Contains("hasDeltaFamilyIntervalCompletion", viewer, StringComparison.Ordinal); + + Assert.Equal(StorageVersion.SchemaVersion, ViewerDataService.RequiredStoreSchemaVersion); + + var method = typeof(ViewerDataService) + .GetMethod("MapProbedSchemaVersion", BindingFlags.NonPublic | BindingFlags.Static)!; + var arity = method.GetParameters().Length; + + /* The top rung's sentinel IS the last argument. */ + Assert.Equal(ProbeOrdinal, arity - 1); + + /* Every sentinel true = a fully-migrated store, which must map to exactly this version. Built by + reflection so the arity tracks the signature. */ + var all = Enumerable.Repeat((object)true, arity).ToArray(); + Assert.Equal(StorageVersion.SchemaVersion, (int)method.Invoke(null, all)!); + + /* This rung's own arm answers for a store that stopped here. Expressed as "false above" rather than + as one named ordinal, so a rung landing on top of this one does not quietly turn this case into a + test of that rung. */ + var atThisRung = Enumerable.Range(0, arity).Select(i => (object)(i <= ProbeOrdinal)).ToArray(); + Assert.Equal(RungVersion, (int)method.Invoke(null, atThisRung)!); + + /* One rung behind: the same store WITHOUT this rung's sentinel reports the previous rung. */ + var behind = (object[])atThisRung.Clone(); + behind[ProbeOrdinal] = false; + Assert.Equal(PreviousVersion, (int)method.Invoke(null, behind)!); + + /* And in the source, the arm sits ABOVE the previous rung's — newest-first is the whole contract of + that method — and returns this build's version rather than a literal that could drift from it. */ + var thisArm = viewer.IndexOf("if (hasDeltaFamilyIntervalCompletion)", StringComparison.Ordinal); + var previousArm = viewer.IndexOf("if (hasDeltaFamilyIntervalColumns)", StringComparison.Ordinal); + Assert.True(thisArm >= 0, "the viewer has no V128 sentinel arm — a fully-migrated store would map one rung low"); + Assert.True(previousArm >= 0, "the previous rung's arm is gone, so this pin is comparing against nothing"); + Assert.True(thisArm < previousArm, "the V128 arm sits below the previous rung's, so a current store maps one rung low"); + Assert.Contains( + "return " + StorageVersion.SchemaVersion.ToString(CultureInfo.InvariantCulture) + ";", + viewer[thisArm..], StringComparison.Ordinal); + } + + /* ---- the writers ---------------------------------------------------------------------------------- */ + + /// + /// The four collectors now WRITE the interval — through CalculateDeltaWithInterval for every group, + /// never the bare CalculateDelta — as the minimum over each row's delta groups (the V127 rule). The + /// two SQL Server collectors take the minimum in WritePayload; the two PostgreSQL collectors compute + /// their deltas in ReadAsync (the idle-row skip needs them before the row exists), so the minimum + /// rides the Row and WritePayload writes it from there. The column without the writer would be a + /// NULL forever; the writer taking one headline group's interval would let an independently reset sibling + /// counter's 0 read as idle over a real interval. + /// + [Fact] + public void TheFourCollectors_WriteTheIntervalAsTheMinimumOverTheirDeltaGroups() + { + foreach (var (file, groups, written) in new[] + { + ("ProcedureStatsCollector.cs", 7, ".Value(sampleIntervalSeconds);"), + ("MemoryGrantsCollector.cs", 2, ".Value(sampleIntervalSeconds);"), + ("PgWaitStatsCollector.cs", 2, ".Value(row.SampleIntervalSeconds);"), + ("PgStatementStatsCollector.cs", 3, ".Value(row.SampleIntervalSeconds);"), + }) + { + var source = RepoFile.ReadRepoFile("PerformanceMonitor.Collectors", file); + + Assert.Equal(groups, CountOf(source, "context.Deltas.CalculateDeltaWithInterval(")); + Assert.DoesNotContain("context.Deltas.CalculateDelta(", source, StringComparison.Ordinal); + Assert.Contains("var sampleIntervalSeconds = Math.Min(", source, StringComparison.Ordinal); + Assert.Contains(written, source, StringComparison.Ordinal); + } + } + + /// + /// query_stats WRITES the two offsets it keys on, raw, from the same Row fields the key is built from — so + /// the stored pair and the key's pair are one value, not two that agree today. + /// + [Fact] + public void TheQueryStatsCollector_WritesTheOffsetsItKeysOn_Raw() + { + var source = RepoFile.ReadRepoFile("PerformanceMonitor.Collectors", "QueryStatsCollector.cs"); + + Assert.Contains("$\"{row.SqlHandle}:{row.StatementStartOffset}:{row.StatementEndOffset}:{row.PlanHandle}\"", source, StringComparison.Ordinal); + Assert.Contains(".Value(row.StatementStartOffset)", source, StringComparison.Ordinal); + Assert.Contains(".Value(row.StatementEndOffset);", source, StringComparison.Ordinal); + /* No arithmetic on the way to the store: the collector's SUBSTRING divides by two to SLICE the text, + and that is the only place the byte offsets are ever transformed. */ + Assert.DoesNotContain(".Value(row.StatementStartOffset / ", source, StringComparison.Ordinal); + Assert.DoesNotContain(".Value(row.StatementEndOffset / ", source, StringComparison.Ordinal); + } + + /// + /// The calculator's doc claim is now true for ALL families and says so: every member of + /// DeltaFamilyCollectors persists the interval, and the sentence names the census that pins it + /// (Lite.Tests' DeltaFamilyIntervalColumnTests, whose still-naked list is empty), so it cannot + /// silently go false again by an eleventh family shipping naked. + /// + [Fact] + public void TheCalculatorDoc_SaysEveryFamilyStoresTheInterval_AndNamesTheCensus() + { + var source = RepoFile.ReadRepoFile("PerformanceMonitor.Collectors", "CollectorDeltaCalculator.cs"); + + Assert.Contains("for EVERY delta family: all ten members of DeltaFamilyCollectors persist a", source, StringComparison.Ordinal); + Assert.Contains("procedure_stats, memory_grant_stats, pg_wait_stats and pg_statement_stats since", source, StringComparison.Ordinal); + Assert.Contains("Darling V128 / Lite v61, #3540)", source, StringComparison.Ordinal); + Assert.Contains("DeltaFamilyIntervalColumnTests is the census", source, StringComparison.Ordinal); + /* The "still persist no interval" sentence V127 wrote is gone with the list it described. */ + Assert.DoesNotContain("persist no interval", source, StringComparison.Ordinal); + Assert.DoesNotContain("naked list shrinks", source, StringComparison.Ordinal); + } + + private static int CountOf(string haystack, string needle) + { + var count = 0; + for (var at = haystack.IndexOf(needle, StringComparison.Ordinal); + at >= 0; + at = haystack.IndexOf(needle, at + needle.Length, StringComparison.Ordinal)) + { + count++; + } + + return count; + } +} diff --git a/Darling/Darling.Tests/OversizedPlanBacklogPins.cs b/Darling/Darling.Tests/OversizedPlanBacklogPins.cs index d0bd089bf..80ec51f7e 100644 --- a/Darling/Darling.Tests/OversizedPlanBacklogPins.cs +++ b/Darling/Darling.Tests/OversizedPlanBacklogPins.cs @@ -204,28 +204,36 @@ from the constants the backlog's own reads filter on rather than as fresh litera } [Fact] - public void BothCollectors_DeclareTheSizeAsTheirLastPayloadColumn() + public void BothCollectors_DeclareTheSizeAtTheOrdinalItWasAppendedAt() { - /* Appended LAST on both, which is what keeps every earlier ordinal — and therefore every existing - store column's position, the positional binary COPY and the positional DuckDB appender — stable. - BigInt because DATALENGTH over an nvarchar(max) expression returns bigint, and a narrower store - column would silently overflow on the megabyte-scale plans this exists to describe. + /* Appended LAST on both when #3392 landed, which is what keeps every earlier ordinal — and therefore + every existing store column's position, the positional binary COPY and the positional DuckDB + appender — stable. BigInt because DATALENGTH over an nvarchar(max) expression returns bigint, and + a narrower store column would silently overflow on the megabyte-scale plans this exists to + describe. + + No longer the LAST column: V128 / Lite v61 (#3540) appended behind it — the interval on + procedure_stats, the two statement offsets on query_stats — by the same append-only rule. The + claim that outlives that is the one the stores depend on: this column's ORDINAL never moved. + Pinned as the ordinal (51 and 35), which is what "appended last at #3392" means once later rungs + exist; a `names[^1]` pin here would assert #3392 is still the newest appender, which is how the + next rung's build goes red. This is the declaration half. The SELECT-ordinal-to-payload-slot agreement is driven through the real shredder in Lite.Tests' two collector-definition suites, which own the reader fakes. */ - Assert.Equal(52, QueryStatsCollector.Instance.PayloadColumns.Count); - Assert.Equal(36, ProcedureStatsCollector.Instance.PayloadColumns.Count); + Assert.Equal(54, QueryStatsCollector.Instance.PayloadColumns.Count); + Assert.Equal(37, ProcedureStatsCollector.Instance.PayloadColumns.Count); - foreach (ICollectorSchemaInfo collector in new ICollectorSchemaInfo[] + foreach (var (collector, ordinal) in new (ICollectorSchemaInfo, int)[] { - QueryStatsCollector.Instance, - ProcedureStatsCollector.Instance, + (QueryStatsCollector.Instance, 51), + (ProcedureStatsCollector.Instance, 35), }) { var names = collector.PayloadColumns.Select(c => c.Name).ToArray(); - Assert.Equal("query_plan_xml_bytes", names[^1]); - Assert.Equal(CollectorColumnType.BigInt, collector.PayloadColumns[^1].Type); + Assert.Equal("query_plan_xml_bytes", names[ordinal]); + Assert.Equal(CollectorColumnType.BigInt, collector.PayloadColumns[ordinal].Type); /* The gated content column is still there and still AHEAD of the size, which is the pair a reader tests as "measured, not captured". query_stats keeps other columns between them, so @@ -895,19 +903,23 @@ comparison alone passes when the primary arm is gone entirely. */ } [Fact] - public void TheQueryStatsFallback_KeysOnTheHash_BecauseTheFactRowCarriesNoOffsets() + public void TheQueryStatsFallback_KeysOnTheHash_TheGrainItsReadersAskAt() { - /* This is the REASON, pinned. query_stats reads the statement offsets for its delta key and never - stores them, so a join from a stored fact row could only match plan_handle + sql_handle — which - for a multi-statement plan is several backlog rows describing DIFFERENT statements' plans. Serving - one of those as "the plan for this query" is worse than serving nothing. If the offsets ever DO - become stored columns, this pin fails and the fallback can become an exact join. */ + /* This was "…BecauseTheFactRowCarriesNoOffsets" and pinned the absence of the two offset columns, + with the note that the day they became stored columns the pin would fail and the fallback could + become an exact join. V128 (#3540) stored them, the pin failed as designed, and the decision is + recorded here rather than taken silently: the fallback STAYS keyed on query_hash. Two reasons, + both in the SQL's own doc — the readers ask at the query_hash grain, which the backlog row serves + directly; and every row written before V128 carries NULL offsets, so an exact join would go dark + on an upgraded store for a raw retention's worth of history. The offsets are asserted PRESENT and + trailing so this record cannot drift back into the old claim. */ var stored = QueryStatsCollector.Instance.PayloadColumns.Select(c => c.Name).ToArray(); - Assert.DoesNotContain("statement_start_offset", stored); - Assert.DoesNotContain("statement_end_offset", stored); + Assert.Equal("statement_start_offset", stored[^2]); + Assert.Equal("statement_end_offset", stored[^1]); Assert.Contains("query_hash = $2", OversizedPlanBacklog.QueryStatsFallbackSql, StringComparison.Ordinal); Assert.DoesNotContain("plan_handle", OversizedPlanBacklog.QueryStatsFallbackSql, StringComparison.Ordinal); + Assert.DoesNotContain("statement_start_offset", OversizedPlanBacklog.QueryStatsFallbackSql, StringComparison.Ordinal); /* procedure_stats CAN join exactly, and does: its three DMVs expose no offsets at all, so the plan apply passes fixed literals and every backlog row for it carries that same pair. */ diff --git a/Darling/Darling.Tests/PayloadDimensionTests.cs b/Darling/Darling.Tests/PayloadDimensionTests.cs index 1b70aa24e..48a19ecad 100644 --- a/Darling/Darling.Tests/PayloadDimensionTests.cs +++ b/Darling/Darling.Tests/PayloadDimensionTests.cs @@ -767,10 +767,12 @@ must define it as the RESOLVING one. */ exactly this tripwire's regression in review: its first cut was a SELECT * passthrough. The shipped V51 DROPs the view (the new column lands mid-list, which CREATE OR REPLACE refuses) and re-emits the generator's resolving definition. V121 re-defines it again for - query_plan_xml_bytes, by the same DROP-then-re-emit route and for the same reason. - The literal is deliberate: a rung that redefines this view has to change this line, which - is what brings a human to the paragraph above. */ - Assert.Equal(121, definers[^1].Version); + query_plan_xml_bytes, by the same DROP-then-re-emit route and for the same reason. V128 + (#3540) re-defines it a fourth time for statement_start_offset / statement_end_offset — the + delta key's two halves — again DROP-then-re-emit, because the offsets land ahead of the digest + columns. The literal is deliberate: a rung that redefines this view has to change this line, + which is what brings a human to the paragraph above. */ + Assert.Equal(128, definers[^1].Version); Assert.Contains( "COALESCE(f.query_text, qtd.query_text) AS query_text", definers[^1].Sql, diff --git a/Darling/Darling.Tests/PgTrendReaderTests.cs b/Darling/Darling.Tests/PgTrendReaderTests.cs index 7852fc5c2..78461cae0 100644 --- a/Darling/Darling.Tests/PgTrendReaderTests.cs +++ b/Darling/Darling.Tests/PgTrendReaderTests.cs @@ -94,6 +94,36 @@ public void AnIntervalWithNoCallsHasANullMeanRatherThanZero() Assert.Contains("WHEN coalesce(calls, 0) > 0", sql, StringComparison.Ordinal); } + /// + /// #3540 (V128): the same reasoning the trend already applied to the DELTAS now applies to the INTERVAL. + /// pg_statement_stats stores sample_interval_seconds beside its deltas — the span the delta accrued + /// over, which for a collector that skips idle rows is NOT the gap between the rows it left behind. Per + /// snapshot the interval is MAX over the queryid's rows (0 only when every row was the unknowable marker), + /// 0 → NULL through NULLIF, NULL (pre-V128) → the LAG this read always used. No ELSE 0 on + /// calls_per_second: a NULL rate is dropped by the reader rather than plotted as 0.00 calls/sec at a + /// restart, and the first pre-V128 snapshot is absent rather than a fabricated 0.0. + /// + [Fact] + public void TheQueryTrendPrefersTheStoredInterval_AndNeverFabricatesZeroCallsPerSecond() + { + var sql = DarlingPgTrendReader.QueryDurationTrendSql; + + Assert.Contains("CASE WHEN MAX(sample_interval_seconds) IS NULL", sql, StringComparison.Ordinal); + Assert.Contains("THEN extract(epoch FROM (collection_time - LAG(collection_time) OVER (ORDER BY collection_time)))", sql, StringComparison.Ordinal); + Assert.Contains("ELSE NULLIF(MAX(sample_interval_seconds), 0)", sql, StringComparison.Ordinal); + Assert.DoesNotContain("ELSE 0", sql, StringComparison.Ordinal); + Assert.Contains("WHEN interval_seconds > 0", sql, StringComparison.Ordinal); + Assert.Contains("/ interval_seconds", sql, StringComparison.Ordinal); + + /* The reader drops the NULL-rate point; it does not read it as 0. */ + var source = RepoFile.ReadRepoFile("Darling", "PerformanceMonitor.Darling.Storage", "DarlingPgTrendReader.cs"); + var reader = source[source.IndexOf("GetQueryDurationTrendAsync(", StringComparison.Ordinal)..]; + reader = reader[..reader.IndexOf("return points;", StringComparison.Ordinal)]; + Assert.Contains("if (reader.IsDBNull(4))", reader, StringComparison.Ordinal); + Assert.Contains("continue;", reader, StringComparison.Ordinal); + Assert.DoesNotContain("reader.IsDBNull(4) ? 0", reader, StringComparison.Ordinal); + } + /// /// The automatic choice must skip the CPU class. pg_wait_sampling's Running means the /// backend was NOT waiting and dominates any healthy server's profile — measured on the rig it grew by diff --git a/Darling/Darling.Tests/StartupCommandTimeoutTests.cs b/Darling/Darling.Tests/StartupCommandTimeoutTests.cs index 488e0a1e8..74661e816 100644 --- a/Darling/Darling.Tests/StartupCommandTimeoutTests.cs +++ b/Darling/Darling.Tests/StartupCommandTimeoutTests.cs @@ -130,7 +130,7 @@ private static readonly (string File, string Member, int Bootstrap, int ConnectP ("DarlingDeltaCalculator.cs", "SeedLatchStatsAsync", 1, 0, 0, 0), ("DarlingDeltaCalculator.cs", "SeedSpinlockStatsAsync", 1, 0, 0, 0), ("DarlingDeltaCalculator.cs", "SeedProcedureStatsAsync", 1, 0, 0, 0), - ("DarlingDeltaCalculator.cs", "SeedQueryStatsPassesAsync", 1, 0, 0, 0), + ("DarlingDeltaCalculator.cs", "SeedQueryStatsAsync", 1, 0, 0, 0), ("DarlingDeltaCalculator.cs", "SeedPgWaitStatsAsync", 1, 0, 0, 0), ("DarlingDeltaCalculator.cs", "SeedPgStatementStatsAsync", 1, 0, 0, 0), ("StoreConfigProvider.cs", "WarnAboutFileOnlyServersAsync", 1, 0, 0, 0), diff --git a/Darling/Darling.Tests/ViewerHistoryWindowTests.cs b/Darling/Darling.Tests/ViewerHistoryWindowTests.cs index 34bb065fa..5629852d3 100644 --- a/Darling/Darling.Tests/ViewerHistoryWindowTests.cs +++ b/Darling/Darling.Tests/ViewerHistoryWindowTests.cs @@ -46,7 +46,7 @@ public void QueryStatsHistorySql_FiltersOneQueryHashOverTheWindow_OrderedByTime( } [Fact] - public void ProcStatsHistorySql_FiltersOneSchemaObjectOverTheWindow_AndDerivesTheInterval() + public void ProcStatsHistorySql_FiltersOneSchemaObjectOverTheWindow_AndPrefersTheStoredInterval() { var sql = ViewerDataService.ProcStatsHistorySql; Assert.Contains("FROM procedure_stats", sql, StringComparison.Ordinal); @@ -57,7 +57,12 @@ public void ProcStatsHistorySql_FiltersOneSchemaObjectOverTheWindow_AndDerivesTh Assert.Contains("collection_time >= $5", sql, StringComparison.Ordinal); Assert.Contains("collection_time <= $6", sql, StringComparison.Ordinal); Assert.Contains("ORDER BY collection_time", sql, StringComparison.Ordinal); - /* procedure_stats has no sample_interval_seconds column — it's derived from the previous row's gap. */ + /* #3540 (V128): procedure_stats carries sample_interval_seconds now. The STORED value is shown where + the row has one — a 0 included, the Interval (sec) 0 the query-stats history grid has always shown + for an unknowable row — and a pre-V128 row (NULL) keeps the interval this read always derived from + the previous row's gap. COALESCE, not CASE: a displayed interval is not a rate, so the stored 0 + stays a 0 here rather than becoming NULL. */ + Assert.Contains("COALESCE(sample_interval_seconds, CAST(extract(epoch FROM", sql, StringComparison.Ordinal); Assert.Contains("LAG(collection_time)", sql, StringComparison.Ordinal); Assert.Contains("total_spills", sql, StringComparison.Ordinal); AssertPgPositionalDialect(sql); diff --git a/Darling/Darling.Tests/ViewerQueriesRestTests.cs b/Darling/Darling.Tests/ViewerQueriesRestTests.cs index 73090f708..b10372c1c 100644 --- a/Darling/Darling.Tests/ViewerQueriesRestTests.cs +++ b/Darling/Darling.Tests/ViewerQueriesRestTests.cs @@ -53,6 +53,33 @@ public void TrendSql_ComputesPerSecondRate_ViaLagInterval_BaseTable(string sqlNa Assert.Contains("ORDER BY collection_time", sql, StringComparison.Ordinal); } + /// + /// #3540 (V128): the procedure trend reads the collection's STORED interval — MAX over the collection's + /// rows, 0 → NULL through NULLIF so a restart's marker collection drops rather than plotting 0.00 — and + /// falls back to the LAG derivation only for a pre-V128 collection (NULL). No ELSE 0 anywhere in it: the + /// rate is NULL when the interval is unknowable or absent and the reader drops the point. Its + /// query-stats sibling deliberately keeps the LAG-only form (the A11a residual, reported not rewritten). + /// + [Fact] + public void ProcedureDurationTrendSql_PrefersTheStoredInterval_AndNeverFabricatesZero() + { + var sql = ViewerDataService.ProcedureDurationTrendSql; + Assert.Contains("CASE WHEN MAX(sample_interval_seconds) IS NULL", sql, StringComparison.Ordinal); + Assert.Contains("ELSE NULLIF(MAX(sample_interval_seconds), 0)", sql, StringComparison.Ordinal); + Assert.Contains("THEN extract(epoch FROM (date_trunc('second', collection_time) - date_trunc('second', LAG(collection_time) OVER (ORDER BY collection_time))))", sql, StringComparison.Ordinal); + Assert.DoesNotContain("ELSE 0", sql, StringComparison.Ordinal); + Assert.Contains("CASE WHEN interval_seconds > 0 THEN total_elapsed_ms / interval_seconds END AS elapsed_ms_per_second", sql, StringComparison.Ordinal); + Assert.Contains("CASE WHEN interval_seconds > 0 THEN CAST(total_executions AS DOUBLE PRECISION) / interval_seconds END AS executions_per_second", sql, StringComparison.Ordinal); + + /* And the shared reader DROPS a NULL-rate row rather than reading it as 0 — the C# half of the idiom. */ + var source = RepoFile.ReadRepoFile("Darling", "PerformanceMonitor.Darling.Viewer", "ViewerDataService.QueryTrends.cs"); + var reader = source[source.IndexOf("private async Task> ReadDurationTrendAsync(", StringComparison.Ordinal)..]; + reader = reader[..reader.IndexOf("return items;", StringComparison.Ordinal)]; + Assert.Contains("if (reader.IsDBNull(1))", reader, StringComparison.Ordinal); + Assert.Contains("continue;", reader, StringComparison.Ordinal); + Assert.DoesNotContain("reader.IsDBNull(1) ? 0", reader, StringComparison.Ordinal); + } + [Fact] public void DurationTrendSql_SumsElapsedMs_ExecutionTrendSql_OnlyExecutions() { diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingDeltaCalculator.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingDeltaCalculator.cs index d0ea495b7..11e1fc888 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingDeltaCalculator.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingDeltaCalculator.cs @@ -47,11 +47,12 @@ throws away anything older than it anyway — see CollectorDeltaCalculator.SeedL - Latest collection per server (the original four, latch_stats, spinlock_stats): these collectors write every key they read on every pass, so the newest collection holds every key's current counter and the row-value probe is the cheapest exact read. - - Latest row per key (procedure_stats, pg_wait_stats, pg_statement_stats): these collectors do - NOT write every key every pass — procedure_stats is a TOP (150) that churns, and the two - PostgreSQL collectors skip idle rows at the write — so the newest collection is missing keys - whose counters are nevertheless unchanged, and a key seeded from nothing takes the - first-sighting path. DISTINCT ON (server_id, key) ... ORDER BY collection_time DESC over the + - Latest row per key (procedure_stats, query_stats, pg_wait_stats, pg_statement_stats): these + collectors do NOT write every key every pass — procedure_stats and query_stats are TOP (n) + reads that churn, and the two PostgreSQL collectors skip idle rows at the write — so the newest + collection is missing keys whose counters are nevertheless unchanged, and a key seeded from + nothing takes the first-sighting path. DISTINCT ON (server_id, key) ... ORDER BY + collection_time DESC over the cutoff window returns each key's newest row instead. Its bound is the single collection_time >= $1 on its only table read: there is no inner aggregate to bind a second time. On the hypertable that bound is what keeps the read to the window's chunk(s) through @@ -63,13 +64,18 @@ cutoff window returns each key's newest row instead. Its bound is the single measured; the value is exact (the counter was idle in between, which is why no newer row exists) and the gap policy still bounds the span. - query_stats has NO key seed on either host, and the reason is stated rather than left as a gap: - its delta key is sql_handle:statement_start_offset:statement_end_offset:plan_handle and the - store persists neither offset, so no row in query_stats can reproduce the key the collector - will present, and a seed under any other key seeds nothing. Its PASS WINDOW is seeded below - (QueryStatsPassSeedSql), which is what the #2235 series-age rescue reads — so on the first - post-restart pass a plan that compiled since the last pre-restart pass is credited in full even - though older plans baseline. Persisting the offsets is a rung, tracked on #3540. */ + query_stats joined the per-key shape with Darling V128 / Lite v61 (#3540). Its delta key is + sql_handle:statement_start_offset:statement_end_offset:plan_handle, and until V128 the store + persisted neither offset, so no row could reproduce the key the collector presents and only the + family's PASS WINDOW could be seeded (the #2235 series-age rescue's input). The offsets are + stored now, raw (-1 = "to the end of the batch", byte offsets into the nvarchar batch text) and + the seed rebuilds the key from them with the collector's own interpolation. Two rules, both in + the seeder rather than the SQL: a row whose offsets are NULL — every row written before V128 — + seeds NO key, because a key built from a fabricated 0/-1 would be one nothing ever presents and + the baseline under it would sit unread until it aged out; and EVERY row, NULL offsets or not, + still feeds the pass window, so the first restart after the upgrade (when the whole window is + pre-V128 rows) keeps the rescue armed exactly as #3614 left it. One read serves both, which is + why the offset filter is not in the WHERE. */ public const string WaitStatsSeedSql = @" SELECT server_id, wait_type, waiting_tasks_count, wait_time_ms, signal_wait_time_ms, collection_time @@ -154,14 +160,20 @@ FROM procedure_stats ) AS recent ORDER BY server_id, delta_key, collection_time DESC"; - /* The pass window only (see the header): every distinct collection time per server inside the - cutoff, which is a handful of rows per server from the (server_id, collection_time) index. - GROUP BY rather than DISTINCT so the shape reads as the aggregate it is. */ - public const string QueryStatsPassSeedSql = @" -SELECT server_id, collection_time + /* QueryStatsCollector keys on the dm_exec_query_stats row identity — sql_handle, both statement + offsets, plan_handle — and its TOP (n) churns like procedure_stats', so: latest row per that + identity. DISTINCT ON treats NULL offsets as one group, which is harmless: those are pre-V128 rows + the seeder reads for the pass window only (see the header). The eight counters are the ones the + collector's eight CalculateDeltaWithSeriesAge calls difference. */ + public const string QueryStatsSeedSql = @" +SELECT DISTINCT ON (server_id, sql_handle, statement_start_offset, statement_end_offset, plan_handle) + server_id, sql_handle, statement_start_offset, statement_end_offset, plan_handle, + execution_count, total_worker_time, total_elapsed_time, + total_logical_reads, total_logical_writes, total_physical_reads, total_rows, total_spills, + collection_time FROM query_stats WHERE collection_time >= $1 -GROUP BY server_id, collection_time"; +ORDER BY server_id, sql_handle, statement_start_offset, statement_end_offset, plan_handle, collection_time DESC"; /* PgWaitStatsCollector keys on the numeric wait_event_id (never the name — it changes case across Aurora majors) and skips idle rows at the write, so: latest row per (server, event id). */ @@ -242,7 +254,7 @@ public async Task SeedFromStoreAsync(NpgsqlDataSource postgres, ILogger? logger, await SeedFamilyAsync("latch_stats", () => SeedLatchStatsAsync(connection, cutoff, logger, cancellationToken), logger); await SeedFamilyAsync("spinlock_stats", () => SeedSpinlockStatsAsync(connection, cutoff, logger, cancellationToken), logger); await SeedFamilyAsync("procedure_stats", () => SeedProcedureStatsAsync(connection, cutoff, logger, cancellationToken), logger); - await SeedFamilyAsync("query_stats", () => SeedQueryStatsPassesAsync(connection, cutoff, logger, cancellationToken), logger); + await SeedFamilyAsync("query_stats", () => SeedQueryStatsAsync(connection, cutoff, logger, cancellationToken), logger); await SeedFamilyAsync("pg_wait_stats", () => SeedPgWaitStatsAsync(connection, cutoff, logger, cancellationToken), logger); await SeedFamilyAsync("pg_statement_stats", () => SeedPgStatementStatsAsync(connection, cutoff, logger, cancellationToken), logger); @@ -438,19 +450,50 @@ private async Task SeedProcedureStatsAsync(NpgsqlConnection connection, DateTime if (count > 0) logger?.LogDebug("Seeded {Count} procedure_stats baseline rows", count); } - /* Pass window only — see the header for why query_stats has no key seed. */ - private async Task SeedQueryStatsPassesAsync(NpgsqlConnection connection, DateTime cutoff, ILogger? logger, CancellationToken cancellationToken) + private async Task SeedQueryStatsAsync(NpgsqlConnection connection, DateTime cutoff, ILogger? logger, CancellationToken cancellationToken) { - using var cmd = new NpgsqlCommand(QueryStatsPassSeedSql, connection) { CommandTimeout = ServiceCommandDeadlines.BootstrapSeconds }; + using var cmd = new NpgsqlCommand(QueryStatsSeedSql, connection) { CommandTimeout = ServiceCommandDeadlines.BootstrapSeconds }; cmd.Parameters.AddWithValue(cutoff); using var reader = await cmd.ExecuteReaderAsync(cancellationToken); + var count = 0; + var preV128 = 0; var passes = new SeedPassTracker(); while (await reader.ReadAsync(cancellationToken)) { - passes.Observe(reader.GetInt32(0), reader.IsDBNull(1) ? (DateTime?)null : reader.GetDateTime(1)); + var serverId = reader.GetInt32(0); + var ts = reader.IsDBNull(13) ? (DateTime?)null : reader.GetDateTime(13); + + /* The pass window takes EVERY row, offsets or not — see the header. */ + passes.Observe(serverId, ts); + + /* A pre-V128 row never recorded its offsets. Its key cannot be rebuilt, and a key built from a + guessed pair would be a baseline nothing ever reads — so it seeds nothing. */ + if (reader.IsDBNull(2) || reader.IsDBNull(3)) + { + preV128++; + continue; + } + + /* The key exactly as QueryStatsCollector.WritePayload spells it: the same interpolation over + the same raw parts, so a null handle formats as empty and the offsets — -1 included — are + spelled by the same int formatting on both sides. Never normalized. */ + var sqlHandle = reader.IsDBNull(1) ? null : reader.GetString(1); + var planHandle = reader.IsDBNull(4) ? null : reader.GetString(4); + var deltaKey = $"{sqlHandle}:{reader.GetInt32(2)}:{reader.GetInt32(3)}:{planHandle}"; + + Seed(serverId, "query_stats_exec", deltaKey, reader.IsDBNull(5) ? 0 : reader.GetInt64(5), ts); + Seed(serverId, "query_stats_worker", deltaKey, reader.IsDBNull(6) ? 0 : reader.GetInt64(6), ts); + Seed(serverId, "query_stats_elapsed", deltaKey, reader.IsDBNull(7) ? 0 : reader.GetInt64(7), ts); + Seed(serverId, "query_stats_reads", deltaKey, reader.IsDBNull(8) ? 0 : reader.GetInt64(8), ts); + Seed(serverId, "query_stats_writes", deltaKey, reader.IsDBNull(9) ? 0 : reader.GetInt64(9), ts); + Seed(serverId, "query_stats_phys_reads", deltaKey, reader.IsDBNull(10) ? 0 : reader.GetInt64(10), ts); + Seed(serverId, "query_stats_rows", deltaKey, reader.IsDBNull(11) ? 0 : reader.GetInt64(11), ts); + Seed(serverId, "query_stats_spills", deltaKey, reader.IsDBNull(12) ? 0 : reader.GetInt64(12), ts); + count++; } SeedPasses(passes, QueryStatsGroups); - if (passes.Count > 0) logger?.LogDebug("Seeded the query_stats pass window for {Count} servers", passes.Count); + if (count > 0) logger?.LogDebug("Seeded {Count} query_stats baseline rows", count); + if (preV128 > 0) logger?.LogDebug("Skipped {Count} query_stats rows with no stored statement offsets (pre-V128); their collection times still seeded the pass window", preV128); } private async Task SeedPgWaitStatsAsync(NpgsqlConnection connection, DateTime cutoff, ILogger? logger, CancellationToken cancellationToken) diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpMemoryGrantTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpMemoryGrantTools.cs index f40694756..0076e0920 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpMemoryGrantTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpMemoryGrantTools.cs @@ -30,7 +30,7 @@ namespace PerformanceMonitor.Darling.Service.Mcp; [McpServerToolType] public sealed class DarlingMcpMemoryGrantTools { - [McpServerTool(Name = "get_resource_semaphore"), Description("Gets resource semaphore statistics showing granted vs available workspace memory against the target/max-target ceiling, waiter counts, and timeout/forced grant pressure indicators. High waiter counts or rising timeout/forced deltas indicate memory grant pressure affecting query performance.")] + [McpServerTool(Name = "get_resource_semaphore"), Description("Gets resource semaphore statistics showing granted vs available workspace memory against the target/max-target ceiling, waiter counts, and timeout/forced grant pressure indicators. High waiter counts or rising timeout/forced deltas indicate memory grant pressure affecting query performance. sample_interval_seconds is the measured seconds the two deltas accrued over; it is null with interval_known false when the row is a restart marker (no delta was knowable, so the zero deltas beside it are not 'no timeouts') or predates the column.")] public static async Task GetResourceSemaphore( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -68,7 +68,14 @@ public static async Task GetResourceSemaphore( timeout_error_count = r.TimeoutErrorCount, forced_grant_count = r.ForcedGrantCount, timeout_error_count_delta = r.TimeoutErrorCountDelta, - forced_grant_count_delta = r.ForcedGrantCountDelta + forced_grant_count_delta = r.ForcedGrantCountDelta, + /* #3540 (V128): the interval the deltas accrued over, the way the perfmon and file-I/O tools + hand it over. A stored 0 is the calculator's no-delta-knowable marker (a restart, not a + quiet semaphore) and is reported as null rather than 0 — 0 seconds is not a measurement; + a pre-V128 row that never recorded one is null too. interval_known states the one thing + both nulls have in common: the two *_delta zeros beside them are not "none this interval". */ + sample_interval_seconds = r.SampleIntervalSeconds is > 0 ? r.SampleIntervalSeconds : null, + interval_known = r.SampleIntervalSeconds is > 0 }); return JsonSerializer.Serialize(new diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMemoryGrantReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMemoryGrantReader.cs index 093ec0d0c..8d8d53aff 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMemoryGrantReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMemoryGrantReader.cs @@ -25,8 +25,11 @@ namespace PerformanceMonitor.Darling.Service.Mcp; /// get_resource_semaphore is the semaphore/ceiling lens: one row per (resource_semaphore_id, pool_id) with the /// full workspace-memory sizing — target / max_target (the hard ceiling) / total / /// available / granted / used plus grantee/waiter/timeout/forced counts and deltas. It -/// carries max_target_memory_mb (present in the store, the Dashboard tool omitted it) and drops the -/// Dashboard's sample_interval_seconds (Darling's delta collector stores no sample interval). get_memory_grants +/// carries max_target_memory_mb (present in the store, the Dashboard tool omitted it) and, since V128 +/// (#3540), the Dashboard's sample_interval_seconds too — the measured seconds the two deltas accrued +/// over, which this tool dropped while the collector stored none; 0 is the calculator's "no delta +/// knowable" marker (a restart, not a quiet semaphore) and the tool reports it as null, and a pre-V128 +/// row that never recorded one reads as null with interval_known = false. get_memory_grants /// is Lite's pool-detail lens: the sizing + activity SUMMED per pool (Lite's GetMemoryGrantChartDataAsync /// shape). Every SQL string is a public const so Darling.Tests can pin the dialect + columns without a live Postgres. /// @@ -36,11 +39,19 @@ internal static class DarlingMemoryGrantReader /* ─────────────────────────── result rows ─────────────────────────── */ /// One resource semaphore at the latest snapshot — the workspace-memory ceiling lens. + /// #3540 (V128): the measured seconds the two deltas accrued over; + /// 0 is the calculator's "no delta knowable" marker (first sighting, counter reset, a gap past the + /// policy — a restart, not a quiet semaphore); null is a pre-V128 row that never recorded one. public sealed record ResourceSemaphoreRow( DateTime CollectionTime, short ResourceSemaphoreId, int PoolId, double TargetMemoryMb, double MaxTargetMemoryMb, double TotalMemoryMb, double AvailableMemoryMb, double GrantedMemoryMb, double UsedMemoryMb, int GranteeCount, int WaiterCount, long TimeoutErrorCount, long ForcedGrantCount, - long TimeoutErrorCountDelta, long ForcedGrantCountDelta); + long TimeoutErrorCountDelta, long ForcedGrantCountDelta, int? SampleIntervalSeconds) + { + /// True when the row's deltas are the calculator's (0, 0) marker: no delta was knowable, so + /// the two *_delta zeros beside it are not "no timeouts this interval". + public bool IsUnknowable => SampleIntervalSeconds == 0; + } /// One resource pool at the latest snapshot (summed across its semaphores) — Lite's grant lens. public sealed record MemoryGrantRow( @@ -51,9 +62,9 @@ public sealed record MemoryGrantRow( /// /// The latest snapshot's per-(semaphore, pool) rows — the Dashboard's get_resource_semaphore shape - /// over Darling's store (plus max_target_memory_mb from the store; minus the Dashboard's - /// unstored sample_interval_seconds). MB columns are numeric(18,2) → double precision. - /// $1 server_id, $2 window start, $3 window end (naive UTC). + /// over Darling's store (plus max_target_memory_mb from the store; plus, since V128, the + /// Dashboard's sample_interval_seconds, trailing so every existing ordinal is stable). MB columns + /// are numeric(18,2) → double precision. $1 server_id, $2 window start, $3 window end (naive UTC). /// public const string ResourceSemaphoreLatestSql = """ WITH latest AS @@ -79,7 +90,8 @@ FROM v_memory_grant_stats timeout_error_count, forced_grant_count, timeout_error_count_delta, - forced_grant_count_delta + forced_grant_count_delta, + sample_interval_seconds FROM v_memory_grant_stats WHERE server_id = $1 AND collection_time = (SELECT mx FROM latest) @@ -111,7 +123,10 @@ public static async Task> GetResourceSemaphoreLatestA reader.IsDBNull(11) ? 0 : reader.GetInt64(11), reader.IsDBNull(12) ? 0 : reader.GetInt64(12), reader.IsDBNull(13) ? 0 : reader.GetInt64(13), - reader.IsDBNull(14) ? 0 : reader.GetInt64(14))); + reader.IsDBNull(14) ? 0 : reader.GetInt64(14), + /* NULL stays NULL: a pre-V128 row never recorded its interval, and that is a different + statement from the 0 the calculator writes when no delta was knowable. */ + reader.IsDBNull(15) ? null : reader.GetInt32(15))); } return rows; diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingTrendReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingTrendReader.cs index 56fc9a14e..5b1b8689f 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingTrendReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingTrendReader.cs @@ -594,6 +594,13 @@ public static Task GetQueryDurationTrendAsync( /// shows up smeared across however many statements it runs. This charges the whole call to the /// procedure. When both are available, the pair answers "did ad-hoc SQL regress, or did a procedure?" - /// which one series alone never can. $1 server_id, $2/$3 window (naive UTC). + /// #3540 (V128): the interval is the collection's STORED one where the rows have it — MAX + /// over the collection's rows, because a plan first seen in an otherwise steady pass carries 0 beside + /// its siblings' real interval and contributes 0 to the sums; MAX is 0 only when EVERY row was + /// unknowable (a restart), and that 0 becomes NULL through NULLIF so the rates are NULL and the + /// reader drops the point rather than rendering 0.00 ms/sec. NULL (a pre-V128 collection) falls back to + /// the LAG this read always used. No ELSE 0. Verbatim from the viewer's copy apart from the + /// database filter, as before. /// public const string ProcedureDurationTrendSql = """ WITH raw AS @@ -602,7 +609,10 @@ WITH raw AS collection_time, SUM(delta_elapsed_time) / 1000.0 AS total_elapsed_ms, SUM(delta_execution_count) AS total_executions, - extract(epoch FROM (date_trunc('second', collection_time) - date_trunc('second', LAG(collection_time) OVER (ORDER BY collection_time)))) AS interval_seconds + CASE WHEN MAX(sample_interval_seconds) IS NULL + THEN extract(epoch FROM (date_trunc('second', collection_time) - date_trunc('second', LAG(collection_time) OVER (ORDER BY collection_time)))) + ELSE NULLIF(MAX(sample_interval_seconds), 0) + END AS interval_seconds FROM procedure_stats WHERE server_id = $1 AND collection_time >= $2 @@ -611,8 +621,8 @@ GROUP BY collection_time ) SELECT collection_time, - CASE WHEN interval_seconds > 0 THEN total_elapsed_ms / interval_seconds ELSE 0 END AS elapsed_ms_per_second, - CASE WHEN interval_seconds > 0 THEN CAST(total_executions AS DOUBLE PRECISION) / interval_seconds ELSE 0 END AS executions_per_second + CASE WHEN interval_seconds > 0 THEN total_elapsed_ms / interval_seconds END AS elapsed_ms_per_second, + CASE WHEN interval_seconds > 0 THEN CAST(total_executions AS DOUBLE PRECISION) / interval_seconds END AS executions_per_second FROM raw ORDER BY collection_time """; @@ -796,10 +806,18 @@ private static async Task> ReadDurationPointsAsync await using var reader = await command.ExecuteReaderAsync(cancellationToken); while (await reader.ReadAsync(cancellationToken)) { + /* A NULL rate is an unknowable interval (#3540, V128): the point is dropped, not read as 0. Only + the procedure trend emits one today (its query-stats and Query Store siblings keep ELSE 0), so + this is a no-op for them and the missing-sample posture for it. */ + if (reader.IsDBNull(1)) + { + continue; + } + var executionsPerSecond = reader.IsDBNull(2) ? 0 : Convert.ToDouble(reader.GetValue(2)); items.Add(new QueryDurationTrendPoint( reader.GetDateTime(0), - reader.IsDBNull(1) ? 0 : Convert.ToDouble(reader.GetValue(1)), + Convert.ToDouble(reader.GetValue(1)), (long)executionsPerSecond, executionsPerSecond)); } diff --git a/Darling/PerformanceMonitor.Darling.Storage/DarlingPgTrendReader.cs b/Darling/PerformanceMonitor.Darling.Storage/DarlingPgTrendReader.cs index 600c92624..f88ea8229 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/DarlingPgTrendReader.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/DarlingPgTrendReader.cs @@ -107,6 +107,19 @@ ORDER BY collection_time /// cumulative ones again. They are written on collection, where the previous reading is in hand, and /// re-deriving them here would give a DIFFERENT answer whenever a snapshot is missing — the collector's /// delta spans the gap it actually observed, and a LAG here would span the gap in the stored data. + /// + /// The same reasoning now applies to the INTERVAL (#3540, V128). The collector stores + /// sample_interval_seconds beside the deltas — the span the delta actually accrued over, which + /// is not the gap between stored snapshots either: this collector skips idle rows at the write, so for + /// a statement that ran, went quiet for three passes and ran again, the stored delta spans four + /// intervals and a LAG over the rows it left behind spans one. Per snapshot the interval is MAX + /// over the queryid's rows (one per database/user/toplevel entry) — an entry first seen in an + /// otherwise steady pass carries 0 and contributes 0 to the sums, so MAX is 0 only when every row was + /// unknowable (a restart or a pg_stat_statements_reset()), and that 0 becomes NULL through + /// NULLIF; a NULL rate drops the point in the reader rather than plotting 0.00 calls/sec. NULL + /// (a pre-V128 row that never recorded one) falls back to the LAG this read always used, so history + /// renders as it did. No ELSE 0: the first snapshot of a pre-V128 series is absent rather than + /// a fabricated 0.0. /// public const string QueryDurationTrendSql = """ WITH per_snapshot AS ( @@ -114,7 +127,10 @@ WITH per_snapshot AS ( collection_time, SUM(delta_calls) AS calls, SUM(delta_total_exec_time_ms) AS total_exec_ms, - extract(epoch FROM (collection_time - LAG(collection_time) OVER (ORDER BY collection_time))) AS interval_seconds + CASE WHEN MAX(sample_interval_seconds) IS NULL + THEN extract(epoch FROM (collection_time - LAG(collection_time) OVER (ORDER BY collection_time))) + ELSE NULLIF(MAX(sample_interval_seconds), 0) + END AS interval_seconds FROM pg_statement_stats WHERE server_id = $1 AND queryid = $2 @@ -133,9 +149,9 @@ CASE WHEN coalesce(calls, 0) > 0 THEN coalesce(total_exec_ms, 0)::double precision / calls ELSE NULL END AS mean_exec_ms, + /* NULL, not 0, when the interval is unknowable or absent — the reader drops the point. */ CASE WHEN interval_seconds > 0 THEN coalesce(calls, 0)::double precision / interval_seconds - ELSE 0 END AS calls_per_second FROM per_snapshot ORDER BY collection_time @@ -266,12 +282,21 @@ public static async Task> GetQueryDurationTrendA await using var reader = await command.ExecuteReaderAsync(cancellationToken); while (await reader.ReadAsync(cancellationToken)) { + /* A NULL rate is an unknowable interval (#3540, V128) — every row of the snapshot stored 0, a + restart or a stats reset — or a pre-V128 first snapshot with nothing to LAG against. The + point is dropped: its calls and time are the calculator's fabricated zeros, and a point that + says "0 calls, 0 ms" at the moment of a restart is the lie this column exists to stop. */ + if (reader.IsDBNull(4)) + { + continue; + } + points.Add(new PgQueryDurationTrendPoint( reader.GetDateTime(0), reader.IsDBNull(1) ? 0 : reader.GetInt64(1), reader.IsDBNull(2) ? 0 : reader.GetDouble(2), reader.IsDBNull(3) ? 0 : reader.GetDouble(3), - reader.IsDBNull(4) ? 0 : reader.GetDouble(4))); + reader.GetDouble(4))); } return points; diff --git a/Darling/PerformanceMonitor.Darling.Storage/OversizedPlanBacklog.cs b/Darling/PerformanceMonitor.Darling.Storage/OversizedPlanBacklog.cs index 270c12b33..03a299bdb 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/OversizedPlanBacklog.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/OversizedPlanBacklog.cs @@ -230,12 +230,17 @@ UPDATE collect.oversized_plan_backlog /// (server, query_hash) — optionally narrowed to one database, the shape both the MCP read (database /// optional) and the viewer's (database required) need from one statement. /// - /// Read from the backlog, not joined to the fact table. query_stats does not store - /// the statement offsets — they are read for the delta key and never persisted — so a join from a stored - /// fact row could only match on plan_handle + sql_handle, which for a multi-statement plan - /// is several backlog rows describing DIFFERENT statements' plans. Serving one of those as "the plan for - /// this query" is worse than serving nothing. query_hash on the row keys the fallback at exactly - /// the grain the readers already ask at. + /// Read from the backlog, not joined to the fact table. When this was written + /// query_stats did not store the statement offsets — they were read for the delta key and never + /// persisted — so a join from a stored fact row could only match on plan_handle + sql_handle, + /// which for a multi-statement plan is several backlog rows describing DIFFERENT statements' plans. + /// Serving one of those as "the plan for this query" is worse than serving nothing. V128 (#3540) stores + /// the offsets, so an exact join is POSSIBLE for rows written since; it is not taken here because the + /// readers ask at the query_hash grain (the Dashboard's and viewer's key), which + /// query_hash on the backlog row already serves, and because every row written before V128 + /// carries NULL offsets, so an exact join would go dark on an upgraded store for a raw retention's + /// worth of history. If a reader ever asks at the statement grain, the join is now available to + /// it. /// /// A non-null plan_xml is itself the "this row was capped" test the caller would otherwise /// make against query_plan_xml_bytes: the row exists only because the measurement exceeded the diff --git a/Darling/PerformanceMonitor.Darling.Storage/PgMigrations.cs b/Darling/PerformanceMonitor.Darling.Storage/PgMigrations.cs index 4028d4ea4..4dbec7958 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/PgMigrations.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/PgMigrations.cs @@ -200,6 +200,14 @@ Generated from the collector definition rather than by threading a specific late new Migration(125, "collector-database-scope", V125Sql), new Migration(126, "self-disk-warn-gb-floor", V126Sql), new Migration(127, "delta-family-interval-columns", V127Sql), + /* V128 re-emits the payload-resolving v_query_stats (its two new query_stats columns land mid-list, + ahead of the digests, which CREATE OR REPLACE VIEW refuses), so it rides the V121 idiom: V54's + gz pre-add, then every payload column's pre-add, then the regenerated view. MigrationLadderPins + holds the ordering. */ + new Migration(128, "delta-family-interval-completion", + V128Sql + "\n" + V54Sql + "\n" + + PgSchemaGenerator.GenerateQueryStatsPayloadColumnPreAdds() + "\n" + + PgSchemaGenerator.GenerateQueryStatsResolvingView()), }; /// @@ -745,6 +753,107 @@ ALTER TABLE collect.spinlock_stats CREATE OR REPLACE VIEW collect.v_latch_stats AS SELECT * FROM collect.latch_stats; CREATE OR REPLACE VIEW collect.v_spinlock_stats AS SELECT * FROM collect.spinlock_stats;"; + /// + /// V128 — the completion of V127 (#3540): sample_interval_seconds on the four delta families + /// V127 left naked — procedure_stats, memory_grant_stats, pg_wait_stats, + /// pg_statement_stats — and the two statement offsets on query_stats that its delta key + /// is made of. After this rung EVERY member of CollectorDeltaCalculator.DeltaFamilyCollectors + /// stores the interval its deltas accrued over, so the calculator's (delta 0, interval 0) "no delta + /// knowable" marker reaches the store from every family and no restart zero reads as a measurement + /// anywhere; Lite.Tests' DeltaFamilyIntervalColumnTests census asserts the set with nothing + /// left on its still-naked list. Same integer type as the six that already carry it, so the + /// one NULLIF(sample_interval_seconds, 0) idiom reads all ten. Pinned by + /// DeltaFamilyIntervalCompletionRungTests. + /// + /// Why five tables in one rung. The repo allows one un-landed rung at a time, and these + /// five changes are one change: every column here exists so the same reader idiom can be applied + /// uniformly (stored interval → NULLIF(…, 0); NULL → the LAG derivation; no ELSE 0), + /// and the offsets exist so the restart seed can rebuild the one delta key the store could not + /// reproduce. Five rungs would have been five fleet schema hops carrying one idea. + /// + /// Nullable, no DEFAULT, no backfill on all six columns, matching V127 and every + /// column-adding rung on a collector table (V80, V81, V121): a historical row never recorded its + /// interval or its offsets, so NULL is the honest value. A backfilled 0 interval would stamp every + /// pre-V128 row "unknowable" and blank 30 days of rate history; a backfilled 0/-1 offset pair would + /// build a delta key nothing will ever present, and the seed would restore baselines under it + /// silently. Readers treat the three interval states distinctly: n > 0 measured; 0 + /// the unknowable marker, mapped to NULL (the point is absent, never 0.00); NULL a pre-V128 + /// row, falling back to the LAG-over-collection_time derivation those readers always used. The seed + /// consumes only rows whose offsets are NOT NULL for keys, and every row for the pass window. A + /// nullable no-default ADD COLUMN is catalog-only in PostgreSQL and TimescaleDB accepts it on a + /// compressed hypertable with continuous aggregates attached (verified live on 2.28.1 against + /// procedure_stats with its hourly/daily aggregates and query_stats with its hourly one), + /// so this stays instant on a multi-hundred-GB store. + /// + /// The offsets' semantics, stated here because this is where the next reader will look. + /// statement_start_offset and statement_end_offset are sys.dm_exec_query_stats's + /// own columns: the statement's position inside its batch text in BYTES of the + /// nvarchar text, not characters — so slicing the text at them divides by two (the + /// collector's SUBSTRING(st.text, (statement_start_offset / 2) + 1, …)), and a reader who + /// forgets the Unicode factor lands halfway into the wrong statement. statement_end_offset = -1 + /// means "to the end of the batch"; (0, -1) is the whole batch. They are stored + /// verbatim as the DMV reports them, -1 included and never normalized to a length, + /// because the collector's delta key is $"{sql_handle}:{start}:{end}:{plan_handle}" over the raw + /// ints and the seed has to spell the same string byte for byte. + /// + /// The PostgreSQL pair's columns are added in TWO places and both are required — the V101 + /// rule. A store's tables come from one of two texts depending on when it was created: a fresh store + /// builds every table from V1's generated schema, walked from the collector catalog, while a store that + /// predates V63/V64 has whatever those rungs built. So the pg_wait_stats and + /// pg_statement_stats CREATE TABLE rungs gain the column for the population that first meets them + /// (PgSchemaGeneratorTests enforces the rung text against the generator, column for column) and + /// THIS rung's ALTER carries the existing one. Neither is redundant: the CREATE is IF NOT EXISTS + /// and never re-runs on a store that already has the table, and the ALTER is ADD COLUMN IF NOT + /// EXISTS and is a no-op wherever the column already exists. The three SQL Server tables need no + /// second site: they are generated at V1 and only ever ALTERed. + /// + /// Two view treatments. v_memory_grant_stats is a SELECT * passthrough + /// and is refreshed here for the V14/V80/V81/V127 reason (Postgres freezes the column list at CREATE; + /// appending is the one alteration CREATE OR REPLACE VIEW permits). procedure_stats, + /// pg_wait_stats and pg_statement_stats have no v_* view (their readers hit the + /// base table; PgSchemaGenerator.AllPassthroughViews pins the set). v_query_stats is + /// the #1767 payload-RESOLVING view, not a passthrough, and the generator emits payload columns BEFORE + /// the trailing digest columns, so the two offsets land mid-list — an alteration + /// CREATE OR REPLACE VIEW refuses. Hence DROP VIEW here and the regenerated resolving + /// definition concatenated after this constant in the ladder entry, exactly as V51 and V121 did, with + /// V54Sql and GenerateQueryStatsPayloadColumnPreAdds() ahead of it so a store climbing + /// from below those rungs has every column the view names (MigrationLadderPins). Plain DROP, + /// no CASCADE: nothing persistent depends on the view. + /// + /// What this rung deliberately does NOT do. It does not touch + /// collect.wait_stats_baseline (V127's stated follow-up is a new aggregate under a new name — + /// an aggregate-plus-retirement operation, not a column, and not this rung). It does not backfill. It + /// does not change procedure_stats_hourly/_daily or query_stats_hourly: those sum + /// deltas, and a fabricated 0 adds nothing to a sum. + /// + private const string V128Sql = @" +ALTER TABLE collect.procedure_stats + ADD COLUMN IF NOT EXISTS sample_interval_seconds integer; +ALTER TABLE collect.memory_grant_stats + ADD COLUMN IF NOT EXISTS sample_interval_seconds integer; +ALTER TABLE collect.pg_wait_stats + ADD COLUMN IF NOT EXISTS sample_interval_seconds integer; +ALTER TABLE collect.pg_statement_stats + ADD COLUMN IF NOT EXISTS sample_interval_seconds integer; + +/* The delta key's two halves that were never stored. BYTE offsets into the batch's nvarchar text + (a character position is offset / 2); statement_end_offset = -1 means ""to the end of the batch""; + stored raw, -1 included, because the key string carries the raw values. */ +ALTER TABLE collect.query_stats + ADD COLUMN IF NOT EXISTS statement_start_offset integer; +ALTER TABLE collect.query_stats + ADD COLUMN IF NOT EXISTS statement_end_offset integer; + +/* Postgres FREEZES a view's SELECT * column list at CREATE, so the passthrough would keep serving the + pre-V128 column list forever — the V14 lesson, restated by V80, V81 and V127. The other three interval + tables have no v_ view. */ +CREATE OR REPLACE VIEW collect.v_memory_grant_stats AS SELECT * FROM collect.memory_grant_stats; + +/* v_query_stats is the payload-RESOLVING view (#1767), and its two new columns land ahead of the digest + columns — mid-list, which CREATE OR REPLACE VIEW refuses. Dropped here; the ladder entry concatenates + the regenerated resolving definition after the pre-adds, the V51/V121 idiom. */ +DROP VIEW IF EXISTS collect.v_query_stats;"; + /// /// V2 — the service's observability store: the servers registry (upserted on every /// successful connect) and the per-run collection_log. Column names deliberately mirror @@ -1909,7 +2018,8 @@ CREATE TABLE IF NOT EXISTS collect.pg_wait_stats ( waits bigint, wait_time_us bigint, delta_waits bigint, - delta_wait_time_us bigint + delta_wait_time_us bigint, + sample_interval_seconds integer ); CREATE INDEX IF NOT EXISTS idx_pg_wait_stats_time @@ -1972,7 +2082,8 @@ CREATE TABLE IF NOT EXISTS collect.pg_statement_stats ( max_exec_peakmem_bytes bigint, delta_calls bigint, delta_total_exec_time_ms bigint, - delta_rows bigint + delta_rows bigint, + sample_interval_seconds integer ); CREATE INDEX IF NOT EXISTS idx_pg_statement_stats_time diff --git a/Darling/PerformanceMonitor.Darling.Storage/StorageVersion.cs b/Darling/PerformanceMonitor.Darling.Storage/StorageVersion.cs index a5ada99f7..f50b18e9e 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/StorageVersion.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/StorageVersion.cs @@ -16,5 +16,5 @@ namespace PerformanceMonitor.Darling.Storage; /// public static class StorageVersion { - public const int SchemaVersion = 127; + public const int SchemaVersion = 128; } diff --git a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.ItemHistory.cs b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.ItemHistory.cs index 689efb97b..c45505ba3 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.ItemHistory.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.ItemHistory.cs @@ -182,8 +182,10 @@ public async Task> GetQueryStatsHistoryAsync( /// /// Every collected procedure_stats snapshot for one (database, schema, object) over the window — Lite's /// GetProcedureStatsHistoryAsync column list against the base procedure_stats table. - /// procedure_stats has no sample_interval_seconds column (unlike query_stats), so — exactly as Lite - /// does — the interval is derived per row from the gap to the previous collection via + /// procedure_stats carries sample_interval_seconds since V128 (#3540; query_stats always has), so + /// — exactly as Lite does — the row's STORED interval is shown where it has one (a 0 is the "Interval + /// (sec)" 0 the query-stats history has always shown for an unknowable row), and a pre-V128 row (NULL) + /// keeps the interval this read always derived from the gap to the previous collection via /// LAG(collection_time) (identical syntax in Postgres). $1 server_id, $2 database_name, $3 /// schema_name, $4 object_name, $5 window start, $6 window end (naive UTC). /// @@ -221,7 +223,7 @@ public async Task> GetQueryStatsHistoryAsync( total_physical_reads, total_logical_writes, delta_spills, - CAST(extract(epoch FROM (date_trunc('second', collection_time) - date_trunc('second', LAG(collection_time) OVER (ORDER BY collection_time)))) AS bigint) AS sample_interval_seconds + COALESCE(sample_interval_seconds, CAST(extract(epoch FROM (date_trunc('second', collection_time) - date_trunc('second', LAG(collection_time) OVER (ORDER BY collection_time)))) AS bigint)) AS sample_interval_seconds FROM procedure_stats WHERE server_id = $1 AND database_name = $2 diff --git a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.QueryTrends.cs b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.QueryTrends.cs index 6ce84610c..6e03e7ac5 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.QueryTrends.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.QueryTrends.cs @@ -68,7 +68,18 @@ FROM raw ORDER BY collection_time """; - /// Procedure-stats duration trend: elapsed ms/sec + executions/sec per collection snapshot. + /// + /// Procedure-stats duration trend: elapsed ms/sec + executions/sec per collection snapshot. + /// + /// #3540 (V128): the interval is the collection's STORED one where the rows have it — MAX + /// over the collection's rows, because a plan first seen in an otherwise steady pass (a TOP (150) + /// readmission) carries 0 beside its siblings' real interval and contributes 0 to the sums; MAX is 0 + /// only when EVERY row was unknowable (a restart), and that 0 becomes NULL through NULLIF so the + /// rates are NULL and the reader drops the point rather than rendering 0.00 ms/sec. NULL (a pre-V128 + /// collection that never recorded one) falls back to the LAG over collection_time this read always + /// used, so history renders exactly as it did. No ELSE 0: the first row of a pre-V128 series is + /// absent rather than a fabricated 0.0, the same correction V127 made for the wait trends. + /// public const string ProcedureDurationTrendSql = """ WITH raw AS ( @@ -76,7 +87,10 @@ WITH raw AS collection_time, SUM(delta_elapsed_time) / 1000.0 AS total_elapsed_ms, SUM(delta_execution_count) AS total_executions, - extract(epoch FROM (date_trunc('second', collection_time) - date_trunc('second', LAG(collection_time) OVER (ORDER BY collection_time)))) AS interval_seconds + CASE WHEN MAX(sample_interval_seconds) IS NULL + THEN extract(epoch FROM (date_trunc('second', collection_time) - date_trunc('second', LAG(collection_time) OVER (ORDER BY collection_time)))) + ELSE NULLIF(MAX(sample_interval_seconds), 0) + END AS interval_seconds FROM procedure_stats WHERE server_id = $1 AND collection_time >= $2 @@ -86,8 +100,8 @@ GROUP BY collection_time ) SELECT collection_time, - CASE WHEN interval_seconds > 0 THEN total_elapsed_ms / interval_seconds ELSE 0 END AS elapsed_ms_per_second, - CASE WHEN interval_seconds > 0 THEN CAST(total_executions AS DOUBLE PRECISION) / interval_seconds ELSE 0 END AS executions_per_second + CASE WHEN interval_seconds > 0 THEN total_elapsed_ms / interval_seconds END AS elapsed_ms_per_second, + CASE WHEN interval_seconds > 0 THEN CAST(total_executions AS DOUBLE PRECISION) / interval_seconds END AS executions_per_second FROM raw ORDER BY collection_time """; @@ -300,10 +314,18 @@ private async Task> ReadDurationTrendAsync( await using var reader = await command.ExecuteReaderAsync(cancellationToken); while (await reader.ReadAsync(cancellationToken)) { + /* A NULL rate is an unknowable interval (#3540, V128): the point is dropped, not read as 0. Only + the procedure trend emits one today (its query-stats and Query Store siblings keep ELSE 0), so + this is a no-op for them and the missing-sample posture for it. */ + if (reader.IsDBNull(1)) + { + continue; + } + items.Add(new QueryTrendPoint { CollectionTime = reader.GetDateTime(0), - Value = reader.IsDBNull(1) ? 0 : Convert.ToDouble(reader.GetValue(1)), + Value = Convert.ToDouble(reader.GetValue(1)), ExecutionCount = reader.IsDBNull(2) ? 0 : (long)Convert.ToDouble(reader.GetValue(2)), }); } diff --git a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.cs b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.cs index 71d14f4c2..c98e7c9f2 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.cs @@ -819,6 +819,15 @@ table existence cannot separate the rungs. The rung adds the same column to four looking for a significance it does not have. Named only in this probe line, never in prose, per the V71 finding: the coverage ratchet strips information_schema lines but cannot strip a comment. */ EXISTS (SELECT 1 FROM information_schema.columns WHERE table_name = 'wait_stats' + AND column_name = 'sample_interval_seconds'), + /* V128 probes a COLUMN for the same reason as V127: its five tables have existed since V4/V10 and + V63/V64, so table existence cannot separate the rungs. The rung adds the same interval column to + four tables and two offset columns to a fifth in one transaction, so any of the six would answer; + the procedure table is chosen because it is the rung's first ALTER, and that choice is stated so + nobody goes looking for a significance it does not have. The column NAME is shared with V127's + sentinel — only the TABLE differs — which is exactly why the table is part of the predicate. Named + only in this probe line, never in prose, per the V71 finding. */ + EXISTS (SELECT 1 FROM information_schema.columns WHERE table_name = 'procedure_stats' AND column_name = 'sample_interval_seconds')"; /// The store schema version this viewer build requires — the highest migration it knows @@ -841,7 +850,7 @@ table existence cannot separate the rungs. The rung adds the same column to four await using var reader = await command.ExecuteReaderAsync(cancellationToken); if (await reader.ReadAsync(cancellationToken)) { - return MapProbedSchemaVersion(reader.GetBoolean(0), reader.GetBoolean(1), reader.GetBoolean(2), reader.GetBoolean(3), reader.GetBoolean(4), reader.GetBoolean(5), reader.GetBoolean(6), reader.GetBoolean(7), reader.GetBoolean(8), reader.GetBoolean(9), reader.GetBoolean(10), reader.GetBoolean(11), reader.GetBoolean(12), reader.GetBoolean(13), reader.GetBoolean(14), reader.GetBoolean(15), reader.GetBoolean(16), reader.GetBoolean(17), reader.GetBoolean(18), reader.GetBoolean(19), reader.GetBoolean(20), reader.GetBoolean(21), reader.GetBoolean(22), reader.GetBoolean(23), reader.GetBoolean(24), reader.GetBoolean(25), reader.GetBoolean(26), reader.GetBoolean(27), reader.GetBoolean(28), reader.GetBoolean(29), reader.GetBoolean(30), reader.GetBoolean(31), reader.GetBoolean(32), reader.GetBoolean(33), reader.GetBoolean(34), reader.GetBoolean(35), reader.GetBoolean(36), reader.GetBoolean(37), reader.GetBoolean(38), reader.GetBoolean(39), reader.GetBoolean(40), reader.GetBoolean(41), reader.GetBoolean(42), reader.GetBoolean(43), reader.GetBoolean(44), reader.GetBoolean(45), reader.GetBoolean(46), reader.GetBoolean(47), reader.GetBoolean(48), reader.GetBoolean(49), reader.GetBoolean(50), reader.GetBoolean(51), reader.GetBoolean(52), reader.GetBoolean(53), reader.GetBoolean(54), reader.GetBoolean(55), reader.GetBoolean(56), reader.GetBoolean(57), reader.GetBoolean(58), reader.GetBoolean(59), reader.GetBoolean(60), reader.GetBoolean(61), reader.GetBoolean(62), reader.GetBoolean(63), reader.GetBoolean(64), reader.GetBoolean(65), reader.GetBoolean(66), reader.GetBoolean(67), reader.GetBoolean(68), reader.GetBoolean(69), reader.GetBoolean(70), reader.GetBoolean(71), reader.GetBoolean(72), reader.GetBoolean(73), reader.GetBoolean(74), reader.GetBoolean(75), reader.GetBoolean(76), reader.GetBoolean(77), reader.GetBoolean(78), reader.GetBoolean(79), reader.GetBoolean(80), reader.GetBoolean(81), reader.GetBoolean(82), reader.GetBoolean(83), reader.GetBoolean(84), reader.GetBoolean(85), reader.GetBoolean(86), reader.GetBoolean(87), reader.GetBoolean(88), reader.GetBoolean(89), reader.GetBoolean(90), reader.GetBoolean(91), reader.GetBoolean(92), reader.GetBoolean(93), reader.GetBoolean(94), reader.GetBoolean(95), reader.GetBoolean(96), reader.GetBoolean(97), reader.GetBoolean(98), reader.GetBoolean(99), reader.GetBoolean(100), reader.GetBoolean(101), reader.GetBoolean(102)); + return MapProbedSchemaVersion(reader.GetBoolean(0), reader.GetBoolean(1), reader.GetBoolean(2), reader.GetBoolean(3), reader.GetBoolean(4), reader.GetBoolean(5), reader.GetBoolean(6), reader.GetBoolean(7), reader.GetBoolean(8), reader.GetBoolean(9), reader.GetBoolean(10), reader.GetBoolean(11), reader.GetBoolean(12), reader.GetBoolean(13), reader.GetBoolean(14), reader.GetBoolean(15), reader.GetBoolean(16), reader.GetBoolean(17), reader.GetBoolean(18), reader.GetBoolean(19), reader.GetBoolean(20), reader.GetBoolean(21), reader.GetBoolean(22), reader.GetBoolean(23), reader.GetBoolean(24), reader.GetBoolean(25), reader.GetBoolean(26), reader.GetBoolean(27), reader.GetBoolean(28), reader.GetBoolean(29), reader.GetBoolean(30), reader.GetBoolean(31), reader.GetBoolean(32), reader.GetBoolean(33), reader.GetBoolean(34), reader.GetBoolean(35), reader.GetBoolean(36), reader.GetBoolean(37), reader.GetBoolean(38), reader.GetBoolean(39), reader.GetBoolean(40), reader.GetBoolean(41), reader.GetBoolean(42), reader.GetBoolean(43), reader.GetBoolean(44), reader.GetBoolean(45), reader.GetBoolean(46), reader.GetBoolean(47), reader.GetBoolean(48), reader.GetBoolean(49), reader.GetBoolean(50), reader.GetBoolean(51), reader.GetBoolean(52), reader.GetBoolean(53), reader.GetBoolean(54), reader.GetBoolean(55), reader.GetBoolean(56), reader.GetBoolean(57), reader.GetBoolean(58), reader.GetBoolean(59), reader.GetBoolean(60), reader.GetBoolean(61), reader.GetBoolean(62), reader.GetBoolean(63), reader.GetBoolean(64), reader.GetBoolean(65), reader.GetBoolean(66), reader.GetBoolean(67), reader.GetBoolean(68), reader.GetBoolean(69), reader.GetBoolean(70), reader.GetBoolean(71), reader.GetBoolean(72), reader.GetBoolean(73), reader.GetBoolean(74), reader.GetBoolean(75), reader.GetBoolean(76), reader.GetBoolean(77), reader.GetBoolean(78), reader.GetBoolean(79), reader.GetBoolean(80), reader.GetBoolean(81), reader.GetBoolean(82), reader.GetBoolean(83), reader.GetBoolean(84), reader.GetBoolean(85), reader.GetBoolean(86), reader.GetBoolean(87), reader.GetBoolean(88), reader.GetBoolean(89), reader.GetBoolean(90), reader.GetBoolean(91), reader.GetBoolean(92), reader.GetBoolean(93), reader.GetBoolean(94), reader.GetBoolean(95), reader.GetBoolean(96), reader.GetBoolean(97), reader.GetBoolean(98), reader.GetBoolean(99), reader.GetBoolean(100), reader.GetBoolean(101), reader.GetBoolean(102), reader.GetBoolean(103)); } return null; @@ -866,7 +875,7 @@ table existence cannot separate the rungs. The rung adds the same column to four /// is unit-tested without a live store; any schema bump past the newest arm trips the pinning test that keeps /// this in step with . /// - internal static int MapProbedSchemaVersion(bool hasConfigControlPlane, bool hasAlertDeliveryOverride, bool hasAnalysisState, bool hasAlertTuningKnobs, bool hasDefaultTraceEvents, bool hasIndexObjectStatsLatestIndex, bool hasCollectionLogHypertableOrPlainPg, bool hasJobHistory, bool hasAgentStatus, bool hasGenericWebhook, bool hasDeadlocksDatabaseName, bool hasQueryStoreReplicaRole, bool hasLongQueryCompletions, bool hasWebDashboardConfig, bool hasCustomViews, bool hasServerTags, bool hasConnectionRefireKnobs = false, bool hasAgCollectors = false, bool hasAgAlertKnobs = false, bool hasAgLatencyColumns = false, bool hasAgDisconnectRefire = false, bool hasPayloadDimensions = false, bool hasDimFloorIndexes = false, bool hasBlockingWaitThreshold = false, bool hasQueryStoreIntervalIdentity = false, bool hasPagerDutyWebhook = false, bool hasPagerDutyProxy = false, bool hasCollectorState = false, bool hasPlanCorrection = false, bool hasPvsStats = false, bool hasPvsPressureKnobs = false, bool hasDatabaseStateAlert = false, bool hasServerTagColour = false, bool hasQueryStatsHostObject = false, bool hasFindingDrillDown = false, bool hasStoreMetrics = false, bool hasPlanDimGzip = false, bool hasSelfAlertKnobs = false, bool hasJobMetricsColumns = false, bool hasJobCadenceKnob = false, bool hasBackfillSwitch = false, bool hasCollectorMemoryKnobs = false, bool hasDatabaseStateEdgeMemory = false, bool hasIncidentOccurrences = false, bool hasPlanXmlCompressionKnob = false, bool hasMonitoredServerEngine = false, bool hasPgBlockingEdges = false, bool hasQueryStorePlanMap = false, bool hasPgStatementText = false, bool hasQueryStoreText = false, bool hasPlanContentRetentionKnob = false, bool hasQueryStoreHealth = false, bool hasQueryStoreTextHash = false, bool hasComposeTimeoutKnob = false, bool hasFileGrowthAlert = false, bool hasCollectionLogFanoutRollup = false, bool hasTempDbMaxSize = false, bool hasServerEngineKind = false, bool hasPgDatabaseStats = false, bool hasPgIndexUsageStats = false, bool hasPgTableBloatStats = false, bool hasPgSessionStates = false, bool hasPgPlanCaptureReadiness = false, bool hasPgWriteStats = false, bool hasPgExtensionAvailability = false, bool hasPgLockStats = false, bool hasPgColumnStats = false, bool hasPgReplicationStats = false, bool hasPgBufferUsage = false, bool hasPgIndexBloat = false, bool hasPgPerDatabaseAttribution = false, bool hasPgWaitSampling = false, bool hasPgKernelStats = false, bool hasPgPredicateStats = false, bool hasPgPlanCapture = false, bool hasPgMajorVersion = false, bool hasPg18IoBytes = false, bool hasPgServerConfig = false, bool hasPgDeadlocks = false, bool hasPgDeadlockIdentity = false, bool hasCollectorCost = false, bool hasPgCpuUtilization = false, bool hasPlanForceActions = false, bool hasCollectionLogPhaseSplit = false, bool hasCollectionLogDrainForensics = false, bool hasCollectionLogFetchPhaseSums = false, bool hasStoreLogSelfMonitoring = false, bool hasCollectorStallProbes = false, bool hasRemediationCredentialAndActor = false, bool hasPgIndexBloatEstimate = false, bool hasPgCpuCapacityHeadroom = false, bool hasCustomAlertCore = false, bool hasMuteRuleReloadBeacon = false, bool hasBuiltinAlertPersistence = false, bool hasRetentionHoldRatioKnobs = false, bool hasDeadlockRateBandKnobs = false, bool hasOversizedPlanBacklog = false, bool hasPgAlertCountKnobs = false, bool hasFleetSweepState = false, bool hasFleetSweepCadenceKnobs = false, bool hasCollectorScheduleDatabases = false, bool hasSelfDiskWarnGbFloor = false, bool hasDeltaFamilyIntervalColumns = false) + internal static int MapProbedSchemaVersion(bool hasConfigControlPlane, bool hasAlertDeliveryOverride, bool hasAnalysisState, bool hasAlertTuningKnobs, bool hasDefaultTraceEvents, bool hasIndexObjectStatsLatestIndex, bool hasCollectionLogHypertableOrPlainPg, bool hasJobHistory, bool hasAgentStatus, bool hasGenericWebhook, bool hasDeadlocksDatabaseName, bool hasQueryStoreReplicaRole, bool hasLongQueryCompletions, bool hasWebDashboardConfig, bool hasCustomViews, bool hasServerTags, bool hasConnectionRefireKnobs = false, bool hasAgCollectors = false, bool hasAgAlertKnobs = false, bool hasAgLatencyColumns = false, bool hasAgDisconnectRefire = false, bool hasPayloadDimensions = false, bool hasDimFloorIndexes = false, bool hasBlockingWaitThreshold = false, bool hasQueryStoreIntervalIdentity = false, bool hasPagerDutyWebhook = false, bool hasPagerDutyProxy = false, bool hasCollectorState = false, bool hasPlanCorrection = false, bool hasPvsStats = false, bool hasPvsPressureKnobs = false, bool hasDatabaseStateAlert = false, bool hasServerTagColour = false, bool hasQueryStatsHostObject = false, bool hasFindingDrillDown = false, bool hasStoreMetrics = false, bool hasPlanDimGzip = false, bool hasSelfAlertKnobs = false, bool hasJobMetricsColumns = false, bool hasJobCadenceKnob = false, bool hasBackfillSwitch = false, bool hasCollectorMemoryKnobs = false, bool hasDatabaseStateEdgeMemory = false, bool hasIncidentOccurrences = false, bool hasPlanXmlCompressionKnob = false, bool hasMonitoredServerEngine = false, bool hasPgBlockingEdges = false, bool hasQueryStorePlanMap = false, bool hasPgStatementText = false, bool hasQueryStoreText = false, bool hasPlanContentRetentionKnob = false, bool hasQueryStoreHealth = false, bool hasQueryStoreTextHash = false, bool hasComposeTimeoutKnob = false, bool hasFileGrowthAlert = false, bool hasCollectionLogFanoutRollup = false, bool hasTempDbMaxSize = false, bool hasServerEngineKind = false, bool hasPgDatabaseStats = false, bool hasPgIndexUsageStats = false, bool hasPgTableBloatStats = false, bool hasPgSessionStates = false, bool hasPgPlanCaptureReadiness = false, bool hasPgWriteStats = false, bool hasPgExtensionAvailability = false, bool hasPgLockStats = false, bool hasPgColumnStats = false, bool hasPgReplicationStats = false, bool hasPgBufferUsage = false, bool hasPgIndexBloat = false, bool hasPgPerDatabaseAttribution = false, bool hasPgWaitSampling = false, bool hasPgKernelStats = false, bool hasPgPredicateStats = false, bool hasPgPlanCapture = false, bool hasPgMajorVersion = false, bool hasPg18IoBytes = false, bool hasPgServerConfig = false, bool hasPgDeadlocks = false, bool hasPgDeadlockIdentity = false, bool hasCollectorCost = false, bool hasPgCpuUtilization = false, bool hasPlanForceActions = false, bool hasCollectionLogPhaseSplit = false, bool hasCollectionLogDrainForensics = false, bool hasCollectionLogFetchPhaseSums = false, bool hasStoreLogSelfMonitoring = false, bool hasCollectorStallProbes = false, bool hasRemediationCredentialAndActor = false, bool hasPgIndexBloatEstimate = false, bool hasPgCpuCapacityHeadroom = false, bool hasCustomAlertCore = false, bool hasMuteRuleReloadBeacon = false, bool hasBuiltinAlertPersistence = false, bool hasRetentionHoldRatioKnobs = false, bool hasDeadlockRateBandKnobs = false, bool hasOversizedPlanBacklog = false, bool hasPgAlertCountKnobs = false, bool hasFleetSweepState = false, bool hasFleetSweepCadenceKnobs = false, bool hasCollectorScheduleDatabases = false, bool hasSelfDiskWarnGbFloor = false, bool hasDeltaFamilyIntervalColumns = false, bool hasDeltaFamilyIntervalCompletion = false) { /* V71 (the PostgreSQL blocking-edges rung): a table-existence sentinel and now the newest-first arm. A collector table would ordinarily get no arm at all — see the V63-V69 note below — but the TOP @@ -1015,15 +1024,35 @@ information_schema lines but cannot strip a comment. */ StorageVersion.SchemaVersion (116) rather than falling through to 115 and showing a spurious upgrade banner on a store that is current. The table is named only in the probe line, not this prose, per the V71 finding (the coverage ratchet strips information_schema lines but cannot strip a comment). */ + /* V128 (#3540): the completion of V127 — the measured sample interval on the four delta families + V127 left naked, and the two statement offsets on the query table that its delta key is made of. + After this rung every delta family stores the interval, so a restart's fabricated (0, 0) row can + be told from a genuinely idle one at EVERY read, and the restart seed can rebuild the one key + the store could not reproduce. COLUMN-existence sentinel (the five tables have existed since + V4/V10 and V63/V64, so table existence cannot separate the rungs), newest-first, and the TOP + rung, so a fully-migrated store maps to EXACTLY StorageVersion.SchemaVersion rather than falling + through to the rung below and showing a spurious upgrade banner on a store that is current. + + The gate earns its place beyond that standing invariant: the viewer's procedure duration trend + and procedure history reads now name the column (the stored interval is preferred over the LAG + derivation), so a viewer pointed below this rung would throw a raw 42703 on the Performance + Trends and Top Procedures surfaces — the banner has to fire before those do. The column and its + tables are named only in the probe line, not this prose, per the V71 finding: the coverage + ratchet strips information_schema lines but cannot strip a comment. */ + if (hasDeltaFamilyIntervalCompletion) + { + return 128; + } + /* V127 (#3540): the measured sample interval on the four delta families that persisted their deltas naked — the measurement-layer keystone, so a restart's fabricated (0, 0) row can be told from a genuinely idle one at every read. COLUMN-existence sentinel (the four tables have existed - since V4/V10, so table existence cannot separate the rungs), newest-first, and the TOP rung, so a - fully-migrated store maps to EXACTLY StorageVersion.SchemaVersion rather than falling through to - the rung below and showing a spurious upgrade banner on a store that is current. + since V4/V10, so table existence cannot separate the rungs), newest-first, one rung behind the + top since V128 landed — a store carrying this and not V128 above maps to 127, which is the + honest answer for it and also what makes the upgrade banner correct in both directions. The gate earns its place beyond that standing invariant: every viewer trend read over these four - families now names the column (the stored interval is preferred over the LAG derivation), so a + families names the column (the stored interval is preferred over the LAG derivation), so a viewer pointed below this rung would throw a raw 42703 on the Waits, File I/O and Latch/Spinlock tabs — the banner has to fire before those tabs do. The column and its tables are named only in the probe line, not this prose, per the V71 finding: the coverage ratchet strips diff --git a/Darling/README.md b/Darling/README.md index b81ec16be..fddde95c0 100644 --- a/Darling/README.md +++ b/Darling/README.md @@ -525,7 +525,7 @@ The embedded MCP server, over Streamable HTTP bound to `localhost` by default (s - *Plan cache / scheduler* — `get_plan_cache_bloat` (single-use vs multi-use + bloat level), `get_cpu_scheduler_pressure` (runnable queue, worker utilization, pressure level). - *Jobs* — `get_running_jobs` (running SQL Agent jobs vs historical average / p95). - The Dashboard's per-class latch `severity` / `description` / `recommendation`, spinlock `description`, plan-cache `bloat_level`, and CPU-scheduler `pressure_level` / `recommendation` are the Dashboard / reporting-view CASE derivations (not collected columns), reproduced service-side so the full result shape is served. Per-second latch/spinlock rates divide by each row's stored `sample_interval_seconds` (the measured seconds its deltas accrued over, stored beside the deltas on every SQL Server delta family except procedure and memory-grant stats) and are `null` rather than 0 when that interval was unknowable — a restart, not quiet. The memory-grant collector stores no interval, so the Dashboard's `get_resource_semaphore` `sample_interval_seconds` is not emitted (`max_target_memory_mb`, the workspace-memory ceiling, is added since the store carries it). + The Dashboard's per-class latch `severity` / `description` / `recommendation`, spinlock `description`, plan-cache `bloat_level`, and CPU-scheduler `pressure_level` / `recommendation` are the Dashboard / reporting-view CASE derivations (not collected columns), reproduced service-side so the full result shape is served. Per-second latch/spinlock rates divide by each row's stored `sample_interval_seconds` (the measured seconds its deltas accrued over, stored beside the deltas on every delta family since V128 — SQL Server and PostgreSQL alike) and are `null` rather than 0 when that interval was unknowable — a restart, not quiet. `get_resource_semaphore` emits the Dashboard's `sample_interval_seconds` again now that the memory-grant collector stores it (V128), `null` with `interval_known: false` on a restart marker or a pre-V128 row; `max_target_memory_mb`, the workspace-memory ceiling, is added since the store carries it. - **PostgreSQL data-read tools** — the read surface for a PostgreSQL target's collectors, each a stored read. The `get_pg_*` family is larger than the reads called out here (see [PostgreSQL targets](#postgresql-targets)): - *Waits and queries* — `get_pg_wait_stats` (top wait events in the window, decoded to type + event name), `get_pg_top_queries` (query shapes by total execution time, carrying Aurora's storage-vs-cache I/O split and per-statement peak memory). @@ -1029,7 +1029,7 @@ The viewer is read-only over collected data, but it does perform a small set of The service is built to restart cleanly, any time: -- **Delta continuity** — every delta-based collector (wait stats, file I/O, perfmon, memory grants, latch and spinlock stats, procedure stats, and the PostgreSQL wait and statement stats) re-seeds its baselines from the store at startup — the latest row per key inside a 15-minute window — along with the per-collector pass window the recompiled-plan credit reads, so the first cycle after a restart produces real deltas instead of zeroes. The one exception is stated rather than hidden: `query_stats` keys its deltas on statement offsets the store does not persist, so only its pass window is restored and plans older than the restart gap baseline on the first cycle. +- **Delta continuity** — every delta-based collector (wait stats, file I/O, perfmon, memory grants, latch and spinlock stats, procedure stats, query stats, and the PostgreSQL wait and statement stats) re-seeds its baselines from the store at startup — the latest row per key inside a 15-minute window — along with the per-collector pass window the recompiled-plan credit reads, so the first cycle after a restart produces real deltas instead of zeroes. `query_stats` joined the key seed with V128: its delta key is built from the two statement offsets, which the store persists since that rung, so a row written before it (NULL offsets) restores only the pass window and every row written since restores its baseline too. - **Alert no-re-fire** — edge-trigger watermarks and the failed-job watermark persist in `config_edge_trigger_watermarks`, and per-alert cooldowns re-seed from `config_alert_log`, so a restart does not replay alerts you already received. - **Idempotent store setup** — migrations are versioned and skip what is already applied; TimescaleDB conversion and compression policies re-converge as no-ops. - **Per-connect snapshots** — the on-connect config snapshot collectors run once per (re)connect, mirroring Lite's server-open behavior. diff --git a/Lite.Tests/DeltaFamilyIntervalColumnTests.cs b/Lite.Tests/DeltaFamilyIntervalColumnTests.cs index 56ebe91b0..c83dfcc48 100644 --- a/Lite.Tests/DeltaFamilyIntervalColumnTests.cs +++ b/Lite.Tests/DeltaFamilyIntervalColumnTests.cs @@ -22,46 +22,53 @@ namespace Lite.Tests; /// the interval is the ONLY thing that tells those apart. Four of the six SQL Server delta families /// (wait_stats, file_io_stats, latch_stats, spinlock_stats) discarded it at the /// write until Darling V127 / Lite v60, so a restart's fabricated zero survived as a measured one and every -/// per-second reader LAG-divided it into a confident 0.00. This is the census that keeps a seventh family -/// from shipping naked, and keeps the four that still are named rather than assumed. +/// per-second reader LAG-divided it into a confident 0.00; the remaining four (procedure_stats, +/// memory_grant_stats, pg_wait_stats, pg_statement_stats) followed at Darling V128 / +/// Lite v61. This is the census that keeps an eleventh family from shipping naked. Rule 4 is COMPLETE: the +/// still-naked list below is empty, and asserted empty, so the claim "every delta family stores its +/// interval" is a test rather than a sentence. /// public sealed class DeltaFamilyIntervalColumnTests { private const string IntervalColumn = "sample_interval_seconds"; /// - /// The delta families that persist NO interval today, named so the list can only SHRINK deliberately. - /// procedure_stats and memory_grant_stats take the calculator's bare long; - /// pg_wait_stats and pg_statement_stats ask for the interval only to skip idle rows at the - /// write and store nothing. Each is a follow-up in the #3540 campaign, not a permanent exemption — a - /// rung that gives one of them the column must remove it from here, and the reverse-direction assertion - /// below is what makes forgetting to loud. + /// The delta families that persist NO interval, named so the list can only SHRINK deliberately. EMPTY + /// since Darling V128 / Lite v61 (#3540). From V127 / v60 until then it named four: procedure_stats + /// and memory_grant_stats took the calculator's bare long; pg_wait_stats and + /// pg_statement_stats asked for the interval only to skip idle rows at the write and stored + /// nothing. Each was a follow-up in the #3540 campaign, not a permanent exemption, and the rung that + /// dressed them removed them from here — the reverse-direction assertion below is what made forgetting + /// to loud. Kept declared, and asserted empty, so a twelfth family that ships naked has to name itself + /// here to pass and the diff says so. /// - private static readonly HashSet StillNaked = new(StringComparer.OrdinalIgnoreCase) - { - "procedure_stats", - "memory_grant_stats", - "pg_wait_stats", - "pg_statement_stats", - }; + private static readonly HashSet StillNaked = new(StringComparer.OrdinalIgnoreCase); - /// The families this PR dressed, plus the two that were never naked. + /// Every delta family: the two that were never naked, the four V127 / v60 dressed, and the four + /// V128 / v61 dressed. Stated as a literal so the PR that adds a family has to say so here too. private static readonly string[] Dressed = { "wait_stats", "file_io_stats", "latch_stats", "spinlock_stats", "perfmon_stats", "query_stats", + "procedure_stats", "memory_grant_stats", "pg_wait_stats", "pg_statement_stats", }; /// /// Every member of either carries the /// interval column or is on the named still-naked list — and nothing on that list carries it. Both /// directions, so a new delta family cannot ship without the column by omission, and a family that gains - /// the column cannot keep claiming it has not. + /// the column cannot keep claiming it has not. Since V128 / v61 the list is empty, so the "named as still + /// naked" branch is the historical record of how the four got here and the census reduces to: every + /// family carries the column. /// [Fact] public void EveryDeltaFamily_PersistsItsInterval_OrIsNamedAsStillNaked() { var families = CollectorDeltaCalculator.DeltaFamilyCollectors.OrderBy(f => f, StringComparer.Ordinal).ToList(); Assert.NotEmpty(families); + Assert.Equal(10, families.Count); /* the floor that makes "every family" mean something */ + + /* Rule 4 is complete: nothing is exempt. A family added to the list has to be a deliberate diff. */ + Assert.Empty(StillNaked); foreach (var family in families) { @@ -109,19 +116,27 @@ public void TheIntervalColumn_IsTheSameIntegerOnEveryFamilyThatCarriesIt() } } + /// The families the two rungs dressed: V127 / v60's four, then V128 / v61's four. On every one + /// the column was ADDED by ALTER TABLE, so it must be the tail — see the test below. + private static readonly string[] DressedByRung = + { + "wait_stats", "file_io_stats", "latch_stats", "spinlock_stats", + "procedure_stats", "memory_grant_stats", "pg_wait_stats", "pg_statement_stats", + }; + /// - /// On the four families this PR dressed, the column is the LAST payload column. Both stores' writers are - /// positional — the DuckDB appender writes one value per declared column in order, the PostgreSQL COPY - /// writer likewise — and an existing database receives the column by ALTER TABLE ADD COLUMN, which - /// can only ever land at the end. A column declared anywhere else would shift every later ordinal on an - /// upgraded store and write deltas into the wrong columns. (perfmon_stats and query_stats are not held to - /// the tail: they were extracted with the column already in place and query_stats has since appended - /// others behind it.) + /// On the eight families the two rungs dressed, the column is the LAST payload column. Both stores' + /// writers are positional — the DuckDB appender writes one value per declared column in order, the + /// PostgreSQL COPY writer likewise — and an existing database receives the column by + /// ALTER TABLE ADD COLUMN, which can only ever land at the end. A column declared anywhere else + /// would shift every later ordinal on an upgraded store and write deltas into the wrong columns. + /// (perfmon_stats and query_stats are not held to the tail: they were extracted with the column already + /// in place and query_stats has since appended others behind it — the V128 offsets among them.) /// [Fact] - public void OnTheFourNewlyDressedFamilies_TheIntervalIsTheTrailingColumn() + public void OnTheEightRungDressedFamilies_TheIntervalIsTheTrailingColumn() { - foreach (var family in new[] { "wait_stats", "file_io_stats", "latch_stats", "spinlock_stats" }) + foreach (var family in DressedByRung) { var columns = CollectorCatalog.Find(family)!.PayloadColumns; Assert.Equal(IntervalColumn, columns[^1].Name); @@ -129,14 +144,22 @@ public void OnTheFourNewlyDressedFamilies_TheIntervalIsTheTrailingColumn() } /// - /// The DuckDB generator carries the column into a fresh store's DDL for each of the four, as the trailing - /// column and nullable — the same shape the v60 ALTER TABLE ... ADD COLUMN IF NOT EXISTS - /// sample_interval_seconds INTEGER gives an upgraded store, so fresh and upgraded databases agree. + /// The DuckDB generator carries the column into a fresh store's DDL for each family Lite stores, as the + /// trailing column and nullable — the same shape the v60 / v61 ALTER TABLE ... ADD COLUMN IF NOT + /// EXISTS sample_interval_seconds INTEGER gives an upgraded store, so fresh and upgraded databases + /// agree. The PostgreSQL pair is not a DuckDB table (Lite monitors no PostgreSQL), so it is not here; + /// DeltaFamilyIntervalCompletionRungTests pins its generated PostgreSQL DDL. /// [Fact] - public void TheDuckDbGenerator_EmitsTheIntervalAsTheTrailingNullableColumn_OnTheFour() + public void TheDuckDbGenerator_EmitsTheIntervalAsTheTrailingNullableColumn_OnTheSixLiteStores() { - foreach (var family in new[] { "wait_stats", "file_io_stats", "latch_stats", "spinlock_stats" }) + var liteTables = DuckDbSchemaGenerator.CollectorTableNames().ToHashSet(StringComparer.Ordinal); + var lite = DressedByRung.Where(liteTables.Contains).ToList(); + Assert.Equal(6, lite.Count); + Assert.DoesNotContain("pg_wait_stats", lite); + Assert.DoesNotContain("pg_statement_stats", lite); + + foreach (var family in lite) { var ddl = DuckDbSchemaGenerator.CreateTable(CollectorCatalog.Find(family)!); var lines = ddl.Split('\n').Select(l => l.Trim().TrimEnd(',')).Where(l => l.Length > 0).ToList(); @@ -169,6 +192,40 @@ public void TheLiteMigration_AddsTheColumnToAllFour_AtSchemaVersion60() source, StringComparison.Ordinal); } + /// + /// The v61 twin of Darling's V128: the interval on the two remaining Lite-stored families, the two + /// statement offsets on query_stats, and the version bump. Four (table, column) pairs, every one an + /// idempotent INTEGER ADD COLUMN, held by the same parity rule as v60 — and the offsets' semantics + /// (BYTE offsets into the batch's nvarchar text; -1 = end of batch; stored raw) stated in the block so the + /// next reader finds them where they will look. + /// + [Fact] + public void TheLiteMigration_CompletesTheIntervalAndStoresTheOffsets_AtSchemaVersion61() + { + Assert.True(DuckDbInitializer.CurrentSchemaVersion >= 61); + + var source = RepoSource.Read("Lite", "Database", "DuckDbInitializer.cs"); + var block = source[source.IndexOf("if (fromVersion < 61)", StringComparison.Ordinal)..]; + + foreach (var pair in new[] + { + "(\"procedure_stats\", \"sample_interval_seconds\")", + "(\"memory_grant_stats\", \"sample_interval_seconds\")", + "(\"query_stats\", \"statement_start_offset\")", + "(\"query_stats\", \"statement_end_offset\")", + }) + { + Assert.Contains(pair, block, StringComparison.Ordinal); + } + + Assert.Contains("$\"ALTER TABLE {table} ADD COLUMN IF NOT EXISTS {column} INTEGER\"", block, StringComparison.Ordinal); + + /* The offsets' semantics, in the block, in these words: the pair a future reader second-guesses. */ + Assert.Contains("BYTES", block, StringComparison.Ordinal); + Assert.Contains("statement_end_offset = -1 means \"to the end of the batch\"", block, StringComparison.Ordinal); + Assert.Contains("VERBATIM", block, StringComparison.Ordinal); + } + private static class RepoSource { public static string Read(params string[] parts) diff --git a/Lite.Tests/DeltaFamilySeedingCensusTests.cs b/Lite.Tests/DeltaFamilySeedingCensusTests.cs index c3d015839..f69436bc7 100644 --- a/Lite.Tests/DeltaFamilySeedingCensusTests.cs +++ b/Lite.Tests/DeltaFamilySeedingCensusTests.cs @@ -35,8 +35,10 @@ namespace PerformanceMonitorLite.Tests; /// and per host that monitors it: a seed read /// over the family's table; a delta-group list equal to the groups the collector actually passes; a /// SeedPasses over that list; and a per-family guard, so one family's failure cannot cost the rest -/// their continuity. The one family that cannot be key-seeded is named, with the store fact that makes it -/// so asserted beside it, so the exemption dies the day the fact does. +/// their continuity. Until Darling V128 / Lite v61 one family could not be key-seeded and was named here +/// with the store fact that made it so; that fact died with the rung, the exemption set is EMPTY, and the +/// record of why it existed stays below so the next family that cannot be seeded has the idiom to +/// follow. /// public sealed class DeltaFamilySeedingCensusTests { @@ -45,12 +47,15 @@ public sealed class DeltaFamilySeedingCensusTests private const string CollectorsDir = "PerformanceMonitor.Collectors"; /// - /// Families whose KEYS the store cannot reproduce, so the host seeds their pass window only. query_stats - /// keys its deltas on sql_handle:statement_start_offset:statement_end_offset:plan_handle and the - /// store persists neither offset — asserted below against the catalog, so a rung that adds them turns - /// this exemption red and demands the key seed. Not a permanent exemption: a follow-up on #3540. + /// Families whose KEYS the store cannot reproduce, so the host seeds their pass window only. EMPTY since + /// Darling V128 / Lite v61 (#3540): query_stats sat here from #3614 until then, because its delta + /// key is sql_handle:statement_start_offset:statement_end_offset:plan_handle and the store + /// persisted neither offset; the exemption was asserted against the catalog so the rung that added them + /// would turn it red and demand the key seed — which is exactly what happened. The set is kept, empty, + /// because the census arms below are the idiom a future un-seedable family would use, and because an + /// empty exemption list is itself the claim: every delta family on both hosts is key-seeded. /// - private static readonly HashSet PassWindowOnly = new(StringComparer.Ordinal) { "query_stats" }; + private static readonly HashSet PassWindowOnly = new(StringComparer.Ordinal); private static readonly Regex s_seedFamilyCall = new(@"SeedFamilyAsync\(\s*""([^""]+)""\s*,", RegexOptions.Compiled); private static readonly Regex s_seedSqlConst = new(@"public const string (\w+SeedSql) = @""([^""]*)"";", RegexOptions.Compiled | RegexOptions.Singleline); @@ -188,7 +193,10 @@ public void TheSharedSeedReads_AreByteIdenticalAcrossTheTwoHosts() var darling = s_seedSqlConst.Matches(ReadRepoFile(DarlingSeeder)).ToDictionary(m => m.Groups[1].Value, m => m.Groups[2].Value, StringComparer.Ordinal); var shared = lite.Keys.Intersect(darling.Keys, StringComparer.Ordinal).OrderBy(k => k, StringComparer.Ordinal).ToList(); - Assert.Equal(8, shared.Count); /* the six SQL Server key seeds, the memory-grant seed and the query_stats pass seed */ + Assert.Equal(8, shared.Count); /* the eight SQL Server key seeds — query_stats' replaced its pass-window-only read at V128 / v61 */ + Assert.Contains("QueryStatsSeedSql", shared); + Assert.DoesNotContain("QueryStatsPassSeedSql", lite.Keys); + Assert.DoesNotContain("QueryStatsPassSeedSql", darling.Keys); foreach (var name in shared) { @@ -218,21 +226,42 @@ public void EachHost_GuardsEveryFamilyIndividually() } /// - /// The pass-window-only exemption rests on a store fact: query_stats persists neither statement offset. - /// Asserted against the catalog so the day a rung adds them, this fails and the key seed is owed. + /// The record of the exemption that used to live here, and the fact that retired it. From #3614 until + /// Darling V128 / Lite v61 this test asserted that query_stats persisted NEITHER statement offset, so + /// the day a rung added them the exemption would go red and the key seed would be owed. The rung added + /// them; this now asserts the inverse — both offsets are stored, as the Integer the DMV reports, at the + /// TAIL of the payload (both stores' writers are positional) — and that the collector still keys on + /// exactly the string the seeders rebuild, so the key seed is about THIS key and not a guess. The + /// exemption set is empty, and asserted so. /// [Fact] - public void ThePassWindowOnlyExemption_RestsOnTheOffsetsNotBeingStored() + public void TheOffsetsAreStored_SoNoFamilyIsPassWindowOnly() { - Assert.All(PassWindowOnly, family => Assert.Contains(family, CollectorDeltaCalculator.DeltaFamilyCollectors)); + Assert.Empty(PassWindowOnly); var queryStats = CollectorCatalog.Find("query_stats")!; - Assert.DoesNotContain(queryStats.PayloadColumns, c => c.Name == "statement_start_offset"); - Assert.DoesNotContain(queryStats.PayloadColumns, c => c.Name == "statement_end_offset"); - - /* And the collector really does key on them, so the exemption is about THIS key and not a guess. */ + var start = queryStats.PayloadColumns.Single(c => c.Name == "statement_start_offset"); + var end = queryStats.PayloadColumns.Single(c => c.Name == "statement_end_offset"); + Assert.Equal(CollectorColumnType.Integer, start.Type); + Assert.Equal(CollectorColumnType.Integer, end.Type); + Assert.Equal(queryStats.PayloadColumns.Count - 2, queryStats.PayloadColumns.ToList().IndexOf(start)); + Assert.Equal(queryStats.PayloadColumns.Count - 1, queryStats.PayloadColumns.ToList().IndexOf(end)); + + /* The collector's key, verbatim, and both seeders rebuilding it with the same interpolation over the + same raw parts — so a null handle formats as empty on both sides and the offsets (-1 included) are + spelled by the same int formatting. A normalized form on either side would seed nothing, silently. */ var source = ReadRepoFile(CollectorsDir + "/QueryStatsCollector.cs"); Assert.Contains("$\"{row.SqlHandle}:{row.StatementStartOffset}:{row.StatementEndOffset}:{row.PlanHandle}\"", source, StringComparison.Ordinal); + foreach (var host in Hosts()) + { + Assert.Contains("$\"{sqlHandle}:{reader.GetInt32(2)}:{reader.GetInt32(3)}:{planHandle}\"", host.Source, StringComparison.Ordinal); + /* NULL offsets (pre-V128 / pre-v61 rows) seed no key but still feed the pass window: the + skip sits AFTER the Observe, in the seeder, not in the SQL. */ + var seeder = host.Source[host.Source.IndexOf("SeedQueryStatsAsync(", StringComparison.Ordinal)..]; + var observe = seeder.IndexOf("passes.Observe(serverId, ts);", StringComparison.Ordinal); + var skip = seeder.IndexOf("if (reader.IsDBNull(2) || reader.IsDBNull(3))", StringComparison.Ordinal); + Assert.True(observe >= 0 && skip > observe, $"{host.Label}: the query_stats seeder must observe the pass window from every row BEFORE skipping a NULL-offset row's key"); + } } /// diff --git a/Lite.Tests/DeltaFamilyUnknowableRowReadTests.cs b/Lite.Tests/DeltaFamilyUnknowableRowReadTests.cs index be81f92a4..2d0dada51 100644 --- a/Lite.Tests/DeltaFamilyUnknowableRowReadTests.cs +++ b/Lite.Tests/DeltaFamilyUnknowableRowReadTests.cs @@ -238,6 +238,95 @@ public async Task LatchAndSpinlockTrends_DropTheUnknowableRow_PreferTheStoredInt Assert.Equal(10.0, spin[1].CollisionsPerSecond, precision: 6); } + /// + /// #3540 (v61): the procedure duration trend, the read that LAG-divided procedure_stats' fabricated + /// zero into a confident 0.00 ms/sec. Four collections five minutes apart: t1/t2 are pre-v61 collections + /// (NULL interval) — t1 has no prior and is not a point, t2 divides by the LAG's 300 s. t3 is a restart: + /// every row stores 0, so MAX is 0 and the collection is absent. t4 is a steady pass with a plan the + /// TOP (150) just readmitted (its row stores 0 beside a 0 delta) beside a measured row (120 s), so MAX is + /// 120 — the stored interval wins over the LAG's 300 — and the readmitted plan adds nothing to the sums. + /// + [Fact] + public async Task ProcedureDurationTrend_DropsTheUnknowableCollection_PrefersTheStoredInterval_KeepsPreV61History() + { + var t1 = Truncate(DateTime.UtcNow.AddHours(-2)); + var t2 = t1.AddMinutes(5); + var t3 = t2.AddMinutes(5); + var t4 = t3.AddMinutes(5); + + await SeedProcedureAsync(t1, "usp_A", deltaExecutions: 5, deltaElapsedUs: 100_000, interval: null); + await SeedProcedureAsync(t2, "usp_A", deltaExecutions: 30, deltaElapsedUs: 600_000, interval: null); + await SeedProcedureAsync(t3, "usp_A", deltaExecutions: 0, deltaElapsedUs: 0, interval: 0); + await SeedProcedureAsync(t4, "usp_A", deltaExecutions: 24, deltaElapsedUs: 1_200_000, interval: 120); + await SeedProcedureAsync(t4, "usp_New", deltaExecutions: 0, deltaElapsedUs: 0, interval: 0); + + var points = await _dataService.GetProcedureDurationTrendAsync(ServerId, hoursBack: 3); + + Assert.Equal(new[] { t2, t4 }, points.Select(p => p.CollectionTime).ToArray()); + + /* t2 (pre-v61): 600 ms / 300 s = 2.0 ms/sec; 30 / 300 = 0.1 executions/sec. */ + Assert.Equal(2.0, points[0].Value, precision: 6); + Assert.Equal(0.1, points[0].ExecutionsPerSecond, precision: 6); + + /* t4: the STORED 120 s — 1200 / 120 = 10.0, not the LAG's 1200 / 300 = 4.0; 24 / 120 = 0.2. */ + Assert.Equal(10.0, points[1].Value, precision: 6); + Assert.Equal(0.2, points[1].ExecutionsPerSecond, precision: 6); + } + + /// + /// #3540 (v61): the procedure history grid's "Interval (sec)" column shows the row's STORED interval where + /// it has one — the marker's 0 INCLUDED, which is what the query-stats history has always shown for an + /// unknowable row (a displayed interval is not a rate, so 0 is honest here where it would be a lie in a + /// division) — and the LAG-derived gap for a pre-v61 row that never recorded one. + /// + [Fact] + public async Task ProcedureHistory_ShowsTheStoredIntervalIncludingTheMarker_DerivesOnlyForPreV61Rows() + { + var t1 = Truncate(DateTime.UtcNow.AddHours(-2)); + var t2 = t1.AddMinutes(5); + var t3 = t2.AddMinutes(5); + var t4 = t3.AddMinutes(5); + + await SeedProcedureAsync(t1, "usp_H", 5, 100_000, interval: null); + await SeedProcedureAsync(t2, "usp_H", 30, 600_000, interval: null); + await SeedProcedureAsync(t3, "usp_H", 0, 0, interval: 0); + await SeedProcedureAsync(t4, "usp_H", 24, 1_200_000, interval: 120); + + var rows = await _dataService.GetProcedureStatsHistoryAsync(ServerId, "AppDb", "dbo", "usp_H", hoursBack: 3); + + Assert.Equal(new[] { t1, t2, t3, t4 }, rows.Select(r => r.CollectionTime).ToArray()); + Assert.Null(rows[0].SampleIntervalSeconds); /* pre-v61, no prior: nothing to derive from */ + Assert.Equal(300, rows[1].SampleIntervalSeconds); /* pre-v61: the LAG gap */ + Assert.Equal(0, rows[2].SampleIntervalSeconds); /* the marker, shown as the 0 it is */ + Assert.Equal(120, rows[3].SampleIntervalSeconds); /* stored, not the LAG's 300 */ + } + + /// + /// #3540 (v61): the resource-semaphore snapshot carries the stored interval and IsUnknowable is + /// true ONLY for the marker — never for a pre-v61 NULL, which is "never recorded" rather than + /// "unknowable". Three semaphores at the latest collection: a marker, a measured row, a pre-v61 row. + /// + [Fact] + public async Task ResourceSemaphoreSnapshot_CarriesTheStoredInterval_UnknowableOnlyForTheMarker() + { + var t = Truncate(DateTime.UtcNow.AddMinutes(-2)); + + await SeedMemoryGrantAsync(t, poolId: 1, interval: 0); + await SeedMemoryGrantAsync(t, poolId: 2, interval: 120); + await SeedMemoryGrantAsync(t, poolId: 3, interval: null); + + var rows = await _dataService.GetResourceSemaphoreSnapshotAsync(ServerId, hoursBack: 1); + var byPool = rows.ToDictionary(r => r.PoolId); + Assert.Equal(3, byPool.Count); + + Assert.Equal(0, byPool[1].SampleIntervalSeconds); + Assert.True(byPool[1].IsUnknowable); + Assert.Equal(120, byPool[2].SampleIntervalSeconds); + Assert.False(byPool[2].IsUnknowable); + Assert.Null(byPool[3].SampleIntervalSeconds); + Assert.False(byPool[3].IsUnknowable); + } + /* ---- seeding ---------------------------------------------------------------------------------------- */ private static DateTime Truncate(DateTime value) => @@ -293,6 +382,43 @@ private async Task SeedFileIoAsync(DateTime at, long reads, long writes, long st await cmd.ExecuteNonQueryAsync(); } + private async Task SeedProcedureAsync(DateTime at, string objectName, long deltaExecutions, long deltaElapsedUs, int? interval) + { + using var readLock = _duckDb.AcquireReadLock(); + var conn = await SeedConnectionAsync(); + using var cmd = conn.CreateCommand(); + cmd.CommandText = @"INSERT INTO procedure_stats + (collection_id, collection_time, server_id, server_name, database_name, schema_name, object_name, object_type, + execution_count, total_worker_time, total_elapsed_time, total_logical_reads, total_physical_reads, total_logical_writes, + delta_execution_count, delta_worker_time, delta_elapsed_time, sample_interval_seconds) + VALUES ($1, $2, $3, $4, 'AppDb', 'dbo', $5, 'PROCEDURE', 0, 0, 0, 0, 0, 0, $6, 0, $7, $8)"; + foreach (var v in new object[] { _nextId--, at, ServerId, ServerName, objectName, deltaExecutions, deltaElapsedUs, IntervalValue(interval) }) + { + cmd.Parameters.Add(new DuckDBParameter { Value = v }); + } + + await cmd.ExecuteNonQueryAsync(); + } + + private async Task SeedMemoryGrantAsync(DateTime at, int poolId, int? interval) + { + using var readLock = _duckDb.AcquireReadLock(); + var conn = await SeedConnectionAsync(); + using var cmd = conn.CreateCommand(); + cmd.CommandText = @"INSERT INTO memory_grant_stats + (collection_id, collection_time, server_id, server_name, resource_semaphore_id, pool_id, + target_memory_mb, max_target_memory_mb, total_memory_mb, available_memory_mb, granted_memory_mb, used_memory_mb, + grantee_count, waiter_count, timeout_error_count, forced_grant_count, + timeout_error_count_delta, forced_grant_count_delta, sample_interval_seconds) + VALUES ($1, $2, $3, $4, 0, $5, 100, 200, 90, 80, 10, 8, 3, 1, 5, 2, 0, 0, $6)"; + foreach (var v in new object[] { _nextId--, at, ServerId, ServerName, poolId, IntervalValue(interval) }) + { + cmd.Parameters.Add(new DuckDBParameter { Value = v }); + } + + await cmd.ExecuteNonQueryAsync(); + } + private async Task SeedLatchAsync(DateTime at, long deltaWait, int? interval) { using var readLock = _duckDb.AcquireReadLock(); diff --git a/Lite.Tests/DuckDbSchemaEquivalenceTests.cs b/Lite.Tests/DuckDbSchemaEquivalenceTests.cs index 3813d3cbc..c85a39c7f 100644 --- a/Lite.Tests/DuckDbSchemaEquivalenceTests.cs +++ b/Lite.Tests/DuckDbSchemaEquivalenceTests.cs @@ -193,6 +193,17 @@ fail until the oracle is extended by hand. That is the whole point of an oracle. /// 's v60 migration adds it to existing databases; /// DeltaFamilyIntervalColumnTests is the census that keeps a seventh delta family from shipping /// without it. + /// + /// #3540 / schema v61: the completion. sample_interval_seconds on procedure_stats and + /// memory_grant_stats — the last two Lite-stored delta families without it, so every family Lite + /// stores now carries the interval and the census's still-naked list is empty. And + /// query_stats.statement_start_offset / statement_end_offset: the two halves of that + /// family's delta key the store never persisted, so the restart seed could restore its pass window but + /// not one baseline. INTEGER (the DMV's type), nullable, trailing; BYTE offsets into the batch's nvarchar + /// text, -1 as the end offset meaning "to the end of the batch", stored raw because the key string + /// is built over the raw values. NULL on every pre-v61 row: "never recorded", which the seed reads as + /// "no key can be rebuilt from this row". 's v61 migration adds all four + /// to existing databases. /// private static readonly HashSet IntentionalAppendedColumns = new(StringComparer.Ordinal) { @@ -203,6 +214,10 @@ fail until the oracle is extended by hand. That is the whole point of an oracle. "file_io_stats.sample_interval_seconds", "latch_stats.sample_interval_seconds", "spinlock_stats.sample_interval_seconds", + "procedure_stats.sample_interval_seconds", + "memory_grant_stats.sample_interval_seconds", + "query_stats.statement_start_offset", + "query_stats.statement_end_offset", }; [Fact] diff --git a/Lite.Tests/LiteDeltaSeederTests.cs b/Lite.Tests/LiteDeltaSeederTests.cs index 255aa7eb2..2b57e0ada 100644 --- a/Lite.Tests/LiteDeltaSeederTests.cs +++ b/Lite.Tests/LiteDeltaSeederTests.cs @@ -96,20 +96,24 @@ public void ProcedureStatsSeedSql_BuildsTheCollectorsKey_AndBoundsItsOnlyTableRe } /// - /// query_stats seeds its PASS WINDOW only: the store persists neither statement offset the delta key - /// carries, so no row can reproduce the key. The read is the distinct collection times per server - /// inside the cutoff and nothing else. + /// query_stats is key-seeded since v61 (#3540): the store persists both statement offsets now, so the + /// read partitions by the collector's FULL key — sql_handle, both offsets, plan_handle — and returns + /// each key's latest row inside the window. The offsets are selected RAW (no COALESCE, no arithmetic) + /// because the seeder rebuilds the key from them with the collector's own interpolation, and there is + /// no offset filter in the SQL: a pre-v61 row (NULL offsets) is read for the pass window and skipped + /// for keys in C#, so one read serves both halves. /// [Fact] - public void QueryStatsPassSeedSql_IsThePassWindowOnly() + public void QueryStatsSeedSql_PartitionsByTheFullDeltaKeyAndSelectsTheOffsetsRaw() { - var sql = DeltaCalculator.QueryStatsPassSeedSql; - Assert.Contains("SELECT server_id, collection_time", sql, StringComparison.Ordinal); + var sql = DeltaCalculator.QueryStatsSeedSql; + Assert.Contains("SELECT DISTINCT ON (server_id, sql_handle, statement_start_offset, statement_end_offset, plan_handle)", sql, StringComparison.Ordinal); Assert.Contains("FROM query_stats", sql, StringComparison.Ordinal); - Assert.Contains("GROUP BY server_id, collection_time", sql, StringComparison.Ordinal); + Assert.Contains("ORDER BY server_id, sql_handle, statement_start_offset, statement_end_offset, plan_handle, collection_time DESC", sql, StringComparison.Ordinal); Assert.Equal(1, CountOccurrences(sql, "collection_time >= $1")); - Assert.DoesNotContain("sql_handle", sql, StringComparison.Ordinal); - Assert.DoesNotContain("plan_handle", sql, StringComparison.Ordinal); + Assert.DoesNotContain("IS NOT NULL", sql, StringComparison.Ordinal); + Assert.DoesNotContain("COALESCE", sql, StringComparison.Ordinal); + Assert.DoesNotContain("GROUP BY", sql, StringComparison.Ordinal); } /// @@ -311,13 +315,71 @@ NOT from the stale row (1), which would make this 8. */ } /// - /// THE pass-window pin (#3540 A4 / #2235). query_stats has no key seed — the store cannot reproduce - /// its key — but its pass window is seeded from the table's collection times, so on the FIRST - /// post-restart pass a plan compiled since the last pre-restart pass is credited in full with a + /// #3540 (v61): query_stats is KEY-seeded from the rows that carry the offsets, with the key spelled exactly + /// as the collector spells it — the raw offsets, -1 included. A key present only in the OLDER pass + /// (it fell out of the TOP (n)) is restored from that pass, the per-key shape's whole point. A pre-v61 row + /// (NULL offsets) seeds NO key — not even under a normalized guess — while its collection time still + /// feeds the pass window (the next test). Every expected value was worked by hand from the rows and + /// executed against the real seeder on DuckDB before this was written. + /// + [Fact] + public async Task Seed_QueryStats_RestoresKeysFromRowsWithOffsets_SpelledAsTheCollectorSpellsThem() + { + var older = DateTime.UtcNow.AddMinutes(-4); + var latest = DateTime.UtcNow.AddMinutes(-2); + + /* The whole-batch statement (0, -1) in both passes; a second statement of the same batch/plan + (100, 240) only in the older pass; a null-handle row; and a pre-v61 row with no offsets. */ + await InsertKeyedQueryStatsAsync(RecentServerId, older, "0xSH1", 0, -1, "0xPH1", 10); + await InsertKeyedQueryStatsAsync(RecentServerId, latest, "0xSH1", 0, -1, "0xPH1", 15); + await InsertKeyedQueryStatsAsync(RecentServerId, older, "0xSH1", 100, 240, "0xPH1", 7); + await InsertKeyedQueryStatsAsync(RecentServerId, latest, null, 0, -1, null, 3); + await InsertQueryStatsAsync(RecentServerId, latest); + + var deltas = new DeltaCalculator(NullLogger.Instance); + await deltas.SeedFromDatabaseAsync(_duckDb); + + var now = DateTime.UtcNow; + const int Gap = CollectorDeltaCalculator.DefaultMaxGapSeconds; + + /* The latest pass is the baseline for the key both passes wrote: 18 - 15, over ~120 s. The key + carries the raw -1, as the collector's $"{sh}:{start}:{end}:{ph}" does. */ + Assert.Equal(3, deltas.CalculateDeltaWithInterval(RecentServerId, "query_stats_exec", "0xSH1:0:-1:0xPH1", 18, out var interval, now, Gap)); + Assert.InRange(interval, 118, 122); + /* Every one of the eight groups, from the same row (counters are multiples of the execution count). */ + Assert.Equal(30, deltas.CalculateDelta(RecentServerId, "query_stats_worker", "0xSH1:0:-1:0xPH1", 180, now, Gap)); + Assert.Equal(60, deltas.CalculateDelta(RecentServerId, "query_stats_elapsed", "0xSH1:0:-1:0xPH1", 360, now, Gap)); + Assert.Equal(90, deltas.CalculateDelta(RecentServerId, "query_stats_reads", "0xSH1:0:-1:0xPH1", 540, now, Gap)); + Assert.Equal(120, deltas.CalculateDelta(RecentServerId, "query_stats_writes", "0xSH1:0:-1:0xPH1", 720, now, Gap)); + Assert.Equal(150, deltas.CalculateDelta(RecentServerId, "query_stats_phys_reads", "0xSH1:0:-1:0xPH1", 900, now, Gap)); + Assert.Equal(180, deltas.CalculateDelta(RecentServerId, "query_stats_rows", "0xSH1:0:-1:0xPH1", 1080, now, Gap)); + Assert.Equal(210, deltas.CalculateDelta(RecentServerId, "query_stats_spills", "0xSH1:0:-1:0xPH1", 1260, now, Gap)); + + /* The statement that fell out of the TOP (n) on the latest pass: seeded from its OLDER row, over + ~240 s. The latest-collection shape would have missed it and this would be 0. */ + Assert.Equal(2, deltas.CalculateDeltaWithInterval(RecentServerId, "query_stats_exec", "0xSH1:100:240:0xPH1", 9, out var span, now, Gap)); + Assert.InRange(span, 238, 242); + + /* A null handle formats as EMPTY on both sides — the collector's interpolation and the seeder's. */ + Assert.Equal(2, deltas.CalculateDelta(RecentServerId, "query_stats_exec", ":0:-1:", 5, now, Gap)); + + /* The pre-v61 row seeded nothing: not under the guess a normalizing seeder would have made (0, 0), + nor under the whole-batch pair. Gap policy OFF so a first sighting is the only way to read 0 here — + a seeded baseline of 1 would return 4. */ + Assert.Equal(0, deltas.CalculateDelta(RecentServerId, "query_stats_exec", "sh:0:0:ph", 5, now, 0)); + Assert.Equal(0, deltas.CalculateDelta(RecentServerId, "query_stats_exec", "sh:0:-1:ph", 5, now, 0)); + } + + /// + /// THE pass-window pin (#3540 A4 / #2235). Until v61 query_stats had no key seed — the store could not + /// reproduce its key — and only its pass window was seeded from the table's collection times, so on the + /// FIRST post-restart pass a plan compiled since the last pre-restart pass is credited in full with a /// real interval, while a plan older than that gap baselines honestly. On an unseeded calculator both - /// are (0, 0): the rescue was inert on exactly the cycle it exists for. The window is also seeded - /// for the ORIGINAL families (wait_stats here), proven through the same path; and a server whose - /// only rows predate the window gets no pass window. + /// are (0, 0): the rescue was inert on exactly the cycle it exists for. The rows here are PRE-v61 rows + /// (no offsets), which is what makes this also the pin that the v61 key seed still feeds the pass window + /// from every row: the first restart after the upgrade sees only such rows. The window is also seeded + /// for the ORIGINAL families (wait_stats here), proven through the same path; and a server whose only + /// rows predate the window gets no pass window. /// [Fact] public async Task Seed_PassWindow_ArmsTheSeriesAgeRescueOnTheFirstPostRestartPass() @@ -500,6 +562,7 @@ INSERT INTO procedure_stats await cmd.ExecuteNonQueryAsync(); } + /// A pre-v61 query_stats row: no offsets stored, so it can feed the pass window and nothing else. private async Task InsertQueryStatsAsync(int serverId, DateTime collectionTimeUtc) { using var readLock = _duckDb.AcquireReadLock(); @@ -515,4 +578,36 @@ INSERT INTO query_stats cmd.Parameters.Add(new DuckDBParameter { Value = "delta-seed-window" }); await cmd.ExecuteNonQueryAsync(); } + + /// A v61 query_stats row: the two offsets stored raw, the eight counters as multiples of the + /// execution count so every group's expected delta is derivable. + private async Task InsertKeyedQueryStatsAsync(int serverId, DateTime collectionTimeUtc, string? sqlHandle, int start, int end, string? planHandle, long executions) + { + using var readLock = _duckDb.AcquireReadLock(); + var connection = await SeedConnectionAsync(); + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +INSERT INTO query_stats + (collection_id, collection_time, server_id, server_name, query_hash, sql_handle, plan_handle, + statement_start_offset, statement_end_offset, + execution_count, total_worker_time, total_elapsed_time, total_logical_reads, total_logical_writes, total_physical_reads, total_rows, total_spills) +VALUES ($1, $2, $3, $4, 'qh', $5, $6, $7, $8, $9, $10, $11, $12, $13, $14, $15, $16)"; + cmd.Parameters.Add(new DuckDBParameter { Value = _nextId++ }); + cmd.Parameters.Add(new DuckDBParameter { Value = Truncate(collectionTimeUtc) }); + cmd.Parameters.Add(new DuckDBParameter { Value = serverId }); + cmd.Parameters.Add(new DuckDBParameter { Value = "delta-seed-window" }); + cmd.Parameters.Add(new DuckDBParameter { Value = (object?)sqlHandle ?? DBNull.Value }); + cmd.Parameters.Add(new DuckDBParameter { Value = (object?)planHandle ?? DBNull.Value }); + cmd.Parameters.Add(new DuckDBParameter { Value = start }); + cmd.Parameters.Add(new DuckDBParameter { Value = end }); + cmd.Parameters.Add(new DuckDBParameter { Value = executions }); + cmd.Parameters.Add(new DuckDBParameter { Value = executions * 10 }); + cmd.Parameters.Add(new DuckDBParameter { Value = executions * 20 }); + cmd.Parameters.Add(new DuckDBParameter { Value = executions * 30 }); + cmd.Parameters.Add(new DuckDBParameter { Value = executions * 40 }); + cmd.Parameters.Add(new DuckDBParameter { Value = executions * 50 }); + cmd.Parameters.Add(new DuckDBParameter { Value = executions * 60 }); + cmd.Parameters.Add(new DuckDBParameter { Value = executions * 70 }); + await cmd.ExecuteNonQueryAsync(); + } } diff --git a/Lite.Tests/MemoryGrantsCollectorDefinitionTests.cs b/Lite.Tests/MemoryGrantsCollectorDefinitionTests.cs index 5a0f54def..12b5f78c4 100644 --- a/Lite.Tests/MemoryGrantsCollectorDefinitionTests.cs +++ b/Lite.Tests/MemoryGrantsCollectorDefinitionTests.cs @@ -44,8 +44,19 @@ public void PayloadColumns_MatchSchemaOrder() "forced_grant_count", "timeout_error_count_delta", "forced_grant_count_delta", + "sample_interval_seconds", }, names); + + /* #3540 (v61): the interval is the TRAILING column, in the same INTEGER type perfmon_stats and + query_stats have always used, so the two stores' positional writers land it after every + pre-existing column and one NULLIF idiom reads every family. */ + var interval = MemoryGrantsCollector.Instance.PayloadColumns[^1]; + Assert.Equal("sample_interval_seconds", interval.Name); + Assert.Equal(CollectorColumnType.Integer, interval.Type); + Assert.Equal( + PerfmonStatsCollector.Instance.PayloadColumns.Single(c => c.Name == "sample_interval_seconds").Type, + interval.Type); } [Fact] @@ -80,8 +91,10 @@ public void WritePayload_EmitsSchemaOrder_AndPinsCompositeDeltaKey() MemoryGrantsCollector.Instance.WritePayload(row, writer, context); + /* Payload order: raw values, the two deltas (recording calculator returns value * 10), then the + measured interval (#3540, v61) — the fake reports 0, the calculator's "no delta knowable" marker. */ Assert.Equal( - new object?[] { (short)1, 2, 100.5m, 200.5m, 90.25m, 80.75m, 10.5m, 8.25m, 3, 4, 5L, 6L, 50L, 60L }, + new object?[] { (short)1, 2, 100.5m, 200.5m, 90.25m, 80.75m, 10.5m, 8.25m, 3, 4, 5L, 6L, 50L, 60L, 0 }, writer.Values); /* Delta contract: composite key "{pool}_{semaphore}", both groups, the shared gap policy. */ @@ -89,4 +102,39 @@ public void WritePayload_EmitsSchemaOrder_AndPinsCompositeDeltaKey() Assert.Equal(("memory_grants_timeouts", "2_1", 5L, context.CollectionTime, CollectorDeltaCalculator.DefaultMaxGapSeconds), deltas.Calls[0]); Assert.Equal(("memory_grants_forced", "2_1", 6L, context.CollectionTime, CollectorDeltaCalculator.DefaultMaxGapSeconds), deltas.Calls[1]); } + + /// + /// #3540 (v61): the interval reaches the payload MEASURED, not as a constant. A distinctive value (neither + /// 0 nor a plausible cadence) so this can only pass if what the calculator reported is what was written. + /// + [Fact] + public void WritePayload_WritesTheMeasuredInterval() + { + var deltas = new RecordingCollectorDeltaCalculator { ReportedInterval = 137 }; + var context = CollectorTestContext.Make(deltas); + var writer = new RecordingCollectorRowWriter(); + + MemoryGrantsCollector.Instance.WritePayload(new MemoryGrantsCollector.Row(1, 2, 100.5m, 200.5m, 90.25m, 80.75m, 10.5m, 8.25m, 3, 4, 5L, 6L), writer, context); + + Assert.Equal(137, writer.Values[^1]); + } + + /// + /// #3540: one interval per ROW, the MINIMUM over the row's two delta groups. If either group's delta is + /// unknowable (interval 0) the row is stored as (…, 0), so no reader divides a reset counter's 0 by its + /// sibling's real interval and reads it as a quiet semaphore. + /// + [Fact] + public void WritePayload_StoresTheMinimumIntervalAcrossTheRowsDeltaGroups() + { + var deltas = new RecordingCollectorDeltaCalculator { ReportedInterval = 300 }; + deltas.IntervalByGroup["memory_grants_forced"] = 0; + var context = CollectorTestContext.Make(deltas); + var writer = new RecordingCollectorRowWriter(); + + MemoryGrantsCollector.Instance.WritePayload(new MemoryGrantsCollector.Row(1, 2, 100.5m, 200.5m, 90.25m, 80.75m, 10.5m, 8.25m, 3, 4, 5L, 6L), writer, context); + + Assert.Equal(0, writer.Values[^1]); + Assert.Equal(2, deltas.Calls.Count); + } } diff --git a/Lite.Tests/PerformanceTrendsToolTests.cs b/Lite.Tests/PerformanceTrendsToolTests.cs index e9e04b5ef..c957ffc26 100644 --- a/Lite.Tests/PerformanceTrendsToolTests.cs +++ b/Lite.Tests/PerformanceTrendsToolTests.cs @@ -135,7 +135,11 @@ public async Task TheExecutionRate_SurvivesBeingBelowOnePerSecond() { var service = new LocalDataService(_duckDb); - /* Two snapshots five minutes apart, two executions between them: 0.0067/sec. */ + /* Two snapshots five minutes apart, two executions between them: 0.0067/sec. The rows carry no + sample_interval_seconds (the pre-v61 shape), so the read LAG-derives the interval — and since v61 + (#3540) the FIRST snapshot, which has nothing to LAG against, is absent rather than a fabricated + 0.0 point (the correction v60 made for the wait trends). One point comes back: the second snapshot, + whose rate is the thing under test. */ var baseNow = Truncate(DateTime.UtcNow); await SeedProcedureAsync(baseNow.AddMinutes(-20), executions: 0, elapsedUs: 0); await SeedProcedureAsync(baseNow.AddMinutes(-15), executions: 2, elapsedUs: 600_000); @@ -143,9 +147,9 @@ public async Task TheExecutionRate_SurvivesBeingBelowOnePerSecond() var hit = await McpQueryTools.GetProcedureDurationTrend(service, _serverManager, ServerName, 4); var root = JsonDocument.Parse(hit).RootElement; var trend = root.GetProperty("trend"); - Assert.Equal(2, trend.GetArrayLength()); + Assert.Equal(1, trend.GetArrayLength()); - var second = trend[1]; + var second = trend[0]; Assert.True(second.GetProperty("value").GetDouble() > 0, "elapsed ms/sec must be a real rate"); /* The shipped integer field rounds this to an idle server. The double is why it is here. */ @@ -160,9 +164,10 @@ execution_count precedent above. */ /* #3541 A2: the disclosure block, with Lite's truth. One tier (raw, per-collection, no aggregate - note), and the series the store held begins at the 20-minutes-ago seed — effective_start says so, - and because that head sits three-plus hours past the requested 4-hour start, `truncated` is true. - The label describes the data, not the request; that is the whole contract. + note), and the series the read SERVED begins at its first point — the 15-minutes-ago seed, since + v61 dropped the prior-less first snapshot — effective_start says so, and because that head sits + three-plus hours past the requested 4-hour start, `truncated` is true. The label describes the + data, not the request; that is the whole contract. */ AssertDisclosureBlock(root); Assert.Equal("raw", root.GetProperty("source").GetString()); @@ -192,7 +197,9 @@ public async Task ASeriesThatBeginsWhereItWasAskedTo_IsNotTruncated() var root = JsonDocument.Parse(await McpQueryTools.GetProcedureDurationTrend(service, _serverManager, ServerName, 1)).RootElement; - Assert.Equal(2, root.GetProperty("trend").GetArrayLength()); + /* One point, not two: these pre-v61 rows LAG-derive, and the prior-less first snapshot is absent since + v61 (#3540). The head is the 50-minutes-ago point, inside the slack — the property under test. */ + Assert.Equal(1, root.GetProperty("trend").GetArrayLength()); Assert.False(root.GetProperty("truncated").GetBoolean()); Assert.Equal(root.GetProperty("trend")[0].GetProperty("time").GetString(), root.GetProperty("effective_start").GetString()); Assert.InRange(root.GetProperty("effective_hours_back").GetDouble(), 0.8, 1.0); diff --git a/Lite.Tests/PgStatementStatsDeltaSkipTests.cs b/Lite.Tests/PgStatementStatsDeltaSkipTests.cs index c2ca01009..da4c21354 100644 --- a/Lite.Tests/PgStatementStatsDeltaSkipTests.cs +++ b/Lite.Tests/PgStatementStatsDeltaSkipTests.cs @@ -80,7 +80,9 @@ public async Task FirstSighting_ShipsEvenWithZeroCalls() var rows = await ReadAsync(Row(queryId: 1, calls: 0, totalExecTimeMs: 0), T0, deltas); - Assert.Single(rows); + /* #3540 (V128): and it ships as the (0, 0) marker — interval 0 is what tells a reader this row's + zero deltas are unknowable rather than a confirmed idle interval. */ + Assert.Equal(0, Assert.Single(rows).SampleIntervalSeconds); } [Fact] @@ -106,6 +108,8 @@ public async Task RepeatWithNewCalls_ShipsWithTheMeasuredDeltas() Assert.Equal(50, row.DeltaCalls); Assert.Equal(500, row.DeltaTotalExecTimeMs); Assert.Equal(15, row.DeltaRows); + /* #3540 (V128): the measured span the three deltas accrued over, stored beside them. */ + Assert.Equal(60, row.SampleIntervalSeconds); } /// @@ -122,7 +126,9 @@ public async Task CounterReset_Ships() var afterReset = await ReadAsync(Row(queryId: 4, calls: 40, totalExecTimeMs: 300), T0.AddSeconds(60), deltas); - Assert.Single(afterReset); + /* #3540 (V128): a reset row ships with interval 0, not the 60 s that elapsed — the elapsed span is + real, but no delta is knowable over it, and a stored 60 beside a 0 delta would read as idle. */ + Assert.Equal(0, Assert.Single(afterReset).SampleIntervalSeconds); } /// diff --git a/Lite.Tests/PgStatementStatsFlavorTests.cs b/Lite.Tests/PgStatementStatsFlavorTests.cs index 5e1c39b31..6428f1a3c 100644 --- a/Lite.Tests/PgStatementStatsFlavorTests.cs +++ b/Lite.Tests/PgStatementStatsFlavorTests.cs @@ -100,7 +100,9 @@ public void BothFlavorsSelectTheSameColumnsInTheSameOrder() var vanilla = SelectAliases(Sql(isAurora: false)); Assert.Equal(aurora, vanilla); - Assert.Equal(PgStatementStatsCollector.Instance.PayloadColumns.Count - 3, aurora.Count); + /* The SELECT list is the payload minus the four columns computed on the client: the three deltas + and, since V128 (#3540), the interval they accrued over. */ + Assert.Equal(PgStatementStatsCollector.Instance.PayloadColumns.Count - 4, aurora.Count); } /// diff --git a/Lite.Tests/PgWaitStatsCollectorDefinitionTests.cs b/Lite.Tests/PgWaitStatsCollectorDefinitionTests.cs index 67d4f3d6a..41d2dadcc 100644 --- a/Lite.Tests/PgWaitStatsCollectorDefinitionTests.cs +++ b/Lite.Tests/PgWaitStatsCollectorDefinitionTests.cs @@ -25,13 +25,13 @@ public class PgWaitStatsCollectorDefinitionTests { private static readonly RecordingCollectorDeltaCalculator s_deltas = new(); - private static CollectorContext MakeContext() + private static CollectorContext MakeContext(RecordingCollectorDeltaCalculator? deltas = null) => new() { ServerId = 42, ServerName = "aurora-writer", CollectionTime = new DateTime(2026, 8, 11, 12, 0, 0, DateTimeKind.Utc), - Deltas = s_deltas, + Deltas = deltas ?? s_deltas, Target = new CollectorTargetInfo { Engine = CollectorTargetEngine.PostgreSql, @@ -90,6 +90,9 @@ public void PayloadColumns_OrderAndTypes_Pinned() ("wait_time_us", CollectorColumnType.BigInt), ("delta_waits", CollectorColumnType.BigInt), ("delta_wait_time_us", CollectorColumnType.BigInt), + /* #3540 (Darling V128): the TRAILING column, in the same Integer perfmon_stats and query_stats + have always used, so the positional COPY writer lands it after every pre-existing column. */ + ("sample_interval_seconds", CollectorColumnType.Integer), }; var actual = PgWaitStatsCollector.Instance.PayloadColumns; @@ -99,6 +102,10 @@ public void PayloadColumns_OrderAndTypes_Pinned() Assert.Equal(expected[i].Name, actual[i].Name); Assert.Equal(expected[i].Type, actual[i].Type); } + + Assert.Equal( + PerfmonStatsCollector.Instance.PayloadColumns.Single(c => c.Name == "sample_interval_seconds").Type, + actual[^1].Type); } /// @@ -225,7 +232,7 @@ public void WritePayload_EmitsEveryDeclaredColumnInOrder() { var writer = new RecordingCollectorRowWriter(); PgWaitStatsCollector.Instance.WritePayload( - new PgWaitStatsCollector.Row(10, 167772160L, "IO", "DataFileRead", 100L, 5000L, DeltaWaits: 7L, DeltaWaitTime: 250L), + new PgWaitStatsCollector.Row(10, 167772160L, "IO", "DataFileRead", 100L, 5000L, DeltaWaits: 7L, DeltaWaitTime: 250L, SampleIntervalSeconds: 137), writer, MakeContext()); @@ -238,6 +245,35 @@ public void WritePayload_EmitsEveryDeclaredColumnInOrder() Assert.Equal(5000L, writer.Values[5]); Assert.Equal(7L, writer.Values[6]); Assert.Equal(250L, writer.Values[7]); + /* #3540 (V128): the row's measured interval reaches the payload as read — a distinctive value, so + this passes only if the ROW's interval (computed in ReadAsync) is what is written, not a constant. */ + Assert.Equal(137, writer.Values[8]); + } + + /// + /// #3540 (V128): the interval ReadAsync computes beside the deltas is the MINIMUM over the row's two + /// groups, so a row whose wait-time series alone is unknowable is stored as (…, 0) and no reader divides + /// that group's 0 by the waits series' real span. Reads through the recording fake with a per-group + /// override, the way the V127 collectors' minimum rule is pinned. + /// + [Fact] + public async Task ReadAsync_CarriesTheMinimumIntervalOverTheRowsDeltaGroups() + { + var deltas = new RecordingCollectorDeltaCalculator { ReportedInterval = 300 }; + deltas.IntervalByGroup["pg_wait_stats_time"] = 0; + + var reader = new FakeCollectorDataReader( + new object[] { 10, 167772160L, "IO", "DataFileRead", 100L, 5000L }); + var rows = await PgWaitStatsCollector.Instance.ReadAsync(reader, MakeContext(deltas), CancellationToken.None); + + Assert.Equal(0, Assert.Single(rows).SampleIntervalSeconds); + + /* And with both groups agreeing, the measured value itself. */ + var steady = new RecordingCollectorDeltaCalculator { ReportedInterval = 137 }; + var steadyRows = await PgWaitStatsCollector.Instance.ReadAsync( + new FakeCollectorDataReader(new object[] { 10, 167772160L, "IO", "DataFileRead", 100L, 5000L }), + MakeContext(steady), CancellationToken.None); + Assert.Equal(137, Assert.Single(steadyRows).SampleIntervalSeconds); } [Fact] diff --git a/Lite.Tests/PgWaitStatsDeltaSkipTests.cs b/Lite.Tests/PgWaitStatsDeltaSkipTests.cs index 18004ed0f..ffb88b154 100644 --- a/Lite.Tests/PgWaitStatsDeltaSkipTests.cs +++ b/Lite.Tests/PgWaitStatsDeltaSkipTests.cs @@ -67,7 +67,9 @@ public async Task FirstSighting_ShipsEvenWithZeroWaits() var rows = await ReadAsync(Row(eventId: 1, waits: 0, waitTimeMicroseconds: 0), T0, deltas); - Assert.Single(rows); + /* #3540 (V128): and it ships as the (0, 0) marker — interval 0 is what tells a reader this row's + zero deltas are unknowable rather than a confirmed idle interval. */ + Assert.Equal(0, Assert.Single(rows).SampleIntervalSeconds); } [Fact] @@ -92,6 +94,8 @@ public async Task RepeatWithNewWaits_ShipsWithTheMeasuredDeltas() var row = Assert.Single(repeat); Assert.Equal(50, row.DeltaWaits); Assert.Equal(3000, row.DeltaWaitTime); + /* #3540 (V128): the measured span the two deltas accrued over, stored beside them. */ + Assert.Equal(60, row.SampleIntervalSeconds); } /// @@ -108,7 +112,9 @@ public async Task CounterReset_Ships() var afterReset = await ReadAsync(Row(eventId: 4, waits: 40, waitTimeMicroseconds: 1200), T0.AddSeconds(60), deltas); - Assert.Single(afterReset); + /* #3540 (V128): a reset row ships with interval 0, not the 60 s that elapsed — the elapsed span is + real, but no delta is knowable over it, and a stored 60 beside a 0 delta would read as idle. */ + Assert.Equal(0, Assert.Single(afterReset).SampleIntervalSeconds); } /// diff --git a/Lite.Tests/ProcedureStatsCollectorDefinitionTests.cs b/Lite.Tests/ProcedureStatsCollectorDefinitionTests.cs index ff9dc61cf..dc72a5b42 100644 --- a/Lite.Tests/ProcedureStatsCollectorDefinitionTests.cs +++ b/Lite.Tests/ProcedureStatsCollectorDefinitionTests.cs @@ -22,7 +22,7 @@ namespace Lite.Tests; /// Pins the parity contract of the extracted procedure_stats definition: the dynamic-SQL /// standard variant with double-escaped literal exclusions, the Azure single-database variant /// (token intentionally left unreplaced, as the original did), the plan_handle delta key with -/// its db.schema.object fallback, the 35-column payload, and the #1262 gated whole-module plan +/// its db.schema.object fallback, the 37-column payload (the #3540 trailing interval last), and the #1262 gated whole-module plan /// capture (off = byte-identical to the no-plan form; on = the text DMV keyed on 0, -1). /// public sealed class ProcedureStatsCollectorDefinitionTests @@ -115,10 +115,10 @@ UNION branch in the standard body and the single-proc Azure variant. */ } [Fact] - public void PayloadColumns_MatchSchema_36Columns() + public void PayloadColumns_MatchSchema_37Columns() { var names = ProcedureStatsCollector.Instance.PayloadColumns.Select(c => c.Name).ToArray(); - Assert.Equal(36, names.Length); + Assert.Equal(37, names.Length); Assert.Equal("database_name", names[0]); Assert.Equal("plan_handle", names[26]); Assert.Equal("delta_spills", names[33]); @@ -127,6 +127,15 @@ public void PayloadColumns_MatchSchema_36Columns() nvarchar(max) expression returns bigint, and a module-grain plan is measured in megabytes. */ Assert.Equal("query_plan_xml_bytes", names[35]); Assert.Equal(CollectorColumnType.BigInt, ProcedureStatsCollector.Instance.PayloadColumns[35].Type); + /* #3540 (Darling V128 / Lite v61): the interval is the TRAILING column, in the same INTEGER type + perfmon_stats and query_stats have always used, so the two stores' positional writers land it + after every pre-existing column and one NULLIF idiom reads every family. */ + var interval = ProcedureStatsCollector.Instance.PayloadColumns[^1]; + Assert.Equal("sample_interval_seconds", interval.Name); + Assert.Equal(CollectorColumnType.Integer, interval.Type); + Assert.Equal( + PerfmonStatsCollector.Instance.PayloadColumns.Single(c => c.Name == "sample_interval_seconds").Type, + interval.Type); } [Fact] @@ -211,9 +220,10 @@ query_plan_xml_bytes at 28. */ var writer = new RecordingCollectorRowWriter(); ProcedureStatsCollector.Instance.WritePayload(Assert.Single(rows), writer, context); - Assert.Equal(36, writer.Values.Count); + Assert.Equal(37, writer.Values.Count); Assert.Equal("proc", writer.Values[34]); /* query_plan_xml payload slot */ Assert.Equal(9_000_000L, writer.Values[35]); /* #3392: query_plan_xml_bytes */ + Assert.Equal(0, writer.Values[36]); /* #3540: interval — the fake reports 0, the unknowable marker */ /* #3392: 9 MB is over the cap, so this row IS a backlog candidate — and the offsets it hands back are the module-grain literals the plan apply passes, never per-statement values this DMV family @@ -242,9 +252,10 @@ public async Task WritePayload_UsesPlanHandleDeltaKey_WithFallback() var writer = new RecordingCollectorRowWriter(); ProcedureStatsCollector.Instance.WritePayload(rows[0], writer, context); - Assert.Equal(36, writer.Values.Count); + Assert.Equal(37, writer.Values.Count); Assert.Null(writer.Values[34]); /* query_plan_xml null when the flag is off */ Assert.Null(writer.Values[35]); /* #3392: and no measurement either */ + Assert.Equal(0, writer.Values[36]); /* #3540: the interval, trailing */ /* No measurement is not an oversized plan: null is "nobody measured", which is what a plan-capture-off host and an aged-out handle both produce. */ @@ -259,6 +270,49 @@ public async Task WritePayload_UsesPlanHandleDeltaKey_WithFallback() Assert.All(deltas.Calls, c => Assert.Equal("SO.dbo.usp_GetUser", c.Key)); } + /// + /// #3540 (V128 / v61): the interval reaches the payload MEASURED, not as a constant. A distinctive value + /// (neither 0 nor a plausible cadence) so this can only pass if what the calculator reported is what was + /// written. + /// + [Fact] + public async Task WritePayload_WritesTheMeasuredInterval() + { + var deltas = new RecordingCollectorDeltaCalculator { ReportedInterval = 137 }; + var context = CollectorTestContext.Make(deltas); + + using var reader = new FakeCollectorDataReader(MakeSqlRow(planHandle: "0x0600")); + var rows = await ProcedureStatsCollector.Instance.ReadAsync(reader, context, CancellationToken.None); + + var writer = new RecordingCollectorRowWriter(); + ProcedureStatsCollector.Instance.WritePayload(Assert.Single(rows), writer, context); + + Assert.Equal(137, writer.Values[^1]); + Assert.Equal(7, deltas.Calls.Count); + } + + /// + /// #3540: one interval per ROW, the MINIMUM over the row's seven delta groups. If any one group's delta + /// is unknowable (interval 0) the row is stored as (…, 0), so no reader divides a reset counter's 0 by a + /// sibling's real interval and reads it as a procedure that ran zero times. + /// + [Fact] + public async Task WritePayload_StoresTheMinimumIntervalAcrossTheRowsDeltaGroups() + { + var deltas = new RecordingCollectorDeltaCalculator { ReportedInterval = 300 }; + deltas.IntervalByGroup["proc_stats_spills"] = 0; + var context = CollectorTestContext.Make(deltas); + + using var reader = new FakeCollectorDataReader(MakeSqlRow(planHandle: "0x0600")); + var rows = await ProcedureStatsCollector.Instance.ReadAsync(reader, context, CancellationToken.None); + + var writer = new RecordingCollectorRowWriter(); + ProcedureStatsCollector.Instance.WritePayload(Assert.Single(rows), writer, context); + + Assert.Equal(0, writer.Values[^1]); + Assert.Equal(7, deltas.Calls.Count); + } + private static object[] MakeSqlRow(string? planHandle) => new object[] { "SO", "dbo", "usp_GetUser", "PROCEDURE", diff --git a/Lite.Tests/QueryStatsCollectorDefinitionTests.cs b/Lite.Tests/QueryStatsCollectorDefinitionTests.cs index f9363ed7e..dda0e2166 100644 --- a/Lite.Tests/QueryStatsCollectorDefinitionTests.cs +++ b/Lite.Tests/QueryStatsCollectorDefinitionTests.cs @@ -21,8 +21,9 @@ namespace Lite.Tests; /// Pins the parity contract of the extracted query_stats definition: the full row-identity delta /// key (sql_handle:start:end:plan_handle — the multi-statement cross-contamination fix), the /// interval-captured worker delta feeding sample_interval_seconds, the two query variants, and -/// the 52-column payload with the query_plan_xml placeholder, the trailing host_object_name -/// (#2012 stage 2) and query_plan_xml_bytes (#3392). +/// the 54-column payload with the query_plan_xml placeholder, the trailing host_object_name +/// (#2012 stage 2), query_plan_xml_bytes (#3392) and the two statement offsets (#3540, stored raw so +/// the restart seed can rebuild the key). /// public sealed class QueryStatsCollectorDefinitionTests { @@ -213,17 +214,19 @@ compile_age_seconds at 43 (#2235, inside SelectColumnsText so it is present in B var writer = new RecordingCollectorRowWriter(); QueryStatsCollector.Instance.WritePayload(Assert.Single(rows), writer, context); - Assert.Equal(52, writer.Values.Count); + Assert.Equal(54, writer.Values.Count); Assert.Equal("captured", writer.Values[37]); /* query_plan_xml payload slot */ Assert.Equal("dbo.HostProc", writer.Values[50]); /* host_object_name payload slot */ Assert.Equal(37L, writer.Values[51]); /* #3392: query_plan_xml_bytes */ + Assert.Equal(66, writer.Values[52]); /* #3540: statement_start_offset, raw */ + Assert.Equal(512, writer.Values[53]); /* #3540: statement_end_offset, raw */ /* #3392: a measurement at or under the cap is not a backlog candidate, and a captured plan of 37 bytes is comfortably under it. Strictly-greater is what the SQL CASE does, so this side of the boundary must agree. */ Assert.Null(QueryStatsCollector.Instance.DescribeOversizedPlan(rows[0])); - /* #2235: the compile age reaches the delta calculator and is NOT stored — 52 payload values, as + /* #2235: the compile age reaches the delta calculator and is NOT stored — 54 payload values, as pinned above, and one age per delta'd counter. Nine, because crediting only some of them would make one row's metrics disagree about how much work it did. */ Assert.Equal(8, deltas.SeriesAges.Count); @@ -270,10 +273,10 @@ public void BuildQuery_HandlesConvertToVarchar130_HashesStayVarchar64() } [Fact] - public void PayloadColumns_MatchSchemaOrder_52Columns() + public void PayloadColumns_MatchSchemaOrder_54Columns() { var names = QueryStatsCollector.Instance.PayloadColumns.Select(c => c.Name).ToArray(); - Assert.Equal(52, names.Length); + Assert.Equal(54, names.Length); Assert.Equal("database_name", names[0]); Assert.Equal("query_plan_xml", names[37]); Assert.Equal("sample_interval_seconds", names[49]); @@ -284,6 +287,14 @@ public void PayloadColumns_MatchSchemaOrder_52Columns() megabyte-scale plans this column exists to describe. */ Assert.Equal("query_plan_xml_bytes", names[51]); Assert.Equal(CollectorColumnType.BigInt, QueryStatsCollector.Instance.PayloadColumns[51].Type); + /* #3540 (Darling V128 / Lite v61): the delta key's two statement offsets, appended after it for the + same reason — start then end, the DMV's order — as Integer, the DMV's own type. Byte offsets into + the batch's nvarchar text; -1 as the end offset is "to the end of the batch"; stored raw so the + restart seed spells the collector's key. */ + Assert.Equal("statement_start_offset", names[52]); + Assert.Equal("statement_end_offset", names[53]); + Assert.Equal(CollectorColumnType.Integer, QueryStatsCollector.Instance.PayloadColumns[52].Type); + Assert.Equal(CollectorColumnType.Integer, QueryStatsCollector.Instance.PayloadColumns[53].Type); } [Fact] @@ -313,11 +324,15 @@ check that the #3392 read short-circuits on the flag rather than indexing past t var writer = new RecordingCollectorRowWriter(); QueryStatsCollector.Instance.WritePayload(Assert.Single(rows), writer, context); - Assert.Equal(52, writer.Values.Count); + Assert.Equal(54, writer.Values.Count); Assert.Null(writer.Values[37]); /* query_plan_xml placeholder */ Assert.Equal(0, writer.Values[49]); /* interval from recording fake */ Assert.Equal("dbo.HostProc", writer.Values[50]); /* host_object_name appended last */ Assert.Null(writer.Values[51]); /* #3392: no measurement with plans off */ + /* #3540: the offsets are stored exactly as read — the same 66 and 512 the delta key below is built + from — so the seed's rebuilt key and the collector's key are the same string. */ + Assert.Equal(66, writer.Values[52]); + Assert.Equal(512, writer.Values[53]); /* A row with no measurement is not a backlog candidate: null is "nobody measured", which is what a plan-capture-off host and an aged-out handle both produce, and neither is an oversized plan. */ diff --git a/Lite/Database/DuckDbInitializer.cs b/Lite/Database/DuckDbInitializer.cs index 8b4d6c55f..df7f42ea6 100644 --- a/Lite/Database/DuckDbInitializer.cs +++ b/Lite/Database/DuckDbInitializer.cs @@ -271,7 +271,7 @@ public void Dispose() /// /// Current schema version. Increment this when schema changes require table rebuilds. /// - internal const int CurrentSchemaVersion = 60; + internal const int CurrentSchemaVersion = 61; private readonly string _archivePath; @@ -1649,6 +1649,68 @@ await ExecuteNonQueryAsync(connection, } } } + + if (fromVersion < 61) + { + /* v61 (#3540): the completion of v60, twinning Darling's V128. procedure_stats and + memory_grant_stats gain sample_interval_seconds — the measured seconds each row's deltas + accrued over — so EVERY delta family Lite stores now carries the interval, and the shared + calculator's (delta 0, interval 0) "no delta knowable" marker survives every family's write. + These two took the calculator's bare long and discarded the interval, so a restart's + fabricated zero read as a measured idle one: the procedure duration trend LAG-divided it into + a confident 0.00 ms/sec, and a memory-grant row's 0 timeouts over a restart read as a quiet + semaphore. Darling's V128 also dresses pg_wait_stats and pg_statement_stats; Lite stores no + pg_* tables (DuckDbSchemaGenerator.StoredCollectors), so this side has only the two. + + query_stats gains statement_start_offset and statement_end_offset — the two halves of its + delta key (sql_handle:start:end:plan_handle) the store never persisted, so the restart seed + could restore this family's pass window but not one baseline (#3614 named it as the residual). + Their semantics, stated here because this is where the next reader will look: they are + sys.dm_exec_query_stats' own columns, the statement's position inside its batch text in BYTES + of the nvarchar text, not characters — a character slice divides by two, which is the + collector's SUBSTRING(st.text, (statement_start_offset / 2) + 1, ...) — and + statement_end_offset = -1 means "to the end of the batch" ((0, -1) is the whole batch). They + are stored VERBATIM as the DMV reports them, -1 included and never normalized to a length, + because the collector's key is built over the raw ints and the seed has to spell the same + string byte for byte (DeltaCalculator.QueryStatsSeedSql). + + All three appended at the end of their PayloadColumns lists, so the positional appender and + old parquet are unaffected. Nothing to backfill and nothing that COULD be: a row collected + before the upgrade never recorded its interval or its offsets, so NULL is the honest value — + the readers treat a NULL interval as "pre-v61, derive from the previous collection_time" + (exactly what they always did) and 0 as "unknowable, render nothing"; the seed treats NULL + offsets as "no key can be rebuilt from this row" and still takes its collection time for the + pass window. A backfilled 0 interval would stamp all of history unknowable and blank every + procedure rate chart for 30 days; a backfilled 0/-1 offset pair would seed baselines under a + key nothing will ever present. + + REQUIRED on this side for the v60 reason: the appender writes one value per declared payload + column, so a database without these columns fails EndRow() on the first batch of either + collector — the whole batch, not the column. Fresh installs get them from + DuckDbSchemaGenerator; these ALTERs are for an existing database and are idempotent. The v_ + passthrough views need no work here: Lite rebuilds every v_ view on start + (CreateArchiveViewsAsync, called after this). Non-fatal per statement, matching v59/v60. */ + _logger?.LogInformation("Running migration to v61: every delta family stores its interval, and query_stats stores the statement offsets its delta key is made of"); + + foreach (var (table, column) in new[] + { + ("procedure_stats", "sample_interval_seconds"), + ("memory_grant_stats", "sample_interval_seconds"), + ("query_stats", "statement_start_offset"), + ("query_stats", "statement_end_offset"), + }) + { + try + { + await ExecuteNonQueryAsync(connection, + $"ALTER TABLE {table} ADD COLUMN IF NOT EXISTS {column} INTEGER"); + } + catch (Exception ex) + { + _logger?.LogWarning("Migration to v61 on {Table}.{Column} encountered an error (non-fatal): {Error}", table, column, ex.Message); + } + } + } } /// diff --git a/Lite/Mcp/McpMemoryTools.cs b/Lite/Mcp/McpMemoryTools.cs index af766f3ad..561298853 100644 --- a/Lite/Mcp/McpMemoryTools.cs +++ b/Lite/Mcp/McpMemoryTools.cs @@ -278,7 +278,7 @@ public static async Task GetMemoryPressureEvents( } } - [McpServerTool(Name = "get_resource_semaphore"), Description("Gets resource semaphore statistics from the latest snapshot: granted vs available workspace memory against the target/max-target ceiling, per resource semaphore, with waiter counts and cumulative + per-interval timeout/forced-grant pressure indicators. High waiter counts or rising timeout/forced deltas indicate memory grant pressure affecting query performance.")] + [McpServerTool(Name = "get_resource_semaphore"), Description("Gets resource semaphore statistics from the latest snapshot: granted vs available workspace memory against the target/max-target ceiling, per resource semaphore, with waiter counts and cumulative + per-interval timeout/forced-grant pressure indicators. High waiter counts or rising timeout/forced deltas indicate memory grant pressure affecting query performance. sample_interval_seconds is the measured seconds the two deltas accrued over; it is null with interval_known false when the row is a restart marker (no delta was knowable, so the zero deltas beside it are not 'no timeouts') or predates the column.")] public static async Task GetResourceSemaphore( LocalDataService dataService, ServerManager serverManager, @@ -317,7 +317,14 @@ public static async Task GetResourceSemaphore( timeout_error_count = r.TimeoutErrorCount, forced_grant_count = r.ForcedGrantCount, timeout_error_count_delta = r.TimeoutErrorCountDelta, - forced_grant_count_delta = r.ForcedGrantCountDelta + forced_grant_count_delta = r.ForcedGrantCountDelta, + /* #3540 (v61): the interval the deltas accrued over, the way the perfmon and file-I/O tools + hand it over. A stored 0 is the calculator's no-delta-knowable marker (a restart, not a + quiet semaphore) and is reported as null rather than 0 — 0 seconds is not a measurement; + a pre-v61 row that never recorded one is null too. interval_known states the one thing + both nulls have in common: the two *_delta zeros beside them are not "none this interval". */ + sample_interval_seconds = r.SampleIntervalSeconds is > 0 ? r.SampleIntervalSeconds : null, + interval_known = r.SampleIntervalSeconds is > 0 }); return JsonSerializer.Serialize(new diff --git a/Lite/Services/DeltaCalculator.cs b/Lite/Services/DeltaCalculator.cs index dd8bcb2b4..d36b3deb6 100644 --- a/Lite/Services/DeltaCalculator.cs +++ b/Lite/Services/DeltaCalculator.cs @@ -42,11 +42,12 @@ either one left unbounded reads the whole table. #1772 was the Postgres half of - Latest collection per server (the original four, latch_stats, spinlock_stats): these collectors write every key they read on every pass, so the newest collection holds every key's current counter and the row-value probe is the cheapest exact read. - - Latest row per key (procedure_stats here; the PostgreSQL pair on Darling): these collectors - do NOT write every key every pass — procedure_stats is a TOP (150) that churns, and the - PostgreSQL collectors skip idle rows at the write — so the newest collection is missing keys - whose counters are nevertheless unchanged, and a key seeded from nothing takes the - first-sighting path. DISTINCT ON (server_id, key) ... ORDER BY collection_time DESC over the + - Latest row per key (procedure_stats and query_stats here; the PostgreSQL pair on Darling): + these collectors do NOT write every key every pass — procedure_stats and query_stats are + TOP (n) reads that churn, and the PostgreSQL collectors skip idle rows at the write — so the + newest collection is missing keys whose counters are nevertheless unchanged, and a key seeded + from nothing takes the first-sighting path. DISTINCT ON (server_id, key) ... ORDER BY + collection_time DESC over the cutoff window returns each key's newest row instead. Its bound is the single collection_time >= $1 on its only table read: there is no inner aggregate to bind a second time, and the window's rows are the whole working set (one sort over fifteen minutes of one @@ -55,13 +56,18 @@ cutoff window returns each key's newest row instead. Its bound is the single value is exact (the counter was idle in between, which is why no newer row exists) and the gap policy still bounds the span. - query_stats has NO key seed on either host, and the reason is stated rather than left as a gap: - its delta key is sql_handle:statement_start_offset:statement_end_offset:plan_handle and the - store persists neither offset, so no row in query_stats can reproduce the key the collector - will present, and a seed under any other key seeds nothing. Its PASS WINDOW is seeded below - (QueryStatsPassSeedSql), which is what the #2235 series-age rescue reads — so on the first - post-restart pass a plan that compiled since the last pre-restart pass is credited in full even - though older plans baseline. Persisting the offsets is a rung, tracked on #3540. */ + query_stats joined the per-key shape with Lite v61 / Darling V128 (#3540). Its delta key is + sql_handle:statement_start_offset:statement_end_offset:plan_handle, and until v61 the store + persisted neither offset, so no row could reproduce the key the collector presents and only the + family's PASS WINDOW could be seeded (the #2235 series-age rescue's input). The offsets are + stored now, raw (-1 = "to the end of the batch", byte offsets into the nvarchar batch text) and + the seed rebuilds the key from them with the collector's own interpolation. Two rules, both in + the seeder rather than the SQL: a row whose offsets are NULL — every row written before v61 — + seeds NO key, because a key built from a fabricated 0/-1 would be one nothing ever presents and + the baseline under it would sit unread until it aged out; and EVERY row, NULL offsets or not, + still feeds the pass window, so the first restart after the upgrade (when the whole window is + pre-v61 rows) keeps the rescue armed exactly as #3614 left it. One read serves both, which is + why the offset filter is not in the WHERE. */ public const string WaitStatsSeedSql = @" SELECT server_id, wait_type, waiting_tasks_count, wait_time_ms, signal_wait_time_ms, collection_time @@ -146,14 +152,20 @@ FROM procedure_stats ) AS recent ORDER BY server_id, delta_key, collection_time DESC"; - /* The pass window only (see the header): every distinct collection time per server inside the - cutoff, which is a handful of rows per server from an index on (server_id, collection_time). - GROUP BY rather than DISTINCT so the shape reads as the aggregate it is. */ - public const string QueryStatsPassSeedSql = @" -SELECT server_id, collection_time + /* QueryStatsCollector keys on the dm_exec_query_stats row identity — sql_handle, both statement + offsets, plan_handle — and its TOP (n) churns like procedure_stats', so: latest row per that + identity. DISTINCT ON treats NULL offsets as one group, which is harmless: those are pre-v61 rows + the seeder reads for the pass window only (see the header). The eight counters are the ones the + collector's eight CalculateDeltaWithSeriesAge calls difference. */ + public const string QueryStatsSeedSql = @" +SELECT DISTINCT ON (server_id, sql_handle, statement_start_offset, statement_end_offset, plan_handle) + server_id, sql_handle, statement_start_offset, statement_end_offset, plan_handle, + execution_count, total_worker_time, total_elapsed_time, + total_logical_reads, total_logical_writes, total_physical_reads, total_rows, total_spills, + collection_time FROM query_stats WHERE collection_time >= $1 -GROUP BY server_id, collection_time"; +ORDER BY server_id, sql_handle, statement_start_offset, statement_end_offset, plan_handle, collection_time DESC"; /* The delta GROUP names each family's collector passes — the pass window is keyed by these, not by the collector name, so a seeder that seeded "query_stats" would arm nothing. Spelled once per @@ -214,7 +226,7 @@ public async Task SeedFromDatabaseAsync(DuckDbInitializer duckDb) await SeedFamilyAsync("latch_stats", () => SeedLatchStatsAsync(connection, cutoff)); await SeedFamilyAsync("spinlock_stats", () => SeedSpinlockStatsAsync(connection, cutoff)); await SeedFamilyAsync("procedure_stats", () => SeedProcedureStatsAsync(connection, cutoff)); - await SeedFamilyAsync("query_stats", () => SeedQueryStatsPassesAsync(connection, cutoff)); + await SeedFamilyAsync("query_stats", () => SeedQueryStatsAsync(connection, cutoff)); _logger?.LogInformation( "Delta calculator seeded from database (baselines from the last {Minutes} minutes)", @@ -414,19 +426,50 @@ private async Task SeedProcedureStatsAsync(DuckDBConnection connection, DateTime if (count > 0) _logger?.LogDebug("Seeded {Count} procedure_stats baseline rows", count); } - /* Pass window only — see the header for why query_stats has no key seed. */ - private async Task SeedQueryStatsPassesAsync(DuckDBConnection connection, DateTime cutoff) + private async Task SeedQueryStatsAsync(DuckDBConnection connection, DateTime cutoff) { using var cmd = connection.CreateCommand(); - cmd.CommandText = QueryStatsPassSeedSql; + cmd.CommandText = QueryStatsSeedSql; cmd.Parameters.Add(new DuckDBParameter { Value = cutoff }); using var reader = await cmd.ExecuteReaderAsync(); + var count = 0; + var preV61 = 0; var passes = new SeedPassTracker(); while (await reader.ReadAsync()) { - passes.Observe(reader.GetInt32(0), reader.IsDBNull(1) ? (DateTime?)null : reader.GetDateTime(1)); + var serverId = reader.GetInt32(0); + var ts = reader.IsDBNull(13) ? (DateTime?)null : reader.GetDateTime(13); + + /* The pass window takes EVERY row, offsets or not — see the header. */ + passes.Observe(serverId, ts); + + /* A pre-v61 row never recorded its offsets. Its key cannot be rebuilt, and a key built from a + guessed pair would be a baseline nothing ever reads — so it seeds nothing. */ + if (reader.IsDBNull(2) || reader.IsDBNull(3)) + { + preV61++; + continue; + } + + /* The key exactly as QueryStatsCollector.WritePayload spells it: the same interpolation over + the same raw parts, so a null handle formats as empty and the offsets — -1 included — are + spelled by the same int formatting on both sides. Never normalized. */ + var sqlHandle = reader.IsDBNull(1) ? null : reader.GetString(1); + var planHandle = reader.IsDBNull(4) ? null : reader.GetString(4); + var deltaKey = $"{sqlHandle}:{reader.GetInt32(2)}:{reader.GetInt32(3)}:{planHandle}"; + + Seed(serverId, "query_stats_exec", deltaKey, reader.IsDBNull(5) ? 0 : reader.GetInt64(5), ts); + Seed(serverId, "query_stats_worker", deltaKey, reader.IsDBNull(6) ? 0 : reader.GetInt64(6), ts); + Seed(serverId, "query_stats_elapsed", deltaKey, reader.IsDBNull(7) ? 0 : reader.GetInt64(7), ts); + Seed(serverId, "query_stats_reads", deltaKey, reader.IsDBNull(8) ? 0 : reader.GetInt64(8), ts); + Seed(serverId, "query_stats_writes", deltaKey, reader.IsDBNull(9) ? 0 : reader.GetInt64(9), ts); + Seed(serverId, "query_stats_phys_reads", deltaKey, reader.IsDBNull(10) ? 0 : reader.GetInt64(10), ts); + Seed(serverId, "query_stats_rows", deltaKey, reader.IsDBNull(11) ? 0 : reader.GetInt64(11), ts); + Seed(serverId, "query_stats_spills", deltaKey, reader.IsDBNull(12) ? 0 : reader.GetInt64(12), ts); + count++; } SeedPasses(passes, QueryStatsGroups); - if (passes.Count > 0) _logger?.LogDebug("Seeded the query_stats pass window for {Count} servers", passes.Count); + if (count > 0) _logger?.LogDebug("Seeded {Count} query_stats baseline rows", count); + if (preV61 > 0) _logger?.LogDebug("Skipped {Count} query_stats rows with no stored statement offsets (pre-v61); their collection times still seeded the pass window", preV61); } } diff --git a/Lite/Services/LocalDataService.MemoryGrants.cs b/Lite/Services/LocalDataService.MemoryGrants.cs index 9c2543b65..9de994e15 100644 --- a/Lite/Services/LocalDataService.MemoryGrants.cs +++ b/Lite/Services/LocalDataService.MemoryGrants.cs @@ -148,7 +148,8 @@ FROM v_memory_grant_stats timeout_error_count, forced_grant_count, timeout_error_count_delta, - forced_grant_count_delta + forced_grant_count_delta, + sample_interval_seconds FROM v_memory_grant_stats WHERE server_id = $1 AND collection_time = (SELECT mx FROM latest) @@ -178,7 +179,10 @@ FROM v_memory_grant_stats TimeoutErrorCount = reader.IsDBNull(11) ? 0 : ToInt64(reader.GetValue(11)), ForcedGrantCount = reader.IsDBNull(12) ? 0 : ToInt64(reader.GetValue(12)), TimeoutErrorCountDelta = reader.IsDBNull(13) ? 0 : ToInt64(reader.GetValue(13)), - ForcedGrantCountDelta = reader.IsDBNull(14) ? 0 : ToInt64(reader.GetValue(14)) + ForcedGrantCountDelta = reader.IsDBNull(14) ? 0 : ToInt64(reader.GetValue(14)), + /* NULL stays NULL: a pre-v61 row never recorded its interval, and that is a different + statement from the 0 the calculator writes when no delta was knowable. */ + SampleIntervalSeconds = reader.IsDBNull(15) ? null : (int)ToInt64(reader.GetValue(15)) }); } return items; @@ -201,7 +205,9 @@ public class MemoryGrantChartPoint /// One resource-semaphore latest-snapshot row (the get_resource_semaphore MCP lens): one /// (pool_id, resource_semaphore_id) semaphore's full ceiling metrics at the most recent collection in the /// window — target / max-target / total workspace memory, granted vs available/used, grantee/waiter counts, -/// and the cumulative + per-interval-delta timeout/forced-grant pressure counters. +/// the cumulative + per-interval-delta timeout/forced-grant pressure counters, and (since v61, #3540) the +/// measured seconds those deltas accrued over: 0 is the calculator's "no delta knowable" marker (a +/// restart, not a quiet semaphore), null a pre-v61 row that never recorded one. public class ResourceSemaphoreRow { public DateTime CollectionTime { get; set; } @@ -219,4 +225,9 @@ public class ResourceSemaphoreRow public long ForcedGrantCount { get; set; } public long TimeoutErrorCountDelta { get; set; } public long ForcedGrantCountDelta { get; set; } + public int? SampleIntervalSeconds { get; set; } + + /// True when the row's deltas are the calculator's (0, 0) marker: no delta was knowable, so + /// the two *Delta zeros beside it are not "no timeouts this interval". + public bool IsUnknowable => SampleIntervalSeconds == 0; } diff --git a/Lite/Services/LocalDataService.QueryStats.cs b/Lite/Services/LocalDataService.QueryStats.cs index 5cf380f52..b9fc19889 100644 --- a/Lite/Services/LocalDataService.QueryStats.cs +++ b/Lite/Services/LocalDataService.QueryStats.cs @@ -572,7 +572,10 @@ public async Task> GetProcedureStatsHistoryAsync( total_physical_reads, total_logical_writes, delta_spills, - CAST(extract(epoch FROM (date_trunc('second', collection_time) - date_trunc('second', LAG(collection_time) OVER (ORDER BY collection_time)))) AS BIGINT) AS sample_interval_seconds + /* #3540 (v61): the STORED interval where the row has one — including 0, which the grid shows as the + Interval (sec) 0 the query_stats history has always shown for an unknowable row — and the LAG over + collection_time this read always derived for a pre-v61 row (NULL) that never recorded one. */ + COALESCE(sample_interval_seconds, CAST(extract(epoch FROM (date_trunc('second', collection_time) - date_trunc('second', LAG(collection_time) OVER (ORDER BY collection_time)))) AS BIGINT)) AS sample_interval_seconds FROM v_procedure_stats WHERE server_id = $1 AND database_name = $2 @@ -1189,6 +1192,15 @@ FROM v_query_stats /// /// Gets procedure duration trend — elapsed time per second per collection snapshot. + /// + /// #3540 (v61): the interval is the collection's STORED one where the rows have it — MAX + /// over the collection's rows, because a plan first seen in an otherwise steady pass (a TOP (150) + /// readmission) carries 0 beside its siblings' real interval and contributes 0 to the sums; MAX is 0 + /// only when EVERY row was unknowable (a restart), and that 0 becomes NULL through NULLIF so the + /// rates are NULL and the point is dropped rather than rendered as 0.00 ms/sec. NULL (a pre-v61 + /// collection that never recorded one) falls back to the LAG over collection_time this read always + /// used, so history renders exactly as it did. No ELSE 0: the first row of a pre-v61 series is + /// absent rather than a fabricated 0.0, the same correction v60 made for the wait trends. /// public async Task> GetProcedureDurationTrendAsync(int serverId, int hoursBack = 24, DateTime? fromDate = null, DateTime? toDate = null, IReadOnlyList? databaseNames = null, DateTime? asOfUtc = null) { @@ -1205,7 +1217,10 @@ WITH raw AS collection_time, SUM(delta_elapsed_time) / 1000.0 AS total_elapsed_ms, SUM(delta_execution_count) AS total_executions, - extract(epoch FROM (date_trunc('second', collection_time) - date_trunc('second', LAG(collection_time) OVER (ORDER BY collection_time)))) AS interval_seconds + CASE WHEN MAX(sample_interval_seconds) IS NULL + THEN extract(epoch FROM (date_trunc('second', collection_time) - date_trunc('second', LAG(collection_time) OVER (ORDER BY collection_time)))) + ELSE NULLIF(MAX(sample_interval_seconds), 0) + END AS interval_seconds FROM v_procedure_stats WHERE server_id = $1 AND collection_time >= $2 @@ -1214,8 +1229,8 @@ GROUP BY collection_time ) SELECT collection_time, - CASE WHEN interval_seconds > 0 THEN total_elapsed_ms / interval_seconds ELSE 0 END AS elapsed_ms_per_second, - CASE WHEN interval_seconds > 0 THEN CAST(total_executions AS DOUBLE PRECISION) / interval_seconds ELSE 0 END AS executions_per_second + CASE WHEN interval_seconds > 0 THEN total_elapsed_ms / interval_seconds END AS elapsed_ms_per_second, + CASE WHEN interval_seconds > 0 THEN CAST(total_executions AS DOUBLE PRECISION) / interval_seconds END AS executions_per_second FROM raw ORDER BY collection_time"; @@ -1229,10 +1244,16 @@ FROM raw using var reader = await command.ExecuteReaderAsync(); while (await reader.ReadAsync()) { + /* A NULL rate is an unknowable interval (#3540, v61): the point is dropped, not read as 0. */ + if (reader.IsDBNull(1)) + { + continue; + } + items.Add(new QueryTrendPoint { CollectionTime = reader.GetDateTime(0), - Value = reader.IsDBNull(1) ? 0 : ToDouble(reader.GetValue(1)), + Value = ToDouble(reader.GetValue(1)), ExecutionCount = reader.IsDBNull(2) ? 0 : (long)ToDouble(reader.GetValue(2)), ExecutionsPerSecond = reader.IsDBNull(2) ? 0 : ToDouble(reader.GetValue(2)) }); diff --git a/PerformanceMonitor.Collectors/CollectorDeltaCalculator.cs b/PerformanceMonitor.Collectors/CollectorDeltaCalculator.cs index 7d65405d9..5319a20b4 100644 --- a/PerformanceMonitor.Collectors/CollectorDeltaCalculator.cs +++ b/PerformanceMonitor.Collectors/CollectorDeltaCalculator.cs @@ -27,12 +27,15 @@ namespace PerformanceMonitor.Collectors; /// and memory_grant_stats; latch_stats, spinlock_stats, query_stats, procedure_stats and the PostgreSQL /// pair took the first-sighting path after every restart or deploy, so each fabricated one full interval /// of quiet per restart — and the pass window was never seeded at all, which left the #2235 series-age -/// rescue inert on exactly the cycle it exists for. The one family a host cannot key-seed is named where -/// it is not seeded rather than left to be discovered: query_stats keys its deltas on -/// sql_handle:statement_start_offset:statement_end_offset:plan_handle and the store persists -/// neither offset, so no store row can reproduce the key; its PASS WINDOW is seeded (which is what the -/// series-age rescue reads), and the offsets are a rung. Lite.Tests' DeltaFamilySeedingCensusTests -/// enumerates the family against both hosts' seeders so an eleventh family cannot ship unseeded. +/// rescue inert on exactly the cycle it exists for. query_stats was the one family a host could not +/// key-seed until Darling V128 / Lite v61: it keys its deltas on +/// sql_handle:statement_start_offset:statement_end_offset:plan_handle and the store persisted +/// neither offset, so no store row could reproduce the key and only its PASS WINDOW was seeded. The +/// offsets are stored now and both hosts key-seed it from rows that carry them; a pre-V128 row (NULL +/// offsets) still contributes its collection time to the pass window and seeds no key, because a key +/// built from a fabricated offset is one nothing will ever present. Lite.Tests' +/// DeltaFamilySeedingCensusTests enumerates the family against both hosts' seeders so an eleventh +/// family cannot ship unseeded. /// public class CollectorDeltaCalculator : ICollectorDeltaCalculator { @@ -310,20 +313,23 @@ the wall-clock span this delta accrued over. */ a 0 delta over a REAL interval is a claim that nothing happened for that long, and this is the one case where that claim would be false. That invariant (interval 0 <=> no delta knowable) is what lets a reader tell a fabricated zero - from an idle one — but only where the interval REACHES the store. Every SQL Server - delta family that persists a sample_interval_seconds column beside its deltas - (perfmon_stats and query_stats from the start; wait_stats, file_io_stats, - latch_stats and spinlock_stats since Darling V127 / Lite v60, #3540) maps the 0 to - NULL at the read via NULLIF(sample_interval_seconds, 0), or filters it out of an - aggregate with sample_interval_seconds IS DISTINCT FROM 0. Before #3540 this comment - claimed "every consumer" while four of those six families discarded the interval at - the write, so the fabricated zero survived as a measured one and their readers - LAG-divided it into a confident 0.00 at exactly the moments it was unknowable. The - remaining delta families persist no interval: procedure_stats and memory_grant_stats - take CalculateDelta's bare long, and the PostgreSQL pair asks for the interval only - to skip idle rows at the write. A delta-family census in Lite.Tests - (DeltaFamilyIntervalColumnTests) names those four so the naked list shrinks - deliberately rather than growing by accident. */ + from an idle one — and since Darling V128 / Lite v61 the interval REACHES the store + for EVERY delta family: all ten members of DeltaFamilyCollectors persist a + sample_interval_seconds column beside their deltas (perfmon_stats and query_stats + from the start; wait_stats, file_io_stats, + latch_stats and spinlock_stats since Darling V127 / Lite v60, #3540; + procedure_stats, memory_grant_stats, pg_wait_stats and pg_statement_stats since + Darling V128 / Lite v61, #3540), and every per-second reader maps the 0 to NULL + via NULLIF(sample_interval_seconds, 0) or filters it out of an aggregate with + sample_interval_seconds IS DISTINCT FROM 0. Before #3540 this comment + claimed "every consumer" while four of six SQL Server families discarded the + interval at the write, so the fabricated zero survived as a measured one and their + readers LAG-divided it into a confident 0.00 at exactly the moments it was + unknowable; two more took the bare long and the PostgreSQL pair asked for the + interval only to skip idle rows. The claim is pinned rather than trusted: + Lite.Tests' DeltaFamilyIntervalColumnTests is the census that asserts every member + of DeltaFamilyCollectors carries the column, so an eleventh family cannot ship + naked and this sentence cannot silently go false again. */ delta = 0; interval = 0; } diff --git a/PerformanceMonitor.Collectors/MemoryGrantsCollector.cs b/PerformanceMonitor.Collectors/MemoryGrantsCollector.cs index c671b86ad..7517fff52 100644 --- a/PerformanceMonitor.Collectors/MemoryGrantsCollector.cs +++ b/PerformanceMonitor.Collectors/MemoryGrantsCollector.cs @@ -90,6 +90,13 @@ WHERE deqrs.max_target_memory_kb IS NOT NULL new CollectorColumn("forced_grant_count", CollectorColumnType.BigInt), new CollectorColumn("timeout_error_count_delta", CollectorColumnType.BigInt), new CollectorColumn("forced_grant_count_delta", CollectorColumnType.BigInt), + /* Appended (Darling V128 / Lite v61, #3540): the measured seconds the row's two deltas accrued + over, or 0 when no delta was knowable. Appended at the END because both stores' writers are + positional — the same rule GoldenCollectorSchema's header states for every column a numbered + migration adds by ALTER TABLE. This is the column the deprecated Dashboard's + get_resource_semaphore always served and Darling's twin dropped as "unstored"; both hosts' + tools emit it again now that it is. */ + new CollectorColumn("sample_interval_seconds", CollectorColumnType.Integer), }; public override async ValueTask> ReadAsync(DbDataReader reader, CollectorContext context, CancellationToken cancellationToken) @@ -118,10 +125,25 @@ public override async ValueTask> ReadAsync(DbDataReader reader, Collec public override void WritePayload(Row row, ICollectorRowWriter writer, CollectorContext context) { - /* Composite delta key and group names are the parity contract — do not change casually. */ + /* Composite delta key and group names are the parity contract — do not change casually. + + The interval is stored beside the deltas (#3540, Darling V128 / Lite v61). The calculator reports + (delta 0, interval 0) when no delta is knowable — first sighting, counter reset, a gap past the + policy — and (0, n) when the interval was genuinely idle; the interval is the ONLY thing that + tells those apart, and this collector took the bare long and discarded it. The two counters here + are exactly the ones where the distinction matters most: grant timeouts and forced grants are + rare, monotonic, and read as "none this interval" whenever the delta is 0 — a restart's + fabricated 0 was indistinguishable from a genuinely quiet semaphore. + + One interval per ROW, the minimum over the two groups (the V127 rule, WaitStatsCollector): the + groups share a key and a collection time, so the two intervals agree in every case but an + independent single-counter reset, which this DMV does not do — a semaphore's counters reset + together on instance restart. The minimum makes the stored pair mean "every delta in this row is + knowable". */ var deltaKey = $"{row.PoolId}_{row.ResourceSemaphoreId}"; - var deltaTimeouts = context.Deltas.CalculateDelta(context.ServerId, "memory_grants_timeouts", deltaKey, row.TimeoutErrorCount, collectionTime: context.CollectionTime, maxGapSeconds: CollectorDeltaCalculator.DefaultMaxGapSeconds); - var deltaForced = context.Deltas.CalculateDelta(context.ServerId, "memory_grants_forced", deltaKey, row.ForcedGrantCount, collectionTime: context.CollectionTime, maxGapSeconds: CollectorDeltaCalculator.DefaultMaxGapSeconds); + var deltaTimeouts = context.Deltas.CalculateDeltaWithInterval(context.ServerId, "memory_grants_timeouts", deltaKey, row.TimeoutErrorCount, out var timeoutsInterval, collectionTime: context.CollectionTime, maxGapSeconds: CollectorDeltaCalculator.DefaultMaxGapSeconds); + var deltaForced = context.Deltas.CalculateDeltaWithInterval(context.ServerId, "memory_grants_forced", deltaKey, row.ForcedGrantCount, out var forcedInterval, collectionTime: context.CollectionTime, maxGapSeconds: CollectorDeltaCalculator.DefaultMaxGapSeconds); + var sampleIntervalSeconds = Math.Min(timeoutsInterval, forcedInterval); writer .Value(row.ResourceSemaphoreId) /* resource_semaphore_id (appended as SHORT, matching the original) */ @@ -137,6 +159,7 @@ public override void WritePayload(Row row, ICollectorRowWriter writer, Collector .Value(row.TimeoutErrorCount) /* timeout_error_count BIGINT */ .Value(row.ForcedGrantCount) /* forced_grant_count BIGINT */ .Value(deltaTimeouts) /* timeout_error_count_delta BIGINT */ - .Value(deltaForced); /* forced_grant_count_delta BIGINT */ + .Value(deltaForced) /* forced_grant_count_delta BIGINT */ + .Value(sampleIntervalSeconds); /* sample_interval_seconds INTEGER — measured, 0 = unknowable */ } } diff --git a/PerformanceMonitor.Collectors/PgStatementStatsCollector.cs b/PerformanceMonitor.Collectors/PgStatementStatsCollector.cs index f15cbcdcb..c877dd5c8 100644 --- a/PerformanceMonitor.Collectors/PgStatementStatsCollector.cs +++ b/PerformanceMonitor.Collectors/PgStatementStatsCollector.cs @@ -6,6 +6,7 @@ * Licensed under the MIT License. See LICENSE file in the project root for full license information. */ +using System; using System.Collections.Generic; using System.Data.Common; using System.Globalization; @@ -120,7 +121,11 @@ public readonly record struct Row( added to the list WritePayload will later be called once per. */ long DeltaCalls, long DeltaTotalExecTimeMs, - long DeltaRows); + long DeltaRows, + /* #3540 (Darling V128): the measured seconds the three deltas accrued over — the minimum over the + row's groups, so 0 means "no delta in this row is knowable". Computed beside the deltas in + ReadAsync because that is where the calculator is called. */ + int SampleIntervalSeconds); /* Column names DIFFER between PostgreSQL 16 and 17 and our fleet spans both, so the query is built per major rather than SELECT *-ed. Verified against live 16.11 and 17.7: @@ -287,6 +292,14 @@ these are signals the SQL Server side cannot offer. */ new CollectorColumn("delta_calls", CollectorColumnType.BigInt), new CollectorColumn("delta_total_exec_time_ms", CollectorColumnType.BigInt), new CollectorColumn("delta_rows", CollectorColumnType.BigInt), + /* Appended (Darling V128, #3540): the measured seconds the row's three deltas accrued over, or 0 + when no delta was knowable. Appended at the END because the COPY writer is positional — the + same rule GoldenCollectorSchema's header states for every column a numbered migration adds by + ALTER TABLE. Until V128 this collector asked the calculator for the interval only to decide + the idle-row skip and stored nothing, so every row it DID ship with interval 0 (first sighting, + reset, gap re-baseline — the three cases the skip deliberately lets through) carried delta 0s + that the per-statement duration trend divided by a LAG-derived span into 0.00 calls/sec. */ + new CollectorColumn("sample_interval_seconds", CollectorColumnType.Integer), }; public override async ValueTask> ReadAsync(DbDataReader reader, CollectorContext context, CancellationToken cancellationToken) @@ -327,12 +340,18 @@ taken on whole milliseconds. Sub-millisecond drift per interval is immaterial ag THIS series stale while the calls series (called for every row, always) stayed fresh — so the next genuinely active cycle could see a gap here that calls never sees, and report a false counter-reset on a real accrual. See the class remarks. */ - var deltaTotalTime = context.Deltas.CalculateDelta( - context.ServerId, "pg_statement_stats_time", key, (long)totalExecTimeMs, + var deltaTotalTime = context.Deltas.CalculateDeltaWithInterval( + context.ServerId, "pg_statement_stats_time", key, (long)totalExecTimeMs, out var timeIntervalSeconds, collectionTime: context.CollectionTime, maxGapSeconds: CollectorDeltaCalculator.DefaultMaxGapSeconds); - var deltaRows = context.Deltas.CalculateDelta( - context.ServerId, "pg_statement_stats_rows", key, rowsReturned, + var deltaRows = context.Deltas.CalculateDeltaWithInterval( + context.ServerId, "pg_statement_stats_rows", key, rowsReturned, out var rowsIntervalSeconds, collectionTime: context.CollectionTime, maxGapSeconds: CollectorDeltaCalculator.DefaultMaxGapSeconds); + /* #3540 (V128): the stored interval is the MINIMUM over the row's three groups — the V127 rule + (WaitStatsCollector). The groups share a key and a collection time, so they agree in every + case but an independent single-counter reset, and pg_stat_statements resets an entry's + counters together; the minimum makes the stored pair mean "every delta in this row is + knowable", so a reader never divides one group's reset 0 by a sibling's real span. */ + var sampleIntervalSeconds = Math.Min(callsIntervalSeconds, Math.Min(timeIntervalSeconds, rowsIntervalSeconds)); /* The skip: a REAL interval (this is not a first sighting, a counter reset, or a gap this pass just re-baselined) with zero new calls means the statement demonstrably did not run, @@ -377,7 +396,8 @@ coalescing them to zero here would erase the distinction the whole change rests MaxExecPeakMemBytes: NullableLong(reader, 26), DeltaCalls: deltaCalls, DeltaTotalExecTimeMs: deltaTotalTime, - DeltaRows: deltaRows)); + DeltaRows: deltaRows, + SampleIntervalSeconds: sampleIntervalSeconds)); } return rows; @@ -421,6 +441,7 @@ public override void WritePayload(Row row, ICollectorRowWriter writer, Collector .Value(row.MaxExecPeakMemBytes) .Value(row.DeltaCalls) .Value(row.DeltaTotalExecTimeMs) - .Value(row.DeltaRows); + .Value(row.DeltaRows) + .Value(row.SampleIntervalSeconds); /* sample_interval_seconds INTEGER — measured, 0 = unknowable */ } } diff --git a/PerformanceMonitor.Collectors/PgWaitStatsCollector.cs b/PerformanceMonitor.Collectors/PgWaitStatsCollector.cs index 9d9e72517..60a81afca 100644 --- a/PerformanceMonitor.Collectors/PgWaitStatsCollector.cs +++ b/PerformanceMonitor.Collectors/PgWaitStatsCollector.cs @@ -71,7 +71,11 @@ public readonly record struct Row( whether this row ships at all, which means it must exist before the row is (or is not) added to the list WritePayload will later be called once per. */ long DeltaWaits, - long DeltaWaitTime); + long DeltaWaitTime, + /* #3540 (Darling V128): the measured seconds the two deltas accrued over — the minimum over the + row's two groups, so 0 means "no delta in this row is knowable". Computed beside the deltas in + ReadAsync because that is where the calculator is called. */ + int SampleIntervalSeconds); /* Wait TYPES whose events are never a finding. Filtered by type rather than by event name so a new background worker in a future Aurora release is excluded automatically instead of arriving @@ -171,6 +175,14 @@ LEFT JOIN aurora_stat_wait_event() AS e(type_id, event_id, event_name) new CollectorColumn("wait_time_us", CollectorColumnType.BigInt), new CollectorColumn("delta_waits", CollectorColumnType.BigInt), new CollectorColumn("delta_wait_time_us", CollectorColumnType.BigInt), + /* Appended (Darling V128, #3540): the measured seconds the row's two deltas accrued over, or 0 + when no delta was knowable. Appended at the END because the COPY writer is positional — the + same rule GoldenCollectorSchema's header states for every column a numbered migration adds by + ALTER TABLE. Until V128 this collector asked the calculator for the interval only to decide + the idle-row skip and stored nothing, so every row it DID ship with interval 0 (first sighting, + reset, gap re-baseline — the three cases the skip deliberately lets through) carried a delta 0 + that read as a measured idle interval. */ + new CollectorColumn("sample_interval_seconds", CollectorColumnType.Integer), }; public override async ValueTask> ReadAsync(DbDataReader reader, CollectorContext context, CancellationToken cancellationToken) @@ -206,10 +218,13 @@ so the id also carries the type. */ context.ServerId, "pg_wait_stats_waits", key, waits, out var waitsIntervalSeconds, collectionTime: context.CollectionTime, maxGapSeconds: CollectorDeltaCalculator.DefaultMaxGapSeconds); /* Computed unconditionally, even on a row this pass goes on to skip — see the class remarks - on why the wait-time series' baseline must stay fresh regardless. */ - var deltaWaitTime = context.Deltas.CalculateDelta( - context.ServerId, "pg_wait_stats_time", key, waitTimeMicroseconds, + on why the wait-time series' baseline must stay fresh regardless. With the interval (#3540, + V128): the stored interval is the MINIMUM over the row's two groups (the V127 rule, + WaitStatsCollector), so a reader never divides one group's reset 0 by the other's real span. */ + var deltaWaitTime = context.Deltas.CalculateDeltaWithInterval( + context.ServerId, "pg_wait_stats_time", key, waitTimeMicroseconds, out var timeIntervalSeconds, collectionTime: context.CollectionTime, maxGapSeconds: CollectorDeltaCalculator.DefaultMaxGapSeconds); + var sampleIntervalSeconds = Math.Min(waitsIntervalSeconds, timeIntervalSeconds); /* The skip: a REAL interval (this is not a first sighting, a counter reset, or a gap this pass just re-baselined) with zero new waits means the event demonstrably did not fire, @@ -228,7 +243,8 @@ and Aurora only advances wait_time when a wait completes — so nothing on this Waits: waits, WaitTimeMicroseconds: waitTimeMicroseconds, DeltaWaits: deltaWaits, - DeltaWaitTime: deltaWaitTime)); + DeltaWaitTime: deltaWaitTime, + SampleIntervalSeconds: sampleIntervalSeconds)); } return rows; @@ -244,6 +260,7 @@ public override void WritePayload(Row row, ICollectorRowWriter writer, Collector .Value(row.Waits) /* waits BIGINT */ .Value(row.WaitTimeMicroseconds) /* wait_time_us BIGINT */ .Value(row.DeltaWaits) /* delta_waits BIGINT */ - .Value(row.DeltaWaitTime); /* delta_wait_time_us BIGINT */ + .Value(row.DeltaWaitTime) /* delta_wait_time_us BIGINT */ + .Value(row.SampleIntervalSeconds); /* sample_interval_seconds INTEGER — measured, 0 = unknowable */ } } diff --git a/PerformanceMonitor.Collectors/ProcedureStatsCollector.cs b/PerformanceMonitor.Collectors/ProcedureStatsCollector.cs index 10daf0496..4281b53de 100644 --- a/PerformanceMonitor.Collectors/ProcedureStatsCollector.cs +++ b/PerformanceMonitor.Collectors/ProcedureStatsCollector.cs @@ -424,6 +424,12 @@ doubled escaping because @sql is itself a single-quoted T-SQL string. The plan f expression returns bigint, and a module-grain plan is measured in megabytes on the tail this exists to describe. */ new CollectorColumn("query_plan_xml_bytes", CollectorColumnType.BigInt), + /* Appended (Darling V128 / Lite v61, #3540): the measured seconds the row's seven deltas accrued + over, or 0 when no delta was knowable. Appended at the END because both stores' writers are + positional — the same rule GoldenCollectorSchema's header states for every column a numbered + migration adds by ALTER TABLE. The same integer perfmon_stats and query_stats have carried from + the start and the four V127 families gained, so one NULLIF idiom reads all ten. */ + new CollectorColumn("sample_interval_seconds", CollectorColumnType.Integer), }; public override async ValueTask> ReadAsync(DbDataReader reader, CollectorContext context, CancellationToken cancellationToken) @@ -478,15 +484,33 @@ a strict accessor. */ public override void WritePayload(Row row, ICollectorRowWriter writer, CollectorContext context) { /* Delta key: plan_handle to prevent cross-contamination when multiple plans exist for the - same object; the db.schema.object fallback and the seven group names are the parity contract. */ + same object; the db.schema.object fallback and the seven group names are the parity contract. + + The interval is stored beside the deltas (#3540, Darling V128 / Lite v61). The calculator + reports (delta 0, interval 0) when no delta is knowable — first sighting, counter reset, a gap + past the policy — and (0, n) when the interval was genuinely idle, and that pairing is the ONLY + way a reader can tell the two apart. This collector took the bare long and discarded the + interval at the write, so a restart's fabricated zero survived as a measured one and the + procedure duration trend LAG-divided it into a confident 0.00 ms/sec. This family also has the + highest first-sighting rate of the ten: a TOP (150) that churns readmits plans that fell out, + and every readmission is a first sighting whose 0 used to read as "ran zero times". + + One interval per ROW, the minimum over the row's seven groups — the V127 rule + (WaitStatsCollector). The groups share a key and a collection time, so first-sighting, + gap-policy and seeding decisions are identical across them and the intervals agree in every + case but an independent single-counter reset, which the module DMVs never do (a plan eviction + resets all seven together). Taking the minimum makes the stored pair mean "every delta in this + row is knowable", so a reader never divides a reset counter's 0 by a sibling's real interval + and reads it as idle. */ var deltaKey = row.PlanHandle ?? $"{row.DatabaseName}.{row.SchemaName}.{row.ObjectName}"; - var deltaExec = context.Deltas.CalculateDelta(context.ServerId, "proc_stats_exec", deltaKey, row.ExecutionCount, collectionTime: context.CollectionTime, maxGapSeconds: CollectorDeltaCalculator.DefaultMaxGapSeconds); - var deltaWorker = context.Deltas.CalculateDelta(context.ServerId, "proc_stats_worker", deltaKey, row.TotalWorkerTime, collectionTime: context.CollectionTime, maxGapSeconds: CollectorDeltaCalculator.DefaultMaxGapSeconds); - var deltaElapsed = context.Deltas.CalculateDelta(context.ServerId, "proc_stats_elapsed", deltaKey, row.TotalElapsedTime, collectionTime: context.CollectionTime, maxGapSeconds: CollectorDeltaCalculator.DefaultMaxGapSeconds); - var deltaReads = context.Deltas.CalculateDelta(context.ServerId, "proc_stats_reads", deltaKey, row.TotalLogicalReads, collectionTime: context.CollectionTime, maxGapSeconds: CollectorDeltaCalculator.DefaultMaxGapSeconds); - var deltaWrites = context.Deltas.CalculateDelta(context.ServerId, "proc_stats_writes", deltaKey, row.TotalLogicalWrites, collectionTime: context.CollectionTime, maxGapSeconds: CollectorDeltaCalculator.DefaultMaxGapSeconds); - var deltaPhysReads = context.Deltas.CalculateDelta(context.ServerId, "proc_stats_phys_reads", deltaKey, row.TotalPhysicalReads, collectionTime: context.CollectionTime, maxGapSeconds: CollectorDeltaCalculator.DefaultMaxGapSeconds); - var deltaSpills = context.Deltas.CalculateDelta(context.ServerId, "proc_stats_spills", deltaKey, row.TotalSpills, collectionTime: context.CollectionTime, maxGapSeconds: CollectorDeltaCalculator.DefaultMaxGapSeconds); + var deltaExec = context.Deltas.CalculateDeltaWithInterval(context.ServerId, "proc_stats_exec", deltaKey, row.ExecutionCount, out var execInterval, collectionTime: context.CollectionTime, maxGapSeconds: CollectorDeltaCalculator.DefaultMaxGapSeconds); + var deltaWorker = context.Deltas.CalculateDeltaWithInterval(context.ServerId, "proc_stats_worker", deltaKey, row.TotalWorkerTime, out var workerInterval, collectionTime: context.CollectionTime, maxGapSeconds: CollectorDeltaCalculator.DefaultMaxGapSeconds); + var deltaElapsed = context.Deltas.CalculateDeltaWithInterval(context.ServerId, "proc_stats_elapsed", deltaKey, row.TotalElapsedTime, out var elapsedInterval, collectionTime: context.CollectionTime, maxGapSeconds: CollectorDeltaCalculator.DefaultMaxGapSeconds); + var deltaReads = context.Deltas.CalculateDeltaWithInterval(context.ServerId, "proc_stats_reads", deltaKey, row.TotalLogicalReads, out var readsInterval, collectionTime: context.CollectionTime, maxGapSeconds: CollectorDeltaCalculator.DefaultMaxGapSeconds); + var deltaWrites = context.Deltas.CalculateDeltaWithInterval(context.ServerId, "proc_stats_writes", deltaKey, row.TotalLogicalWrites, out var writesInterval, collectionTime: context.CollectionTime, maxGapSeconds: CollectorDeltaCalculator.DefaultMaxGapSeconds); + var deltaPhysReads = context.Deltas.CalculateDeltaWithInterval(context.ServerId, "proc_stats_phys_reads", deltaKey, row.TotalPhysicalReads, out var physReadsInterval, collectionTime: context.CollectionTime, maxGapSeconds: CollectorDeltaCalculator.DefaultMaxGapSeconds); + var deltaSpills = context.Deltas.CalculateDeltaWithInterval(context.ServerId, "proc_stats_spills", deltaKey, row.TotalSpills, out var spillsInterval, collectionTime: context.CollectionTime, maxGapSeconds: CollectorDeltaCalculator.DefaultMaxGapSeconds); + var sampleIntervalSeconds = Math.Min(execInterval, Math.Min(workerInterval, Math.Min(elapsedInterval, Math.Min(readsInterval, Math.Min(writesInterval, Math.Min(physReadsInterval, spillsInterval)))))); writer .Value(row.DatabaseName) @@ -524,7 +548,8 @@ public override void WritePayload(Row row, ICollectorRowWriter writer, Collector .Value(deltaPhysReads) .Value(deltaSpills) .Value(row.QueryPlanXml) /* null unless CapturePlanXml captured it (Darling) */ - .Value(row.QueryPlanXmlBytes); /* #3392: measured size, never gated by the cap */ + .Value(row.QueryPlanXmlBytes) /* #3392: measured size, never gated by the cap */ + .Value(sampleIntervalSeconds); /* sample_interval_seconds INTEGER — measured, 0 = unknowable */ } /// diff --git a/PerformanceMonitor.Collectors/QueryStatsCollector.cs b/PerformanceMonitor.Collectors/QueryStatsCollector.cs index b008ac6e0..ad6353881 100644 --- a/PerformanceMonitor.Collectors/QueryStatsCollector.cs +++ b/PerformanceMonitor.Collectors/QueryStatsCollector.cs @@ -86,7 +86,20 @@ public sealed class Row public long? QueryPlanXmlBytes { get; set; } public long PlanGenerationNum { get; set; } + + /// + /// sys.dm_exec_query_stats.statement_start_offset / statement_end_offset: the statement's + /// position inside its batch text, in BYTES of the nvarchar text, so a character slice divides + /// by two (the SUBSTRING(st.text, (start / 2) + 1, …) in the SELECT above). -1 as the end + /// offset means "to the end of the batch"; (0, -1) is the whole batch. Half of the delta key + /// (sql_handle:start:end:plan_handle) — a multi-statement batch shares one sql_handle and one + /// plan_handle across its statements, and only the offsets tell them apart. Stored since Darling + /// V128 / Lite v61 (#3540) exactly as read, -1 included, so the restart seed can rebuild the + /// key from the store. + /// public int StatementStartOffset { get; set; } + + /// See . public int StatementEndOffset { get; set; } /// The statement's host object (schema.name) from sys.dm_exec_sql_text.objectid; @@ -383,6 +396,17 @@ public override CollectorQuery BuildQuery(CollectorContext context) expression returns bigint, and a plan XML document is measured in megabytes on the tail this exists to describe. */ new CollectorColumn("query_plan_xml_bytes", CollectorColumnType.BigInt), + /* #3540 (Darling V128 / Lite v61), appended after it for the same reason: the two statement + offsets the delta key is made of. Until these were stored no row in query_stats could + reproduce the key WritePayload builds, so the restart seed could restore this family's pass + window but not one baseline, and every plan older than the restart gap re-baselined on every + deploy. Integer, the DMV's own type. BYTE offsets into the batch's nvarchar text (Unicode, so a + character position is offset / 2), and statement_end_offset = -1 means "to the end of the + batch" — stored verbatim, -1 included, because the key string carries the raw values and the + seed has to spell the same string. NULL on every pre-V128 row: the offsets were never + recorded, and a fabricated 0/-1 would build a key nothing will ever present. */ + new CollectorColumn("statement_start_offset", CollectorColumnType.Integer), + new CollectorColumn("statement_end_offset", CollectorColumnType.Integer), }; public override async ValueTask> ReadAsync(DbDataReader reader, CollectorContext context, CancellationToken cancellationToken) @@ -461,7 +485,11 @@ capture modes and its ordinal is fixed. That pushes the plan XML to 44. */ public override void WritePayload(Row row, ICollectorRowWriter writer, CollectorContext context) { /* Delta key = the dm_exec_query_stats row identity (sql_handle + offsets + plan_handle). - Keying on plan_handle alone cross-contaminated multi-statement plans — parity contract. */ + Keying on plan_handle alone cross-contaminated multi-statement plans — parity contract. + The two hosts' restart seeds (DeltaCalculator / DarlingDeltaCalculator, QueryStatsSeedSql) + rebuild THIS string from the stored handles and offsets with the same interpolation — a null + handle formats as empty here and there, and the raw offsets (-1 included) are spelled by the + same int formatting — so the seeded key is the one this line presents. */ var deltaKey = $"{row.SqlHandle}:{row.StatementStartOffset}:{row.StatementEndOffset}:{row.PlanHandle}"; /* #2235: plan_handle is in the key above, and it changes on every recompile — so a churning plan @@ -540,7 +568,9 @@ ALL EIGHT delta'd counters take the same rule. Crediting only some would make on .Value(row.PlanGenerationNum) .Value(sampleIntervalSeconds) /* sample_interval_seconds INTEGER */ .Value(row.HostObjectName) /* #2012 stage 2: NULL for ad-hoc text */ - .Value(row.QueryPlanXmlBytes); /* #3392: measured size, never gated by the cap */ + .Value(row.QueryPlanXmlBytes) /* #3392: measured size, never gated by the cap */ + .Value(row.StatementStartOffset) /* #3540: the delta key's offsets, raw, -1 included */ + .Value(row.StatementEndOffset); } /// From ee07b5cf6a1539a0caa093b67d55f907bae739ca Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 15:25:38 -0400 Subject: [PATCH 57/69] Database File Growth's rise arm fires once per hourly observation, not once per cooldown: the adapter returns the collection that produced the growth and the engine remembers which one it already reported per file (#3636) (#3639) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The rise gate's input is a stored delta between two collections of database_size_stats, which lands hourly; the engine re-asked every ~30 s and re-fired on every 5-minute cooldown against the same two rows — up to twelve cards per growth event. #3579's mechanism, one condition over, at the worse ratio. Both SKUs' reads project the newest sample's collection_time as observed_at, appended last so the fourteen bound ordinals do not move. DatabaseFileGrowthInfo gains a nullable ObservedAtUtc. The engine keeps, per (server, database, file), the stamp it last fired on beside the per-server cooldown clock and active flag, and declines to send the card when no breached file is news: a level breach is always news (standing level, re-fired by design); a rise-only file is news only with a stamp newer than the one it last fired on. Stamped on fire only, including muted fires; pruned as files leave the breached set; cleared on recovery. In-memory like the forced-plan memory — a restart may re-fire once, which the emptied cooldown clock already did. --- Darling/Darling.Tests/AlertEngineTests.cs | 309 +++++++++++++++++- .../DatabaseFileGrowthReadTests.cs | 195 +++++++++++ .../DarlingAlertReadAdapter.cs | 17 +- Lite.Tests/FileGrowthAlertTests.cs | 69 ++++ Lite.Tests/FileGrowthReadTests.cs | 131 ++++++++ Lite/Services/LocalDataService.FileGrowth.cs | 30 +- .../AlertContextBuilders.cs | 19 +- PerformanceMonitor.Alerting/AlertEngine.cs | 76 ++++- .../DatabaseFileGrowthInfo.cs | 29 ++ 9 files changed, 861 insertions(+), 14 deletions(-) create mode 100644 Darling/Darling.Tests/DatabaseFileGrowthReadTests.cs create mode 100644 Lite.Tests/FileGrowthReadTests.cs diff --git a/Darling/Darling.Tests/AlertEngineTests.cs b/Darling/Darling.Tests/AlertEngineTests.cs index a2dfb366f..46fb5b418 100644 --- a/Darling/Darling.Tests/AlertEngineTests.cs +++ b/Darling/Darling.Tests/AlertEngineTests.cs @@ -159,14 +159,18 @@ public Task> GetVolumeFreeSpaceAsync(string serverKey, /* #2349: EMPTY by default. These tests mostly exercise other alerts, and a fabricated file would make the file-growth gate fire inside an unrelated scenario; the #3539 A8c pins below plant rows - and read back the lookback the engine asked for. */ + and read back the lookback the engine asked for. #3636: a fetch counter beside it, like the + forced-plan seam, so the once-per-observation pins can assert both what fired and that the read + still happened on every pass. */ public List Files { get; } = new(); public int? FileGrowthLookbackAsked { get; private set; } + public int FileGrowthFetches { get; private set; } public Task> GetDatabaseFileGrowthAsync( string serverKey, int lookbackMinutes, CancellationToken cancellationToken = default) { FileGrowthLookbackAsked = lookbackMinutes; + FileGrowthFetches++; return Task.FromResult(new List(Files)); } @@ -2174,6 +2178,309 @@ public async Task FileGrowth_ARateUnderTheBar_IsSilentOnAnyLookback(int lookback Assert.Empty(h.Deliverer.Outcomes); } + /* ---------------- #3636: the rise arm fires once per hourly observation, not once per cooldown ---------------- */ + + /// Two consecutive hourly collections of database_size_stats, so the pins below replay the + /// shape the issue describes rather than an invented one: a rise observed at the top of the hour, twelve + /// five-minute cooldowns of re-reads against the same two rows, the next collection an hour later. + private static readonly DateTime HourlyCollection0 = new(2026, 9, 18, 6, 0, 0, DateTimeKind.Utc); + private static readonly DateTime HourlyCollection1 = HourlyCollection0.AddHours(1); + + /// A rise-only file (2% of a 4 TB volume — the level gate cannot see it) stamped with the collection + /// that produced it. 20 GB in the 60-minute window is twice the default 10,240 MB/hr bar. + private static DatabaseFileGrowthInfo RiseOnlyFile(DateTime? observedAt, double growthMb = 20_480, string fileName = "tempdev") + { + var f = GrowingFile(growthMb, windowMinutes: 60); + f.FileName = fileName; + f.ObservedAtUtc = observedAt; + return f; + } + + /// A level-only file: 80% of a small volume and not growing at all. The standing-level shape that + /// re-fires on the cooldown by design. + private static DatabaseFileGrowthInfo LevelOnlyFile(DateTime? observedAt) => new() + { + DatabaseName = "Sales", FileName = "Sales_log", PhysicalName = @"L:\log\Sales_log.ldf", FileTypeDesc = "LOG", + TotalSizeMb = 400_000, GrowthMb = 0, GrowthWindowMinutes = 60, + VolumeMountPoint = @"L:\", VolumeTotalMb = 500_000, VolumeFreeMb = 90_000, ObservedAtUtc = observedAt, + }; + + [Fact] + public async Task FileGrowth_SameHourlyObservationAcrossTwelveCooldowns_FiresOnce_ThenResolvesOnTheNextCollection() + { + /* #3636's shape: the collector landed a 20 GB rise at 06:00 and nothing else until 07:00. Between those + two collections the adapter returned the SAME row (newest = 06:00, baseline = the window's far edge) + on every ~30 s pass, and the pre-#3636 engine fired on every cooldown expiry — up to twelve cards for + one growth event. The cooldown elapsing is not proof a new observation exists. */ + var h = new Harness(); + h.Settings.FileGrowthEnabled = true; + h.Settings.CooldownMinutes = 5; + h.Now = HourlyCollection0.AddSeconds(40); + h.Adapter.Files.Add(RiseOnlyFile(HourlyCollection0)); + var engine = h.Build(); + + await engine.EvaluateServerAsync(Harness.Snapshot()); + var first = Assert.Single(h.Deliverer.Outcomes); + Assert.Equal("Database File Growth", first.MetricName); + + /* Eleven more passes, each past the cooldown, none a new observation — the twelve-card loop. */ + for (var pass = 1; pass <= 11; pass++) + { + h.Now = HourlyCollection0.AddMinutes(pass * 5).AddSeconds(40); + await engine.EvaluateServerAsync(Harness.Snapshot()); + } + + Assert.Single(h.Deliverer.Outcomes); + /* And the read still happened on every pass — the guard is on the FIRE, never on the fetch, so the + recovery arm keeps seeing fresh evidence. */ + Assert.Equal(12, h.Adapter.FileGrowthFetches); + + /* 07:00: the next collection. The file did not grow in the new window; the read returns it under the + bar, and the recovery is announced exactly as before #3636. */ + h.Adapter.Files.Clear(); + h.Adapter.Files.Add(RiseOnlyFile(HourlyCollection1, growthMb: 0)); + h.Now = HourlyCollection1.AddSeconds(40); + await engine.EvaluateServerAsync(Harness.Snapshot()); + + Assert.Single(h.Deliverer.Outcomes); + var resolution = Assert.Single(h.Resolutions, r => r.MetricName == "Database File Growth"); + Assert.Contains("no file is growing past the threshold", resolution.Message, StringComparison.Ordinal); + } + + [Fact] + public async Task FileGrowth_NewerObservationWithARise_FiresAgain_CarryingItsOwnNumbers() + { + /* A file that keeps growing across successive hourly collections is still a standing condition: each + collection is a NEW observation with a new rise, and it re-fires — the guard removes repeats of one + observation, not the second card for a second hour of growth. The card carries the new observation's + growth, not the first one's. */ + var h = new Harness(); + h.Settings.FileGrowthEnabled = true; + h.Settings.CooldownMinutes = 5; + h.Now = HourlyCollection0.AddSeconds(40); + h.Adapter.Files.Add(RiseOnlyFile(HourlyCollection0, growthMb: 20_480)); + var engine = h.Build(); + + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Single(h.Deliverer.Outcomes); + Assert.Contains("grew 20.0 GB in 60 min", h.Deliverer.Outcomes[0].ShortMessage, StringComparison.Ordinal); + + /* The next collection: another 30 GB in the new window. Same file key, newer stamp. */ + h.Adapter.Files.Clear(); + h.Adapter.Files.Add(RiseOnlyFile(HourlyCollection1, growthMb: 30_720)); + h.Now = HourlyCollection1.AddSeconds(40); + await engine.EvaluateServerAsync(Harness.Snapshot()); + + Assert.Equal(2, h.Deliverer.Outcomes.Count); + Assert.Contains("grew 30.0 GB in 60 min", h.Deliverer.Outcomes[1].ShortMessage, StringComparison.Ordinal); + + /* And that observation, re-read past another cooldown, is one card too. */ + h.Now = h.Now.AddMinutes(6); + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Equal(2, h.Deliverer.Outcomes.Count); + Assert.Empty(h.Resolutions); + } + + [Fact] + public async Task FileGrowth_NewerObservationWithoutARise_DoesNotFire_AndResolves() + { + /* The third arm: a newer observation in which the file did NOT grow past the bar is not news for the + rise gate — it is the falling edge. No second card; the recovery is announced. */ + var h = new Harness(); + h.Settings.FileGrowthEnabled = true; + h.Settings.CooldownMinutes = 5; + h.Now = HourlyCollection0.AddSeconds(40); + h.Adapter.Files.Add(RiseOnlyFile(HourlyCollection0)); + var engine = h.Build(); + + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Single(h.Deliverer.Outcomes); + + h.Adapter.Files.Clear(); + h.Adapter.Files.Add(RiseOnlyFile(HourlyCollection1, growthMb: 512)); /* a twentieth of the bar */ + h.Now = HourlyCollection1.AddSeconds(40); + await engine.EvaluateServerAsync(Harness.Snapshot()); + + Assert.Single(h.Deliverer.Outcomes); + Assert.Single(h.Resolutions, r => r.MetricName == "Database File Growth"); + } + + [Fact] + public async Task FileGrowth_TheLevelArm_StillRefiresEveryCooldown_OnTheSameObservation() + { + /* The guard is on the RISE gate only. A file at 80% of its volume is at 80% on every pass whether or + not a new collection has landed — a standing level, re-fired on the cooldown by design (#2349), and + #3636 leaves that alone. Same stamp on every read; it fires on every cooldown regardless. */ + var h = new Harness(); + h.Settings.FileGrowthEnabled = true; + h.Settings.CooldownMinutes = 5; + h.Now = HourlyCollection0.AddSeconds(40); + h.Adapter.Files.Add(LevelOnlyFile(HourlyCollection0)); + var engine = h.Build(); + + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Single(h.Deliverer.Outcomes); + + h.Now = h.Now.AddMinutes(6); + await engine.EvaluateServerAsync(Harness.Snapshot()); + h.Now = h.Now.AddMinutes(6); + await engine.EvaluateServerAsync(Harness.Snapshot()); + + Assert.Equal(3, h.Deliverer.Outcomes.Count); + + /* And a rise-only file riding on the same server's card does not silence the level file: the card is + per server, one file with news is enough, and the level file is always news. */ + h.Adapter.Files.Add(RiseOnlyFile(HourlyCollection0)); + h.Now = h.Now.AddMinutes(6); + await engine.EvaluateServerAsync(Harness.Snapshot()); + h.Now = h.Now.AddMinutes(6); + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Equal(5, h.Deliverer.Outcomes.Count); + } + + [Fact] + public async Task FileGrowth_NewObservationInsideTheCooldown_FiresOnceTheCooldownElapses() + { + /* The memory is 'last ALERTED observation', not 'last SEEN' (the #3579 lesson): a collection that lands + while the cooldown from the previous card is still running has not been reported, so when the + cooldown elapses and the row is still that observation, it fires. Folding the two into one 'last + seen' stamp would record it as seen on the quiet pass and then never fire it. The hourly collector + makes this rare; a shortened cadence or a manual collection makes it real. */ + var h = new Harness(); + h.Settings.FileGrowthEnabled = true; + h.Settings.CooldownMinutes = 5; + h.Now = HourlyCollection0.AddSeconds(40); + h.Adapter.Files.Add(RiseOnlyFile(HourlyCollection0)); + var engine = h.Build(); + + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Single(h.Deliverer.Outcomes); + + var quickCollection = HourlyCollection0.AddMinutes(2); + h.Adapter.Files.Clear(); + h.Adapter.Files.Add(RiseOnlyFile(quickCollection, growthMb: 25_600)); + h.Now = quickCollection.AddSeconds(40); + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Single(h.Deliverer.Outcomes); /* cooldown holds it — rate limiting is still the cooldown's job */ + + h.Now = HourlyCollection0.AddMinutes(5).AddSeconds(50); + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Equal(2, h.Deliverer.Outcomes.Count); + + h.Now = h.Now.AddMinutes(6); + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Equal(2, h.Deliverer.Outcomes.Count); + } + + [Fact] + public async Task FileGrowth_MutedFire_StampsTheObservation_SoUnmutingDoesNotReplayIt() + { + /* A muted fire is still a fire: delivered flagged Muted, it stamps the cooldown, and since #3636 it + stamps the observation. A mute rule lifted mid-hour must not turn the same 06:00 rise into a fresh + card — the operator muted the server's file growth, not the engine's memory of it. */ + var h = new Harness(); + h.Settings.FileGrowthEnabled = true; + h.Settings.CooldownMinutes = 5; + h.Muted = true; + h.Now = HourlyCollection0.AddSeconds(40); + h.Adapter.Files.Add(RiseOnlyFile(HourlyCollection0)); + var engine = h.Build(); + + await engine.EvaluateServerAsync(Harness.Snapshot()); + var muted = Assert.Single(h.Deliverer.Outcomes); + Assert.True(muted.Muted); + + h.Muted = false; + h.Now = h.Now.AddMinutes(6); + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Single(h.Deliverer.Outcomes); + } + + [Fact] + public async Task FileGrowth_RecoveryForgetsTheObservation_SoANewEpisodeFires() + { + /* The falling edge clears the memory with the server (the #2166 lesson, at file grain): a file that + recovers and later grows again is a new episode and its first rise fires, whatever the stamp. */ + var h = new Harness(); + h.Settings.FileGrowthEnabled = true; + h.Settings.CooldownMinutes = 5; + h.Now = HourlyCollection0.AddSeconds(40); + h.Adapter.Files.Add(RiseOnlyFile(HourlyCollection0)); + var engine = h.Build(); + + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Single(h.Deliverer.Outcomes); + + h.Adapter.Files.Clear(); + h.Now = HourlyCollection1.AddSeconds(40); + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Single(h.Resolutions, r => r.MetricName == "Database File Growth"); + + var laterCollection = HourlyCollection1.AddHours(3); + h.Adapter.Files.Add(RiseOnlyFile(laterCollection)); + h.Now = laterCollection.AddSeconds(40); + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Equal(2, h.Deliverer.Outcomes.Count); + } + + [Fact] + public async Task FileGrowth_AFileThatLeavesTheBreachedSet_LosesItsMemory_WhileAnotherKeepsTheCardActive() + { + /* The memory is per FILE, pruned as files leave the breached set, not per server. tempdev fires at + 06:00; at 07:00 tempdev is quiet and templog has the rise — templog has never fired, so the card goes + (no recovery: the server still has a breaching file). At 08:00 tempdev is back with a fresh stamp + and templog is quiet: tempdev's 06:00 memory went with it when it left, and it fires as a new + episode would anyway — asserted through the card's headline naming the file. */ + var h = new Harness(); + h.Settings.FileGrowthEnabled = true; + h.Settings.CooldownMinutes = 5; + h.Now = HourlyCollection0.AddSeconds(40); + h.Adapter.Files.Add(RiseOnlyFile(HourlyCollection0, fileName: "tempdev")); + h.Adapter.Files.Add(RiseOnlyFile(HourlyCollection0, growthMb: 0, fileName: "templog")); + var engine = h.Build(); + + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Single(h.Deliverer.Outcomes); + Assert.StartsWith("tempdb.tempdev", h.Deliverer.Outcomes[0].ShortMessage, StringComparison.Ordinal); + + h.Adapter.Files.Clear(); + h.Adapter.Files.Add(RiseOnlyFile(HourlyCollection1, growthMb: 0, fileName: "tempdev")); + h.Adapter.Files.Add(RiseOnlyFile(HourlyCollection1, fileName: "templog")); + h.Now = HourlyCollection1.AddSeconds(40); + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Equal(2, h.Deliverer.Outcomes.Count); + Assert.StartsWith("tempdb.templog", h.Deliverer.Outcomes[1].ShortMessage, StringComparison.Ordinal); + Assert.Empty(h.Resolutions); + + /* Same 07:00 observation re-read past the cooldown: templog was reported; nothing new. */ + h.Now = h.Now.AddMinutes(6); + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Equal(2, h.Deliverer.Outcomes.Count); + } + + [Fact] + public async Task FileGrowth_StamplessRow_KeepsThePre3636CooldownRepeat() + { + /* The stated fallback for an adapter that supplies no observation stamp (the shipped two always do): + a null never matches a remembered stamp, so every read counts as new and the cooldown alone + rate-limits it — the pre-#3636 behaviour, degraded towards repetition rather than silence, the same + direction IAlertStateStore's no-op fallbacks degrade. Pinned so the fallback is a decision and not + an accident of null comparison. */ + var h = new Harness(); + h.Settings.FileGrowthEnabled = true; + h.Settings.CooldownMinutes = 5; + h.Now = HourlyCollection0.AddSeconds(40); + h.Adapter.Files.Add(RiseOnlyFile(observedAt: null)); + var engine = h.Build(); + + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Single(h.Deliverer.Outcomes); + + h.Now = h.Now.AddMinutes(6); + await engine.EvaluateServerAsync(Harness.Snapshot()); + Assert.Equal(2, h.Deliverer.Outcomes.Count); + } + /* ---------------- persistent version store (#1984) ---------------- */ [Fact] diff --git a/Darling/Darling.Tests/DatabaseFileGrowthReadTests.cs b/Darling/Darling.Tests/DatabaseFileGrowthReadTests.cs new file mode 100644 index 000000000..89937ccf1 --- /dev/null +++ b/Darling/Darling.Tests/DatabaseFileGrowthReadTests.cs @@ -0,0 +1,195 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Globalization; +using System.Threading; +using System.Threading.Tasks; +using Npgsql; +using PerformanceMonitor.Darling.Service; +using PerformanceMonitor.Darling.Storage; +using Xunit; + +namespace Darling.Tests; + +/// +/// #3636: the alerting pass's file-growth read (#2349) carries the observation's identity. The read's growth +/// figure is a fact about two COLLECTIONS — the newest sample and the oldest inside the window — and the +/// database_size_stats collector lands one per hour, so the row reads byte-identical on every ~30 s alert +/// pass in between. Without the newest sample's collection_time on the row, the engine's rise arm re-fired +/// on every 5-minute cooldown against the same observation: up to twelve cards per growth event. #3579 gave the +/// forced-plan read its observed_at for the identical mechanism at a 5-minute cadence; this is the same +/// column on the sibling read. +/// +/// The ungated pins hold the statement's shape — the stamp is LAST, so the fourteen ordinals the reader +/// already binds do not move, and it is the CTE's own collection_time rather than a new source column. +/// The gated one seeds two hourly collections on a live store and reads them back through the real Npgsql path: +/// the growth is the difference, the window is the measured span, and the stamp is the newest collection's +/// clock to the tick, Kind Utc — the engine compares one file's stamps for equality, so a stamp that came back +/// shifted or truncated would either never match (cooldown-repeat returns) or match a different collection. +/// +/* Live-fixture tests share one Postgres store; the collection serializes them so cross-test row churn + cannot race another class's assertions. */ +[Collection("live-postgres")] +public sealed class DatabaseFileGrowthReadTests +{ + /// Distinctive fake id — a real server_id is a storage-name hash, never this. + private const int TestServerId = -363636; + private static readonly string TestServerKey = TestServerId.ToString(CultureInfo.InvariantCulture); + private const string TestServerName = "file-growth-read-e2e"; + + /* ---------------- ungated shape pins ---------------- */ + + /// + /// The observation stamp is the LAST column of the shipped read and is c.collection_time — the newest + /// sample's collector clock, which current_files already selected for the window-width arithmetic and + /// never projected. Last, because the reader binds ordinals 0–13 to the fourteen pre-#3636 columns and an + /// inserted column would silently shift every one of them onto its neighbour's type. Pinned by splitting the + /// final select list into its lines (one column per line; several carry COALESCE(a, b), so a comma + /// split would over-count) rather than by substring, so a column appended AFTER it would fail here too. + /// + [Fact] + public void TheObservationStamp_IsTheLastColumn_AndIsTheNewestSamplesCollectionTime() + { + var sql = DarlingAlertReadAdapter.DatabaseFileGrowthSql; + var selectStart = sql.LastIndexOf("SELECT", StringComparison.Ordinal); + var fromStart = sql.IndexOf("FROM current_files c", StringComparison.Ordinal); + Assert.True(selectStart >= 0 && fromStart > selectStart, "the final select list was not found"); + + var columns = sql[(selectStart + "SELECT".Length)..fromStart] + .Split('\n', StringSplitOptions.TrimEntries | StringSplitOptions.RemoveEmptyEntries); + + Assert.Equal(15, columns.Length); + Assert.Equal("c.max_size_mb,", columns[13]); + Assert.Equal("c.collection_time AS observed_at", columns[14]); + + /* Not a new source column: the CTE selected collection_time before #3636 (the window width needs it). */ + var cte = sql[..sql.IndexOf("baseline AS", StringComparison.Ordinal)]; + Assert.Contains("file_type_desc, collection_time,", cte, StringComparison.Ordinal); + } + + /// The read stays a raw-table, parameter-bound-clock statement like every other alert feed: no bare + /// now() (timestamptz against the naive-UTC columns) and never Lite's v_ view. + [Fact] + public void TheRead_TargetsTheRawTable_AndBindsItsClock() + { + var sql = DarlingAlertReadAdapter.DatabaseFileGrowthSql; + Assert.DoesNotContain("now(", sql.ToLowerInvariant()); + Assert.DoesNotContain("FROM v_", sql, StringComparison.Ordinal); + Assert.Contains("FROM database_size_stats", sql, StringComparison.Ordinal); + Assert.Contains("collection_time >= $2", sql, StringComparison.Ordinal); + } + + /* ---------------- gated: the round trip ---------------- */ + + /// + /// Two hourly collections of one file plus a single-sample neighbour, read back through the real adapter. + /// The stamp is the NEWEST collection's collection_time, tick-equal to what was seeded (the store + /// holds naive UTC; the adapter names the Kind and never shifts the value), a re-read between collections + /// returns the same stamp, and a third collection moves it. + /// + [Fact] + public async Task TheStamp_IsTheNewestCollectionsClock_ToTheTick_AndMovesWithTheNextCollection() + { + var connectionString = Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + Assert.SkipWhen(string.IsNullOrEmpty(connectionString), + "Set DARLING_TEST_PG to a Postgres connection string to run the live file-growth read test."); + + var ct = TestContext.Current.CancellationToken; + + using var connection = new NpgsqlConnection(connectionString); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + await DeleteTestRowsAsync(connection, ct); + + await using var postgres = NpgsqlDataSource.Create(connectionString!); + var adapter = new DarlingAlertReadAdapter(postgres); + + var bodySucceeded = false; + try + { + /* Floored to whole microseconds (#3579's lesson): PostgreSQL timestamp is microsecond-resolution and + .NET ticks are 100 ns, so a raw UtcNow does not survive the round trip and the tick-equality below + would fail on any clock that is not itself microsecond-aligned. Kind Unspecified: naive-UTC storage. */ + var rawNow = DateTime.UtcNow; + var utcNow = DateTime.SpecifyKind(new DateTime(rawNow.Ticks - (rawNow.Ticks % 10)), DateTimeKind.Unspecified); + var collection0 = utcNow.AddMinutes(-70); + var collection1 = utcNow.AddMinutes(-10); + + /* tempdev: 100 GB at the top of the hour, 120 GB an hour later. templog: one sample only. */ + await SeedFileAsync(connection, ct, 1L, collection0, "tempdb", "tempdev", 102_400m); + await SeedFileAsync(connection, ct, 2L, collection1, "tempdb", "tempdev", 122_880m); + await SeedFileAsync(connection, ct, 2L, collection1, "tempdb", "templog", 4_096m); + + var files = await adapter.GetDatabaseFileGrowthAsync(TestServerKey, lookbackMinutes: 120, ct); + Assert.Equal(2, files.Count); + + var tempdev = Assert.Single(files, f => f.FileName == "tempdev"); + Assert.Equal(122_880d, tempdev.TotalSizeMb, precision: 3); + Assert.Equal(20_480d, tempdev.GrowthMb, precision: 3); + Assert.Equal(60d, tempdev.GrowthWindowMinutes, precision: 3); + /* #3636: the observation's identity is the newest sample's collection_time, to the tick, Kind Utc. */ + Assert.Equal((DateTime?)collection1, tempdev.ObservedAtUtc); + Assert.Equal(DateTimeKind.Utc, tempdev.ObservedAtUtc!.Value.Kind); + + /* The single-sample neighbour: no rise observed, and its own stamp all the same. */ + var templog = Assert.Single(files, f => f.FileName == "templog"); + Assert.Equal(0d, templog.GrowthMb, precision: 3); + Assert.Equal(0d, templog.GrowthWindowMinutes, precision: 3); + Assert.Equal((DateTime?)collection1, templog.ObservedAtUtc); + + /* Re-reading between collections is the same observation: same row, same stamp. This is the read + the engine used to fire on twelve times; the stamp is what lets it recognise the repeat. */ + var reread = await adapter.GetDatabaseFileGrowthAsync(TestServerKey, lookbackMinutes: 120, ct); + Assert.Equal(tempdev.ObservedAtUtc, Assert.Single(reread, f => f.FileName == "tempdev").ObservedAtUtc); + + /* The next collection lands: the stamp moves with it. */ + var collection2 = utcNow.AddMinutes(-2); + await SeedFileAsync(connection, ct, 3L, collection2, "tempdb", "tempdev", 122_880m); + var after = await adapter.GetDatabaseFileGrowthAsync(TestServerKey, lookbackMinutes: 120, ct); + Assert.Equal((DateTime?)collection2, Assert.Single(after, f => f.FileName == "tempdev").ObservedAtUtc); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(connectionString!, bodySucceeded, async (cleanup, cleanupCt) => + await DeleteTestRowsAsync(cleanup, cleanupCt)); + } + } + + private static async Task SeedFileAsync( + NpgsqlConnection connection, CancellationToken ct, + long collectionId, DateTime collectionTime, string databaseName, string fileName, decimal totalSizeMb) + { + using var command = new NpgsqlCommand(@" +INSERT INTO database_size_stats + (collection_id, collection_time, server_id, server_name, database_name, database_id, + file_id, file_type_desc, file_name, physical_name, total_size_mb, used_size_mb, + volume_mount_point, volume_total_mb, volume_free_mb, auto_growth_mb, is_percent_growth, growth_pct, max_size_mb) +VALUES ($1, $2, $3, $4, $5, 2, $6, $7, $8, $9, $10, $11, 'D:\', 4096000, 3000000, 1024, false, NULL, -1)", connection); + command.Parameters.AddWithValue(collectionId); + command.Parameters.AddWithValue(collectionTime); + command.Parameters.AddWithValue(TestServerId); + command.Parameters.AddWithValue(TestServerName); + command.Parameters.AddWithValue(databaseName); + command.Parameters.AddWithValue(fileName == "tempdev" ? 1 : 2); + command.Parameters.AddWithValue(fileName == "tempdev" ? "ROWS" : "LOG"); + command.Parameters.AddWithValue(fileName); + command.Parameters.AddWithValue(@"D:\data\" + fileName); + command.Parameters.AddWithValue(totalSizeMb); + command.Parameters.AddWithValue(totalSizeMb * 0.8m); + await command.ExecuteNonQueryAsync(ct); + } + + private static async Task DeleteTestRowsAsync(NpgsqlConnection connection, CancellationToken ct) + { + using var cleanup = new NpgsqlCommand($"DELETE FROM database_size_stats WHERE server_id = {TestServerId}", connection); + await cleanup.ExecuteNonQueryAsync(ct); + } +} diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingAlertReadAdapter.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingAlertReadAdapter.cs index f39429fc4..cead42bf8 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingAlertReadAdapter.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingAlertReadAdapter.cs @@ -557,6 +557,15 @@ public async Task> GetLongRunningQueriesAsync( /// chunks rather than scanning retention. The reported window width is measured rather than assumed, so a /// gap in collection cannot make a slow rise look fast. /// + /// The newest sample's collection_time travels with the row (#3636) as observed_at, + /// the last column. The growth is a fact about two COLLECTIONS and reads byte-identical on every alert pass + /// until the next collection lands — and this collector's cadence is an HOUR against a 5-minute cooldown, + /// so the engine needs the observation's identity to fire the rise gate once per collection instead of up to + /// twelve times. current_files already carried collection_time for the window-width arithmetic; + /// it was simply never projected. Appended rather than inserted so the fourteen ordinals the reader already + /// binds do not move. #3579's observed_at on the forced-plan read is the same column for the same + /// reason. + /// /// $1 server_id, $2 window start (naive UTC). /// public const string DatabaseFileGrowthSql = @" @@ -592,7 +601,8 @@ FROM database_size_stats c.auto_growth_mb, COALESCE(c.is_percent_growth, false) AS is_percent_growth, c.growth_pct, - c.max_size_mb + c.max_size_mb, + c.collection_time AS observed_at FROM current_files c LEFT JOIN baseline b ON b.database_name = c.database_name @@ -632,6 +642,11 @@ public async Task> GetDatabaseFileGrowthAsync( IsPercentGrowth = !reader.IsDBNull(11) && reader.GetBoolean(11), GrowthPct = reader.IsDBNull(12) ? null : Convert.ToDouble(reader.GetValue(12)), MaxSizeMb = reader.IsDBNull(13) ? null : Convert.ToDouble(reader.GetValue(13)), + /* #3636: the stored value is naive UTC (the $2 window bound above is built the same way), read + back with Kind Unspecified; stamped Utc because that is what it IS and what the property's name + says. The engine only ever compares one file's stamps with each other, so the Kind is honesty + rather than arithmetic — #3579's forced-plan read does exactly this. */ + ObservedAtUtc = reader.IsDBNull(14) ? null : DateTime.SpecifyKind(reader.GetDateTime(14), DateTimeKind.Utc), }); } diff --git a/Lite.Tests/FileGrowthAlertTests.cs b/Lite.Tests/FileGrowthAlertTests.cs index b18ed4c2d..985890ec4 100644 --- a/Lite.Tests/FileGrowthAlertTests.cs +++ b/Lite.Tests/FileGrowthAlertTests.cs @@ -317,4 +317,73 @@ public void TheCardsRate_UsesTheSharedUnitPhrase() Assert.Equal($"39.1 GB in 60 min (40000 {AlertContextBuilders.FileGrowthRiseUnit})", growth.Item2); Assert.Equal("MB/hr", AlertContextBuilders.FileGrowthRiseUnit); } + /* ---------------- #3636: the rise gate's split-out helpers, and the observation stamp both reads carry ---------------- */ + + /// + /// The two gates as the engine's #3636 guard asks them: + /// and are what + /// applies, so the guard (rise-only files are held to the observation stamp; level files are always news) + /// cannot classify a file differently from the breach list that put it on the card. Held on the three shapes + /// the alert distinguishes: rise-only, level-only, both. + /// + [Fact] + public void TheGateHelpers_AgreeWithTheBreachList_OnRiseOnlyLevelOnlyAndBoth() + { + var riseOnly = File(sizeMb: 90_000, growthMb: 40_000, volumeTotalMb: 4_000_000); + var levelOnly = File(sizeMb: 400_000, growthMb: 0, volumeTotalMb: 500_000); + var both = File(sizeMb: 400_000, growthMb: 40_000, volumeTotalMb: 500_000); + var neither = File(sizeMb: 50_000, growthMb: 100, volumeTotalMb: 500_000); + + Assert.True(AlertContextBuilders.BreachesRiseGate(riseOnly, riseMbPerHour: 10_240, lookbackMinutes: 60)); + Assert.False(AlertContextBuilders.BreachesLevelGate(riseOnly, volumePercent: 60)); + + Assert.False(AlertContextBuilders.BreachesRiseGate(levelOnly, riseMbPerHour: 10_240, lookbackMinutes: 60)); + Assert.True(AlertContextBuilders.BreachesLevelGate(levelOnly, volumePercent: 60)); + + Assert.True(AlertContextBuilders.BreachesRiseGate(both, riseMbPerHour: 10_240, lookbackMinutes: 60)); + Assert.True(AlertContextBuilders.BreachesLevelGate(both, volumePercent: 60)); + + /* Zero disables each helper exactly as it disables the gate in the breach list. */ + Assert.False(AlertContextBuilders.BreachesRiseGate(riseOnly, riseMbPerHour: 0, lookbackMinutes: 60)); + Assert.False(AlertContextBuilders.BreachesLevelGate(levelOnly, volumePercent: 0)); + + /* And the rise helper is the A8c bar, not the raw knob: 40 GB in a 5-minute window is held to 853 MB. */ + Assert.True(AlertContextBuilders.BreachesRiseGate(File(growthMb: 1_024, windowMinutes: 5, volumeTotalMb: 4_000_000), riseMbPerHour: 10_240, lookbackMinutes: 5)); + + foreach (var f in new[] { riseOnly, levelOnly, both, neither }) + { + var inList = AlertContextBuilders.GetBreachedFiles(new List { f }, riseMbPerHour: 10_240, volumePercent: 60, lookbackMinutes: 60).Count == 1; + var byHelpers = AlertContextBuilders.BreachesRiseGate(f, 10_240, 60) || AlertContextBuilders.BreachesLevelGate(f, 60); + Assert.Equal(byHelpers, inList); + } + } + + /// + /// #3636: both SKUs' file-growth reads project the newest sample's collection_time as + /// observed_at, LAST, so the fourteen ordinals both readers already bind do not move. The two texts + /// are not equal (DuckDB has no DISTINCT ON, so Lite's is a ROW_NUMBER() rewrite — the + /// FileGrowthAlertStoreTests pin holds that shape), so this pins the one clause #3636 added to each, read + /// from source on the Darling side through . + /// + [Fact] + public void BothSkusFileGrowthReads_CarryTheObservationStamp_Last() + { + var lite = PerformanceMonitorLite.Services.LocalDataService.DatabaseFileGrowthSql.ReplaceLineEndings("\n"); + Assert.EndsWith( + "c.max_size_mb,\n c.collection_time AS observed_at\nFROM windowed c", + lite[..(lite.IndexOf("FROM windowed c", StringComparison.Ordinal) + "FROM windowed c".Length)], + StringComparison.Ordinal); + + var darling = Lite.Tests.ParitySource.ReadFile("Darling/PerformanceMonitor.Darling.Service/DarlingAlertReadAdapter.cs"); + var darlingSql = darling[(darling.IndexOf("public const string DatabaseFileGrowthSql = @\"", StringComparison.Ordinal) + "public const string DatabaseFileGrowthSql = @\"".Length)..]; + darlingSql = darlingSql[..darlingSql.IndexOf("\";", StringComparison.Ordinal)].ReplaceLineEndings("\n"); + Assert.EndsWith( + "c.max_size_mb,\n c.collection_time AS observed_at\nFROM current_files c", + darlingSql[..(darlingSql.IndexOf("FROM current_files c", StringComparison.Ordinal) + "FROM current_files c".Length)], + StringComparison.Ordinal); + + /* The same window bound on both — the stamp is the newest row INSIDE the lookback, not the newest ever. */ + Assert.Contains("collection_time >= $2", lite, StringComparison.Ordinal); + Assert.Contains("collection_time >= $2", darlingSql, StringComparison.Ordinal); + } } diff --git a/Lite.Tests/FileGrowthReadTests.cs b/Lite.Tests/FileGrowthReadTests.cs new file mode 100644 index 000000000..0e5d0963a --- /dev/null +++ b/Lite.Tests/FileGrowthReadTests.cs @@ -0,0 +1,131 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor Lite. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Linq; +using System.Threading.Tasks; +using DuckDB.NET.Data; +using PerformanceMonitorLite.Database; +using PerformanceMonitorLite.Services; +using PerformanceMonitorLite.Tests; +using Xunit; + +namespace Lite.Tests; + +/// +/// Real-DuckDB round trip for Lite's file-growth read (#2349), replaying the #3636 shape: one file whose +/// database_size_stats row went 100 GB → 120 GB across two hourly collections. The read must return the +/// growth as the difference and the window as the measured span (the pre-#3636 contract), and must carry the +/// NEWEST sample's collection_time as the observation stamp — the fact the engine keys its +/// once-per-observation guard on, so that the same two rows re-read on every alert pass for the rest of the hour +/// produce one card rather than up to twelve. +/// +/// Two things the text pin cannot prove and this does: that DuckDB's TIMESTAMP comes back through +/// GetDateTime(14) at the ordinal the reader binds, and that the value round-trips to the tick — the engine +/// compares one file's stamps for equality, so a stamp that came back shifted or truncated would either never +/// match (cooldown-repeat returns) or match a different collection. ForcePlanFailuresReadTests is the +/// #3579 twin of this file. +/// +public sealed class FileGrowthReadTests : IClassFixture, IDisposable +{ + private const int ServerId = 3636; + + private readonly DuckDbInitializer _duckDb; + private DuckDBConnection? _seedConn; + private long _nextId = 1; + + public FileGrowthReadTests(SharedDuckDbFixture fixture) + { + fixture.ResetData(); + _duckDb = fixture.DuckDb; + } + + public void Dispose() => _seedConn?.Dispose(); + + /* Whole minutes by construction: DuckDB TIMESTAMP is microsecond-resolution, so raw DateTime ticks would + not survive the round trip and the tick-equality below would be testing the wrong thing. Kind + Unspecified, the store's naive-UTC convention. Inside a 120-minute lookback. */ + private static readonly DateTime Collection0 = MinuteFloor(DateTime.UtcNow.AddMinutes(-70)); + private static readonly DateTime Collection1 = Collection0.AddMinutes(60); + + private static DateTime MinuteFloor(DateTime t) => + DateTime.SpecifyKind(new DateTime(t.Ticks - (t.Ticks % TimeSpan.TicksPerMinute)), DateTimeKind.Unspecified); + + private async Task SeedConnectionAsync() + { + if (_seedConn is null) + { + _seedConn = _duckDb.CreateConnection(); + await _seedConn.OpenAsync(); + } + return _seedConn; + } + + private async Task SeedFileAsync(DateTime collectionTime, string fileName, string fileType, double totalSizeMb) + { + using var readLock = _duckDb.AcquireReadLock(); + var connection = await SeedConnectionAsync(); + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +INSERT INTO database_size_stats + (collection_id, collection_time, server_id, server_name, + database_name, database_id, file_id, file_type_desc, file_name, physical_name, + total_size_mb, used_size_mb, + volume_mount_point, volume_total_mb, volume_free_mb) +VALUES ($1, $2, $3, $4, 'tempdb', 2, $5, $6, $7, $8, $9, $10, 'D:\', 4096000, 3000000)"; + cmd.Parameters.Add(new DuckDBParameter { Value = _nextId++ }); + cmd.Parameters.Add(new DuckDBParameter { Value = collectionTime }); + cmd.Parameters.Add(new DuckDBParameter { Value = ServerId }); + cmd.Parameters.Add(new DuckDBParameter { Value = "GrowthSrv" }); + cmd.Parameters.Add(new DuckDBParameter { Value = fileName == "tempdev" ? 1 : 2 }); + cmd.Parameters.Add(new DuckDBParameter { Value = fileType }); + cmd.Parameters.Add(new DuckDBParameter { Value = fileName }); + cmd.Parameters.Add(new DuckDBParameter { Value = @"D:\data\" + fileName }); + cmd.Parameters.Add(new DuckDBParameter { Value = totalSizeMb }); + cmd.Parameters.Add(new DuckDBParameter { Value = totalSizeMb * 0.8 }); + await cmd.ExecuteNonQueryAsync(); + } + + [Fact] + public async Task TheRise_IsStampedWithTheNewestCollection_ToTheTick_AndTheStampMovesWithTheNextOne() + { + var service = new LocalDataService(_duckDb); + + /* Top of the hour: 100 GB. An hour later: 120 GB. templog has one sample only. */ + await SeedFileAsync(Collection0, "tempdev", "ROWS", 102_400); + await SeedFileAsync(Collection1, "tempdev", "ROWS", 122_880); + await SeedFileAsync(Collection1, "templog", "LOG", 4_096); + + var files = await service.GetDatabaseFileGrowthAsync(ServerId, lookbackMinutes: 120); + Assert.Equal(2, files.Count); + + var tempdev = Assert.Single(files, f => f.FileName == "tempdev"); + Assert.Equal(122_880d, tempdev.TotalSizeMb, precision: 3); + Assert.Equal(20_480d, tempdev.GrowthMb, precision: 3); + Assert.Equal(60d, tempdev.GrowthWindowMinutes, precision: 3); + /* #3636: the observation's identity is the newest sample's collection_time, to the tick, Kind Utc. */ + Assert.Equal((DateTime?)Collection1, tempdev.ObservedAtUtc); + Assert.Equal(DateTimeKind.Utc, tempdev.ObservedAtUtc!.Value.Kind); + + /* The single-sample neighbour: no rise observed (not "the whole file appeared"), and a stamp all the same. */ + var templog = Assert.Single(files, f => f.FileName == "templog"); + Assert.Equal(0d, templog.GrowthMb, precision: 3); + Assert.Equal((DateTime?)Collection1, templog.ObservedAtUtc); + + /* Re-reading between collections is the same observation: same row, same stamp. This is the read the + engine used to fire on twelve times; the stamp is what lets it recognise the repeat. */ + var reread = await service.GetDatabaseFileGrowthAsync(ServerId, lookbackMinutes: 120); + Assert.Equal(tempdev.ObservedAtUtc, Assert.Single(reread, f => f.FileName == "tempdev").ObservedAtUtc); + + /* The next collection lands: the stamp moves with it. */ + var collection2 = Collection1.AddMinutes(8); + await SeedFileAsync(collection2, "tempdev", "ROWS", 122_880); + var after = await service.GetDatabaseFileGrowthAsync(ServerId, lookbackMinutes: 120); + Assert.Equal((DateTime?)collection2, Assert.Single(after, f => f.FileName == "tempdev").ObservedAtUtc); + } +} diff --git a/Lite/Services/LocalDataService.FileGrowth.cs b/Lite/Services/LocalDataService.FileGrowth.cs index 7fee8c73c..3f5389831 100644 --- a/Lite/Services/LocalDataService.FileGrowth.cs +++ b/Lite/Services/LocalDataService.FileGrowth.cs @@ -25,15 +25,20 @@ namespace PerformanceMonitorLite.Services; /// DuckDB has no DISTINCT ON, so the same selection is expressed with ROW_NUMBER() /// partitioned by the file key — the idiom the rest of Lite's store SQL already uses where Postgres would use /// DISTINCT ON. +/// +/// The newest sample's collection_time rides along as observed_at, the last column (#3636): +/// the growth is a fact about two collections that reads identically on every alert pass until the next one +/// lands — hourly, for this collector, against a 5-minute cooldown — and the engine needs the observation's +/// identity to fire the rise gate once per collection rather than up to twelve times. Appended so the fourteen +/// ordinals already bound do not move; Darling's twin carries the same column at the same position, and the +/// Lite.Tests pin holds both texts to it. #3579's observed_at on the forced-plan read is the same +/// column for the same reason. /// public partial class LocalDataService { - public async Task> GetDatabaseFileGrowthAsync(int serverId, int lookbackMinutes) - { - using var connection = await OpenConnectionAsync(); - using var command = connection.CreateCommand(); - - command.CommandText = @" + /// The file-growth read's text, exposed like so the tests can + /// pin its shape against Darling's twin without a DuckDB round trip. $1 server_id, $2 window start. + public const string DatabaseFileGrowthSql = @" WITH windowed AS ( SELECT database_name, file_name, physical_name, file_type_desc, collection_time, @@ -59,7 +64,8 @@ FROM v_database_size_stats c.auto_growth_mb, COALESCE(c.is_percent_growth, false) AS is_percent_growth, c.growth_pct, - c.max_size_mb + c.max_size_mb, + c.collection_time AS observed_at FROM windowed c LEFT JOIN windowed b ON b.database_name = c.database_name @@ -69,6 +75,13 @@ LEFT JOIN windowed b AND c.total_size_mb IS NOT NULL ORDER BY c.database_name, c.file_name"; + public async Task> GetDatabaseFileGrowthAsync(int serverId, int lookbackMinutes) + { + using var connection = await OpenConnectionAsync(); + using var command = connection.CreateCommand(); + + command.CommandText = DatabaseFileGrowthSql; + command.Parameters.Add(new DuckDBParameter { Value = serverId }); command.Parameters.Add(new DuckDBParameter { @@ -95,6 +108,9 @@ AND c.total_size_mb IS NOT NULL IsPercentGrowth = !reader.IsDBNull(11) && Convert.ToBoolean(reader.GetValue(11)), GrowthPct = reader.IsDBNull(12) ? null : ToDouble(reader.GetValue(12)), MaxSizeMb = reader.IsDBNull(13) ? null : ToDouble(reader.GetValue(13)), + /* #3636: a naive-UTC TIMESTAMP read back Kind-Unspecified, stamped Utc because that is what it + is. The engine compares one file's stamps only with each other. */ + ObservedAtUtc = reader.IsDBNull(14) ? null : DateTime.SpecifyKind(reader.GetDateTime(14), DateTimeKind.Utc), }); } diff --git a/PerformanceMonitor.Alerting/AlertContextBuilders.cs b/PerformanceMonitor.Alerting/AlertContextBuilders.cs index 71805b5e5..6f8b57fe4 100644 --- a/PerformanceMonitor.Alerting/AlertContextBuilders.cs +++ b/PerformanceMonitor.Alerting/AlertContextBuilders.cs @@ -390,11 +390,8 @@ public static List GetBreachedFiles( { if (files is null || files.Count == 0) return new List(); - var riseBarMb = FileGrowthRiseBarMb(riseMbPerHour, lookbackMinutes); var breached = files - .Where(f => - (riseMbPerHour > 0 && f.GrowthMb >= riseBarMb) - || (volumePercent > 0 && f.VolumeTotalMb > 0 && f.VolumePercent >= volumePercent)) + .Where(f => BreachesRiseGate(f, riseMbPerHour, lookbackMinutes) || BreachesLevelGate(f, volumePercent)) .OrderByDescending(f => f.VolumePercent) .ThenByDescending(f => f.GrowthMb) .ToList(); @@ -402,6 +399,20 @@ public static List GetBreachedFiles( return breached; } + /// The RISE gate on its own: grew at least — the MB-per-hour + /// threshold scaled to the configured window (#3539 A8c) — inside the lookback window. Zero disables it. + /// Split out of at #3636 because the engine's once-per-observation guard + /// applies to THIS gate only, and it has to ask the same question the breach list asked rather than a + /// re-typed copy of it. + public static bool BreachesRiseGate(DatabaseFileGrowthInfo f, int riseMbPerHour, int lookbackMinutes) => + riseMbPerHour > 0 && f.GrowthMb >= FileGrowthRiseBarMb(riseMbPerHour, lookbackMinutes); + + /// The LEVEL gate on its own: the file is at least of its volume. + /// Zero disables it; a file with no volume stats (Azure SQL DB) is never level-gated. A standing level that + /// re-fires on the cooldown by design — the #3636 guard never consults it. + public static bool BreachesLevelGate(DatabaseFileGrowthInfo f, int volumePercent) => + volumePercent > 0 && f.VolumeTotalMb > 0 && f.VolumePercent >= volumePercent; + /// #2349: the alert card. Renders the top few by the same order /// produced, and names the fields an operator needs to act without opening the Viewer — including /// is_percent_growth, which surfaces a percent-autogrowth misconfiguration for free. diff --git a/PerformanceMonitor.Alerting/AlertEngine.cs b/PerformanceMonitor.Alerting/AlertEngine.cs index a2962d25e..b04205a54 100644 --- a/PerformanceMonitor.Alerting/AlertEngine.cs +++ b/PerformanceMonitor.Alerting/AlertEngine.cs @@ -192,6 +192,22 @@ without the other is a caller that can lose a signal. private readonly ConcurrentDictionary _activePvsAlert = new(); private readonly ConcurrentDictionary _lastFileGrowthAlert = new(); private readonly ConcurrentDictionary _activeFileGrowthAlert = new(); + + /* #3636: per server, the OBSERVATION each breached file was last alerted on — keyed "database|file", the + same key FileGrowthIncidents fingerprints on — so the RISE gate fires once per collection rather than + once per cooldown. The file-growth condition had no per-file state before this (its DTO says so: "no + per-file state to keep, survive a restart, or leak"), only the per-server cooldown clock and active + flag above; this sits beside them at file grain because files enter and leave the breached set + independently and a per-server stamp would let a threshold change mid-observation silence a file that + had never been reported. Stamped on FIRE only — including a muted fire — never on a pass that merely + saw the file (the #3579 seen ≠ alerted lesson: a collection landing inside the cooldown must still + fire when it elapses). A file that leaves the breached set loses its entry, and the server's recovery + clears the map, so a later episode starts with no memory exactly as the cooldown clock does. + + In-memory only, like the forced-plan memory it copies: a restart empties it, so the first pass after + one may re-fire once for an observation still in the window — which is exactly what the cooldown + clock, also emptied, already did before #3636, so the guard never makes a restart noisier than it was. */ + private readonly ConcurrentDictionary> _lastAlertedFileGrowthObservation = new(); private readonly ConcurrentDictionary _lastAlertedPvsPercent = new(); /* Rolling-count edge-trigger watermarks (#1091) — Lite's MainWindow.xaml.cs:103-104; @@ -1757,6 +1773,23 @@ await NotifyResolutionAsync(new AlertResolution( /// /// Observation sits OUTSIDE the fire branch, like blocking's (#2216/#2362): counting only at delivery /// lets a file that stops breaching during a cooldown mask the next one. + /// + /// The RISE gate fires once per OBSERVATION, not once per cooldown (#3636). The rise is a stored + /// fact about two collections — the newest sample and the oldest inside the window — and the + /// database_size_stats collector lands one per HOUR, while this check runs every ~30 s and the + /// cooldown is 5 minutes: at pass granularity "still growing" and "no new data yet" are indistinguishable, + /// so the pre-#3636 loop re-fired on every cooldown expiry against the SAME two rows — up to twelve cards + /// for one growth event before the next collection replaced the observation. #3579 found this mechanism + /// first in the forced-plan alert (5-minute cadence, six cards measured on one production store) and its + /// CheckForcePlanFailuresAsync remarks carry the fuller explanation; this is that guard one + /// condition over. The contract now: a file whose only breach is the rise gate fires once per new + /// — the engine remembers, per (server, database, + /// file), the stamp it last fired on and declines to fire the same stamp again regardless of cooldown; a + /// newer stamp with a rise fires (a file growing across successive hourly collections still re-fires, each + /// collection being a new observation), the cooldown still rate-limits those, and recovery is unchanged. + /// A file breaching the LEVEL gate is a standing level and re-fires on the cooldown exactly as before — the + /// guard never consults it. A row without a stamp falls back to the pre-#3636 cooldown-repeat rather than to + /// silence. /// private async Task CheckFileGrowthAsync( string key, string serverName, DateTime now, TimeSpan alertCooldown, bool suppressed, CancellationToken ct) @@ -1789,11 +1822,42 @@ a number whose meaning would otherwise change with the other knob. */ var worst = breached[0]; _activeFileGrowthAlert[key] = true; - if (!suppressed && CooldownElapsed(_lastFileGrowthAlert, key, now, alertCooldown)) + /* #3636: the observation guard, per file. A file that leaves the breached set loses its memory + here (a later re-entry is a new episode, like a forced plan that recovers and fails again); a + file that stays is news when it breaches the LEVEL gate — a standing level, re-fired by design + — or when its rise carries a stamp NEWER than the one this file last fired on. "Not newer" + rather than "equal", so an older stamp (nothing produces one today; a deleted newest row or a + clock step would) is not mistaken for news, and a null on either side never matches: a + stampless row keeps the cooldown-repeat, and a file that has not fired yet is always eligible. + The card is per server and carries every breached file, so ONE file with news is enough to + send it and every file on it is then stamped as reported. */ + var alertedObservations = _lastAlertedFileGrowthObservation.GetOrAdd(key, _ => new Dictionary(StringComparer.Ordinal)); + var breachedKeys = new HashSet(breached.Select(FileGrowthObservationKey), StringComparer.Ordinal); + foreach (var departed in alertedObservations.Keys.Where(k => !breachedKeys.Contains(k)).ToList()) + { + alertedObservations.Remove(departed); + } + + bool anyNewObservation = breached.Any(f => + AlertContextBuilders.BreachesLevelGate(f, _settings.FileGrowthVolumePercent) + || !(f.ObservedAtUtc is { } observedAt + && alertedObservations.TryGetValue(FileGrowthObservationKey(f), out var lastAlertedAt) + && observedAt <= lastAlertedAt)); + + if (!suppressed && anyNewObservation && CooldownElapsed(_lastFileGrowthAlert, key, now, alertCooldown)) { var muteCtx = new AlertMuteContext { ServerName = serverName, MetricName = "Database File Growth" }; bool isMuted = _isAlertMuted(muteCtx); _lastFileGrowthAlert[key] = now; + foreach (var f in breached) + { + /* #3636: stamped even when muted, like the cooldown — the operator muted the server's + file growth, not the engine's memory of which observation it already reported. */ + if (f.ObservedAtUtc is { } reported) + { + alertedObservations[FileGrowthObservationKey(f)] = reported; + } + } var context = AlertContextBuilders.BuildFileGrowthContext( serverName, breached, fileGrowthOccurrences.Decorate); @@ -1830,6 +1894,10 @@ await FireAsync(new AlertOutcome( else if (_activeFileGrowthAlert.TryGetValue(key, out var wasGrowing) && wasGrowing) { _activeFileGrowthAlert[key] = false; + /* #3636: the recovery drops the server's observation memory with it, deliberately — a file + that recovers and later grows again is a new episode and starts with no memory, exactly as + the cooldown clock beside it does when the condition next fires. */ + _lastAlertedFileGrowthObservation.TryRemove(key, out _); await ClearOccurrencesAsync(key, FileGrowthWatermarkMetric); readClock.Restart(); @@ -1853,6 +1921,12 @@ await NotifyResolutionAsync(new AlertResolution( } } + /// #3636: the per-file key of the observation memory — database|file, the same key + /// fingerprints on, so the memory and the occurrence + /// accumulator agree about what a "file" is (eight tempdb data files are eight files; a log file running + /// away is a different incident from its data files). + private static string FileGrowthObservationKey(DatabaseFileGrowthInfo f) => f.DatabaseName + "|" + f.FileName; + /* ---------------- anomalous Agent jobs (Lite AlertEngine.cs:557-632) ---------------- */ private async Task CheckAnomalousJobsAsync( diff --git a/PerformanceMonitor.Alerting/DatabaseFileGrowthInfo.cs b/PerformanceMonitor.Alerting/DatabaseFileGrowthInfo.cs index 4a1893fc6..8b1248aab 100644 --- a/PerformanceMonitor.Alerting/DatabaseFileGrowthInfo.cs +++ b/PerformanceMonitor.Alerting/DatabaseFileGrowthInfo.cs @@ -6,6 +6,8 @@ * Licensed under the MIT License. See LICENSE file in the project root for full license information. */ +using System; + namespace PerformanceMonitor.Alerting; /// @@ -21,6 +23,16 @@ namespace PerformanceMonitor.Alerting; /// Both gates come from one read. The store already holds the time series, so /// is measured against a sample from the lookback window rather than tracked in memory /// — no per-file state to keep, survive a restart, or leak. +/// +/// A row is an observation, and it carries its own identity (#3636). "Grew ≥ X MB inside the +/// lookback window" is a fact about two COLLECTIONS — the newest sample and the oldest one in the window — +/// and the database_size_stats collector lands one per HOUR. The engine re-reads the same two rows on +/// every ~30 s alert pass in between, with a 5-minute cooldown, so without the +/// rise arm could not tell "still growing" from "no new data yet": one growth event, one hourly observation, +/// up to TWELVE cards before the next collection replaced it. #3579 found the identical mechanism in the +/// forced-plan alert at a 5-minute cadence and gave ForcePlanFailureInfo its ObservedAtUtc; +/// this is the same stamp one condition over, at the worse ratio. The level gate is not this — a file at +/// 80% of its volume IS still at 80% every pass, and re-fires on the cooldown by design. /// public class DatabaseFileGrowthInfo { @@ -68,4 +80,21 @@ public class DatabaseFileGrowthInfo /// extrapolates — one autogrowth inside a five-minute span reads as twelve an hour (#3539 A8c). public double GrowthMbPerHour => GrowthWindowMinutes > 0 ? GrowthMb / (GrowthWindowMinutes / 60.0) : 0; + + /// + /// The collection_time of the NEWEST sample for this file — the collector's clock, not the alert + /// sweep's — which is the identity of the observation this row reports (#3636). Two reads that return the + /// same stamp for the same file are the same observation surfacing twice, not two growth events; a newer + /// stamp is a new collection and a new measurement. The engine keeps the stamp it last alerted on per + /// (server, database, file) and declines to re-fire the RISE gate on the same one regardless of cooldown + /// — ForcePlanFailureInfo.ObservedAtUtc's guard (#3579), which was itself the poison-wait family's + /// #2704 unrefreshed-source-row guard, at file grain. The LEVEL gate never consults it. + /// + /// Nullable so an adapter that does not supply it degrades to the pre-#3636 cooldown-repeat rather + /// than to silence: a null never equals a remembered stamp, so every read counts as new and the + /// cooldown alone rate-limits it — the same stated fallback gives a host + /// that cannot persist. Both shipped adapters always supply it; collection_time is NOT NULL in both + /// stores. + /// + public DateTime? ObservedAtUtc { get; set; } } From a33eaf393ae44e8ec7b8ab1c3b6d0e1a7368edda Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 15:26:55 -0400 Subject: [PATCH 58/69] DeliverAndReportAsync is a REQUIRED member of IAlertDeliverer, not a defaulted one: CONTRIBUTING's Two-Store Parity rule names this interface, so Lite's deliverer and every test fake now state their (null) answer by hand (#3580 review) (#3640) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Review note taken: a default body compiled and left LiteAlertDeliverer and fourteen fakes quietly inheriting an answer nobody wrote down — the exact pattern CONTRIBUTING forbids for this seam. Lite's implementation returns null and says why (its send seam returns no disposition; it hosts neither daily document); the DarlingAlertingTests wrapper forwards the inner deliverer's report rather than swallowing it. Also: an xUnit2013 analyzer warning in the new two-consults pin (Assert.Single over the Regex matches). --- Darling/Darling.Tests/AlertEngineTests.cs | 9 ++++++++ .../Darling.Tests/AlertStoredValueTests.cs | 9 ++++++++ .../Darling.Tests/CollectorCostDigestTests.cs | 9 ++++++++ .../CustomAlertResolveGrantLiveTests.cs | 10 +++++++++ .../CustomAlertSeverityEscalationLiveTests.cs | 9 ++++++++ .../CustomAlertTeardownLiveTests.cs | 10 +++++++++ Darling/Darling.Tests/DarlingAlertingTests.cs | 9 ++++++++ .../Darling.Tests/DarlingSelfAlertTests.cs | 9 ++++++++ .../Darling.Tests/FleetSweepRollupTests.cs | 9 ++++++++ .../SelfAlertDeliveryStampTests.cs | 19 +++++++++++------ .../Darling.Tests/StoreSelfMetricsTests.cs | 9 ++++++++ Lite.Tests/LiteAlertForwardingTests.cs | 9 ++++++++ Lite/Services/LiteAlertDeliverer.cs | 17 +++++++++++++++ .../IAlertDeliverer.cs | 21 +++++++++---------- 14 files changed, 141 insertions(+), 17 deletions(-) diff --git a/Darling/Darling.Tests/AlertEngineTests.cs b/Darling/Darling.Tests/AlertEngineTests.cs index 46fb5b418..d3bdd3b30 100644 --- a/Darling/Darling.Tests/AlertEngineTests.cs +++ b/Darling/Darling.Tests/AlertEngineTests.cs @@ -343,6 +343,15 @@ public Task DeliverAsync(AlertOutcome outcome, CancellationToken cancellationTok Outcomes.Add(outcome); return Task.CompletedTask; } + + /* #3580: DeliverAndReportAsync is REQUIRED on the seam rather than defaulted (CONTRIBUTING, Two-Store + Parity), so every fake answers it by hand. This one reports nothing: null is "unreported", which the + two daily documents treat as delivered, exactly as every fire before #3580 was. */ + public async Task DeliverAndReportAsync(AlertOutcome outcome, CancellationToken cancellationToken = default) + { + await DeliverAsync(outcome, cancellationToken); + return null; + } } /// One engine + fakes + a controllable clock per test. diff --git a/Darling/Darling.Tests/AlertStoredValueTests.cs b/Darling/Darling.Tests/AlertStoredValueTests.cs index 1df8ae3ad..b75683655 100644 --- a/Darling/Darling.Tests/AlertStoredValueTests.cs +++ b/Darling/Darling.Tests/AlertStoredValueTests.cs @@ -118,6 +118,15 @@ public Task DeliverAsync(AlertOutcome outcome, CancellationToken cancellationTok Outcomes.Add(outcome); return Task.CompletedTask; } + + /* #3580: DeliverAndReportAsync is REQUIRED on the seam rather than defaulted (CONTRIBUTING, Two-Store + Parity), so every fake answers it by hand. This one reports nothing: null is "unreported", which the + two daily documents treat as delivered, exactly as every fire before #3580 was. */ + public async Task DeliverAndReportAsync(AlertOutcome outcome, CancellationToken cancellationToken = default) + { + await DeliverAsync(outcome, cancellationToken); + return null; + } } private sealed class NullHistory : IAlertHistoryStore diff --git a/Darling/Darling.Tests/CollectorCostDigestTests.cs b/Darling/Darling.Tests/CollectorCostDigestTests.cs index 1e883b9ff..631489787 100644 --- a/Darling/Darling.Tests/CollectorCostDigestTests.cs +++ b/Darling/Darling.Tests/CollectorCostDigestTests.cs @@ -265,6 +265,15 @@ public Task DeliverAsync(AlertOutcome outcome, CancellationToken cancellationTok Outcomes.Add(outcome); return Task.CompletedTask; } + + /* #3580: DeliverAndReportAsync is REQUIRED on the seam rather than defaulted (CONTRIBUTING, Two-Store + Parity), so every fake answers it by hand. This one reports nothing: null is "unreported", which the + two daily documents treat as delivered, exactly as every fire before #3580 was. */ + public async Task DeliverAndReportAsync(AlertOutcome outcome, CancellationToken cancellationToken = default) + { + await DeliverAsync(outcome, cancellationToken); + return null; + } } private sealed class RecordingHistoryStore : IAlertHistoryStore diff --git a/Darling/Darling.Tests/CustomAlertResolveGrantLiveTests.cs b/Darling/Darling.Tests/CustomAlertResolveGrantLiveTests.cs index 7a92afa91..f06034a24 100644 --- a/Darling/Darling.Tests/CustomAlertResolveGrantLiveTests.cs +++ b/Darling/Darling.Tests/CustomAlertResolveGrantLiveTests.cs @@ -13,6 +13,7 @@ using Microsoft.Extensions.Logging.Abstractions; using Npgsql; using PerformanceMonitor.Alerting; +using PerformanceMonitor.Notifications; using PerformanceMonitor.Darling.Service; using PerformanceMonitor.Darling.Storage; using Xunit; @@ -41,6 +42,15 @@ public sealed class CustomAlertResolveGrantLiveTests private sealed class NoopDeliverer : IAlertDeliverer { public Task DeliverAsync(AlertOutcome outcome, CancellationToken cancellationToken = default) => Task.CompletedTask; + + /* #3580: DeliverAndReportAsync is REQUIRED on the seam rather than defaulted (CONTRIBUTING, Two-Store + Parity), so every fake answers it by hand. This one reports nothing: null is "unreported", which the + two daily documents treat as delivered, exactly as every fire before #3580 was. */ + public async Task DeliverAndReportAsync(AlertOutcome outcome, CancellationToken cancellationToken = default) + { + await DeliverAsync(outcome, cancellationToken); + return null; + } } private static string RequireLivePostgres() diff --git a/Darling/Darling.Tests/CustomAlertSeverityEscalationLiveTests.cs b/Darling/Darling.Tests/CustomAlertSeverityEscalationLiveTests.cs index f1289030e..d5b3e8ee5 100644 --- a/Darling/Darling.Tests/CustomAlertSeverityEscalationLiveTests.cs +++ b/Darling/Darling.Tests/CustomAlertSeverityEscalationLiveTests.cs @@ -45,6 +45,15 @@ public Task DeliverAsync(AlertOutcome outcome, CancellationToken cancellationTok return Task.CompletedTask; } + + /* #3580: DeliverAndReportAsync is REQUIRED on the seam rather than defaulted (CONTRIBUTING, Two-Store + Parity), so every fake answers it by hand. This one reports nothing: null is "unreported", which the + two daily documents treat as delivered, exactly as every fire before #3580 was. */ + public async Task DeliverAndReportAsync(AlertOutcome outcome, CancellationToken cancellationToken = default) + { + await DeliverAsync(outcome, cancellationToken); + return null; + } } private static string RequireLivePostgres() diff --git a/Darling/Darling.Tests/CustomAlertTeardownLiveTests.cs b/Darling/Darling.Tests/CustomAlertTeardownLiveTests.cs index 7edeaae99..5d38eff04 100644 --- a/Darling/Darling.Tests/CustomAlertTeardownLiveTests.cs +++ b/Darling/Darling.Tests/CustomAlertTeardownLiveTests.cs @@ -13,6 +13,7 @@ using Microsoft.Extensions.Logging.Abstractions; using Npgsql; using PerformanceMonitor.Alerting; +using PerformanceMonitor.Notifications; using PerformanceMonitor.Darling.Service; using PerformanceMonitor.Darling.Storage; using Xunit; @@ -35,6 +36,15 @@ public sealed class CustomAlertTeardownLiveTests private sealed class NoopDeliverer : IAlertDeliverer { public Task DeliverAsync(AlertOutcome outcome, CancellationToken cancellationToken = default) => Task.CompletedTask; + + /* #3580: DeliverAndReportAsync is REQUIRED on the seam rather than defaulted (CONTRIBUTING, Two-Store + Parity), so every fake answers it by hand. This one reports nothing: null is "unreported", which the + two daily documents treat as delivered, exactly as every fire before #3580 was. */ + public async Task DeliverAndReportAsync(AlertOutcome outcome, CancellationToken cancellationToken = default) + { + await DeliverAsync(outcome, cancellationToken); + return null; + } } private static string RequireLivePostgres() diff --git a/Darling/Darling.Tests/DarlingAlertingTests.cs b/Darling/Darling.Tests/DarlingAlertingTests.cs index 11f1ed454..21d3e889d 100644 --- a/Darling/Darling.Tests/DarlingAlertingTests.cs +++ b/Darling/Darling.Tests/DarlingAlertingTests.cs @@ -199,6 +199,15 @@ public async Task DeliverAsync(AlertOutcome outcome, CancellationToken cancellat Outcomes.Add(outcome); await _inner.DeliverAsync(outcome, cancellationToken); } + + /* #3580: REQUIRED on the seam rather than defaulted (CONTRIBUTING, Two-Store Parity). This wrapper + records and forwards, so it forwards the REPORT too — the inner deliverer here is the real + DarlingAlertDeliverer, and swallowing its answer would make the wrapper lie about it. */ + public async Task DeliverAndReportAsync(AlertOutcome outcome, CancellationToken cancellationToken = default) + { + Outcomes.Add(outcome); + return await _inner.DeliverAndReportAsync(outcome, cancellationToken); + } } [Fact] diff --git a/Darling/Darling.Tests/DarlingSelfAlertTests.cs b/Darling/Darling.Tests/DarlingSelfAlertTests.cs index 031d11e22..a3d0aa38d 100644 --- a/Darling/Darling.Tests/DarlingSelfAlertTests.cs +++ b/Darling/Darling.Tests/DarlingSelfAlertTests.cs @@ -106,6 +106,15 @@ public Task DeliverAsync(AlertOutcome outcome, CancellationToken cancellationTok Outcomes.Add(outcome); return Task.CompletedTask; } + + /* #3580: DeliverAndReportAsync is REQUIRED on the seam rather than defaulted (CONTRIBUTING, Two-Store + Parity), so every fake answers it by hand. This one reports nothing: null is "unreported", which the + two daily documents treat as delivered, exactly as every fire before #3580 was. */ + public async Task DeliverAndReportAsync(AlertOutcome outcome, CancellationToken cancellationToken = default) + { + await DeliverAsync(outcome, cancellationToken); + return null; + } } private sealed class FakeHistoryStore : IAlertHistoryStore diff --git a/Darling/Darling.Tests/FleetSweepRollupTests.cs b/Darling/Darling.Tests/FleetSweepRollupTests.cs index 179199a55..10e479824 100644 --- a/Darling/Darling.Tests/FleetSweepRollupTests.cs +++ b/Darling/Darling.Tests/FleetSweepRollupTests.cs @@ -95,6 +95,15 @@ public Task DeliverAsync(AlertOutcome outcome, CancellationToken cancellationTok Outcomes.Add(outcome); return Task.CompletedTask; } + + /* #3580: DeliverAndReportAsync is REQUIRED on the seam rather than defaulted (CONTRIBUTING, Two-Store + Parity), so every fake answers it by hand. This one reports nothing: null is "unreported", which the + two daily documents treat as delivered, exactly as every fire before #3580 was. */ + public async Task DeliverAndReportAsync(AlertOutcome outcome, CancellationToken cancellationToken = default) + { + await DeliverAsync(outcome, cancellationToken); + return null; + } } private sealed class RecordingHistoryStore : IAlertHistoryStore diff --git a/Darling/Darling.Tests/SelfAlertDeliveryStampTests.cs b/Darling/Darling.Tests/SelfAlertDeliveryStampTests.cs index 67c062249..9fffabf73 100644 --- a/Darling/Darling.Tests/SelfAlertDeliveryStampTests.cs +++ b/Darling/Darling.Tests/SelfAlertDeliveryStampTests.cs @@ -84,8 +84,9 @@ public Task DeliverAsync(AlertOutcome outcome, CancellationToken cancellationTok } } - /// The pre-#3580 shape: records, never reports. What every other suite's fake is, and what - /// Lite's deliverer still is — the default interface method answers null. + /// The pre-#3580 shape: records, never reports — what every other suite's fake is, and what + /// Lite's deliverer is. The report member is REQUIRED on the seam (CONTRIBUTING, Two-Store Parity), so + /// "never reports" is written down here as an explicit null rather than inherited from a default. private sealed class SilentDeliverer : IAlertDeliverer { public List Outcomes { get; } = new(); @@ -95,6 +96,12 @@ public Task DeliverAsync(AlertOutcome outcome, CancellationToken cancellationTok Outcomes.Add(outcome); return Task.CompletedTask; } + + public async Task DeliverAndReportAsync(AlertOutcome outcome, CancellationToken cancellationToken = default) + { + await DeliverAsync(outcome, cancellationToken); + return null; + } } private sealed class RecordingHistoryStore : IAlertHistoryStore @@ -428,7 +435,7 @@ been delivered and the memory gate is open. */ Assert.Equal(2, deliverer.Outcomes.Count); Assert.Equal(1, store.Reads); Assert.Equal(1, h.ReadFailures.ReadInstance().ReadFailures); - Assert.Equal(1, Regex.Matches(h.Log.Joined, "delivery stamp could not be read").Count); + Assert.Single(Regex.Matches(h.Log.Joined, "delivery stamp could not be read")); /* The next hour: still nothing landed, still one read on record — the retry runs from memory. */ h.Now = h.Now.AddHours(1); @@ -475,9 +482,9 @@ public async Task AStampWriteFault_StillGatesThisProcess_AndWarns_Uncounted(stri /* ---------------- what counts as delivered ---------------- */ - /// A deliverer that does not REPORT — the default interface method, every pre-#3580 fake, - /// Lite's deliverer — stamps: null is "unreported", treated as every fire before #3580 was, and never - /// read as "failed". A deliverer that knows a send failed says so. + /// A deliverer that does not REPORT — every pre-#3580 fake, Lite's deliverer — stamps: null is + /// "unreported", treated as every fire before #3580 was, and never read as "failed". A deliverer that + /// knows a send failed says so. [Theory] [InlineData(Digest)] [InlineData(Rollup)] diff --git a/Darling/Darling.Tests/StoreSelfMetricsTests.cs b/Darling/Darling.Tests/StoreSelfMetricsTests.cs index 41e07b21d..ef59e70b2 100644 --- a/Darling/Darling.Tests/StoreSelfMetricsTests.cs +++ b/Darling/Darling.Tests/StoreSelfMetricsTests.cs @@ -975,6 +975,15 @@ public Task DeliverAsync(AlertOutcome outcome, CancellationToken cancellationTok Outcomes.Add(outcome); return Task.CompletedTask; } + + /* #3580: DeliverAndReportAsync is REQUIRED on the seam rather than defaulted (CONTRIBUTING, Two-Store + Parity), so every fake answers it by hand. This one reports nothing: null is "unreported", which the + two daily documents treat as delivered, exactly as every fire before #3580 was. */ + public async Task DeliverAndReportAsync(AlertOutcome outcome, CancellationToken cancellationToken = default) + { + await DeliverAsync(outcome, cancellationToken); + return null; + } } private sealed class CadenceFakeHistoryStore : IAlertHistoryStore diff --git a/Lite.Tests/LiteAlertForwardingTests.cs b/Lite.Tests/LiteAlertForwardingTests.cs index 1e62373d6..90bed2571 100644 --- a/Lite.Tests/LiteAlertForwardingTests.cs +++ b/Lite.Tests/LiteAlertForwardingTests.cs @@ -254,6 +254,15 @@ public Task DeliverAsync(AlertOutcome outcome, CancellationToken cancellationTok Outcomes.Add(outcome); return Task.CompletedTask; } + + /* #3580: DeliverAndReportAsync is REQUIRED on the seam rather than defaulted (CONTRIBUTING, Two-Store + Parity), so every fake answers it by hand. This one reports nothing: null is "unreported", which the + two daily documents treat as delivered, exactly as every fire before #3580 was. */ + public async Task DeliverAndReportAsync(AlertOutcome outcome, CancellationToken cancellationToken = default) + { + await DeliverAsync(outcome, cancellationToken); + return null; + } } /// Engine over Lite's REAL live-settings adapter + fakes + a controllable clock. diff --git a/Lite/Services/LiteAlertDeliverer.cs b/Lite/Services/LiteAlertDeliverer.cs index a9f335500..64576f0cb 100644 --- a/Lite/Services/LiteAlertDeliverer.cs +++ b/Lite/Services/LiteAlertDeliverer.cs @@ -176,6 +176,23 @@ await _sendAlert( } } + /// + /// , reporting nothing (#3580). Lite hosts neither of the two daily documents + /// the report exists for, and its send seam ( over + /// EmailAlertService.TrySendAlertEmailAsync) returns no disposition — the fan-out result stays + /// inside the email service, which writes the history row itself — so there is nothing truthful to + /// report without rewiring that seam, which no Lite caller needs. Written by hand rather than + /// inherited, because the interface declares the member REQUIRED (CONTRIBUTING, Two-Store Parity: a + /// defaulted member leaves the implementer you forgot quietly doing nothing). null is + /// "unreported", never "failed"; if Lite ever grows a document-class self-alert, this is the seam to + /// teach, and the Darling deliverer is the shape. + /// + public async Task DeliverAndReportAsync(AlertOutcome outcome, CancellationToken cancellationToken = default) + { + await DeliverAsync(outcome, cancellationToken); + return null; + } + /// /// The old loop's toast icon table: Error for the two data-corruption-adjacent conditions /// (deadlocks, poison waits), Warning for everything else. diff --git a/PerformanceMonitor.Alerting/IAlertDeliverer.cs b/PerformanceMonitor.Alerting/IAlertDeliverer.cs index 32efe7055..33b31def6 100644 --- a/PerformanceMonitor.Alerting/IAlertDeliverer.cs +++ b/PerformanceMonitor.Alerting/IAlertDeliverer.cs @@ -108,16 +108,15 @@ public interface IAlertDeliverer /// nothing else on the engine's side does; so the report rides a separate method that the two askers /// call and every other caller ignores. /// - /// The default reports nothing. A deliverer that has not been taught to report delivers - /// exactly as before and returns null, which the askers treat the way every fire before #3580 - /// was treated: as delivered. That keeps Lite's deliverer and every test fake compiling and behaving - /// unchanged, and it means null is "unreported", never "failed" — a deliverer that KNOWS a send - /// failed reports , which is the one disposition the askers - /// withhold their delivered-today stamp on. + /// Required, not defaulted — CONTRIBUTING's Two-Store Parity rule, which names this interface. + /// A default body here would have compiled, and would have left Lite's deliverer and fourteen test fakes + /// quietly inheriting an answer nobody wrote down. So every implementer states its answer: Darling's + /// deliverer reports the disposition its history row was written with; Lite's, whose send seam returns + /// no disposition and which hosts neither daily document, returns null by hand and says why; each + /// fake does the same. null means "unreported", never "failed" — the askers treat it the way every + /// fire before #3580 was treated, as delivered — and a deliverer that KNOWS a send failed reports + /// , the one disposition the askers withhold their + /// delivered-today stamp on. /// - async Task DeliverAndReportAsync(AlertOutcome outcome, CancellationToken cancellationToken = default) - { - await DeliverAsync(outcome, cancellationToken); - return null; - } + Task DeliverAndReportAsync(AlertOutcome outcome, CancellationToken cancellationToken = default); } From f50a6ab30ae75f905cfc8d062a5c6319aa8af55a Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 15:37:34 -0400 Subject: [PATCH 59/69] compare_analysis bands each delta by the server's own dispersion and folds one cause into one row, so same-hour-yesterday noise stops reading as a verdict (#3538 A3) (#3634) * compare_analysis bands each delta by the server's own dispersion and folds one cause into one row, so same-hour-yesterday noise stops reading as a verdict (#3538 A3) The tool compared ONE window against ONE window and banded every key by its severity delta on a flat +/-0.1 dead-band. Severity is a threshold-formula artifact: a saturating ladder read a doubling from 30% to 60% of observed time as "stable", a trace of a 1%-bar wait read a few seconds an hour as "worse", one I/O stall surfaced as four to six worse keys, and BAD_ACTOR_ keys churned into new_issues/resolved_issues on plan-cache eviction alone. Shared ComparisonBanding (PerformanceMonitor.Analysis) now decides every verdict for both SKUs: a key with a same-unit per-server baseline (CPU %, read latency, connections) is banded in that bucket's robust sigma (delta_sigma, band_source "baseline", trustworthy buckets only); every other key changes status only when the value moved >= 25% of the larger side AND the larger side sits >= 0.25 up its own Layer-1 ladder (the scorer's base severity, so every concerning bar is reused with no number copied); one-sided keys are new/resolved only when they register on their ladder; BAD_ACTOR_* appearances are plan_cache_churn; rows carry a physical-cause family and the payload carries one family row per cause; the rules are stated in band_rules; every verdict row carries coverage_caveat when either side was partly observed (#3592 composes). ComparePeriodsAsync on both SKUs returns the dispersion map, fenced so a failed baseline read degrades to the absolute rule. * Verdict order is direction before magnitude, so a mixed-direction family's worst member is its regression (#3538 A3 review) The row sort partitioned only stable from non-stable; worse and better shared a bucket ordered by relative move, so a family whose BLOCKING_EVENTS fell 60% while its LCK_M_S rose 30% took the improvement as its worst member and left families_worse. Rows and families now rank worse > better > stable, then move; a mixed-direction family pin covers it. --- .../AnalysisCoverageLivePostgresTests.cs | 21 + Darling/Darling.Tests/DarlingMcpToolsTests.cs | 35 ++ .../DarlingAnalysisService.cs | 44 +- .../Mcp/DarlingMcpInstructions.cs | 2 +- .../Mcp/DarlingMcpTools.cs | 77 ++- Lite.Tests/AnalysisCoverageTests.cs | 10 + Lite.Tests/CompareAnalysisDispersionTests.cs | 261 ++++++++ Lite.Tests/ComparisonBandingTests.cs | 503 +++++++++++++++ Lite.Tests/McpMissMessageParityPinTests.cs | 6 + Lite/Analysis/AnalysisService.cs | 43 +- Lite/Mcp/McpAnalysisTools.cs | 77 ++- Lite/Mcp/McpInstructions.cs | 2 +- .../ComparisonBanding.cs | 584 ++++++++++++++++++ 13 files changed, 1573 insertions(+), 92 deletions(-) create mode 100644 Lite.Tests/CompareAnalysisDispersionTests.cs create mode 100644 Lite.Tests/ComparisonBandingTests.cs create mode 100644 PerformanceMonitor.Analysis/ComparisonBanding.cs diff --git a/Darling/Darling.Tests/AnalysisCoverageLivePostgresTests.cs b/Darling/Darling.Tests/AnalysisCoverageLivePostgresTests.cs index 1219bbf71..beaf997a4 100644 --- a/Darling/Darling.Tests/AnalysisCoverageLivePostgresTests.cs +++ b/Darling/Darling.Tests/AnalysisCoverageLivePostgresTests.cs @@ -162,6 +162,27 @@ rather than as a compared key. */ Assert.True(root.GetProperty("comparison").GetProperty("coverage").GetProperty("partial").GetBoolean()); Assert.DoesNotContain(root.GetProperty("facts").EnumerateArray(), f => f.GetProperty("key").GetString() == WindowCoverage.FactKey); + + /* #3538 A3 composes with the caveat: every verdict row and family carries coverage_caveat, + the storm is a comparison-only row banded by presence (worse: 0.25 saturates CXPACKET's + ladder), the rules are stated, and the summary counts families beside rows. */ + Assert.True(root.GetProperty("summary").GetProperty("coverage_caveat").GetBoolean()); + Assert.All(root.GetProperty("facts").EnumerateArray(), f => Assert.True(f.GetProperty("coverage_caveat").GetBoolean())); + Assert.All(root.GetProperty("families").EnumerateArray(), f => Assert.True(f.GetProperty("coverage_caveat").GetBoolean())); + var cxRow = Assert.Single(root.GetProperty("facts").EnumerateArray(), f => f.GetProperty("key").GetString() == "CXPACKET"); + Assert.Equal("comparison_only", cxRow.GetProperty("presence").GetString()); + Assert.Equal("presence", cxRow.GetProperty("band_source").GetString()); + Assert.Equal("worse", cxRow.GetProperty("status").GetString()); + Assert.Equal("parallelism", cxRow.GetProperty("family").GetString()); + /* BLOCKING_EVENTS at 10/hr (base 0.5) is the other comparison-only key that registers; the + two are two families (parallelism, lock_contention), so families_worse counts causes. */ + var blockingRow = Assert.Single(root.GetProperty("facts").EnumerateArray(), f => f.GetProperty("key").GetString() == "BLOCKING_EVENTS"); + Assert.Equal("worse", blockingRow.GetProperty("status").GetString()); + Assert.Equal("lock_contention", blockingRow.GetProperty("family").GetString()); + Assert.True(root.GetProperty("summary").GetProperty("new_issues").GetInt32() >= 2); + Assert.True(root.GetProperty("summary").GetProperty("families_worse").GetInt32() >= 2); + Assert.Contains("N=1 vs N=1", root.GetProperty("reading").GetString()!, StringComparison.Ordinal); + Assert.Contains("robust-sigma", root.GetProperty("band_rules").GetProperty("baseline").GetString()!, StringComparison.Ordinal); } /* ── zero coverage: a 2h window anchored inside the dead stretch composes with the #3524 diff --git a/Darling/Darling.Tests/DarlingMcpToolsTests.cs b/Darling/Darling.Tests/DarlingMcpToolsTests.cs index 87ea10ba0..e77921dcb 100644 --- a/Darling/Darling.Tests/DarlingMcpToolsTests.cs +++ b/Darling/Darling.Tests/DarlingMcpToolsTests.cs @@ -110,6 +110,41 @@ public void MuteAnalysisFinding_Description_NamesRegistered_MatchedNow_AndTheUnm Assert.Contains("`muted_unmatched`", row, StringComparison.Ordinal); } + /// + /// #3538 A3: compare_analysis's description promises the verdict shape the payload now carries — + /// value-banded rows with band_source / delta_sigma, the rules in band_rules, + /// physical-cause families, plan_cache_churn, coverage_caveat — and says what "worse" does + /// NOT mean (one window against one window is not an experiment). Same words as Lite (its twin pin is + /// CompareAnalysisDispersionTests; the shared sentences are in McpMissMessageParityPinTests), and + /// the instruction table row agrees. The banding arithmetic is pinned on the shared + /// ComparisonBanding in Lite.Tests; this is what a caller reads before trusting a verdict. + /// + [Fact] + public void CompareAnalysis_Description_SaysWhatWorseMeans_AndWhatItDoesNot() + { + var method = typeof(DarlingMcpTools).GetMethods(BindingFlags.Public | BindingFlags.Static) + .Single(m => m.GetCustomAttribute()?.Name == "compare_analysis"); + var description = method.GetCustomAttribute()!.Description; + + foreach (var token in new[] { "delta_sigma", "band_source", "band_rules", "families", "plan_cache_churn", "coverage_caveat", "N=1 vs N=1", "cannot show that a change CAUSED anything" }) + { + Assert.Contains(token, description, StringComparison.Ordinal); + } + + var row = DarlingMcpInstructions.Text.Split('\n').Single(l => l.Contains("| `compare_analysis` |", StringComparison.Ordinal)); + foreach (var token in new[] { "`band_source`", "`families`", "`plan_cache_churn`", "N=1 vs N=1" }) + { + Assert.Contains(token, row, StringComparison.Ordinal); + } + + /* The tool's ComparePeriodsAsync seam returns the dispersion the banding needs — a 5-tuple whose + last item is the per-metric BaselineBucket map. Pinned so a twin that forgot the item would fail + here rather than silently band everything by the absolute rule. */ + var compare = typeof(DarlingAnalysisService).GetMethod(nameof(DarlingAnalysisService.ComparePeriodsAsync))!; + var tuple = compare.ReturnType.GetGenericArguments()[0]; + Assert.Contains(typeof(IReadOnlyDictionary), tuple.GetGenericArguments()); + } + /* ---------------- ungated: config + hosting pins ---------------- */ [Fact] diff --git a/Darling/PerformanceMonitor.Darling.Analysis/DarlingAnalysisService.cs b/Darling/PerformanceMonitor.Darling.Analysis/DarlingAnalysisService.cs index c95de0765..7301e8920 100644 --- a/Darling/PerformanceMonitor.Darling.Analysis/DarlingAnalysisService.cs +++ b/Darling/PerformanceMonitor.Darling.Analysis/DarlingAnalysisService.cs @@ -14,6 +14,7 @@ using Microsoft.Extensions.Logging; using Npgsql; using PerformanceMonitor.Analysis; +using PerformanceMonitor.Analysis.Baselines; using PerformanceMonitor.Common; namespace PerformanceMonitor.Darling.Analysis; @@ -511,8 +512,20 @@ letting the halves drift. */ /// window's observed coverage (#3538 A2) so the caller can say when one side was only partly /// collected — the case the empty-window caveats never reached, where a half-collected window /// produces confident numbers with nothing to flag them. + /// + /// #3538 A3: also returns the stored per-server dispersion for the baselined metrics some + /// compared key is measured in (), keyed by metric + /// name, so compare_analysis can band a CPU or read-latency delta in the server's own robust + /// sigma instead of on a flat severity dead-band. The bucket is the comparison window's START hour + /// × day-of-week — the same coordinate the anomaly detectors read for a pass over that window + /// (PgAnomalyDetector passes context.TimeRangeStart), so the default same-hour-yesterday + /// call reuses the pass's cached buckets and an anchored one recomputes them the way an anchored + /// analyze_server does. The lookups are fenced separately from collection: a baseline read + /// that fails must not cost the caller the comparison it was only meant to refine, so it degrades + /// to an empty map and every key takes the absolute rule — the never-blind fallback the anomaly + /// gate follows. /// - public async Task<(List BaselineFacts, List ComparisonFacts, WindowCoverage? BaselineCoverage, WindowCoverage? ComparisonCoverage)> ComparePeriodsAsync( + public async Task<(List BaselineFacts, List ComparisonFacts, WindowCoverage? BaselineCoverage, WindowCoverage? ComparisonCoverage, IReadOnlyDictionary Dispersion)> ComparePeriodsAsync( int serverId, string serverName, DateTime baselineStart, DateTime baselineEnd, DateTime comparisonStart, DateTime comparisonEnd) @@ -541,14 +554,39 @@ letting the halves drift. */ _scorer.ScoreAll(baselineFacts); _scorer.ScoreAll(comparisonFacts); - return (baselineFacts, comparisonFacts, baselineContext.Coverage, comparisonContext.Coverage); + var dispersion = await LookUpDispersionAsync(serverId, serverName, baselineFacts, comparisonFacts, comparisonStart); + + return (baselineFacts, comparisonFacts, baselineContext.Coverage, comparisonContext.Coverage, dispersion); } catch (Exception ex) { _logger?.LogError("[DarlingAnalysisService] Period comparison failed for {Server}: {Message}", serverName, ex.Message); - return ([], [], null, null); + return ([], [], null, null, new Dictionary()); + } + } + + /// + /// The baseline buckets hands to the comparison, one per metric some + /// compared key is measured in. Its own try: see the summary above for why a failed baseline read + /// degrades to "no dispersion" rather than failing the comparison. + /// + private async Task> LookUpDispersionAsync( + int serverId, string serverName, List baselineFacts, List comparisonFacts, DateTime comparisonStart) + { + var dispersion = new Dictionary(StringComparer.Ordinal); + try + { + foreach (var metric in ComparisonBanding.DispersionMetricsFor(baselineFacts, comparisonFacts)) + dispersion[metric] = await _baselineProvider.GetBaselineAsync(serverId, metric, comparisonStart); + } + catch (Exception ex) + { + _logger?.LogWarning("[DarlingAnalysisService] Baseline dispersion lookup failed for {Server}; compare_analysis bands every key by the absolute rule: {Message}", + serverName, ex.Message); + dispersion.Clear(); } + return dispersion; } /// diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs index da382529b..a52b92dcf 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs @@ -88,7 +88,7 @@ public static string Build(DarlingPeerDirectory.Snapshot peers) |------|---------|----------------| | `analyze_server` | Runs the inference engine: scores facts, traverses relationship graph, returns evidence-backed findings with severity and recommended next tools. Each finding's `confidence` is an EVIDENCE score (0.20 for the symptom alone, more as the root fact's amplifier checks match and the chain deepens — a lone uncorroborated symptom is 0.20, never 1.0) and `confidence_basis` says in words what it rests on; rank by `severity` for impact and read `confidence` as how much of the engine's own corroboration showed up. A remediable finding also carries `remediation_command` — the full copy-paste T-SQL remediation (identical to the viewer card), with a two-sided risk-disclosure header on destructive changes; advisory only, never executed. Force-plan findings additionally carry `structured_remediation`: the verdict as machine-readable fields (eligible + named blockers), evidence, and split force/unforce/verify SQL. With `as_of` it analyzes a PAST window — anomaly baseline included — and is EXPLORATORY: the findings come back in full but are NOT persisted, which `persisted` / `persistence_note` state on every result | `server_name`, `hours_back` (default 4), `as_of` | | `get_analysis_facts` | Exposes raw scored facts from the collect+score pipeline — every observation the engine sees with base severity, amplifiers, and metadata (an ANOMALY_* fact's `baseline_confidence` is the baseline's trustworthiness, not a finding's `confidence`) | `server_name`, `hours_back` (default 4), `source` (filter), `min_severity`, `as_of` | - | `compare_analysis` | Compares two time periods (e.g., peak vs off-peak, before vs after a change) showing severity deltas for each fact. When NEITHER window produced facts the result is `unavailable` rather than an all-zero comparison, because "nothing to compare" is not "nothing changed"; when only ONE window is empty the payload carries a `caveat` saying so, since every fact then counts as new or resolved by default. `baseline_hours_back` is measured from the comparison window's END, so `as_of` moves BOTH windows together | `server_name`, `hours_back` (default 4), `baseline_hours_back` (default 28), `as_of` | + | `compare_analysis` | Compares two time periods (e.g., peak vs off-peak, yesterday vs today, the windows around a change), banding each fact worse / better / stable by how far its VALUE moved on the server's own scale — in the stored per-server baseline's robust sigma where one exists (`delta_sigma`, `band_source` `baseline`), otherwise only when the value moved at least a quarter of the larger side AND registers a quarter of the way up its own severity ladder (`band_source` `absolute`); `band_rules` states the rules on every payload. Rows are grouped into physical-cause `families` (one I/O stall is one family row), and `BAD_ACTOR_` appearances are `plan_cache_churn`, not new or resolved issues. A verdict is a DIFFERENCE, not an experiment: same-hour-yesterday at N=1 vs N=1 cannot show that a change caused anything. When NEITHER window produced facts the result is `unavailable` rather than an all-zero comparison, because "nothing to compare" is not "nothing changed"; when only ONE window is empty the payload carries a `caveat` saying so; a partly collected side flags every verdict row with `coverage_caveat`. `baseline_hours_back` is measured from the comparison window's END, so `as_of` moves BOTH windows together | `server_name`, `hours_back` (default 4), `baseline_hours_back` (default 28), `as_of` | | `audit_config` | Edition-aware configuration audit: evaluates CTFP, MAXDOP, max memory, and max worker threads against best practices | `server_name` | | `get_analysis_findings` | Retrieves persisted findings from previous analysis runs (each with `confidence_basis`: rows persisted before `confidence` measured corroboration are labelled `path-shape (pre-#3538)` — under that formula a lone symptom read 1.0, so do not read those as corroborated) (the service also analyzes on its own schedule, every 30 minutes per server), deduplicated to one entry per diagnostic chain (`story_path_hash` + `incident_id`): the latest occurrence plus `occurrences`/`first_seen`/`last_seen`/`peak_severity` spanning the window; each remediable finding carries `remediation_command` — the full copy-paste T-SQL remediation (identical to the viewer card), rendered from the persisted action, advisory only and never executed; force-plan findings additionally carry `structured_remediation` (verdict + evidence + split artifacts, machine-readable). Its window is on ANALYSIS TIME, so `as_of` asks what analysis was saying then rather than re-analyzing that window now | `server_name`, `hours_back` (default 24), `as_of` | | `mute_analysis_finding` | Mutes a finding pattern by story_path_hash so it won't appear in future runs. Reports what the write did: `registered`, and `matched_now` — how many stored findings in scope carry the hash (status `muted_unmatched` when 0: the mute is kept, but check the hash) | `story_path_hash` (required), `server_name`, `reason` | diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTools.cs index 77ae09866..84f61efc5 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTools.cs @@ -329,7 +329,7 @@ after configuration alone knows audit_config still has it. */ } } - [McpServerTool(Name = "compare_analysis"), Description("Compares two time periods by running the inference engine's fact collection and scoring on each, then showing what changed. Use this to compare peak vs off-peak, before vs after a change, or yesterday vs today. Returns facts from both periods side-by-side with severity deltas. Note: for routine anomaly detection, use analyze_server instead — it automatically compares against 30-day time-bucketed baselines (hour-of-day x day-of-week). This tool is for explicit window-to-window comparisons.")] + [McpServerTool(Name = "compare_analysis"), Description("Compares two time periods by running the inference engine's fact collection and scoring on each, then showing what changed. Use this to compare peak vs off-peak, yesterday vs today, or the windows around a change. Returns facts from both periods side-by-side, each banded worse / better / stable by how far the VALUE moved on the server's own scale, not by the severity formula's slope: a key with a stored per-server baseline (CPU %, read latency, connections) is banded in that baseline's robust sigma for the comparison hour (delta_sigma, band_source \"baseline\"); every other key changes status only when the value moved at least a quarter of the larger side AND registers at least a quarter of the way up its own severity ladder (band_source \"absolute\"); the rules are stated in band_rules. Rows are grouped into physical-cause families (one I/O stall is one family row, not four worse keys), and BAD_ACTOR_ appearances are reported as plan_cache_churn rather than as new or resolved issues. What \"worse\" does NOT mean: this is one window against one window — same-hour-yesterday at N=1 vs N=1 cannot show that a change CAUSED anything (DB time on an unchanged server routinely varies severalfold day to day), and a partly collected side flags every verdict with coverage_caveat. Note: for routine anomaly detection, use analyze_server instead — it automatically compares against 30-day time-bucketed baselines (hour-of-day x day-of-week). This tool is for explicit window-to-window comparisons.")] public static async Task CompareAnalysis( DarlingAnalysisService analysisService, NpgsqlDataSource postgres, @@ -359,7 +359,7 @@ silently change what the two windows are relative to each other. */ var baselineEnd = windowEnd.AddHours(-baseline_hours_back + hours_back); var baselineStart = windowEnd.AddHours(-baseline_hours_back); - var (baselineFacts, comparisonFacts, baselineCoverage, comparisonCoverage) = await analysisService.ComparePeriodsAsync( + var (baselineFacts, comparisonFacts, baselineCoverage, comparisonCoverage, dispersion) = await analysisService.ComparePeriodsAsync( resolved.ServerId, resolved.ServerName, baselineStart, baselineEnd, comparisonStart, comparisonEnd); @@ -370,41 +370,35 @@ silently change what the two windows are relative to each other. */ and pad fact rows with a key no advice speaks to. */ var baselineServerFacts = baselineFacts.Where(f => f.Source != WindowCoverage.FactSource).ToList(); var comparisonServerFacts = comparisonFacts.Where(f => f.Source != WindowCoverage.FactSource).ToList(); - var baselineByKey = baselineServerFacts.ToFactLookup(); - var comparisonByKey = comparisonServerFacts.ToFactLookup(); - var allKeys = baselineByKey.Keys.Union(comparisonByKey.Keys).ToHashSet(); - var comparisons = allKeys - .Select(key => - { - var baseline = baselineByKey.GetValueOrDefault(key); - var comparison = comparisonByKey.GetValueOrDefault(key); - var severityDelta = (comparison?.Severity ?? 0) - (baseline?.Severity ?? 0); + /* + #3538 A3: the coverage caveat is COMPOSED into the verdicts, not restated. The prose below + says which side was partly collected; every verdict row and family row carries + coverage_caveat: true when either side was, so a reader of one row cannot take "worse" at + face value without being told the side it rests on speaks for a fraction of its window. + */ + var baselinePartial = baselineCoverage is not null && (baselineCoverage.IsPartial || !baselineCoverage.IsObserved); + var comparisonPartial = comparisonCoverage is not null && (comparisonCoverage.IsPartial || !comparisonCoverage.IsObserved); - return new - { - key, - source = baseline?.Source ?? comparison?.Source ?? "unknown", - baseline_value = baseline != null ? Math.Round(baseline.Value, 6) : (double?)null, - comparison_value = comparison != null ? Math.Round(comparison.Value, 6) : (double?)null, - baseline_severity = baseline != null ? Math.Round(baseline.Severity, 4) : (double?)null, - comparison_severity = comparison != null ? Math.Round(comparison.Severity, 4) : (double?)null, - severity_delta = Math.Round(severityDelta, 4), - status = severityDelta > 0.1 ? "worse" : severityDelta < -0.1 ? "better" : "stable" - }; - }) - .OrderByDescending(c => Math.Abs(c.severity_delta)) - .ToList(); + /* + Every verdict — sigma-banded where a per-server baseline exists, ladder-banded where it + does not, plan-cache churn kept out of the issue counters, one family row per physical + cause — is decided in the shared ComparisonBanding, so this SKU and its twin cannot band + the same two windows differently. The tool only serializes. + */ + var comparison = ComparisonBanding.Compare( + baselineServerFacts, comparisonServerFacts, dispersion, + coverageCaveat: baselinePartial || comparisonPartial); - if (comparisons.Count == 0) + if (comparison.IsEmpty) { /* Neither window produced a single fact, and the old payload said that with all-zero counters and facts: [] -- which reads as "nothing changed" when it actually means "there was nothing to compare". Those are opposite conclusions about the same server. - No probe is needed to tell them apart: comparisons is the UNION of both windows' keys, - so zero entries is exactly "both fact sets were empty" and the fact_counts already in - hand are the whole answer. + No probe is needed to tell them apart: the comparison is over the UNION of both windows' + keys, so an empty one is exactly "both fact sets were empty" and the fact_counts already + in hand are the whole answer. */ return McpHelpers.Status( "unavailable", @@ -451,10 +445,10 @@ collection log without saying what they will find there. Both sides can earn one : null; var coverageCaveats = new List(2); - if (baselineCoverage is not null && (baselineCoverage.IsPartial || !baselineCoverage.IsObserved)) - coverageCaveats.Add($"The BASELINE window was only partly collected: {baselineCoverage.Describe()}. Its rates are per observed time, and its windowed facts are absent where nothing was observed."); - if (comparisonCoverage is not null && (comparisonCoverage.IsPartial || !comparisonCoverage.IsObserved)) - coverageCaveats.Add($"The COMPARISON window was only partly collected: {comparisonCoverage.Describe()}. Its rates are per observed time, and its windowed facts are absent where nothing was observed."); + if (baselinePartial) + coverageCaveats.Add($"The BASELINE window was only partly collected: {baselineCoverage!.Describe()}. Its rates are per observed time, and its windowed facts are absent where nothing was observed."); + if (comparisonPartial) + coverageCaveats.Add($"The COMPARISON window was only partly collected: {comparisonCoverage!.Describe()}. Its rates are per observed time, and its windowed facts are absent where nothing was observed."); if (coverageCaveats.Count > 0) coverageCaveats.Add("A side that was not fully observed cannot be read as the whole period: a wait that is absent because the collector was down is not a wait that resolved. Confirm coverage (get_collection_log, get_collection_health) before reading worse/better/resolved_issues as change."); @@ -468,6 +462,10 @@ collection log without saying what they will find there. Both sides can earn one /* Null when both windows produced facts at full coverage — the ordinary case, where nothing needs saying. */ caveat, + /* #3538 A3: what a verdict can and cannot carry, stated on every payload because the tool's + description is not in front of the reader when the numbers are. */ + reading = "Each row is banded by how far its VALUE moved on this server's own scale (band_source says which rule; band_rules states them), not by the severity formula's slope. One window against one window cannot show that a change caused anything: a same-hour-yesterday comparison at N=1 vs N=1 is a difference, not an experiment. Count families, not rows, to count causes.", + band_rules = ComparisonBanding.BandRulesPayload, baseline = new { start = baselineStart.ToString("o"), @@ -482,15 +480,10 @@ nothing needs saying. */ fact_count = comparisonServerFacts.Count, coverage = comparisonCoverage?.ToPayload() }, - summary = new - { - worse = comparisons.Count(c => c.status == "worse"), - better = comparisons.Count(c => c.status == "better"), - stable = comparisons.Count(c => c.status == "stable"), - new_issues = comparisons.Count(c => c.baseline_severity == null && c.comparison_severity > 0), - resolved_issues = comparisons.Count(c => c.baseline_severity > 0 && c.comparison_severity == null) - }, - facts = comparisons + summary = comparison.SummaryPayload(), + families = comparison.Families.Select(f => f.ToPayload()).ToList(), + plan_cache_churn = comparison.Churn.ToPayload(), + facts = comparison.Rows.Select(r => r.ToPayload()).ToList() }, McpHelpers.JsonOptions); } catch (Exception ex) diff --git a/Lite.Tests/AnalysisCoverageTests.cs b/Lite.Tests/AnalysisCoverageTests.cs index fbe31c5de..84ea23c3f 100644 --- a/Lite.Tests/AnalysisCoverageTests.cs +++ b/Lite.Tests/AnalysisCoverageTests.cs @@ -296,6 +296,14 @@ window because the collector saw a quarter as much time — and THAT is what the Assert.InRange(cx!.Value.GetProperty("baseline_value").GetDouble(), 0.06, 0.065); Assert.InRange(cx.Value.GetProperty("comparison_value").GetDouble(), 0.24, 0.26); + /* #3538 A3 composes with this: the fourfold value move IS banded worse (75% of the larger side, and + the key saturates its ladder), and the row carries coverage_caveat: true so that verdict cannot + be read without the sentence above — it is the collector's quarter, not the server's storm. */ + Assert.Equal("worse", cx.Value.GetProperty("status").GetString()); + Assert.True(cx.Value.GetProperty("coverage_caveat").GetBoolean()); + Assert.True(root.GetProperty("summary").GetProperty("coverage_caveat").GetBoolean()); + Assert.All(root.GetProperty("families").EnumerateArray(), f => Assert.True(f.GetProperty("coverage_caveat").GetBoolean())); + /* The COLLECTION_GAP fact is reported through the coverage blocks, not as a compared key. */ Assert.DoesNotContain(WindowCoverage.FactKey, result, StringComparison.Ordinal); } @@ -314,6 +322,8 @@ public async Task CompareAnalysis_BothWindowsFullyCollected_HasNoCaveat() Assert.Equal(JsonValueKind.Null, root.GetProperty("caveat").ValueKind); Assert.InRange(root.GetProperty("baseline").GetProperty("coverage").GetProperty("observed_fraction").GetDouble(), 0.99, 1.0); Assert.InRange(root.GetProperty("comparison").GetProperty("coverage").GetProperty("observed_fraction").GetDouble(), 0.99, 1.0); + Assert.False(root.GetProperty("summary").GetProperty("coverage_caveat").GetBoolean()); + Assert.All(root.GetProperty("facts").EnumerateArray(), f => Assert.False(f.GetProperty("coverage_caveat").GetBoolean())); } /* ── WindowCoverage arithmetic, no store ── */ diff --git a/Lite.Tests/CompareAnalysisDispersionTests.cs b/Lite.Tests/CompareAnalysisDispersionTests.cs new file mode 100644 index 000000000..54c26da5b --- /dev/null +++ b/Lite.Tests/CompareAnalysisDispersionTests.cs @@ -0,0 +1,261 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor Lite. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.ComponentModel; +using System.IO; +using System.Linq; +using System.Reflection; +using System.Text.Json; +using System.Threading.Tasks; +using DuckDB.NET.Data; +using ModelContextProtocol.Server; +using PerformanceMonitor.Analysis; +using PerformanceMonitorLite.Analysis; +using PerformanceMonitorLite.Database; +using PerformanceMonitorLite.Mcp; +using PerformanceMonitorLite.Models; +using PerformanceMonitorLite.Services; +using Xunit; + +namespace PerformanceMonitorLite.Tests; + +/// +/// #3538 A3 through the real compare_analysis tool over a DuckDB store: what a CALLER sees when +/// the same-hour-yesterday window is compared with today's. The banding arithmetic itself is pinned on +/// the shared helper in ; this class pins the payload contract — the +/// rules stated on every result, one family row per physical cause, the summary's family counts beside +/// its row counts, and the coverage caveat riding on every verdict row when a side was partly seen. +/// +/// Rows are planted as the collector writes them: a baseline reading at each window's start whose +/// delta is unknowable, then one collection every fifteen minutes carrying the delta since the previous +/// one, across the whole four-hour window (full coverage, so nothing here is about A2's divisor). +/// +public sealed class CompareAnalysisDispersionTests : IClassFixture, IDisposable +{ + private readonly string _tempDir; + private readonly DuckDbInitializer _duckDb; + private readonly ServerManager _serverManager; + private readonly int _serverId; + private long _nextId = -3_538_300_000; + private DuckDBConnection? _seedConn; + + public CompareAnalysisDispersionTests(SharedDuckDbFixture fixture) + { + fixture.ResetData(); + _duckDb = fixture.DuckDb; + + _tempDir = Path.Combine(Path.GetTempPath(), "CompareAnalysisDispersionTests_" + Guid.NewGuid().ToString("N")[..8]); + var configDir = Path.Combine(_tempDir, "config"); + Directory.CreateDirectory(configDir); + + /* Windows auth so AddServer never touches the credential store — no DPAPI side effects. */ + _serverManager = new ServerManager(configDir); + var server = new ServerConnection { ServerName = "TestServer", DisplayName = "TestServer" }; + _serverManager.AddServer(server); + + _serverId = RemoteCollectorService.GetDeterministicHashCode( + RemoteCollectorService.GetServerNameForStorage(server)); + } + + public void Dispose() + { + _seedConn?.Dispose(); + try { if (Directory.Exists(_tempDir)) Directory.Delete(_tempDir, recursive: true); } + catch { /* best-effort cleanup */ } + } + + /// + /// The review's I/O stall, yesterday vs today: PAGEIOLATCH_SH 10% → 30% and PAGEIOLATCH_EX 5% → 20% + /// of observed time, with CXPACKET flat at 10%. Two worse ROWS, one worse FAMILY (io_pressure, + /// both latch keys as members, the larger relative move as the worst member), one stable family; the + /// rules are stated on the payload; nothing is caveated, because both windows were fully observed. + /// + [Fact] + public async Task OneIoStall_IsOneWorseFamily_AndTheRulesAreStated() + { + var now = DateTime.UtcNow; + await PlantWindowAsync(now.AddHours(-24), ("PAGEIOLATCH_SH", 0.10), ("PAGEIOLATCH_EX", 0.05), ("CXPACKET", 0.10)); + await PlantWindowAsync(now, ("PAGEIOLATCH_SH", 0.30), ("PAGEIOLATCH_EX", 0.20), ("CXPACKET", 0.10)); + + var result = await McpAnalysisTools.CompareAnalysis(CreateService(), _serverManager, null, 4, 28); + + using var doc = JsonDocument.Parse(result); + var root = doc.RootElement; + Assert.Equal(JsonValueKind.Null, root.GetProperty("caveat").ValueKind); + + /* The rules travel with the numbers. */ + Assert.Contains("N=1 vs N=1", root.GetProperty("reading").GetString()!, StringComparison.Ordinal); + var rules = root.GetProperty("band_rules"); + Assert.Contains("robust-sigma", rules.GetProperty("baseline").GetString()!, StringComparison.Ordinal); + Assert.Contains("25%", rules.GetProperty("absolute").GetString()!, StringComparison.Ordinal); + Assert.Contains("plan_cache_churn", rules.GetProperty("presence").GetString()!, StringComparison.Ordinal); + + var summary = root.GetProperty("summary"); + Assert.Equal(2, summary.GetProperty("worse").GetInt32()); + Assert.Equal(1, summary.GetProperty("stable").GetInt32()); + Assert.Equal(1, summary.GetProperty("families_worse").GetInt32()); + Assert.Equal(1, summary.GetProperty("families_stable").GetInt32()); + Assert.Equal(0, summary.GetProperty("new_issues").GetInt32()); + Assert.Equal(0, summary.GetProperty("plan_cache_churn_appeared").GetInt32()); + Assert.False(summary.GetProperty("coverage_caveat").GetBoolean()); + + var families = root.GetProperty("families").EnumerateArray().ToList(); + Assert.Equal(2, families.Count); + var io = families[0]; // changed families first + Assert.Equal("io_pressure", io.GetProperty("family").GetString()); + Assert.Equal("worse", io.GetProperty("status").GetString()); + Assert.Equal("PAGEIOLATCH_EX", io.GetProperty("worst_key").GetString()); // 0.05 → 0.20 is the larger relative move + Assert.Equal(new[] { "PAGEIOLATCH_EX", "PAGEIOLATCH_SH" }, io.GetProperty("members").EnumerateArray().Select(m => m.GetString()).ToArray()); + Assert.Equal(2, io.GetProperty("worse").GetInt32()); + Assert.Equal("parallelism", families[1].GetProperty("family").GetString()); + Assert.Equal("stable", families[1].GetProperty("status").GetString()); + + var rows = root.GetProperty("facts").EnumerateArray().ToDictionary(r => r.GetProperty("key").GetString()!); + Assert.Equal(3, rows.Count); + foreach (var key in new[] { "PAGEIOLATCH_SH", "PAGEIOLATCH_EX" }) + { + var row = rows[key]; + Assert.Equal("worse", row.GetProperty("status").GetString()); + Assert.Equal("absolute", row.GetProperty("band_source").GetString()); + Assert.Equal("io_pressure", row.GetProperty("family").GetString()); + Assert.Equal("both", row.GetProperty("presence").GetString()); + Assert.False(row.GetProperty("coverage_caveat").GetBoolean()); + Assert.Equal(JsonValueKind.Null, row.GetProperty("delta_sigma").ValueKind); // no per-type wait baseline exists + } + Assert.InRange(rows["PAGEIOLATCH_EX"].GetProperty("relative_move").GetDouble(), 0.74, 0.76); + Assert.InRange(rows["PAGEIOLATCH_SH"].GetProperty("relative_move").GetDouble(), 0.66, 0.67); + + /* PAGEIOLATCH_SH saturates its (0.25, null) ladder at 30%: severity 1.0 on the comparison side, so + the old ±0.1 severity band read 0.10 → 0.30 as +0.6 and would have called 0.30 → 0.60 stable. */ + Assert.Equal(1.0, rows["PAGEIOLATCH_SH"].GetProperty("ladder_position").GetDouble(), precision: 4); + + var cx = rows["CXPACKET"]; + Assert.Equal("stable", cx.GetProperty("status").GetString()); + Assert.Equal(0.0, cx.GetProperty("relative_move").GetDouble(), precision: 4); + + var churn = root.GetProperty("plan_cache_churn"); + Assert.Empty(churn.GetProperty("appeared").EnumerateArray()); + Assert.Empty(churn.GetProperty("disappeared").EnumerateArray()); + Assert.Contains("plan-cache identities", churn.GetProperty("note").GetString()!, StringComparison.Ordinal); + } + + /// + /// A key present in one window only is a new issue only when it registers on its ladder: WRITELOG + /// appearing at 30% (base 0.6 on the (0.25, 0.50) ramp) is; LCK_M_IS appearing at 0.1% (base 0.02) is + /// a trace, stable, and not counted. The rows say presence and band_source so the reader + /// knows why one counted and the other did not. + /// + [Fact] + public async Task AKeyAppearingAtATrace_IsNotANewIssue_ButOneRegisteringOnItsLadderIs() + { + var now = DateTime.UtcNow; + await PlantWindowAsync(now.AddHours(-24), ("CXPACKET", 0.10)); + await PlantWindowAsync(now, ("CXPACKET", 0.10), ("WRITELOG", 0.30), ("LCK_M_IS", 0.001)); + + var result = await McpAnalysisTools.CompareAnalysis(CreateService(), _serverManager, null, 4, 28); + + using var doc = JsonDocument.Parse(result); + var root = doc.RootElement; + var summary = root.GetProperty("summary"); + Assert.Equal(1, summary.GetProperty("new_issues").GetInt32()); + Assert.Equal(1, summary.GetProperty("worse").GetInt32()); + Assert.Equal(2, summary.GetProperty("stable").GetInt32()); + + var rows = root.GetProperty("facts").EnumerateArray().ToDictionary(r => r.GetProperty("key").GetString()!); + var writelog = rows["WRITELOG"]; + Assert.Equal("worse", writelog.GetProperty("status").GetString()); + Assert.Equal("comparison_only", writelog.GetProperty("presence").GetString()); + Assert.Equal("presence", writelog.GetProperty("band_source").GetString()); + Assert.Equal(JsonValueKind.Null, writelog.GetProperty("baseline_value").ValueKind); + Assert.Equal("log_io", writelog.GetProperty("family").GetString()); + + var trace = rows["LCK_M_IS"]; + Assert.Equal("stable", trace.GetProperty("status").GetString()); + Assert.Equal("comparison_only", trace.GetProperty("presence").GetString()); + Assert.InRange(trace.GetProperty("ladder_position").GetDouble(), 0.0, 0.03); + + Assert.Equal("WRITELOG", root.GetProperty("facts")[0].GetProperty("key").GetString()); // the one changed row leads + } + + /// + /// The tool's description and the instructions row promise what the payload now carries — and say what + /// "worse" does not mean. A caller reads these before deciding to trust a verdict. + /// + [Fact] + public void Description_AndInstructionsRow_SayWhatWorseMeans_AndWhatItDoesNot() + { + var method = typeof(McpAnalysisTools).GetMethods(BindingFlags.Public | BindingFlags.Static) + .Single(m => m.GetCustomAttribute()?.Name == "compare_analysis"); + var description = method.GetCustomAttribute()!.Description; + + foreach (var token in new[] { "delta_sigma", "band_source", "band_rules", "families", "plan_cache_churn", "coverage_caveat", "N=1 vs N=1", "cannot show that a change CAUSED anything" }) + { + Assert.Contains(token, description, StringComparison.Ordinal); + } + + var row = McpInstructions.Text.Split('\n').Single(l => l.Contains("| `compare_analysis` |", StringComparison.Ordinal)); + foreach (var token in new[] { "`band_source`", "`families`", "`plan_cache_churn`", "N=1 vs N=1" }) + { + Assert.Contains(token, row, StringComparison.Ordinal); + } + } + + /* ── helpers ── */ + + private AnalysisService CreateService() => new(_duckDb) { MinimumDataHours = 0 }; + + /// + /// A fully collected 4h window ending at : a baseline reading at the start + /// (delta 0, unknowable) and sixteen 15-minute deltas per wait type, summing to the wait type's + /// fraction of the window. + /// + private async Task PlantWindowAsync(DateTime windowEnd, params (string WaitType, double Fraction)[] waits) + { + var start = windowEnd.AddHours(-4); + const int points = 16; + foreach (var (waitType, fraction) in waits) + { + var perPoint = (long)Math.Round(fraction * 4 * 3_600_000 / points); + await PlantWaitAsync(start, waitType, 0L); + for (var i = 1; i <= points; i++) + await PlantWaitAsync(start.AddMinutes(15 * i), waitType, perPoint); + } + } + + private async Task SeedConnectionAsync() + { + if (_seedConn is null) + { + _seedConn = _duckDb.CreateConnection(); + await _seedConn.OpenAsync(); + } + return _seedConn; + } + + private async Task PlantWaitAsync(DateTime at, string waitType, long deltaWaitMs) + { + using var readLock = _duckDb.AcquireReadLock(); + var conn = await SeedConnectionAsync(); + using var cmd = conn.CreateCommand(); + cmd.CommandText = @" +INSERT INTO wait_stats + (collection_id, collection_time, server_id, server_name, wait_type, + waiting_tasks_count, wait_time_ms, signal_wait_time_ms, + delta_waiting_tasks, delta_wait_time_ms, delta_signal_wait_time_ms) +VALUES ($1, $2, $3, 'TestServer', $4, 50, 1000000, 0, 4, $5, 0)"; + void P(object v) => cmd.Parameters.Add(new DuckDBParameter { Value = v }); + P(_nextId--); + P(at); + P(_serverId); + P(waitType); + P(deltaWaitMs); + await cmd.ExecuteNonQueryAsync(); + } +} diff --git a/Lite.Tests/ComparisonBandingTests.cs b/Lite.Tests/ComparisonBandingTests.cs new file mode 100644 index 000000000..5441eccd7 --- /dev/null +++ b/Lite.Tests/ComparisonBandingTests.cs @@ -0,0 +1,503 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor Lite. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Linq; +using PerformanceMonitor.Analysis; +using PerformanceMonitor.Analysis.Baselines; +using Xunit; + +namespace PerformanceMonitorLite.Tests; + +/// +/// #3538 A3: the arithmetic behind compare_analysis's verdicts, pinned on the shared +/// both SKUs serialize. Every fact here is scored by the REAL +/// before it is compared, so the ladder positions the absolute rule reads are the +/// scorer's own and a threshold change upstream moves these pins with it rather than past them. The +/// scenarios are the review's: the flat ±0.1 severity dead-band read a saturated doubling as "stable" and a +/// trace-to-trace wobble as "worse"; the same value delta banded by a tight and a loose baseline must +/// land on different sides; one I/O stall must be one family row; a plan-cache hash swap must be churn. +/// +public sealed class ComparisonBandingTests +{ + private static readonly IReadOnlyDictionary NoDispersion = new Dictionary(); + + /* ── the distortion the dead-band produced ── */ + + /// + /// PAGEIOLATCH_SH's ladder saturates at 25% of observed time ((0.25, null)), so 30% and 60% both + /// score 1.0 and the old band called a DOUBLING of I/O latch time "stable" (severity delta 0.0). The + /// value moved by half of the larger side and the key sits at the top of its ladder: worse. + /// + [Fact] + public void ASaturatedLadderDoubling_IsWorse_NotStable() + { + var (baseline, comparison) = Scored( + [Wait("PAGEIOLATCH_SH", 0.30)], + [Wait("PAGEIOLATCH_SH", 0.60)]); + Assert.Equal(1.0, baseline[0].Severity, precision: 6); + Assert.Equal(1.0, comparison[0].Severity, precision: 6); + + var row = Assert.Single(ComparisonBanding.Compare(baseline, comparison, NoDispersion, coverageCaveat: false).Rows); + + Assert.Equal(0.0, row.SeverityDelta, precision: 6); // what the old band read + Assert.Equal(ComparisonBanding.StatusWorse, row.Status); // what the value says + Assert.Equal(ComparisonBanding.BandSourceAbsolute, row.BandSource); + Assert.Equal(0.5, row.RelativeMove!.Value, precision: 6); + Assert.Equal(1.0, row.LadderPosition, precision: 6); + } + + /// + /// The RESOURCE_SEMAPHORE-vs-CPU distortion. RESOURCE_SEMAPHORE's ramp is (0.01, 0.10), so a trace + /// doubling from 0.2% to 0.45% of observed time (7 s/hr → 16 s/hr of grant queueing) moved severity + /// +0.125 — over the old dead-band, "worse". A 24-point CPU rise from 50% to 74% moved severity +0.16 + /// on the (75, 95) ramp — nearly the same number for an incomparably larger physical change. The + /// absolute rule reads each on its own ladder: the trace never climbed a quarter of the way up its + /// ladder (0.225), so it is stable; CPU did (0.493), and moved a third of the larger side, so it is + /// worse. Ordering follows the value's relative move, not the formula's slope. + /// + [Fact] + public void ATraceDoubling_IsStable_WhileARealCpuRise_IsWorse_WhateverTheSeverityDeltasSaid() + { + var (baseline, comparison) = Scored( + [Wait("RESOURCE_SEMAPHORE", 0.002), Cpu(50)], + [Wait("RESOURCE_SEMAPHORE", 0.0045), Cpu(74)]); + + var result = ComparisonBanding.Compare(baseline, comparison, NoDispersion, coverageCaveat: false); + var rs = result.Rows.Single(r => r.Key == "RESOURCE_SEMAPHORE"); + var cpu = result.Rows.Single(r => r.Key == "CPU_SQL_PERCENT"); + + /* The formula's slopes — the numbers the old band decided on. */ + Assert.Equal(0.125, rs.SeverityDelta, precision: 3); + Assert.Equal(0.16, cpu.SeverityDelta, precision: 2); + + Assert.Equal(ComparisonBanding.StatusStable, rs.Status); + Assert.InRange(rs.LadderPosition, 0.22, 0.23); + Assert.InRange(rs.RelativeMove!.Value, 0.55, 0.56); // a big RELATIVE move alone is not a verdict + + Assert.Equal(ComparisonBanding.StatusWorse, cpu.Status); + Assert.InRange(cpu.LadderPosition, 0.49, 0.50); + Assert.InRange(cpu.RelativeMove!.Value, 0.32, 0.33); + + Assert.Equal("CPU_SQL_PERCENT", result.Rows[0].Key); // changed rows first + Assert.Equal(1, result.Worse); + Assert.Equal(1, result.Stable); + } + + /// The absolute rule refuses a 1% wobble on a key that is otherwise high on its ladder. + [Fact] + public void AOnePercentWobble_IsStable_HoweverHighTheKeySits() + { + var (baseline, comparison) = Scored([Cpu(80)], [Cpu(80.8)]); + var row = Assert.Single(ComparisonBanding.Compare(baseline, comparison, NoDispersion, coverageCaveat: false).Rows); + + Assert.Equal(ComparisonBanding.StatusStable, row.Status); + Assert.InRange(row.LadderPosition, 0.6, 0.7); // well up the ladder … + Assert.InRange(row.RelativeMove!.Value, 0.0, 0.011); // … but it did not move + } + + /// + /// Both arms are required. Read latency doubling from 4 ms to 8 ms is a 50% relative move between two + /// values the scorer grades as healthy (base 0.2 at 8 ms on the (20, 50) ramp): stable. Doubling + /// from 12 ms to 24 ms crosses the concerning bar: worse. + /// + [Fact] + public void ADoublingBetweenTwoHealthyValues_IsStable_ADoublingIntoConcerning_IsWorse() + { + var (b1, c1) = Scored([Io(4)], [Io(8)]); + var healthy = Assert.Single(ComparisonBanding.Compare(b1, c1, NoDispersion, coverageCaveat: false).Rows); + Assert.Equal(ComparisonBanding.StatusStable, healthy.Status); + Assert.Equal(0.2, healthy.LadderPosition, precision: 6); + + var (b2, c2) = Scored([Io(12)], [Io(24)]); + var concerning = Assert.Single(ComparisonBanding.Compare(b2, c2, NoDispersion, coverageCaveat: false).Rows); + Assert.Equal(ComparisonBanding.StatusWorse, concerning.Status); + Assert.InRange(concerning.LadderPosition, 0.56, 0.57); + } + + /// A fall is "better" by the same two arms, and the direction comes from the scorer's ladder. + [Fact] + public void AFallByBothArms_IsBetter() + { + var (baseline, comparison) = Scored([Cpu(90)], [Cpu(45)]); + var row = Assert.Single(ComparisonBanding.Compare(baseline, comparison, NoDispersion, coverageCaveat: false).Rows); + Assert.Equal(ComparisonBanding.StatusBetter, row.Status); + Assert.Equal(0.5, row.RelativeMove!.Value, precision: 6); + } + + /// + /// DISK_SPACE is the free fraction and the scorer inverts it. With both sides saturated (under 5% free + /// scores 1.0 on either side) the value has to decide, and less free space must still read worse. + /// + [Fact] + public void FreeSpaceFalling_OnASaturatedLadder_IsWorse() + { + var (baseline, comparison) = Scored( + [new Fact { Source = "disk", Key = "DISK_SPACE", Value = 0.04 }], + [new Fact { Source = "disk", Key = "DISK_SPACE", Value = 0.02 }]); + Assert.Equal(1.0, baseline[0].BaseSeverity, precision: 6); + Assert.Equal(1.0, comparison[0].BaseSeverity, precision: 6); + + var row = Assert.Single(ComparisonBanding.Compare(baseline, comparison, NoDispersion, coverageCaveat: false).Rows); + Assert.Equal(ComparisonBanding.StatusWorse, row.Status); + } + + /* ── sigma banding ── */ + + /// + /// The same +10-point CPU delta, banded by two servers' own dispersion. A tight bucket (MAD 2 → robust + /// sigma 2.97, floored to the CPU model's 5-point absolute floor) reads it as 2σ: worse. A loose bucket + /// (MAD 10 → robust sigma 14.8) reads it as 0.67σ: stable. Neither verdict used the severity ladder, + /// and the baseline's confidence and tier travel with the row. + /// + [Fact] + public void TheSameDelta_BandsDifferently_ByTheServersOwnDispersion() + { + var (baseline, comparison) = Scored([Cpu(50)], [Cpu(60)]); + + var tight = Compare(baseline, comparison, CpuBucket(median: 50, mad: 2)); + Assert.Equal(ComparisonBanding.BandSourceBaseline, tight.BandSource); + Assert.Equal(ComparisonBanding.StatusWorse, tight.Status); + Assert.Equal(5.0, tight.BaselineSigma!.Value, precision: 4); // the 5-point CPU floor, not 2.97 + Assert.Equal(2.0, tight.DeltaSigma!.Value, precision: 2); + Assert.Equal(MetricNames.Cpu, tight.BaselineMetric); + Assert.Equal(nameof(BaselineTier.Full), tight.BaselineTier); + Assert.Equal(1.0, tight.BaselineConfidence!.Value, precision: 2); // 20 samples = 2x the Full floor + Assert.False(tight.BeyondAnomalyCutoff!.Value); // 2σ is under the 3.5σ robust cutoff + + var loose = Compare(baseline, comparison, CpuBucket(median: 50, mad: 10)); + Assert.Equal(ComparisonBanding.BandSourceBaseline, loose.BandSource); + Assert.Equal(ComparisonBanding.StatusStable, loose.Status); + Assert.InRange(loose.BaselineSigma!.Value, 14.82, 14.83); + Assert.InRange(loose.DeltaSigma!.Value, 0.67, 0.68); + } + + /// A move past the metric's own anomaly cutoff says so — and the display sigma is capped. + [Fact] + public void ABigSigmaMove_FlagsTheAnomalyCutoff_AndCapsTheDisplay() + { + var (baseline, comparison) = Scored([Cpu(20)], [Cpu(95)]); + var row = Compare(baseline, comparison, CpuBucket(median: 20, mad: 0.5)); + + Assert.Equal(5.0, row.BaselineSigma!.Value, precision: 4); // floored + Assert.Equal(15.0, row.DeltaSigma!.Value, precision: 2); + Assert.True(row.BeyondAnomalyCutoff!.Value); + Assert.Equal(ComparisonBanding.StatusWorse, row.Status); + + /* A session bucket has no absolute floor: MAD 0 floors at 1% of the median = 1.0 connection, so a + +75 move is 75σ raw — decided on the raw value, displayed at the detectors' 25σ cap. */ + var (sb, sc) = Scored([Sessions(100)], [Sessions(175)]); + var sessions = Compare(sb, sc, SessionsBucket(median: 100, mad: 0.0)); + Assert.Equal(1.0, sessions.BaselineSigma!.Value, precision: 4); + Assert.Equal(AnomalyThresholds.SigmaDisplayCap, sessions.DeltaSigma!.Value, precision: 2); + Assert.True(sessions.BeyondAnomalyCutoff!.Value); + Assert.Equal(ComparisonBanding.StatusWorse, sessions.Status); + } + + /// + /// The never-blind rule: an untrustworthy bucket (too few distinct days) routes the key to the + /// absolute rule with band_source: "absolute" rather than to a sigma nobody should trust — and + /// a key with no baseline metric at all never looks one up. + /// + [Fact] + public void AnUntrustworthyBaseline_FallsBackToTheAbsoluteRule() + { + var (baseline, comparison) = Scored([Cpu(50)], [Cpu(60)]); + var thin = new BaselineBucket + { + HourOfDay = 9, DayOfWeek = 2, Tier = BaselineTier.Full, + Mean = 50, StdDev = 5, Median = 50, Mad = 2, SampleCount = 20, DistinctDays = 1, + AbsStdDevFloor = BaselineMath.AbsStdDevFloorFor(MetricNames.Cpu) + }; + Assert.False(thin.IsTrustworthy); + + var row = Compare(baseline, comparison, thin); + Assert.Equal(ComparisonBanding.BandSourceAbsolute, row.BandSource); + Assert.Null(row.DeltaSigma); + Assert.Equal(MetricNames.Cpu, row.BaselineMetric); // the row still says which baseline WOULD apply + Assert.Equal(ComparisonBanding.StatusStable, row.Status); // +10 on 60 is a 17% move: under the quarter + + Assert.Null(ComparisonBanding.BaselinedMetricFor("PAGEIOLATCH_SH")); + Assert.Null(ComparisonBanding.BaselinedMetricFor("BLOCKING_EVENTS")); + Assert.Null(ComparisonBanding.BaselinedMetricFor("IO_WRITE_LATENCY_MS")); + } + + [Fact] + public void DispersionMetrics_AreOnlyTheOnesSomePresentKeyIsMeasuredIn() + { + var baseline = new List { Wait("CXPACKET", 0.1), Io(5) }; + var comparison = new List { Cpu(40), Wait("CXPACKET", 0.2) }; + Assert.Equal(new[] { MetricNames.Cpu, MetricNames.IoLatency }, ComparisonBanding.DispersionMetricsFor(baseline, comparison)); + Assert.Empty(ComparisonBanding.DispersionMetricsFor([Wait("CXPACKET", 0.1)], [Wait("LCK", 0.1)])); + } + + /* ── families ── */ + + /// + /// One I/O stall: PAGEIOLATCH_SH and PAGEIOLATCH_EX up, read latency up. Three worse rows, ONE worse + /// family, whose worst member is the one that moved most and whose members are all three named. + /// + [Fact] + public void OneIoStall_IsOneFamilyRow_WithThreeMembers() + { + var (baseline, comparison) = Scored( + [Wait("PAGEIOLATCH_SH", 0.10), Wait("PAGEIOLATCH_EX", 0.05), Io(12), Wait("CXPACKET", 0.10)], + [Wait("PAGEIOLATCH_SH", 0.30), Wait("PAGEIOLATCH_EX", 0.20), Io(30), Wait("CXPACKET", 0.10)]); + + var result = ComparisonBanding.Compare(baseline, comparison, NoDispersion, coverageCaveat: false); + + Assert.Equal(3, result.Worse); + Assert.Equal(1, result.FamiliesWorse); + Assert.Equal(1, result.FamiliesStable); + + var io = Assert.Single(result.Families, f => f.Family == "io_pressure"); + Assert.Equal(ComparisonBanding.StatusWorse, io.Status); + Assert.Equal("PAGEIOLATCH_EX", io.WorstKey); // 0.05 → 0.20 is the largest relative move (0.75) + Assert.Equal(new[] { "PAGEIOLATCH_EX", "PAGEIOLATCH_SH", "IO_READ_LATENCY_MS" }, io.Members); + Assert.Equal(3, io.Worse); + Assert.Equal(0, io.Stable); + + Assert.Equal("io_pressure", result.Families[0].Family); // changed families first + Assert.Equal("parallelism", result.Families[1].Family); + Assert.All(result.Rows.Where(r => r.Family == "io_pressure"), r => Assert.Equal(ComparisonBanding.StatusWorse, r.Status)); + } + + /// + /// A family with members moving in BOTH directions: BLOCKING_EVENTS falls 60% (better, the larger + /// move) while LCK_M_S rises to a scored level (worse, the smaller move). The regression is the + /// family's worst member and the family counts in families_worse — direction outranks magnitude, so + /// a large improvement in a sibling symptom cannot hide a real degradation in the same cause. The + /// review's catch on the first head; ordered rows read worse, then better, then stable. + /// + [Fact] + public void AMixedDirectionFamily_IsWorse_WhenAnyMemberIs_WhateverTheLargerMoveDid() + { + var (baseline, comparison) = Scored( + [Blocking(50), Wait("LCK_M_S", 0.02), Wait("CXPACKET", 0.10)], + [Blocking(20), Wait("LCK_M_S", 0.03), Wait("CXPACKET", 0.10)]); + + var result = ComparisonBanding.Compare(baseline, comparison, NoDispersion, coverageCaveat: false); + + var blocking = result.Rows.Single(r => r.Key == "BLOCKING_EVENTS"); + var lck = result.Rows.Single(r => r.Key == "LCK_M_S"); + Assert.Equal(ComparisonBanding.StatusBetter, blocking.Status); + Assert.Equal(0.6, blocking.RelativeMove!.Value, precision: 6); + Assert.Equal(ComparisonBanding.StatusWorse, lck.Status); + Assert.InRange(lck.RelativeMove!.Value, 0.33, 0.34); + + var family = Assert.Single(result.Families, f => f.Family == "lock_contention"); + Assert.Equal(ComparisonBanding.StatusWorse, family.Status); + Assert.Equal("LCK_M_S", family.WorstKey); + Assert.Equal(new[] { "LCK_M_S", "BLOCKING_EVENTS" }, family.Members); + Assert.Equal(1, family.Worse); + Assert.Equal(1, family.Better); + Assert.Equal(1, result.FamiliesWorse); + Assert.Equal(0, result.FamiliesBetter); + + Assert.Equal(new[] { "LCK_M_S", "BLOCKING_EVENTS", "CXPACKET" }, result.Rows.Select(r => r.Key)); + Assert.Equal("lock_contention", result.Families[0].Family); + } + + /// + /// The family map mirrors the collector's wait grouping and the reconciler's symptom families: every + /// regular key the reconciler folds an anomaly into shares a family with its siblings; a raw CX* or + /// general lock mode lands where the collector would have grouped it; a key with no family is its own. + /// + [Fact] + public void Families_FollowTheCollectorGrouping_AndTheReconcilersSymptomFamilies() + { + /* AnomalyIncidentReconciler.AnomalyToFamilies: CPU_SPIKE → {CPU_SQL_PERCENT, CPU_SPIKE}. */ + Assert.Equal(ComparisonBanding.FamilyFor("CPU_SQL_PERCENT"), ComparisonBanding.FamilyFor("CPU_SPIKE")); + Assert.Equal("cpu_pressure", ComparisonBanding.FamilyFor("SOS_SCHEDULER_YIELD")); + Assert.Equal("io_pressure", ComparisonBanding.FamilyFor("IO_READ_LATENCY_MS")); + Assert.Equal("log_io", ComparisonBanding.FamilyFor("IO_WRITE_LATENCY_MS")); + Assert.Equal("log_io", ComparisonBanding.FamilyFor("WRITELOG")); + Assert.Equal("log_io", ComparisonBanding.FamilyFor("HADR_SYNC_COMMIT")); + Assert.Equal("memory_grants", ComparisonBanding.FamilyFor("RESOURCE_SEMAPHORE")); + Assert.Equal("memory_grants", ComparisonBanding.FamilyFor("MEMORY_GRANT_PENDING")); + Assert.Equal("lock_contention", ComparisonBanding.FamilyFor("BLOCKING_EVENTS")); + Assert.Equal("lock_contention", ComparisonBanding.FamilyFor("LCK_M_S")); + Assert.Equal("lock_contention", ComparisonBanding.FamilyFor("LCK_M_RS_U")); + Assert.Equal("deadlocking", ComparisonBanding.FamilyFor("DEADLOCKS")); // kept apart, as the reconciler keeps it + + /* FactCollectorHelpers.WaitFamilyKey applied first. */ + Assert.Equal("parallelism", ComparisonBanding.FamilyFor("CXCONSUMER")); + Assert.Equal("lock_contention", ComparisonBanding.FamilyFor("LCK_M_IX")); + Assert.Equal("latch_contention", ComparisonBanding.FamilyFor("PAGELATCH_UP")); + + /* Its own family. */ + Assert.Equal("THREADPOOL", ComparisonBanding.FamilyFor("THREADPOOL")); + Assert.Equal("TEMPDB_USAGE", ComparisonBanding.FamilyFor("TEMPDB_USAGE")); + Assert.Equal("CONFIG_MAXDOP", ComparisonBanding.FamilyFor("CONFIG_MAXDOP")); + Assert.Equal("bad_actor", ComparisonBanding.FamilyFor("BAD_ACTOR_0x1234")); + } + + /* ── plan-cache churn and presence ── */ + + /// + /// A hash swap: the same statement under a recompiled plan is BAD_ACTOR_0xA yesterday and + /// BAD_ACTOR_0xB today. Key-set arithmetic called that one new issue and one resolved issue; it is + /// churn, counted nowhere in the issue counters. A hash present on both sides compares normally. + /// A non-bad-actor key that appears at a scored level IS a new issue; one that appears at a trace is not. + /// + [Fact] + public void ABadActorHashSwap_IsChurn_NotANewAndAResolvedIssue() + { + var (baseline, comparison) = Scored( + [BadActor("0xA", avgCpuMs: 500), BadActor("0xC", avgCpuMs: 100), Wait("CXPACKET", 0.10)], + [BadActor("0xB", avgCpuMs: 500), BadActor("0xC", avgCpuMs: 110), Wait("CXPACKET", 0.10), Wait("PAGEIOLATCH_SH", 0.20), Wait("LCK_M_IS", 0.001)]); + Assert.True(baseline[0].Severity > 0 && comparison[0].Severity > 0); + + var result = ComparisonBanding.Compare(baseline, comparison, NoDispersion, coverageCaveat: false); + + Assert.Equal(new[] { "BAD_ACTOR_0xB" }, result.Churn.Appeared.Select(e => e.Key)); + Assert.Equal(new[] { "BAD_ACTOR_0xA" }, result.Churn.Disappeared.Select(e => e.Key)); + Assert.Equal(1, result.Churn.PresentInBoth); + Assert.Equal(500, result.Churn.Appeared[0].Value); + Assert.DoesNotContain(result.Rows, r => r.Key is "BAD_ACTOR_0xA" or "BAD_ACTOR_0xB"); + + var shared = Assert.Single(result.Rows, r => r.Key == "BAD_ACTOR_0xC"); + Assert.Equal(ComparisonBanding.PresenceBoth, shared.Presence); + Assert.Equal(ComparisonBanding.StatusStable, shared.Status); // 100 → 110 is under the quarter + + /* PAGEIOLATCH_SH appeared at 0.20 (base 0.8): a new issue. LCK_M_IS appeared at a trace (base 0.02): stable, not counted. */ + var pageio = Assert.Single(result.Rows, r => r.Key == "PAGEIOLATCH_SH"); + Assert.Equal(ComparisonBanding.PresenceComparisonOnly, pageio.Presence); + Assert.Equal(ComparisonBanding.BandSourcePresence, pageio.BandSource); + Assert.Equal(ComparisonBanding.StatusWorse, pageio.Status); + Assert.Null(pageio.BaselineValue); + Assert.Null(pageio.ValueDelta); + + var trace = Assert.Single(result.Rows, r => r.Key == "LCK_M_IS"); + Assert.Equal(ComparisonBanding.StatusStable, trace.Status); + + Assert.Equal(1, result.NewIssues); + Assert.Equal(0, result.ResolvedIssues); + Assert.Equal(1, result.Worse); + } + + /// A disappearance at a scored level is a resolved issue; the baseline-only row reads better. + [Fact] + public void AScoredKeyThatDisappeared_IsAResolvedIssue() + { + var (baseline, comparison) = Scored([Wait("WRITELOG", 0.30), Cpu(40)], [Cpu(40)]); + var result = ComparisonBanding.Compare(baseline, comparison, NoDispersion, coverageCaveat: false); + + var writelog = Assert.Single(result.Rows, r => r.Key == "WRITELOG"); + Assert.Equal(ComparisonBanding.PresenceBaselineOnly, writelog.Presence); + Assert.Equal(ComparisonBanding.StatusBetter, writelog.Status); + Assert.Equal(1, result.ResolvedIssues); + Assert.Equal(0, result.NewIssues); + } + + /* ── coverage and emptiness ── */ + + [Fact] + public void TheCoverageCaveat_RidesOnEveryVerdictRow_AndFamily_AndTheSummary() + { + var (baseline, comparison) = Scored([Cpu(50), Wait("CXPACKET", 0.1)], [Cpu(74), Wait("CXPACKET", 0.1)]); + + var caveated = ComparisonBanding.Compare(baseline, comparison, NoDispersion, coverageCaveat: true); + Assert.All(caveated.Rows, r => Assert.True(r.CoverageCaveat)); + Assert.All(caveated.Families, f => Assert.True(f.CoverageCaveat)); + Assert.True(caveated.CoverageCaveat); + + var clean = ComparisonBanding.Compare(baseline, comparison, NoDispersion, coverageCaveat: false); + Assert.All(clean.Rows, r => Assert.False(r.CoverageCaveat)); + } + + [Fact] + public void NothingOnEitherSide_IsEmpty_ButChurnAloneIsNot() + { + Assert.True(ComparisonBanding.Compare([], [], NoDispersion, coverageCaveat: false).IsEmpty); + + var (baseline, comparison) = Scored([], [BadActor("0xB", avgCpuMs: 500)]); + var churnOnly = ComparisonBanding.Compare(baseline, comparison, NoDispersion, coverageCaveat: false); + Assert.False(churnOnly.IsEmpty); + Assert.Empty(churnOnly.Rows); + Assert.Single(churnOnly.Churn.Appeared); + } + + /// The rules the payload states name the constants they rest on, so a change to one moves the other. + [Fact] + public void TheStatedRules_NameTheConstants() + { + /* Parsed, not raw: System.Text.Json escapes ± and σ as \uXXXX in the serialized text. */ + using var doc = System.Text.Json.JsonDocument.Parse(System.Text.Json.JsonSerializer.Serialize(ComparisonBanding.BandRulesPayload)); + var baselineRule = doc.RootElement.GetProperty("baseline").GetString()!; + var absoluteRule = doc.RootElement.GetProperty("absolute").GetString()!; + var presenceRule = doc.RootElement.GetProperty("presence").GetString()!; + Assert.Contains("±1σ", baselineRule, StringComparison.Ordinal); + Assert.Contains("beyond_anomaly_cutoff", baselineRule, StringComparison.Ordinal); + Assert.Contains("25%", absoluteRule, StringComparison.Ordinal); + Assert.Contains("0.25", absoluteRule, StringComparison.Ordinal); + Assert.Contains("0.25", presenceRule, StringComparison.Ordinal); + Assert.Contains("plan_cache_churn", presenceRule, StringComparison.Ordinal); + Assert.Equal(1.0, ComparisonBanding.StableWithinRobustSigmas); + Assert.Equal(0.25, ComparisonBanding.MinimumRelativeMove); + Assert.Equal(0.25, ComparisonBanding.MinimumLadderPosition); + } + + /* ── helpers ── */ + + private static Fact Wait(string type, double fraction) => new() + { + Source = "waits", Key = type, Value = fraction, + Metadata = new Dictionary { ["wait_time_ms"] = fraction * 14_400_000, ["period_duration_ms"] = 14_400_000 } + }; + + private static Fact Cpu(double avgPercent) => new() { Source = "cpu", Key = "CPU_SQL_PERCENT", Value = avgPercent }; + + private static Fact Blocking(double eventsPerHour) => new() + { + Source = "blocking", Key = "BLOCKING_EVENTS", Value = eventsPerHour, + Metadata = new Dictionary { ["event_count"] = eventsPerHour * 4, ["period_hours"] = 4, ["observed_hours"] = 4 } + }; + + private static Fact Io(double avgReadMs) => new() { Source = "io", Key = "IO_READ_LATENCY_MS", Value = avgReadMs }; + + private static Fact Sessions(double total) => new() { Source = "sessions", Key = "SESSION_STATS", Value = total }; + + private static Fact BadActor(string hash, double avgCpuMs) => new() + { + Source = "bad_actor", Key = $"BAD_ACTOR_{hash}", Value = avgCpuMs, DatabaseName = "db", + Metadata = new Dictionary { ["execution_count"] = 5_000, ["avg_cpu_ms"] = avgCpuMs, ["avg_reads"] = 100 } + }; + + /// Scores both lists with the real scorer — the ladder positions are the scorer's, never hand-set. + private static (List, List) Scored(List baseline, List comparison) + { + var scorer = new FactScorer(); + scorer.ScoreAll(baseline); + scorer.ScoreAll(comparison); + return (baseline, comparison); + } + + private static ComparisonRow Compare(List baseline, List comparison, BaselineBucket bucket) + { + var metric = ComparisonBanding.BaselinedMetricFor(comparison[0].Key)!; + var dispersion = new Dictionary { [metric] = bucket }; + return Assert.Single(ComparisonBanding.Compare(baseline, comparison, dispersion, coverageCaveat: false).Rows); + } + + /// A trustworthy Full-tier CPU bucket: 20 samples over 5 distinct days (2x the tier's sample floor). + private static BaselineBucket CpuBucket(double median, double mad) => new() + { + HourOfDay = 9, DayOfWeek = 2, Tier = BaselineTier.Full, + Mean = median, StdDev = Math.Max(mad, 1), Median = median, Mad = mad, SampleCount = 20, DistinctDays = 5, + AbsStdDevFloor = BaselineMath.AbsStdDevFloorFor(MetricNames.Cpu) + }; + + private static BaselineBucket SessionsBucket(double median, double mad) => new() + { + HourOfDay = 9, DayOfWeek = 2, Tier = BaselineTier.Full, + Mean = median, StdDev = Math.Max(mad, 1), Median = median, Mad = mad, SampleCount = 20, DistinctDays = 5, + AbsStdDevFloor = BaselineMath.AbsStdDevFloorFor(MetricNames.SessionCount) + }; +} diff --git a/Lite.Tests/McpMissMessageParityPinTests.cs b/Lite.Tests/McpMissMessageParityPinTests.cs index b83afae06..b7ee844ef 100644 --- a/Lite.Tests/McpMissMessageParityPinTests.cs +++ b/Lite.Tests/McpMissMessageParityPinTests.cs @@ -108,6 +108,12 @@ the only shared miss sentence this list did not name. Darling's sentinel populat ". Its rates are per observed time, and its windowed facts are absent where nothing was observed.", "A side that was not fully observed cannot be read as the whole period: a wait that is absent because the collector was down is not a wait that resolved. Confirm coverage (get_collection_log, get_collection_health) before reading worse/better/resolved_issues as change.", + /* compare_analysis's verdict reading (#3538 A3). The band RULES live once, in the shared + ComparisonBanding, and cannot drift; what lives twice is the sentence each tool body puts on the + payload about what a verdict is, and the description that promises it. */ + "Each row is banded by how far its VALUE moved on this server's own scale (band_source says which rule; band_rules states them), not by the severity formula's slope. One window against one window cannot show that a change caused anything: a same-hour-yesterday comparison at N=1 vs N=1 is a difference, not an experiment. Count families, not rows, to count causes.", + "What \\\"worse\\\" does NOT mean: this is one window against one window — same-hour-yesterday at N=1 vs N=1 cannot show that a change CAUSED anything (DB time on an unchanged server routinely varies severalfold day to day), and a partly collected side flags every verdict with coverage_caveat.", + /* get_query_heatmap (#2484) — the three empty branches, one of which (a collected but IDLE window) no other read has. */ "so this is NOT a report of a quiet server — there is nothing to draw. query_stats is a PERIODIC table rather than an edge table: the collector writes rows every cycle for whatever is in the plan cache, so an empty history means nobody looked. Check get_collection_health for this server.", diff --git a/Lite/Analysis/AnalysisService.cs b/Lite/Analysis/AnalysisService.cs index 94c3bfaa1..c6e962371 100644 --- a/Lite/Analysis/AnalysisService.cs +++ b/Lite/Analysis/AnalysisService.cs @@ -5,6 +5,7 @@ using System.Threading.Tasks; using DuckDB.NET.Data; using PerformanceMonitor.Analysis; +using PerformanceMonitor.Analysis.Baselines; using PerformanceMonitor.Common; using PerformanceMonitorLite.Database; using PerformanceMonitorLite.Services; @@ -406,8 +407,20 @@ cancellation really produces — narrowed to the one shape that arises here. */ /// window's observed coverage (#3538 A2) so the caller can say when one side was only partly /// collected — the case the empty-window caveats never reached, where a half-collected window /// produces confident numbers with nothing to flag them. + /// + /// #3538 A3: also returns the stored per-server dispersion for the baselined metrics some + /// compared key is measured in (), keyed by metric + /// name, so compare_analysis can band a CPU or read-latency delta in the server's own robust + /// sigma instead of on a flat severity dead-band. The bucket is the comparison window's START hour + /// × day-of-week — the same coordinate the anomaly detectors read for a pass over that window + /// (AnomalyDetector passes context.TimeRangeStart), so the default same-hour-yesterday + /// call reuses the pass's cached buckets and an anchored one recomputes them the way an anchored + /// analyze_server does. The lookups are fenced separately from collection: a baseline read + /// that fails must not cost the caller the comparison it was only meant to refine, so it degrades + /// to an empty map and every key takes the absolute rule — the never-blind fallback the anomaly + /// gate follows. /// - public async Task<(List BaselineFacts, List ComparisonFacts, WindowCoverage? BaselineCoverage, WindowCoverage? ComparisonCoverage)> ComparePeriodsAsync( + public async Task<(List BaselineFacts, List ComparisonFacts, WindowCoverage? BaselineCoverage, WindowCoverage? ComparisonCoverage, IReadOnlyDictionary Dispersion)> ComparePeriodsAsync( int serverId, string serverName, DateTime baselineStart, DateTime baselineEnd, DateTime comparisonStart, DateTime comparisonEnd) @@ -436,13 +449,37 @@ cancellation really produces — narrowed to the one shape that arises here. */ _scorer.ScoreAll(baselineFacts); _scorer.ScoreAll(comparisonFacts); - return (baselineFacts, comparisonFacts, baselineContext.Coverage, comparisonContext.Coverage); + var dispersion = await LookUpDispersionAsync(serverId, serverName, baselineFacts, comparisonFacts, comparisonStart); + + return (baselineFacts, comparisonFacts, baselineContext.Coverage, comparisonContext.Coverage, dispersion); } catch (Exception ex) { AppLogger.Error("AnalysisService", $"Period comparison failed for {serverName}: {ex.Message}"); - return ([], [], null, null); + return ([], [], null, null, new Dictionary()); + } + } + + /// + /// The baseline buckets hands to the comparison, one per metric some + /// compared key is measured in. Its own try: see the summary above for why a failed baseline read + /// degrades to "no dispersion" rather than failing the comparison. + /// + private async Task> LookUpDispersionAsync( + int serverId, string serverName, List baselineFacts, List comparisonFacts, DateTime comparisonStart) + { + var dispersion = new Dictionary(StringComparer.Ordinal); + try + { + foreach (var metric in ComparisonBanding.DispersionMetricsFor(baselineFacts, comparisonFacts)) + dispersion[metric] = await _baselineProvider.GetBaselineAsync(serverId, metric, comparisonStart); + } + catch (Exception ex) + { + AppLogger.Warn("AnalysisService", $"Baseline dispersion lookup failed for {serverName}; compare_analysis bands every key by the absolute rule: {ex.Message}"); + dispersion.Clear(); } + return dispersion; } /// diff --git a/Lite/Mcp/McpAnalysisTools.cs b/Lite/Mcp/McpAnalysisTools.cs index 9dfd619b7..29fe2ae0a 100644 --- a/Lite/Mcp/McpAnalysisTools.cs +++ b/Lite/Mcp/McpAnalysisTools.cs @@ -304,7 +304,7 @@ after configuration alone knows audit_config still has it. */ } } - [McpServerTool(Name = "compare_analysis"), Description("Compares two time periods by running the inference engine's fact collection and scoring on each, then showing what changed. Use this to compare peak vs off-peak, before vs after a change, or yesterday vs today. Returns facts from both periods side-by-side with severity deltas. Note: for routine anomaly detection, use analyze_server instead — it automatically compares against 30-day time-bucketed baselines (hour-of-day x day-of-week). This tool is for explicit window-to-window comparisons.")] + [McpServerTool(Name = "compare_analysis"), Description("Compares two time periods by running the inference engine's fact collection and scoring on each, then showing what changed. Use this to compare peak vs off-peak, yesterday vs today, or the windows around a change. Returns facts from both periods side-by-side, each banded worse / better / stable by how far the VALUE moved on the server's own scale, not by the severity formula's slope: a key with a stored per-server baseline (CPU %, read latency, connections) is banded in that baseline's robust sigma for the comparison hour (delta_sigma, band_source \"baseline\"); every other key changes status only when the value moved at least a quarter of the larger side AND registers at least a quarter of the way up its own severity ladder (band_source \"absolute\"); the rules are stated in band_rules. Rows are grouped into physical-cause families (one I/O stall is one family row, not four worse keys), and BAD_ACTOR_ appearances are reported as plan_cache_churn rather than as new or resolved issues. What \"worse\" does NOT mean: this is one window against one window — same-hour-yesterday at N=1 vs N=1 cannot show that a change CAUSED anything (DB time on an unchanged server routinely varies severalfold day to day), and a partly collected side flags every verdict with coverage_caveat. Note: for routine anomaly detection, use analyze_server instead — it automatically compares against 30-day time-bucketed baselines (hour-of-day x day-of-week). This tool is for explicit window-to-window comparisons.")] public static async Task CompareAnalysis( AnalysisService analysisService, ServerManager serverManager, @@ -334,7 +334,7 @@ silently change what the two windows are relative to each other. */ var baselineEnd = windowEnd.AddHours(-baseline_hours_back + hours_back); var baselineStart = windowEnd.AddHours(-baseline_hours_back); - var (baselineFacts, comparisonFacts, baselineCoverage, comparisonCoverage) = await analysisService.ComparePeriodsAsync( + var (baselineFacts, comparisonFacts, baselineCoverage, comparisonCoverage, dispersion) = await analysisService.ComparePeriodsAsync( resolved.ServerId, resolved.ServerName, baselineStart, baselineEnd, comparisonStart, comparisonEnd); @@ -345,41 +345,35 @@ silently change what the two windows are relative to each other. */ and pad fact rows with a key no advice speaks to. */ var baselineServerFacts = baselineFacts.Where(f => f.Source != WindowCoverage.FactSource).ToList(); var comparisonServerFacts = comparisonFacts.Where(f => f.Source != WindowCoverage.FactSource).ToList(); - var baselineByKey = baselineServerFacts.ToFactLookup(); - var comparisonByKey = comparisonServerFacts.ToFactLookup(); - var allKeys = baselineByKey.Keys.Union(comparisonByKey.Keys).ToHashSet(); - var comparisons = allKeys - .Select(key => - { - var baseline = baselineByKey.GetValueOrDefault(key); - var comparison = comparisonByKey.GetValueOrDefault(key); - var severityDelta = (comparison?.Severity ?? 0) - (baseline?.Severity ?? 0); + /* + #3538 A3: the coverage caveat is COMPOSED into the verdicts, not restated. The prose below + says which side was partly collected; every verdict row and family row carries + coverage_caveat: true when either side was, so a reader of one row cannot take "worse" at + face value without being told the side it rests on speaks for a fraction of its window. + */ + var baselinePartial = baselineCoverage is not null && (baselineCoverage.IsPartial || !baselineCoverage.IsObserved); + var comparisonPartial = comparisonCoverage is not null && (comparisonCoverage.IsPartial || !comparisonCoverage.IsObserved); - return new - { - key, - source = baseline?.Source ?? comparison?.Source ?? "unknown", - baseline_value = baseline != null ? Math.Round(baseline.Value, 6) : (double?)null, - comparison_value = comparison != null ? Math.Round(comparison.Value, 6) : (double?)null, - baseline_severity = baseline != null ? Math.Round(baseline.Severity, 4) : (double?)null, - comparison_severity = comparison != null ? Math.Round(comparison.Severity, 4) : (double?)null, - severity_delta = Math.Round(severityDelta, 4), - status = severityDelta > 0.1 ? "worse" : severityDelta < -0.1 ? "better" : "stable" - }; - }) - .OrderByDescending(c => Math.Abs(c.severity_delta)) - .ToList(); + /* + Every verdict — sigma-banded where a per-server baseline exists, ladder-banded where it + does not, plan-cache churn kept out of the issue counters, one family row per physical + cause — is decided in the shared ComparisonBanding, so this SKU and its twin cannot band + the same two windows differently. The tool only serializes. + */ + var comparison = ComparisonBanding.Compare( + baselineServerFacts, comparisonServerFacts, dispersion, + coverageCaveat: baselinePartial || comparisonPartial); - if (comparisons.Count == 0) + if (comparison.IsEmpty) { /* Neither window produced a single fact, and the old payload said that with all-zero counters and facts: [] -- which reads as "nothing changed" when it actually means "there was nothing to compare". Those are opposite conclusions about the same server. - No probe is needed to tell them apart: comparisons is the UNION of both windows' keys, - so zero entries is exactly "both fact sets were empty" and the fact_counts already in - hand are the whole answer. + No probe is needed to tell them apart: the comparison is over the UNION of both windows' + keys, so an empty one is exactly "both fact sets were empty" and the fact_counts already + in hand are the whole answer. */ return McpHelpers.Status( "unavailable", @@ -426,10 +420,10 @@ collection log without saying what they will find there. Both sides can earn one : null; var coverageCaveats = new List(2); - if (baselineCoverage is not null && (baselineCoverage.IsPartial || !baselineCoverage.IsObserved)) - coverageCaveats.Add($"The BASELINE window was only partly collected: {baselineCoverage.Describe()}. Its rates are per observed time, and its windowed facts are absent where nothing was observed."); - if (comparisonCoverage is not null && (comparisonCoverage.IsPartial || !comparisonCoverage.IsObserved)) - coverageCaveats.Add($"The COMPARISON window was only partly collected: {comparisonCoverage.Describe()}. Its rates are per observed time, and its windowed facts are absent where nothing was observed."); + if (baselinePartial) + coverageCaveats.Add($"The BASELINE window was only partly collected: {baselineCoverage!.Describe()}. Its rates are per observed time, and its windowed facts are absent where nothing was observed."); + if (comparisonPartial) + coverageCaveats.Add($"The COMPARISON window was only partly collected: {comparisonCoverage!.Describe()}. Its rates are per observed time, and its windowed facts are absent where nothing was observed."); if (coverageCaveats.Count > 0) coverageCaveats.Add("A side that was not fully observed cannot be read as the whole period: a wait that is absent because the collector was down is not a wait that resolved. Confirm coverage (get_collection_log, get_collection_health) before reading worse/better/resolved_issues as change."); @@ -443,6 +437,10 @@ collection log without saying what they will find there. Both sides can earn one /* Null when both windows produced facts at full coverage — the ordinary case, where nothing needs saying. */ caveat, + /* #3538 A3: what a verdict can and cannot carry, stated on every payload because the tool's + description is not in front of the reader when the numbers are. */ + reading = "Each row is banded by how far its VALUE moved on this server's own scale (band_source says which rule; band_rules states them), not by the severity formula's slope. One window against one window cannot show that a change caused anything: a same-hour-yesterday comparison at N=1 vs N=1 is a difference, not an experiment. Count families, not rows, to count causes.", + band_rules = ComparisonBanding.BandRulesPayload, baseline = new { start = baselineStart.ToString("o"), @@ -457,15 +455,10 @@ nothing needs saying. */ fact_count = comparisonServerFacts.Count, coverage = comparisonCoverage?.ToPayload() }, - summary = new - { - worse = comparisons.Count(c => c.status == "worse"), - better = comparisons.Count(c => c.status == "better"), - stable = comparisons.Count(c => c.status == "stable"), - new_issues = comparisons.Count(c => c.baseline_severity == null && c.comparison_severity > 0), - resolved_issues = comparisons.Count(c => c.baseline_severity > 0 && c.comparison_severity == null) - }, - facts = comparisons + summary = comparison.SummaryPayload(), + families = comparison.Families.Select(f => f.ToPayload()).ToList(), + plan_cache_churn = comparison.Churn.ToPayload(), + facts = comparison.Rows.Select(r => r.ToPayload()).ToList() }, McpHelpers.JsonOptions); } catch (Exception ex) diff --git a/Lite/Mcp/McpInstructions.cs b/Lite/Mcp/McpInstructions.cs index 5071344a6..c8e58e723 100644 --- a/Lite/Mcp/McpInstructions.cs +++ b/Lite/Mcp/McpInstructions.cs @@ -220,7 +220,7 @@ You are connected to a SQL Server performance monitoring tool via Performance Mo |------|---------|----------------| | `analyze_server` | Runs the inference engine: scores facts, traverses relationship graph, returns evidence-backed findings with severity and recommended next tools. Each finding's `confidence` is an EVIDENCE score (0.20 for the symptom alone, more as the root fact's amplifier checks match and the chain deepens — a lone uncorroborated symptom is 0.20, never 1.0) and `confidence_basis` says in words what it rests on; rank by `severity` for impact and read `confidence` as how much of the engine's own corroboration showed up. A remediable finding also carries `remediation_command` — the full copy-paste T-SQL remediation (identical to the viewer card), with a two-sided risk-disclosure header on destructive changes; advisory only, never executed. Force-plan findings additionally carry `structured_remediation`: the verdict as machine-readable fields (eligible + named blockers), evidence, and split force/unforce/verify SQL. With `as_of` it analyzes a PAST window — anomaly baseline included — and is EXPLORATORY: the findings come back in full but are NOT persisted, which `persisted` / `persistence_note` state on every result | `server_name`, `hours_back` (default 4), `as_of` | | `get_analysis_facts` | Exposes raw scored facts from the collect+score pipeline — every observation the engine sees with base severity, amplifiers, and metadata (an ANOMALY_* fact's `baseline_confidence` is the baseline's trustworthiness, not a finding's `confidence`) | `server_name`, `hours_back` (default 4), `source` (filter), `min_severity`, `as_of` | - | `compare_analysis` | Compares two time periods (e.g., peak vs off-peak, before vs after a change) showing severity deltas for each fact. When NEITHER window produced facts the result is `unavailable` rather than an all-zero comparison, because "nothing to compare" is not "nothing changed"; when only ONE window is empty the payload carries a `caveat` saying so, since every fact then counts as new or resolved by default. `baseline_hours_back` is measured from the comparison window's END, so `as_of` moves BOTH windows together | `server_name`, `hours_back` (default 4), `baseline_hours_back` (default 28), `as_of` | + | `compare_analysis` | Compares two time periods (e.g., peak vs off-peak, yesterday vs today, the windows around a change), banding each fact worse / better / stable by how far its VALUE moved on the server's own scale — in the stored per-server baseline's robust sigma where one exists (`delta_sigma`, `band_source` `baseline`), otherwise only when the value moved at least a quarter of the larger side AND registers a quarter of the way up its own severity ladder (`band_source` `absolute`); `band_rules` states the rules on every payload. Rows are grouped into physical-cause `families` (one I/O stall is one family row), and `BAD_ACTOR_` appearances are `plan_cache_churn`, not new or resolved issues. A verdict is a DIFFERENCE, not an experiment: same-hour-yesterday at N=1 vs N=1 cannot show that a change caused anything. When NEITHER window produced facts the result is `unavailable` rather than an all-zero comparison, because "nothing to compare" is not "nothing changed"; when only ONE window is empty the payload carries a `caveat` saying so; a partly collected side flags every verdict row with `coverage_caveat`. `baseline_hours_back` is measured from the comparison window's END, so `as_of` moves BOTH windows together | `server_name`, `hours_back` (default 4), `baseline_hours_back` (default 28), `as_of` | | `audit_config` | Edition-aware configuration audit: evaluates CTFP, MAXDOP, max memory, and max worker threads against best practices | `server_name` | | `get_analysis_findings` | Retrieves persisted findings from previous analysis runs (each with `confidence_basis`: rows persisted before `confidence` measured corroboration are labelled `path-shape (pre-#3538)` — under that formula a lone symptom read 1.0, so do not read those as corroborated), deduplicated to one entry per diagnostic chain (`story_path_hash` + `incident_id`): the latest occurrence plus `occurrences`/`first_seen`/`last_seen`/`peak_severity` spanning the window; each remediable finding carries `remediation_command` — the full copy-paste T-SQL remediation (identical to the viewer card), rendered from the persisted action, advisory only and never executed; force-plan findings additionally carry `structured_remediation` (verdict + evidence + split artifacts, machine-readable). Its window is on ANALYSIS TIME, so `as_of` asks what analysis was saying then rather than re-analyzing that window now | `server_name`, `hours_back` (default 24), `as_of` | | `mute_analysis_finding` | Mutes a finding pattern by story_path_hash so it won't appear in future runs. Reports what the write did: `registered`, and `matched_now` — how many stored findings in scope carry the hash (status `muted_unmatched` when 0: the mute is kept, but check the hash) | `story_path_hash` (required), `server_name`, `reason` | diff --git a/PerformanceMonitor.Analysis/ComparisonBanding.cs b/PerformanceMonitor.Analysis/ComparisonBanding.cs new file mode 100644 index 000000000..e46dbc151 --- /dev/null +++ b/PerformanceMonitor.Analysis/ComparisonBanding.cs @@ -0,0 +1,584 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Linq; +using PerformanceMonitor.Analysis.Baselines; + +namespace PerformanceMonitor.Analysis; + +/// +/// The arithmetic behind compare_analysis's verdicts (#3538 A3), shared by Lite and Darling so the +/// two SKUs cannot band the same pair of windows differently. Pure over two scored fact lists and the +/// baseline buckets the caller looked up; the tools only serialize what comes back. +/// +/// The lie this replaces. The tool compared ONE window against ONE window and banded every key +/// by its severity delta on a flat ±0.1 dead-band: worse above +0.1, better below −0.1, +/// stable between. Severity is a threshold-formula artifact, not a measurement — a saturating +/// ladder (critical == null in FactScorer.ApplyThresholdFormula) reads a doubling from 30% to +/// 60% of observed time as +0.0 ("stable") because both sides sit at 1.0, while a trace of a wait whose +/// concerning bar is 1% of the period reads a few seconds an hour as +0.125 ("worse"). None of it used the +/// per-server dispersion the baselines already store, one physical cause (an I/O stall) surfaced as four +/// to six "worse" keys, and BAD_ACTOR_<hash> keys churned into new_issues / +/// resolved_issues whenever the plan cache evicted a plan. The OtterTune field study measured DB +/// time varying 4× on an UNCHANGED configuration; a same-hour-yesterday comparison at N=1 vs N=1 has to +/// be read against that kind of noise, and this class is where the reading is decided. +/// +/// Three bands, one rule per band, every rule stated in the payload. +/// +/// Baseline-banded (): a key whose value is measured in the +/// same unit as one of the stored per-(server, metric, hour × day-of-week) baselines +/// () expresses its value delta in that bucket's robust sigma +/// (, MAD-based with the model's own floors) and is +/// stable inside ±. Used only when the bucket is +/// — the never-blind rule the anomaly gate follows: an +/// untrustworthy baseline routes to the absolute rule rather than to silence or to a giant sigma. +/// Absolute-banded (): every other key present on both sides +/// changes status only when BOTH a relative move of at least of the +/// larger side AND a position of at least on the key's own Layer-1 +/// severity ladder (the larger side's ) occurred. The ladder position is +/// the scorer's — read off the base severity it already computed — so every concerning bar in +/// FactScorer is reused without a single number copied here, including the ones that are not +/// in a table (blocking per hour, I/O latency, CPU %, the step ladders). +/// Present on one side only (): a new or resolved issue only +/// when the present side clears ; a noise-level appearance is +/// stable. BAD_ACTOR_* keys are the exception and never take this path — see +/// . +/// +/// +/// Families. Every row carries a physical-cause family () and the result +/// carries one family row per cause with its worst member, so the reader counts causes, not symptoms. +/// The family map mirrors the collector's wait grouping (, +/// applied first) and the symptom families AnomalyIncidentReconciler folds anomalies into, extended +/// along the RelationshipGraph edges that tie the regular keys of one cause together +/// (PAGEIOLATCH_* ↔ IO_READ_LATENCY_MS, WRITELOG ↔ IO_WRITE_LATENCY_MS ↔ HADR_SYNC_COMMIT, the memory-grant +/// chain, the lock/blocking chain). The whole graph is NOT used as the family relation: union-find over +/// its THREADPOOL bridge merges everything into one component, which is why the reconciler rejected it +/// too. +/// +/// What it does not do. It does not make the comparison a statistical test — one window against +/// one window cannot show that a change caused anything, and the tool's description says so. It does not +/// band wait fractions by sigma: the wait baselines are ALL-TYPES totals (ms per collection, ms per +/// second) and no per-type dispersion is stored, so per-type waits take the absolute rule. It does not +/// change what either window's facts ARE — the collectors' observed-time divisor (#3538 A2) is upstream of +/// it, and it composes with that lane's coverage caveat by flagging every verdict row when either side +/// was partly observed rather than by restating the caveat. +/// +public static class ComparisonBanding +{ + /// + /// The baseline band: a value delta inside ±1 robust sigma of the comparison hour's bucket is + /// stable. One sigma is the UNIT the bucket's dispersion is stored in, not a tuned cutoff — the + /// question this tool answers is "did it move more than this server routinely moves at this hour", + /// and the server's routine movement IS one sigma of its own same-hour history. The anomaly + /// detectors' 3.5σ / 5.0σ cutoffs () + /// answer a different question — "is this value abnormal against the median" — and are reported + /// beside the band as beyond_anomaly_cutoff rather than used as it. The per-sample sigma is an + /// UPPER bound on the dispersion of a window average (averaging removes within-hour jitter and keeps + /// day-to-day level differences), so the band is conservative for the averaged keys (CPU %, read + /// latency) and exact for the single-sample one (session count). + /// + public const double StableWithinRobustSigmas = 1.0; + + /// + /// The absolute band's relative arm: the value must have moved by at least a quarter of the larger + /// side. A judgment about materiality set by the #3538 engine review rather than a fleet measurement: + /// it refuses the 1%-of-value wobbles that the severity dead-band admitted whenever a ladder's slope + /// was steep, and it admits a doubling whatever the ladder did with it. The read that would revise it + /// is the fleet distribution of same-hour-yesterday relative deltas per fact key; until that exists + /// "a quarter" is stated, not hidden. + /// + public const double MinimumRelativeMove = 0.25; + + /// + /// The absolute band's ladder arm: the larger side's Layer-1 base severity must be at least a quarter + /// — a quarter of the way to the concerning bar on a saturating ladder (value / concerning), + /// half of the way on a ramped one (0.5 · value / concerning below concerning), and the second + /// tier or above on the step ladders. A move between two values the scorer itself grades as + /// negligible is not a verdict. The same "quarter" as on purpose: + /// one number to explain. Together the two arms bound the absolute move from below at a sixteenth of + /// a saturating ladder's concerning bar and an eighth of a ramped one's, which is what keeps a + /// trace-to-trace doubling of a wait whose bar is 1% of the period from reading as change. + /// + public const double MinimumLadderPosition = 0.25; + + public const string StatusWorse = "worse"; + public const string StatusBetter = "better"; + public const string StatusStable = "stable"; + + public const string BandSourceBaseline = "baseline"; + public const string BandSourceAbsolute = "absolute"; + public const string BandSourcePresence = "presence"; + + public const string PresenceBoth = "both"; + public const string PresenceComparisonOnly = "comparison_only"; + public const string PresenceBaselineOnly = "baseline_only"; + + /// The BAD_ACTOR_<query_hash> family — per-query identity, reported as churn. + public const string PlanCacheIdentityPrefix = "BAD_ACTOR_"; + + /// + /// The one sentence per band that the payload carries so a reader never has to infer the rule from + /// the numbers. Written once, here, so both SKUs say it in the same words. + /// + public static object BandRulesPayload => new + { + baseline = $"delta_sigma is the value delta in robust-sigma units (MAD-based) of this server's own hour-of-day x day-of-week baseline for the comparison window's hour; stable within ±{StableWithinRobustSigmas:0.#}σ, worse/better beyond. Used only when that baseline is trustworthy (baseline_confidence > 0); beyond_anomaly_cutoff says whether the move also clears the anomaly detector's own cutoff for the metric.", + absolute = $"no per-key dispersion is stored, so a status changes only when BOTH the value moved at least {MinimumRelativeMove:P0} of the larger side (relative_move) AND the larger side sits at least {MinimumLadderPosition:0.##} up the key's own base-severity ladder (ladder_position) — a move between two values the scorer grades as negligible is not a verdict.", + presence = $"a key present in one window only is a new or resolved issue only when its base severity reaches {MinimumLadderPosition:0.##}; a noise-level appearance or disappearance is stable. BAD_ACTOR_* keys never take this path: their appearance is plan-cache identity churn, reported under plan_cache_churn and excluded from new_issues / resolved_issues." + }; + + /// + /// The stored baseline metric a fact key's is measured in, or null when the + /// key has no baseline in the same unit. Only same-unit pairs are mapped: CPU_SQL_PERCENT is the + /// window's average of the same sqlserver_cpu_utilization samples the CPU baseline is built from + /// and CPU_SPIKE is their maximum (a single sample, so per-sample sigma is exactly its scale); + /// IO_READ_LATENCY_MS is stall ÷ reads in ms, the I/O baseline's own ratio; SESSION_STATS + /// is the latest total-connections reading, the session baseline's own sample. Wait fractions are + /// deliberately unmapped (the wait baselines are all-types totals, not per type), as are blocking and + /// deadlock rates (their baselines are events per day with no robust statistics) and + /// IO_WRITE_LATENCY_MS (no write-latency baseline exists). + /// + public static string? BaselinedMetricFor(string key) => key switch + { + "CPU_SQL_PERCENT" => MetricNames.Cpu, + "CPU_SPIKE" => MetricNames.Cpu, + "IO_READ_LATENCY_MS" => MetricNames.IoLatency, + "SESSION_STATS" => MetricNames.SessionCount, + _ => null + }; + + /// + /// The distinct baseline metrics the caller must look up for this pair of fact lists — only the + /// metrics some present key is measured in, so a comparison with no CPU fact costs no CPU baseline + /// read. + /// + public static IReadOnlyList DispersionMetricsFor(IEnumerable baselineFacts, IEnumerable comparisonFacts) => + baselineFacts.Concat(comparisonFacts) + .Select(f => BaselinedMetricFor(f.Key)) + .Where(m => m is not null) + .Select(m => m!) + .Distinct(StringComparer.Ordinal) + .OrderBy(m => m, StringComparer.Ordinal) + .ToList(); + + /// + /// BAD_ACTOR_<query_hash>: the key IS a plan-cache identity. Its appearance in one window and + /// absence in the other says the cache held a different plan for the top-5 cut, not that a problem + /// began or ended — the same statement under a recompiled hash is a "new issue" and a "resolved + /// issue" at once under key-set arithmetic. Reported as plan_cache_churn instead. + /// + public static bool IsPlanCacheIdentityKey(string key) => + key.StartsWith(PlanCacheIdentityPrefix, StringComparison.OrdinalIgnoreCase); + + /// + /// The one key whose value runs the other way: DISK_SPACE is the FREE fraction, so a lower value + /// is worse (FactScorer.ScoreDiskFact inverts it). Direction is taken from the base-severity + /// delta first, which already carries the inversion; this set only matters when both sides sit on the + /// same saturated rung and the value has to decide. + /// + private static readonly HashSet HigherIsBetterKeys = new(StringComparer.Ordinal) { "DISK_SPACE" }; + + /// + /// The physical-cause family a fact key belongs to. Names are the RelationshipGraph edge + /// categories where one exists, so the vocabulary is the engine's own. The collector's wait grouping + /// is applied first (every CX* is already CXPACKET, every general lock mode already LCK by the time a + /// fact reaches this tool; calling keeps that true for a + /// hand-built fact too). A key with no family of its own is its own family, so every row has one and a + /// singleton family is simply a cause with one symptom. + /// + public static string FamilyFor(string key) + { + if (IsPlanCacheIdentityKey(key)) + return "bad_actor"; + + var grouped = FactCollectorHelpers.WaitFamilyKey(key); + return grouped switch + { + // AnomalyIncidentReconciler: ANOMALY_CPU_SPIKE → CPU_SQL_PERCENT | CPU_SPIKE. Graph cpu_pressure: + // CPU_SQL_PERCENT ↔ SOS_SCHEDULER_YIELD; RUNNABLE_TASKS is the scheduler-queue reading of the same. + "CPU_SQL_PERCENT" or "CPU_SPIKE" or "SOS_SCHEDULER_YIELD" or "RUNNABLE_TASKS" => "cpu_pressure", + // Graph parallelism / query_performance: QUERY_HIGH_DOP → CXPACKET. + "CXPACKET" or "QUERY_HIGH_DOP" => "parallelism", + // Reconciler: ANOMALY_READ_LATENCY → IO_READ_LATENCY_MS. Graph io_pressure / memory_pressure: + // IO_READ_LATENCY_MS ↔ PAGEIOLATCH_SH, PAGEIOLATCH_EX → IO_READ_LATENCY_MS. One stall, one row. + "IO_READ_LATENCY_MS" or "PAGEIOLATCH_SH" or "PAGEIOLATCH_EX" => "io_pressure", + // Reconciler: ANOMALY_WRITE_LATENCY → IO_WRITE_LATENCY_MS. Graph log_io: WRITELOG ↔ + // IO_WRITE_LATENCY_MS, HADR_SYNC_COMMIT ↔ WRITELOG. + "IO_WRITE_LATENCY_MS" or "WRITELOG" or "HADR_SYNC_COMMIT" => "log_io", + // Reconciler: ANOMALY_MEMORY_PRESSURE → RESOURCE_SEMAPHORE. Graph memory_grants: + // RESOURCE_SEMAPHORE ↔ MEMORY_GRANT_PENDING → QUERY_SPILLS; RS_QUERY_COMPILE is the compile gateway. + "RESOURCE_SEMAPHORE" or "RESOURCE_SEMAPHORE_QUERY_COMPILE" or "MEMORY_GRANT_PENDING" or "QUERY_SPILLS" => "memory_grants", + // Reconciler: ANOMALY_BLOCKING_SPIKE → BLOCKING_EVENTS. Graph lock_contention / blocking: + // LCK ↔ BLOCKING_EVENTS ↔ BLOCKING_CHAIN; the ungrouped lock modes (S/IS, range, schema) are the + // same physical queue with a different reason. DEADLOCKS stays its own family, as the + // reconciler keeps ANOMALY_DEADLOCK_SPIKE apart from the blocking family. + "LCK" or "LCK_M_S" or "LCK_M_IS" or "SCH_M" or "BLOCKING_EVENTS" or "BLOCKING_CHAIN" => "lock_contention", + _ when grouped.StartsWith("LCK_M_RS_", StringComparison.Ordinal) + || grouped.StartsWith("LCK_M_RIn_", StringComparison.Ordinal) + || grouped.StartsWith("LCK_M_RX_", StringComparison.Ordinal) => "lock_contention", + "DEADLOCKS" => "deadlocking", + // Graph latch_contention: LATCH_EX → TEMPDB_USAGE / CXPACKET; PAGELATCH_UP is the tempdb + // allocation latch. TEMPDB_USAGE itself is a space reading and stays its own family. + "LATCH_EX" or "LATCH_SH" or "PAGELATCH_UP" => "latch_contention", + _ => grouped + }; + } + + /// + /// Bands every key of the two scored fact lists. is keyed by + /// baseline metric name (); a metric absent from it, or present with an + /// untrustworthy bucket, sends its keys to the absolute rule. is true + /// when either window was partly observed or unobserved (#3538 A2) and flags every verdict row. + /// + public static ComparisonResult Compare( + IReadOnlyList baselineFacts, + IReadOnlyList comparisonFacts, + IReadOnlyDictionary dispersionByMetric, + bool coverageCaveat) + { + var baselineByKey = baselineFacts.ToFactLookup(); + var comparisonByKey = comparisonFacts.ToFactLookup(); + var allKeys = baselineByKey.Keys.Union(comparisonByKey.Keys, StringComparer.Ordinal) + .OrderBy(k => k, StringComparer.Ordinal) + .ToList(); + + var rows = new List(allKeys.Count); + var appeared = new List(); + var disappeared = new List(); + + foreach (var key in allKeys) + { + var baseline = baselineByKey.GetValueOrDefault(key); + var comparison = comparisonByKey.GetValueOrDefault(key); + + if (baseline is null || comparison is null) + { + var present = (baseline ?? comparison)!; + if (IsPlanCacheIdentityKey(key)) + { + var entry = new ChurnEntry(key, present.DatabaseName, Math.Round(present.Value, 6), Math.Round(present.Severity, 4)); + (comparison is not null ? appeared : disappeared).Add(entry); + continue; + } + + rows.Add(BandOneSided(key, baseline, comparison, coverageCaveat)); + continue; + } + + var metric = BaselinedMetricFor(key); + var bucket = metric is not null ? dispersionByMetric.GetValueOrDefault(metric) : null; + rows.Add(bucket is { IsTrustworthy: true } && bucket.EffectiveRobustSigma > 0 + ? BandBySigma(key, baseline, comparison, metric!, bucket, coverageCaveat) + : BandByLadder(key, baseline, comparison, coverageCaveat)); + } + + /* Worse rows first, then better, then stable; the larger relative move first within each + group, then the key — a scale-free order in which a regression always outranks an + improvement. Direction before magnitude matters for the family rollup below: a family whose + BLOCKING_EVENTS fell 60% while its LCK_M_S rose 30% has a real regression in it, and ordering by + magnitude alone would have made the improvement its worst member and dropped the family from + families_worse. The old payload ordered by |severity_delta|, which put a saturated ladder's + doubling last and a trace's formula slope first. */ + rows = rows + .OrderBy(r => StatusRank(r.Status)) + .ThenByDescending(r => r.RelativeMove ?? 0) + .ThenBy(r => r.Key, StringComparer.Ordinal) + .ToList(); + + var families = rows + .GroupBy(r => r.Family, StringComparer.Ordinal) + .Select(g => + { + var members = g.ToList(); // in verdict order (worse > better > stable, then move), so First() is the worst member + var worst = members[0]; + return new ComparisonFamily( + g.Key, + worst.Status, + worst.Key, + members.Select(m => m.Key).ToList(), + members.Count(m => m.Status == StatusWorse), + members.Count(m => m.Status == StatusBetter), + members.Count(m => m.Status == StatusStable), + coverageCaveat); + }) + .OrderBy(f => StatusRank(f.Status)) + .ThenByDescending(f => rows.First(r => r.Key == f.WorstKey).RelativeMove ?? 0) + .ThenBy(f => f.Family, StringComparer.Ordinal) + .ToList(); + + var presentInBoth = allKeys.Count(k => IsPlanCacheIdentityKey(k) && baselineByKey.ContainsKey(k) && comparisonByKey.ContainsKey(k)); + + return new ComparisonResult( + rows, + families, + new PlanCacheChurn(appeared, disappeared, presentInBoth), + coverageCaveat); + } + + private static ComparisonRow BandBySigma(string key, Fact baseline, Fact comparison, string metric, BaselineBucket bucket, bool coverageCaveat) + { + var valueDelta = comparison.Value - baseline.Value; + var sigma = bucket.EffectiveRobustSigma; + var rawDeltaSigma = valueDelta / sigma; + /* Display-capped like the detectors' deviation_sigma (#1486): the band is decided on the raw + value, the payload never renders a collapsed-variance thousand-sigma. */ + var deltaSigma = Math.Clamp(rawDeltaSigma, -AnomalyThresholds.SigmaDisplayCap, AnomalyThresholds.SigmaDisplayCap); + var moved = Math.Abs(rawDeltaSigma) > StableWithinRobustSigmas; + /* Every baselined key is higher-is-worse (CPU %, read latency, connections). */ + var status = !moved ? StatusStable : valueDelta > 0 ? StatusWorse : StatusBetter; + + return new ComparisonRow + { + Key = key, + Source = baseline.Source, + Family = FamilyFor(key), + Presence = PresenceBoth, + BaselineValue = Math.Round(baseline.Value, 6), + ComparisonValue = Math.Round(comparison.Value, 6), + BaselineSeverity = Math.Round(baseline.Severity, 4), + ComparisonSeverity = Math.Round(comparison.Severity, 4), + SeverityDelta = Math.Round(comparison.Severity - baseline.Severity, 4), + ValueDelta = Math.Round(valueDelta, 6), + RelativeMove = Math.Round(RelativeMove(baseline.Value, comparison.Value), 4), + LadderPosition = Math.Round(Math.Max(baseline.BaseSeverity, comparison.BaseSeverity), 4), + Status = status, + BandSource = BandSourceBaseline, + DeltaSigma = Math.Round(deltaSigma, 2), + BaselineSigma = Math.Round(sigma, 4), + BaselineMedian = Math.Round(bucket.Median, 4), + BaselineConfidence = Math.Round(bucket.Confidence, 2), + BaselineTier = bucket.Tier.ToString(), + BaselineMetric = metric, + BeyondAnomalyCutoff = Math.Abs(rawDeltaSigma) >= AnomalyThresholds.ModifiedZThresholdFor(metric), + CoverageCaveat = coverageCaveat + }; + } + + private static ComparisonRow BandByLadder(string key, Fact baseline, Fact comparison, bool coverageCaveat) + { + var valueDelta = comparison.Value - baseline.Value; + var relativeMove = RelativeMove(baseline.Value, comparison.Value); + var ladderPosition = Math.Max(baseline.BaseSeverity, comparison.BaseSeverity); + var moved = relativeMove >= MinimumRelativeMove && ladderPosition >= MinimumLadderPosition; + + return new ComparisonRow + { + Key = key, + Source = baseline.Source, + Family = FamilyFor(key), + Presence = PresenceBoth, + BaselineValue = Math.Round(baseline.Value, 6), + ComparisonValue = Math.Round(comparison.Value, 6), + BaselineSeverity = Math.Round(baseline.Severity, 4), + ComparisonSeverity = Math.Round(comparison.Severity, 4), + SeverityDelta = Math.Round(comparison.Severity - baseline.Severity, 4), + ValueDelta = Math.Round(valueDelta, 6), + RelativeMove = Math.Round(relativeMove, 4), + LadderPosition = Math.Round(ladderPosition, 4), + Status = !moved ? StatusStable : Direction(key, baseline, comparison), + BandSource = BandSourceAbsolute, + BaselineMetric = BaselinedMetricFor(key), + CoverageCaveat = coverageCaveat + }; + } + + private static ComparisonRow BandOneSided(string key, Fact? baseline, Fact? comparison, bool coverageCaveat) + { + var present = (baseline ?? comparison)!; + var isNew = comparison is not null; + var registers = present.BaseSeverity >= MinimumLadderPosition; + + return new ComparisonRow + { + Key = key, + Source = present.Source, + Family = FamilyFor(key), + Presence = isNew ? PresenceComparisonOnly : PresenceBaselineOnly, + BaselineValue = baseline is null ? null : Math.Round(baseline.Value, 6), + ComparisonValue = comparison is null ? null : Math.Round(comparison.Value, 6), + BaselineSeverity = baseline is null ? null : Math.Round(baseline.Severity, 4), + ComparisonSeverity = comparison is null ? null : Math.Round(comparison.Severity, 4), + SeverityDelta = Math.Round((comparison?.Severity ?? 0) - (baseline?.Severity ?? 0), 4), + ValueDelta = null, + /* A side with nothing to compare against is a whole move for ordering purposes. */ + RelativeMove = 1.0, + LadderPosition = Math.Round(present.BaseSeverity, 4), + Status = !registers ? StatusStable : isNew ? StatusWorse : StatusBetter, + BandSource = BandSourcePresence, + BaselineMetric = BaselinedMetricFor(key), + CoverageCaveat = coverageCaveat + }; + } + + /// The verdict order: a regression outranks an improvement outranks no change. + private static int StatusRank(string status) => status switch + { + StatusWorse => 0, + StatusBetter => 1, + _ => 2 + }; + + /// |b − a| over the larger magnitude; 0 when both are 0. + private static double RelativeMove(double a, double b) + { + var larger = Math.Max(Math.Abs(a), Math.Abs(b)); + return larger > 0 ? Math.Abs(b - a) / larger : 0.0; + } + + /// + /// Which way a two-sided absolute-banded move went. The base-severity delta decides when it is + /// non-zero — the scorer already knows each ladder's direction, including the inverted free-space + /// one and the step ladders whose rung is set by metadata rather than . When + /// both sides sit on the same rung (a saturated ladder, the doubling-from-30%-to-60% case) the value + /// decides, inverted for the free-space key. + /// + private static string Direction(string key, Fact baseline, Fact comparison) + { + var baseDelta = comparison.BaseSeverity - baseline.BaseSeverity; + if (baseDelta != 0) + return baseDelta > 0 ? StatusWorse : StatusBetter; + + var valueDelta = comparison.Value - baseline.Value; + var higherIsWorse = !HigherIsBetterKeys.Contains(key); + return (valueDelta > 0) == higherIsWorse ? StatusWorse : StatusBetter; + } +} + +/// One compared key. Serialized by so both SKUs emit the same shape. +public sealed class ComparisonRow +{ + public required string Key { get; init; } + public required string Source { get; init; } + public required string Family { get; init; } + public required string Presence { get; init; } + public double? BaselineValue { get; init; } + public double? ComparisonValue { get; init; } + public double? BaselineSeverity { get; init; } + public double? ComparisonSeverity { get; init; } + /// Kept from the pre-#3538 payload for continuity; it no longer decides the band. + public double SeverityDelta { get; init; } + public double? ValueDelta { get; init; } + public double? RelativeMove { get; init; } + public double LadderPosition { get; init; } + public required string Status { get; init; } + public required string BandSource { get; init; } + public double? DeltaSigma { get; init; } + public double? BaselineSigma { get; init; } + public double? BaselineMedian { get; init; } + public double? BaselineConfidence { get; init; } + public string? BaselineTier { get; init; } + public string? BaselineMetric { get; init; } + public bool? BeyondAnomalyCutoff { get; init; } + public bool CoverageCaveat { get; init; } + + public object ToPayload() => new + { + key = Key, + source = Source, + family = Family, + presence = Presence, + baseline_value = BaselineValue, + comparison_value = ComparisonValue, + baseline_severity = BaselineSeverity, + comparison_severity = ComparisonSeverity, + severity_delta = SeverityDelta, + value_delta = ValueDelta, + relative_move = RelativeMove, + ladder_position = LadderPosition, + status = Status, + band_source = BandSource, + delta_sigma = DeltaSigma, + baseline_sigma = BaselineSigma, + baseline_median = BaselineMedian, + baseline_confidence = BaselineConfidence, + baseline_tier = BaselineTier, + baseline_metric = BaselineMetric, + beyond_anomaly_cutoff = BeyondAnomalyCutoff, + coverage_caveat = CoverageCaveat + }; +} + +/// One physical cause: its worst member and every member key, so one I/O stall is one row. +public sealed record ComparisonFamily( + string Family, + string Status, + string WorstKey, + IReadOnlyList Members, + int Worse, + int Better, + int Stable, + bool CoverageCaveat) +{ + public object ToPayload() => new + { + family = Family, + status = Status, + worst_key = WorstKey, + members = Members, + worse = Worse, + better = Better, + stable = Stable, + coverage_caveat = CoverageCaveat + }; +} + +/// A BAD_ACTOR_* key seen on one side only, with the side's reading. +public sealed record ChurnEntry(string Key, string? DatabaseName, double Value, double Severity) +{ + public object ToPayload() => new { key = Key, database = DatabaseName, value = Value, severity = Severity }; +} + +/// Plan-cache identity churn: what the top-5 cut held on one side and not the other. +public sealed record PlanCacheChurn(IReadOnlyList Appeared, IReadOnlyList Disappeared, int PresentInBoth) +{ + public object ToPayload() => new + { + appeared = Appeared.Select(e => e.ToPayload()).ToList(), + disappeared = Disappeared.Select(e => e.ToPayload()).ToList(), + present_in_both = PresentInBoth, + note = "BAD_ACTOR_ keys are plan-cache identities: a hash present in one window only means the cache held a different plan for the top-5 cut, not that a problem began or ended. Not counted in new_issues / resolved_issues." + }; +} + +/// The whole verdict set for one pair of windows. +public sealed record ComparisonResult( + IReadOnlyList Rows, + IReadOnlyList Families, + PlanCacheChurn Churn, + bool CoverageCaveat) +{ + public int Worse => Rows.Count(r => r.Status == ComparisonBanding.StatusWorse); + public int Better => Rows.Count(r => r.Status == ComparisonBanding.StatusBetter); + public int Stable => Rows.Count(r => r.Status == ComparisonBanding.StatusStable); + public int NewIssues => Rows.Count(r => r.Presence == ComparisonBanding.PresenceComparisonOnly && r.Status == ComparisonBanding.StatusWorse); + public int ResolvedIssues => Rows.Count(r => r.Presence == ComparisonBanding.PresenceBaselineOnly && r.Status == ComparisonBanding.StatusBetter); + public int FamiliesWorse => Families.Count(f => f.Status == ComparisonBanding.StatusWorse); + public int FamiliesBetter => Families.Count(f => f.Status == ComparisonBanding.StatusBetter); + public int FamiliesStable => Families.Count(f => f.Status == ComparisonBanding.StatusStable); + + /// True when the union of both windows' keys (churn included) was empty — nothing to compare. + public bool IsEmpty => Rows.Count == 0 && Churn.Appeared.Count == 0 && Churn.Disappeared.Count == 0; + + public object SummaryPayload() => new + { + worse = Worse, + better = Better, + stable = Stable, + new_issues = NewIssues, + resolved_issues = ResolvedIssues, + families_worse = FamiliesWorse, + families_better = FamiliesBetter, + families_stable = FamiliesStable, + plan_cache_churn_appeared = Churn.Appeared.Count, + plan_cache_churn_disappeared = Churn.Disappeared.Count, + /* True on every verdict row too; surfaced here so a reader of the summary alone sees it. */ + coverage_caveat = CoverageCaveat + }; +} From 32389e6b19139cfaa03948bd712fa0eacb77392f Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 16:02:28 -0400 Subject: [PATCH 60/69] Every latest-snapshot MCP read says when it was captured, and no tool accepts a window it does not read (#3541 A10) (#3637) --- .../DarlingMcpConfigHistoryToolsTests.cs | 2 +- .../DarlingMcpPlanCacheSchedulerToolsTests.cs | 13 +- .../McpLatestSnapshotStampTests.cs | 837 ++++++++++++++++++ .../Darling.Tests/RepoFileAdoptionTests.cs | 4 + .../DarlingWebEndpoints.cs | 4 +- .../Mcp/DarlingConfigHistoryReader.cs | 19 +- .../Mcp/DarlingCurrentConfigReader.cs | 41 +- .../Mcp/DarlingDataReader.cs | 34 +- .../Mcp/DarlingHealthReader.cs | 38 +- .../Mcp/DarlingLatchSpinlockReader.cs | 18 +- .../Mcp/DarlingMcpConfigHistoryTools.cs | 19 +- .../Mcp/DarlingMcpConfigTools.cs | 32 +- .../Mcp/DarlingMcpDataTools.cs | 32 +- .../Mcp/DarlingMcpHealthTools.cs | 13 +- .../Mcp/DarlingMcpInstructions.cs | 27 +- .../Mcp/DarlingMcpLatchSpinlockTools.cs | 16 +- .../Mcp/DarlingMcpMemoryGrantTools.cs | 81 +- .../Mcp/DarlingMcpPlanCacheSchedulerTools.cs | 46 +- .../Mcp/DarlingMemoryGrantReader.cs | 196 ++++ .../Mcp/DarlingPlanCacheSchedulerReader.cs | 19 +- .../Mcp/LatestSnapshot.cs | 76 ++ Lite.Tests/McpLatestSnapshotStampTests.cs | 448 ++++++++++ Lite/Mcp/McpConfigTools.cs | 23 +- Lite/Mcp/McpHealthTools.cs | 18 +- Lite/Mcp/McpInstructions.cs | 29 +- Lite/Mcp/McpIoTools.cs | 4 +- Lite/Mcp/McpLatchSpinlockTools.cs | 10 +- Lite/Mcp/McpLatestSnapshotStamp.cs | 28 + Lite/Mcp/McpMemoryTools.cs | 61 +- Lite/Mcp/McpPerfmonTools.cs | 4 +- Lite/Mcp/McpPlanCacheSchedulerTools.cs | 31 +- Lite/Services/LocalDataService.Config.cs | 40 +- Lite/Services/LocalDataService.FileIo.cs | 9 +- .../LocalDataService.LatchSpinlock.cs | 16 +- Lite/Services/LocalDataService.Memory.cs | 9 +- .../Services/LocalDataService.MemoryGrants.cs | 211 +++++ Lite/Services/LocalDataService.Overview.cs | 23 +- Lite/Services/LocalDataService.Perfmon.cs | 8 +- Lite/Services/LocalDataService.PlanCache.cs | 10 +- 39 files changed, 2359 insertions(+), 190 deletions(-) create mode 100644 Darling/Darling.Tests/McpLatestSnapshotStampTests.cs create mode 100644 Darling/PerformanceMonitor.Darling.Service/Mcp/LatestSnapshot.cs create mode 100644 Lite.Tests/McpLatestSnapshotStampTests.cs create mode 100644 Lite/Mcp/McpLatestSnapshotStamp.cs diff --git a/Darling/Darling.Tests/DarlingMcpConfigHistoryToolsTests.cs b/Darling/Darling.Tests/DarlingMcpConfigHistoryToolsTests.cs index 603b9308a..642248029 100644 --- a/Darling/Darling.Tests/DarlingMcpConfigHistoryToolsTests.cs +++ b/Darling/Darling.Tests/DarlingMcpConfigHistoryToolsTests.cs @@ -143,7 +143,7 @@ public void QueryStoreHealthSql_LatestSnapshot_SelectsPayloadOrder() Assert.Contains("MAX(capture_time)", sql, StringComparison.Ordinal); Assert.Contains("ORDER BY database_name", sql, StringComparison.Ordinal); Assert.Contains( - "database_name, actual_state, desired_state, readonly_reason, current_storage_size_mb, max_storage_size_mb, size_based_cleanup_mode, stale_query_threshold_days, max_plans_per_query, interval_length_minutes", + "database_name, actual_state, desired_state, readonly_reason, current_storage_size_mb, max_storage_size_mb, size_based_cleanup_mode, stale_query_threshold_days, max_plans_per_query, interval_length_minutes, capture_time", sql, StringComparison.Ordinal); } diff --git a/Darling/Darling.Tests/DarlingMcpPlanCacheSchedulerToolsTests.cs b/Darling/Darling.Tests/DarlingMcpPlanCacheSchedulerToolsTests.cs index d660da5e0..75be4d9f6 100644 --- a/Darling/Darling.Tests/DarlingMcpPlanCacheSchedulerToolsTests.cs +++ b/Darling/Darling.Tests/DarlingMcpPlanCacheSchedulerToolsTests.cs @@ -73,12 +73,16 @@ public void ParamContract_PlanCacheBloat_ServerHours() Assert.All(p, x => Assert.True(x.Optional)); } + /// #3541 A10: the same (server_name, hours_back, as_of) surface Lite's twin has always had. The + /// tool took server_name alone here — two parameter surfaces under one tool name, on the one tool whose + /// answer is a CRITICAL/HIGH/MEDIUM/NORMAL verdict. McpLatestSnapshotStampTests pins the two SKUs' + /// descriptions equal; this pins the shape. [Fact] - public void ParamContract_CpuSchedulerPressure_ServerNameOnly() + public void ParamContract_CpuSchedulerPressure_ServerHoursAsOf_MatchesLite() { var p = McpParams("get_cpu_scheduler_pressure"); - Assert.Equal(new[] { "server_name" }, p.Select(x => x.Name).ToArray()); - Assert.True(p.Single().Optional); + Assert.Equal(new[] { "server_name", "hours_back", "as_of" }, p.Select(x => x.Name).ToArray()); + Assert.All(p, x => Assert.True(x.Optional)); } [Fact] @@ -103,6 +107,9 @@ public void CpuSchedulerPressureSql_LatestRow() Assert.Contains("worker_thread_exhaustion_warning", sql, StringComparison.Ordinal); Assert.Contains("ORDER BY collection_time DESC", sql, StringComparison.Ordinal); Assert.Contains("LIMIT 1", sql, StringComparison.Ordinal); + /* #3541 A10: the newest row IN THE WINDOW, as Lite reads it — not the newest row the store ever held. */ + Assert.Contains("collection_time >= $2", sql, StringComparison.Ordinal); + Assert.Contains("collection_time <= $3", sql, StringComparison.Ordinal); } [Theory] diff --git a/Darling/Darling.Tests/McpLatestSnapshotStampTests.cs b/Darling/Darling.Tests/McpLatestSnapshotStampTests.cs new file mode 100644 index 000000000..7f3ce34b3 --- /dev/null +++ b/Darling/Darling.Tests/McpLatestSnapshotStampTests.cs @@ -0,0 +1,837 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.ComponentModel; +using System.Linq; +using System.Reflection; +using System.Text.Json; +using System.Text.RegularExpressions; +using System.Threading.Tasks; +using ModelContextProtocol.Server; +using Npgsql; +using PerformanceMonitor.Collectors; +using PerformanceMonitor.Common; +using PerformanceMonitor.Darling.Service; +using PerformanceMonitor.Darling.Service.Mcp; +using PerformanceMonitor.Darling.Storage; +using Xunit; +using static Darling.Tests.RepoFile; + +namespace Darling.Tests; + +/// +/// #3541 A10 — latest is a time. A latest-snapshot MCP read (WHERE collection_time = MAX(...), +/// ORDER BY collection_time DESC LIMIT 1) answers with the newest row a store holds, and until this lane +/// most of them answered with nothing that said WHEN that row was collected: a memory-clerk list, a file-I/O +/// table, a perfmon page, the whole configuration family (captured ON CONNECT, so a "current" setting could be +/// weeks old) — an agent that cannot see the code read every one of them as "now". Two of them accepted +/// hours_back and read only the newest row in it, so a grant storm three hours ago was invisible while +/// the parameter read as a window; one CRITICAL-verdict tool took server_name alone on one SKU and +/// (server_name, hours_back, as_of) on the other; and get_server_summary published ONE clock (the newest +/// collection of ANY collector) beside two figures it did not stamp. +/// +/// This file pins the rule and the three honest shapes a latest read may take, on BOTH SKUs, as one +/// dialect: every latest read publishes captured_at (the snapshot's own stamp) and its description says +/// "LATEST IS A TIME" or names its two reads; a tool that takes NO window carries no hours_back; a tool +/// whose hours_back is the span SEARCHED for the newest snapshot says so in those words and publishes +/// age_seconds against its anchor; a tool whose hours_back is READ publishes a window block +/// beside the snapshot. The two SKUs' descriptions of the two aligned tools are pinned byte-equal. Darling's +/// half is reflected off the assembly; Lite's is read from source, as does. +/// The discriminators are witnessed against literals so a matcher that stops matching cannot report clean. +/// Lite.Tests/McpLatestSnapshotStampTests executes the Lite tools against a real DuckDB; +/// executes the Darling ones against live Postgres. +/// +public sealed class McpLatestSnapshotStampTests +{ + /* ───────────────────────── the roster ───────────────────────── */ + + /// The three honest shapes. Named rather than inferred so a tool's disposition is a decision + /// written down here, and moving one is a visible edit. + public enum Shape + { + /// Reads one snapshot, takes no window: captured_at only. + Stamped, + + /// hours_back bounds the SEARCH for the newest snapshot (described in those words) and + /// as_of anchors it: captured_at + age_seconds. + SearchBound, + + /// hours_back is READ: the newest snapshot (captured_at + age_seconds) + /// beside a window aggregate over every snapshot in it. + Windowed, + } + + /// Every latest-snapshot tool this lane stamped, by SKU file, with its shape. The Lite file and + /// name are given per row because Lite hosts the same tool names in differently-named files. + public static readonly (Type Tools, string ToolName, string LiteFile, Shape Shape)[] LatestTools = + [ + (typeof(DarlingMcpDataTools), "get_memory_stats", "Lite/Mcp/McpMemoryTools.cs", Shape.Stamped), + (typeof(DarlingMcpDataTools), "get_memory_clerks", "Lite/Mcp/McpMemoryTools.cs", Shape.Stamped), + (typeof(DarlingMcpDataTools), "get_file_io_stats", "Lite/Mcp/McpIoTools.cs", Shape.Stamped), + (typeof(DarlingMcpDataTools), "get_perfmon_stats", "Lite/Mcp/McpPerfmonTools.cs", Shape.Stamped), + (typeof(DarlingMcpConfigTools), "get_server_config", "Lite/Mcp/McpConfigTools.cs", Shape.Stamped), + (typeof(DarlingMcpConfigTools), "get_database_config", "Lite/Mcp/McpConfigTools.cs", Shape.Stamped), + (typeof(DarlingMcpConfigTools), "get_trace_flags", "Lite/Mcp/McpConfigTools.cs", Shape.Stamped), + (typeof(DarlingMcpConfigHistoryTools), "get_database_scoped_config", "Lite/Mcp/McpConfigTools.cs", Shape.Stamped), + (typeof(DarlingMcpConfigHistoryTools), "get_query_store_health", "Lite/Mcp/McpConfigTools.cs", Shape.Stamped), + (typeof(DarlingMcpPlanCacheSchedulerTools), "get_plan_cache_bloat", "Lite/Mcp/McpPlanCacheSchedulerTools.cs", Shape.SearchBound), + (typeof(DarlingMcpPlanCacheSchedulerTools), "get_cpu_scheduler_pressure", "Lite/Mcp/McpPlanCacheSchedulerTools.cs", Shape.SearchBound), + (typeof(DarlingMcpMemoryGrantTools), "get_resource_semaphore", "Lite/Mcp/McpMemoryTools.cs", Shape.Windowed), + (typeof(DarlingMcpMemoryGrantTools), "get_memory_grants", "Lite/Mcp/McpMemoryTools.cs", Shape.Windowed), + ]; + + /// + /// Latest reads that stamped themselves BEFORE this lane, under the top-level key collection_time + /// rather than captured_at. The stamp is true; only the spelling predates the vocabulary, and the + /// Darling web surface reads one of them (get_session_stats, server-tabs.js' SESSION_STATS) + /// by that key, which sits outside this lane's boundary. Carried as a stated allowance with the control + /// below rather than silently: the day one of them is renamed, the allowance must shrink (#3541 A15/A16 is + /// the vocabulary lane). + /// + public static readonly string[] StampedUnderCollectionTime = + [ + "get_database_sizes", "get_running_jobs", "get_server_properties", "get_session_stats", + ]; + + /// + /// The latest reads this lane did NOT reach: the object-stats family reads the latest DAILY snapshot per + /// database (DarlingObjectStatsReader.IndexLockingSql takes MAX per database, not one instant) and + /// publishes no stamp at all on either SKU. Named here so the census fails the day one of them gains + /// captured_at without leaving this list, and so the gap is on the record rather than invisible. + /// Reported to #3541 as the A10 residual. + /// + public static readonly string[] UnstampedLatestReadsPendingA10 = + [ + "get_index_usage", "get_object_locking", "get_table_index_sizes", + ]; + + /// + /// Tools the reader-call sweep sees because they call a *Latest*Async reader as an INPUT to a read + /// that is not itself a latest snapshot: the two CPU rankings read the newest server_properties row for the + /// core count their attribution divides by; get_plan_corrections reads the newest automatic-tuning settings + /// beside its paged correction history; get_sweep_reports reads the newest fleet sweep, a worklist rather than + /// a per-server snapshot; get_pvs_stats mixes a latest per-database snapshot (stamped per row) with a trend, + /// and is the named exclusion carries for that reason. None of them is a + /// latest-snapshot tool, and none gets a captured_at here — the control below fails if one gains it, + /// so the decision is revisited rather than drifted into. + /// + public static readonly string[] LatestLookupInsideAnotherRead = + [ + "get_plan_corrections", "get_pvs_stats", "get_sweep_reports", "get_top_procedures_by_cpu", "get_top_queries_by_cpu", + ]; + + /// get_server_summary carries THREE clocks by name (cpu_captured_at, memory_captured_at, + /// last_collection) rather than one captured_at, because its two latest figures come from two + /// collectors that can be arbitrarily far apart. + /// holds the shape; the sweep only needs to know it is accounted for. + public static readonly string[] ThreeClockTools = ["get_server_summary"]; + + /* ───────────────────────── the discriminators ───────────────────────── */ + + /// The stamp, as a payload key. + private static readonly Regex CapturedAtKey = new(@"\bcaptured_at\s*=", RegexOptions.Compiled); + + /// The anchored distance. + private static readonly Regex AgeSecondsKey = new(@"\bage_seconds\s*=", RegexOptions.Compiled); + + /// The window half of a Windowed tool. + private static readonly Regex WindowKey = new(@"\bwindow\s*=\s*window\.Select\(", RegexOptions.Compiled); + + /// A pre-lane stamp under the old spelling, as a TOP-LEVEL key (a per-row collection_time + /// inside a Select(r => new { ... }) is a series column, not the snapshot's stamp). + private static readonly Regex TopLevelCollectionTimeKey = new(@"\n\s{16}collection_time\s*=", RegexOptions.Compiled); + + /// The words a SearchBound tool must use for hours_back. + private const string SearchBoundWords = "search for the latest snapshot"; + + /// The words a Windowed tool must use for hours_back. + private const string WindowedWords = "window[] aggregates every snapshot in these hours"; + + /* ───────────────────────── both SKUs, from source ───────────────────────── */ + + [Fact] + public void EveryLatestSnapshotTool_PublishesCapturedAt_OnBothSkus() + { + foreach (var (label, body, _) in LatestToolBodies()) + { + Assert.True(CapturedAtKey.IsMatch(Strip(body)), + $"{label}: no `captured_at =` on the payload — a latest read that never says when it was captured"); + } + } + + [Fact] + public void EveryLatestSnapshotTool_SaysLatestIsATime_InItsDescription_OnBothSkus() + { + foreach (var (label, body, shape) in LatestToolBodies()) + { + var description = DescriptionOf(body, label); + Assert.Contains("captured_at", description, StringComparison.Ordinal); + if (shape == Shape.Windowed) + { + Assert.Contains("TWO READS UNDER ONE WINDOW", description, StringComparison.Ordinal); + Assert.Contains("window[]", description, StringComparison.Ordinal); + } + else + { + Assert.Contains("LATEST IS A TIME", description, StringComparison.Ordinal); + } + } + } + + /// A parameter that does nothing is a lie: a Stamped tool takes neither knob. + [Fact] + public void StampedTools_TakeNoWindowAndNoAnchor_OnBothSkus() + { + foreach (var (type, name, liteFile, shape) in LatestTools.Where(t => t.Shape == Shape.Stamped)) + { + var darling = ToolMethod(type, name).GetParameters().Select(p => p.Name).ToArray(); + Assert.DoesNotContain("hours_back", darling); + Assert.DoesNotContain("as_of", darling); + + var lite = LiteParamNames(liteFile, name); + Assert.DoesNotContain("hours_back", lite); + Assert.DoesNotContain("as_of", lite); + } + } + + /// + /// A SearchBound tool's hours_back says it is the SEARCH span in the same words on both SKUs, the + /// tool takes the anchor, and the payload carries the anchored distance — the three things that make a + /// latest read with a window parameter honest. + /// + [Fact] + public void SearchBoundTools_DescribeHoursBackAsTheSearchSpan_AndPublishAge_OnBothSkus() + { + foreach (var (type, name, liteFile, _) in LatestTools.Where(t => t.Shape == Shape.SearchBound)) + { + var method = ToolMethod(type, name); + var hours = method.GetParameters().Single(p => p.Name == "hours_back"); + Assert.Contains(SearchBoundWords, hours.GetCustomAttribute()!.Description, StringComparison.Ordinal); + Assert.Contains("as_of", method.GetParameters().Select(p => p.Name)); + Assert.Matches(AgeSecondsKey, Strip(ToolBody(ReadRepoFileLf(DarlingFileOf(type).Split('/')), name))); + + var liteBody = ToolBody(ReadRepoFileLf(liteFile.Split('/')), name); + Assert.Contains(SearchBoundWords, liteBody, StringComparison.Ordinal); + Assert.Contains("as_of", LiteParamNames(liteFile, name)); + Assert.Matches(AgeSecondsKey, Strip(liteBody)); + } + } + + /// A Windowed tool reads its window: the payload carries the window block, the anchored + /// distance, and the window's own bounds, and hours_back says which half is which. + [Fact] + public void WindowedTools_PublishTheWindowBesideTheSnapshot_OnBothSkus() + { + foreach (var (type, name, liteFile, _) in LatestTools.Where(t => t.Shape == Shape.Windowed)) + { + var hours = ToolMethod(type, name).GetParameters().Single(p => p.Name == "hours_back"); + Assert.Contains(WindowedWords, hours.GetCustomAttribute()!.Description, StringComparison.Ordinal); + + foreach (var body in new[] { ToolBody(ReadRepoFileLf(DarlingFileOf(type).Split('/')), name), ToolBody(ReadRepoFileLf(liteFile.Split('/')), name) }) + { + var text = Strip(body); + Assert.Matches(WindowKey, text); + Assert.Matches(AgeSecondsKey, text); + Assert.Contains("window_start =", text, StringComparison.Ordinal); + Assert.Contains("window_end =", text, StringComparison.Ordinal); + Assert.Contains(WindowedWords, body, StringComparison.Ordinal); + } + } + } + + /// The same tool name takes the same parameters on both SKUs — the drift this lane closed on + /// get_cpu_scheduler_pressure, pinned for every tool in the roster so it cannot reopen on another. + [Fact] + public void TheSameToolName_TakesTheSameParameters_OnBothSkus() + { + foreach (var (type, name, liteFile, _) in LatestTools) + { + var darling = ToolMethod(type, name).GetParameters() + .Where(p => p.GetCustomAttribute() is not null) + .Select(p => p.Name!) + .ToArray(); + Assert.Equal(darling, LiteParamNames(liteFile, name)); + } + } + + /// + /// The two tools whose descriptions were rewritten on both SKUs are pinned byte-equal: a shared const + /// cannot cross the two assemblies, so the census holds the two literals together instead. Darling's is + /// reflected; Lite's is the const its attribute names, read from source. + /// + [Theory] + [InlineData(typeof(DarlingMcpPlanCacheSchedulerTools), "get_cpu_scheduler_pressure", "Lite/Mcp/McpPlanCacheSchedulerTools.cs", "CpuSchedulerPressureDescription")] + [InlineData(typeof(DarlingMcpHealthTools), "get_server_summary", "Lite/Mcp/McpHealthTools.cs", "ServerSummaryDescription")] + public void TheAlignedTools_AreDescribedIdentically_OnBothSkus(Type type, string toolName, string liteFile, string liteConst) + { + var darling = ToolMethod(type, toolName).GetCustomAttribute()!.Description; + var lite = LiteConstLiteral(liteFile, liteConst); + Assert.Equal(darling, lite); + + /* And the attribute really does name the const — a literal pasted beside an unused const would pass + the equality above while drifting the day either is edited. */ + var liteBody = ToolBody(ReadRepoFileLf(liteFile.Split('/')), toolName); + Assert.Contains($"Description({liteConst})", liteBody, StringComparison.Ordinal); + } + + /// The verdict Darling always published, now on Lite too: the same tool name answers with a + /// banded pressure_level on both SKUs, from the SHARED classifier. + [Fact] + public void CpuSchedulerPressure_PublishesTheVerdict_OnBothSkus() + { + var lite = Strip(ToolBody(ReadRepoFileLf("Lite", "Mcp", "McpPlanCacheSchedulerTools.cs"), "get_cpu_scheduler_pressure")); + Assert.Contains("CpuSchedulerMetrics.ClassifyCpuPressure(", lite, StringComparison.Ordinal); + Assert.Contains("pressure_level = pressure.Level", lite, StringComparison.Ordinal); + Assert.Contains("recommendation = pressure.Recommendation", lite, StringComparison.Ordinal); + + var darling = Strip(ToolBody(ReadRepoFileLf(DarlingFileOf(typeof(DarlingMcpPlanCacheSchedulerTools)).Split('/')), "get_cpu_scheduler_pressure")); + Assert.Contains("pressure_level = pressureLevel", darling, StringComparison.Ordinal); + } + + /// get_server_summary names its three clocks on both SKUs, and no longer publishes two figures + /// under one stamp. + [Fact] + public void ServerSummary_NamesItsThreeClocks_OnBothSkus() + { + foreach (var body in new[] + { + ToolBody(ReadRepoFileLf(DarlingFileOf(typeof(DarlingMcpHealthTools)).Split('/')), "get_server_summary"), + ToolBody(ReadRepoFileLf("Lite", "Mcp", "McpHealthTools.cs"), "get_server_summary"), + }) + { + var text = Strip(body); + Assert.Contains("cpu_captured_at =", text, StringComparison.Ordinal); + Assert.Contains("memory_captured_at =", text, StringComparison.Ordinal); + Assert.Contains("counts_window_hours =", text, StringComparison.Ordinal); + Assert.Contains("last_collection =", text, StringComparison.Ordinal); + } + } + + /// The latch band names the interval it came from, beside the window totals it did not. + [Fact] + public void LatchSeverity_NamesTheIntervalItWasBandedFrom() + { + var body = Strip(ToolBody(ReadRepoFileLf(DarlingFileOf(typeof(DarlingMcpLatchSpinlockTools)).Split('/')), "get_latch_stats")); + Assert.Contains("severity_banded_from = new", body, StringComparison.Ordinal); + Assert.Contains("delta_wait_time_ms = r.LatestDeltaWaitTimeMs", body, StringComparison.Ordinal); + Assert.Contains("interval_seconds =", body, StringComparison.Ordinal); + Assert.Contains("captured_at = r.LatestCollectionTime", body, StringComparison.Ordinal); + + var sql = DarlingLatchSpinlockReader.LatchStatsTopNSql; + Assert.Contains("AS latest_interval_seconds", sql, StringComparison.Ordinal); + Assert.Contains("l.latest_interval_seconds", sql, StringComparison.Ordinal); + + var description = ToolMethod(typeof(DarlingMcpLatchSpinlockTools), "get_latch_stats").GetCustomAttribute()!.Description; + Assert.Contains("severity_banded_from", description, StringComparison.Ordinal); + Assert.Contains("LATEST interval only", description, StringComparison.Ordinal); + } + + /* ───────────────────────── the sweep: no latest read escapes the roster ───────────────────────── */ + + /// + /// Every Darling tool whose body calls a *Latest*Async / *Snapshot*Async / Current-family + /// reader, or whose reader const is a latest-snapshot read, is in the roster, the pre-lane allowance, or the + /// named residual — and every allowance entry is still needed. A new latest read must pick a shape here + /// rather than ship unstamped. + /// + [Fact] + public void EveryLatestReadTool_IsInTheRoster_OrANamedAllowance_AndEveryAllowanceIsStillNeeded() + { + var rostered = LatestTools.Select(t => t.ToolName).ToHashSet(StringComparer.Ordinal); + var allowancesUsed = new HashSet(StringComparer.Ordinal); + var residualsSeen = new HashSet(StringComparer.Ordinal); + var lookupsSeen = new HashSet(StringComparer.Ordinal); + var threeClocksSeen = new HashSet(StringComparer.Ordinal); + var examined = 0; + + foreach (var (file, source) in AllDarlingToolSources()) + { + var marks = Regex.Matches(source, @"\[McpServerTool\(Name = ""([a-z_0-9]+)"""); + for (var i = 0; i < marks.Count; i++) + { + var end = i + 1 < marks.Count ? marks[i + 1].Index : source.Length; + var body = Strip(source[marks[i].Index..end]); + var toolName = marks[i].Groups[1].Value; + + if (!LatestReaderCall.IsMatch(body)) + { + continue; + } + + examined++; + if (rostered.Contains(toolName)) + { + continue; + } + + if (StampedUnderCollectionTime.Contains(toolName, StringComparer.Ordinal)) + { + Assert.True(TopLevelCollectionTimeKey.IsMatch(body), + $"{file} {toolName}: listed as stamped under collection_time but publishes no top-level collection_time"); + Assert.False(CapturedAtKey.IsMatch(body), + $"{file} {toolName}: now publishes captured_at — move it into the roster and out of StampedUnderCollectionTime"); + allowancesUsed.Add(toolName); + continue; + } + + if (UnstampedLatestReadsPendingA10.Contains(toolName, StringComparer.Ordinal)) + { + Assert.False(CapturedAtKey.IsMatch(body), + $"{file} {toolName}: now publishes captured_at — move it into the roster and out of UnstampedLatestReadsPendingA10"); + residualsSeen.Add(toolName); + continue; + } + + if (LatestLookupInsideAnotherRead.Contains(toolName, StringComparer.Ordinal)) + { + Assert.False(CapturedAtKey.IsMatch(body), + $"{file} {toolName}: now publishes captured_at — it has become a latest-snapshot tool; give it a Shape and remove it from LatestLookupInsideAnotherRead"); + lookupsSeen.Add(toolName); + continue; + } + + if (ThreeClockTools.Contains(toolName, StringComparer.Ordinal)) + { + Assert.Contains("cpu_captured_at =", body, StringComparison.Ordinal); + Assert.Contains("memory_captured_at =", body, StringComparison.Ordinal); + threeClocksSeen.Add(toolName); + continue; + } + + Assert.Fail($"{file} {toolName}: calls a latest-snapshot reader and is in no list here — give it a Shape in LatestTools (and stamp it) or name it as an allowance with its reason"); + } + } + + /* Population controls: a sweep that matched nothing passes for free, and an allowance nobody needs is a + widened exemption. 26 latest-reading tool bodies at the time of writing. */ + Assert.True(examined >= 24, $"only {examined} latest-reading tool bodies were found; the reader-call pattern has stopped matching"); + Assert.True(LatestLookupInsideAnotherRead.ToHashSet(StringComparer.Ordinal).SetEquals(lookupsSeen), + "LatestLookupInsideAnotherRead no longer matches what the sweep finds: " + string.Join(", ", LatestLookupInsideAnotherRead.Except(lookupsSeen))); + Assert.True(ThreeClockTools.ToHashSet(StringComparer.Ordinal).SetEquals(threeClocksSeen), + "ThreeClockTools no longer matches what the sweep finds: " + string.Join(", ", ThreeClockTools.Except(threeClocksSeen))); + Assert.True(StampedUnderCollectionTime.ToHashSet(StringComparer.Ordinal).SetEquals(allowancesUsed), + "StampedUnderCollectionTime no longer matches what the sweep finds: " + string.Join(", ", StampedUnderCollectionTime.Except(allowancesUsed))); + Assert.True(UnstampedLatestReadsPendingA10.ToHashSet(StringComparer.Ordinal).SetEquals(residualsSeen), + "UnstampedLatestReadsPendingA10 no longer matches what the sweep finds: " + string.Join(", ", UnstampedLatestReadsPendingA10.Except(residualsSeen))); + } + + /// + /// A tool body's call into a latest-snapshot reader. The readers name themselves: GetLatest*Async, + /// *LatestAsync, *SnapshotAsync, the plan-cache / scheduler pair, the server summary, and the + /// object-stats trio (GetIndexUsageAsync / GetIndexLockingAsync / GetObjectSizeGrowthAsync, + /// each keyed on a correlated MAX(collection_time)). + /// + private static readonly Regex LatestReaderCall = new( + @"\.(GetLatest\w+Async|Get\w+LatestAsync|Get\w+SnapshotAsync|GetPlanCacheBloatAsync|GetCpuSchedulerPressureAsync|GetServerSummaryAsync|GetIndexUsageAsync|GetIndexLockingAsync|GetObjectSizeGrowthAsync|GetRunningJobsAsync)\(", + RegexOptions.Compiled); + + /* ───────────────────────── the readers ───────────────────────── */ + + /// Every stamped Darling read carries its stamp column ON THE ROW STATEMENT — never a second + /// MAX() read that could stamp the next capture. + [Theory] + [InlineData(nameof(DarlingDataReader.LatestMemoryClerksSql), "collection_time")] + [InlineData(nameof(DarlingDataReader.LatestFileIoStatsSql), "collection_time")] + [InlineData(nameof(DarlingDataReader.LatestPerfmonStatsSql), "collection_time")] + [InlineData(nameof(DarlingCurrentConfigReader.ServerConfigSql), "capture_time")] + [InlineData(nameof(DarlingCurrentConfigReader.DatabaseConfigSql), "capture_time")] + [InlineData(nameof(DarlingCurrentConfigReader.TraceFlagsSql), "capture_time")] + [InlineData(nameof(DarlingConfigHistoryReader.DatabaseScopedConfigSql), "capture_time")] + [InlineData(nameof(DarlingConfigHistoryReader.QueryStoreHealthSql), "capture_time")] + public void EveryStampedRead_SelectsItsStampColumn_OnTheRowStatement(string sqlName, string column) + { + var sql = ReaderSql(sqlName); + var select = sql[..sql.IndexOf("FROM", StringComparison.Ordinal)]; + Assert.Contains(column, select, StringComparison.Ordinal); + } + + /// The scheduler read is bounded the way Lite's is — the newest row IN THE WINDOW, so a week-old + /// snapshot from a dead collector is unavailable rather than served as current. + [Fact] + public void CpuSchedulerRead_IsBoundedByTheWindow() + { + var sql = DarlingPlanCacheSchedulerReader.CpuSchedulerPressureSql; + Assert.Contains("collection_time >= $2", sql, StringComparison.Ordinal); + Assert.Contains("collection_time <= $3", sql, StringComparison.Ordinal); + Assert.Contains("ORDER BY collection_time DESC", sql, StringComparison.Ordinal); + Assert.Contains("LIMIT 1", sql, StringComparison.Ordinal); + } + + /// The two window reads: the peak's instant from a DISTINCT ON over the SAME windowed rows, + /// the deltas SUMmed (no interval arithmetic — the naked-family rung is a rate question), and no literal cap. + [Theory] + [InlineData(nameof(DarlingMemoryGrantReader.ResourceSemaphoreWindowSql))] + [InlineData(nameof(DarlingMemoryGrantReader.MemoryGrantsWindowSql))] + public void MemoryGrantWindowReads_AggregateEverySnapshot_AndNameThePeaksInstant(string sqlName) + { + var sql = ReaderSql(sqlName); + Assert.Contains("collection_time >= $2", sql, StringComparison.Ordinal); + Assert.Contains("collection_time <= $3", sql, StringComparison.Ordinal); + Assert.Contains("COUNT(*) AS snapshots_in_window", sql, StringComparison.Ordinal); + Assert.Contains("MAX(waiter_count) AS bigint) AS peak_waiter_count", sql, StringComparison.Ordinal); + Assert.Contains("SUM(timeout_error_count_delta) AS bigint) AS timeout_errors_in_window", sql, StringComparison.Ordinal); + Assert.Contains("SUM(forced_grant_count_delta) AS bigint) AS forced_grants_in_window", sql, StringComparison.Ordinal); + Assert.Contains("SELECT DISTINCT ON", sql, StringComparison.Ordinal); + Assert.Contains("waiter_count DESC, collection_time DESC", sql, StringComparison.Ordinal); + Assert.DoesNotMatch(@"\bLIMIT\b", sql); + Assert.DoesNotContain("sample_interval_seconds", sql, StringComparison.Ordinal); + } + + /// The web catalogue advertises the two knobs the aligned scheduler tool now takes, and the + /// dispatch forwards them ( holds the general rule; this names the tool). + [Fact] + public void CpuSchedulerPressure_AdvertisesAndForwardsItsWindow_OnTheWebSurface() + { + var descriptor = DarlingWebEndpoints.CatalogDescriptors["get_cpu_scheduler_pressure"]; + Assert.Contains("hours", descriptor.Params.Select(p => p.Name)); + Assert.Contains("as_of", descriptor.Params.Select(p => p.Name)); + + var source = Strip(ReadRepoFileLf("Darling", "PerformanceMonitor.Darling.Service", "DarlingWebEndpoints.cs")); + Assert.Matches(@"\[""get_cpu_scheduler_pressure""\] = \(c, pg, an\) => DarlingMcpPlanCacheSchedulerTools\.GetCpuSchedulerPressure\(pg, Server\(c\), Hours\(c, 24\), as_of: AsOf\(c\)\)", source); + } + + /* ───────────────────────── the pure pieces, executed ───────────────────────── */ + + [Fact] + public void LatestSnapshot_RefusesRowsWithoutAStamp_AndIsEmptyWithoutRows() + { + Assert.Throws(() => new LatestSnapshot(null, new List { 1 })); + + var empty = LatestSnapshot.Empty; + Assert.True(empty.IsEmpty); + Assert.Null(empty.CapturedAt); + Assert.Equal(0, empty.Count); + + var stamped = new LatestSnapshot(new DateTime(2026, 9, 18, 12, 0, 0), new List { 1, 2 }); + Assert.False(stamped.IsEmpty); + Assert.Equal(2, stamped.Count); + } + + /// Whole seconds from stamp to anchor; never negative; Kind-blind (both instants are UTC). + [Fact] + public void AgeSeconds_IsTheWholeSecondDistanceToTheAnchor_AndNeverNegative() + { + var stamp = new DateTime(2026, 9, 18, 12, 0, 0, DateTimeKind.Unspecified); + var anchor = new DateTime(2026, 9, 18, 12, 5, 0, DateTimeKind.Utc); + Assert.Equal(300L, LatestSnapshotStamp.AgeSeconds(stamp, anchor)); + Assert.Equal(300L, LatestSnapshotStamp.AgeSeconds(stamp, anchor.AddTicks(4_000_000))); /* 0.4 s rounds down */ + Assert.Equal(301L, LatestSnapshotStamp.AgeSeconds(stamp, anchor.AddTicks(6_000_000))); /* 0.6 s rounds up */ + Assert.Equal(0L, LatestSnapshotStamp.AgeSeconds(anchor, stamp)); /* clamped */ + Assert.Equal(0L, LatestSnapshotStamp.AgeSeconds(stamp, stamp)); + } + + /* ───────────────────────── the matchers, witnessed ───────────────────────── */ + + [Fact] + public void TheDiscriminators_FlagTheDefectShapes_AndPassTheFixedOnes() + { + Assert.Matches(CapturedAtKey, " captured_at = snapshot.CapturedAt!.Value.ToString(\"o\"),"); + Assert.DoesNotMatch(CapturedAtKey, " last_captured_at = Stamp(c.LastCapturedAt),"); + Assert.DoesNotMatch(CapturedAtKey, " collection_time = stats.CollectionTime.ToString(\"o\"),"); + + Assert.Matches(AgeSecondsKey, " age_seconds = LatestSnapshotStamp.AgeSeconds(rows[0].CollectionTime, now),"); + Assert.Matches(WindowKey, " window = window.Select(WindowShape)"); + Assert.DoesNotMatch(WindowKey, " window_start = windowStart.ToString(\"o\"),"); + + Assert.Matches(TopLevelCollectionTimeKey, "\n collection_time = rows[0].CollectionTime.ToString(\"o\"),"); + Assert.DoesNotMatch(TopLevelCollectionTimeKey, "\n collection_time = r.CollectionTime.ToString(\"o\"),"); + + Assert.Matches(LatestReaderCall, " var snapshot = await DarlingDataReader.GetLatestMemoryClerksAsync(postgres, resolved.ServerId);"); + Assert.Matches(LatestReaderCall, " var rows = await DarlingMemoryGrantReader.GetResourceSemaphoreLatestAsync("); + Assert.Matches(LatestReaderCall, " var rows = await dataService.GetLatchStatsSnapshotAsync(resolved.ServerId, hours_back, asOfUtc: windowEnd);"); + Assert.DoesNotMatch(LatestReaderCall, " var rows = await DarlingDataReader.GetTempDbTrendAsync(postgres, resolved.ServerId, a, b);"); + } + + /* ───────────────────────── plumbing ───────────────────────── */ + + private static IEnumerable<(string Label, string Body, Shape Shape)> LatestToolBodies() + { + foreach (var (type, name, liteFile, shape) in LatestTools) + { + yield return ($"Darling {name}", ToolBody(ReadRepoFileLf(DarlingFileOf(type).Split('/')), name), shape); + yield return ($"Lite {name}", ToolBody(ReadRepoFileLf(liteFile.Split('/')), name), shape); + } + } + + private static IEnumerable<(string File, string Source)> AllDarlingToolSources() + { + var root = RepoFile.PathTo("Darling/PerformanceMonitor.Darling.Service/Mcp"); + foreach (var file in System.IO.Directory.EnumerateFiles(root, "*.cs").Order(StringComparer.Ordinal)) + { + yield return (System.IO.Path.GetFileName(file), System.IO.File.ReadAllText(file).Replace("\r\n", "\n", StringComparison.Ordinal)); + } + } + + private static string ReaderSql(string sqlName) => sqlName switch + { + nameof(DarlingDataReader.LatestMemoryClerksSql) => DarlingDataReader.LatestMemoryClerksSql, + nameof(DarlingDataReader.LatestFileIoStatsSql) => DarlingDataReader.LatestFileIoStatsSql, + nameof(DarlingDataReader.LatestPerfmonStatsSql) => DarlingDataReader.LatestPerfmonStatsSql, + nameof(DarlingCurrentConfigReader.ServerConfigSql) => DarlingCurrentConfigReader.ServerConfigSql, + nameof(DarlingCurrentConfigReader.DatabaseConfigSql) => DarlingCurrentConfigReader.DatabaseConfigSql, + nameof(DarlingCurrentConfigReader.TraceFlagsSql) => DarlingCurrentConfigReader.TraceFlagsSql, + nameof(DarlingConfigHistoryReader.DatabaseScopedConfigSql) => DarlingConfigHistoryReader.DatabaseScopedConfigSql, + nameof(DarlingConfigHistoryReader.QueryStoreHealthSql) => DarlingConfigHistoryReader.QueryStoreHealthSql, + nameof(DarlingMemoryGrantReader.ResourceSemaphoreWindowSql) => DarlingMemoryGrantReader.ResourceSemaphoreWindowSql, + nameof(DarlingMemoryGrantReader.MemoryGrantsWindowSql) => DarlingMemoryGrantReader.MemoryGrantsWindowSql, + _ => throw new ArgumentOutOfRangeException(nameof(sqlName), sqlName, "not a read this census names"), + }; + + private static MethodInfo ToolMethod(Type type, string toolName) => type + .GetMethods(BindingFlags.Public | BindingFlags.Static) + .Single(m => m.GetCustomAttribute()?.Name == toolName); + + private static string DarlingFileOf(Type type) => + $"Darling/PerformanceMonitor.Darling.Service/Mcp/{type.Name}.cs"; + + /// The source from one tool's [McpServerTool(Name = "…")] attribute to the next tool's, or + /// to the end of the file — the span that holds its description, parameters and payload. + private static string ToolBody(string source, string toolName) + { + var marker = $"[McpServerTool(Name = \"{toolName}\")"; + var start = source.IndexOf(marker, StringComparison.Ordinal); + Assert.True(start >= 0, $"no tool named {toolName} in the source"); + var next = source.IndexOf("[McpServerTool(", start + marker.Length, StringComparison.Ordinal); + return next < 0 ? source[start..] : source[start..next]; + } + + /// The tool's description: the attribute's literal, or the const it names (the aligned tools + /// describe themselves through a const so the two SKUs' texts can be pinned equal). + private static string DescriptionOf(string body, string label) + { + /* Anchored on the TOOL attribute the body starts with, so a parameter's [Description("Server name…")] + further down can never be read as the tool's — which is exactly what a bare `Description(` search + did on the const-described tools when this file was first executed. */ + var attribute = Regex.Match(body, @"\A\[McpServerTool\(Name = ""[a-z_0-9]+""\), Description\((?:\s*)(?:""((?:[^""\\]|\\.)*)""|(\w+))\)\]"); + Assert.True(attribute.Success, $"{label}: could not locate the tool's Description on its McpServerTool attribute"); + if (attribute.Groups[1].Success) + { + return attribute.Groups[1].Value; + } + + var constName = attribute.Groups[2].Value; + + /* The const lives in the same file, above the attribute; the body slice starts AT the attribute, so it + is not in the body — resolve it from the file the label points at. */ + var (type, toolName, liteFile, _) = LatestTools.Single(t => label.EndsWith(t.ToolName, StringComparison.Ordinal)); + return label.StartsWith("Darling", StringComparison.Ordinal) + ? (string)type.GetField(constName, BindingFlags.NonPublic | BindingFlags.Public | BindingFlags.Static)!.GetRawConstantValue()! + : LiteConstLiteral(liteFile, constName); + } + + /// A Lite const string's literal, read from source: internal const string Name =\n "…";. + private static string LiteConstLiteral(string liteFile, string constName) + { + var source = ReadRepoFileLf(liteFile.Split('/')); + var m = Regex.Match(source, $@"const string {Regex.Escape(constName)}\s*=\s*""((?:[^""\\]|\\.)*)"";"); + Assert.True(m.Success, $"{liteFile}: no `const string {constName} = \"…\";`"); + return Regex.Unescape(m.Groups[1].Value); + } + + /// Lite's advertised parameter names for a tool, read off the MASKED signature (comments and + /// string literals blanked, so prose inside a [Description] cannot read as a parameter) — the idiom + /// DarlingMcpDataToolsTests.LiteMcpParamNames established. Every MCP parameter carries a default + /// and the injected services do not, so the defaulted ones in order ARE the contract. + private static string[] LiteParamNames(string liteFile, string toolName) + { + var raw = ReadRepoFileLf(liteFile.Split('/')); + var attribute = raw.IndexOf($"Name = \"{toolName}\"", StringComparison.Ordinal); + Assert.True(attribute > 0, $"{liteFile}: no tool named {toolName}"); + + var masked = CSharpSourceWalker.StripCommentsAndStrings(raw); + Assert.Equal(raw.Length, masked.Length); + + var declaration = masked.IndexOf("public static", attribute, StringComparison.Ordinal); + var signature = masked.IndexOf('(', declaration); + var body = masked.IndexOf('{', signature); + Assert.True(body > signature, $"{liteFile}: could not find the end of {toolName}'s signature"); + + return Regex.Matches(masked[signature..body], @"(\w+)\s*=\s*[^,)]+") + .Select(m => m.Groups[1].Value) + .ToArray(); + } + + /// Comments removed, so a comment that NAMES a key is not read as the key. Line and block comments + /// only; string literals stay, because the payload keys under test are not in strings. + private static string Strip(string source) => + Regex.Replace(Regex.Replace(source, @"/\*.*?\*/", string.Empty, RegexOptions.Singleline), @"//[^\n]*", string.Empty); +} + +/// +/// Gated (DARLING_TEST_PG) live round-trips for the stamps: the seeded row's stamp comes back as +/// captured_at, age_seconds is measured against the anchor the caller sent (never the wall clock), +/// a SearchBound tool refuses a snapshot older than its search span, the memory-grant window sees a storm the +/// latest snapshot does not, and the latch band names its interval. Anchored in the past on purpose: the +/// assertions are equalities, and an anchor of "now" would make every age a race. +/// +[Collection("live-postgres")] +public sealed class McpLatestSnapshotStampLivePostgresTests +{ + private const string ServerName = "darling-mcp-latest-stamp-e2e"; + private static readonly int ServerId = ServerIdHelper.GetDeterministicHashCode(ServerName); + private static string? ConnectionString => Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + + private static readonly string[] Tables = + [ + "memory_grant_stats", "cpu_scheduler_stats", "server_config", "trace_flags", "memory_clerks", "latch_stats", + "memory_stats", "cpu_utilization_stats", "collection_log", + ]; + + [Fact] + public async Task LatestReads_SayWhenTheyWereCaptured_AgainstLivePostgres() + { + var cs = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(cs), "Set DARLING_TEST_PG to a Postgres connection string to run the live latest-stamp test."); + + var ct = TestContext.Current.CancellationToken; + using var connection = new NpgsqlConnection(cs); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + await DeleteRowsAsync(connection, ct); + await using var postgres = NpgsqlDataSource.Create(cs!); + + var bodySucceeded = false; + try + { + await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); + + /* Every row sits in the past; the anchor is base + 5 min, so every age below is exact. */ + var @base = DarlingMcpTestData.TruncateToSeconds(DateTime.UtcNow).AddHours(-2); + var anchor = @base.AddMinutes(5).ToString("o") + "Z"; + + /* ── memory grants: a storm 30 minutes before a calm latest snapshot ── */ + foreach (var (t, waiters, timeouts, granted) in new[] { (@base.AddMinutes(-30), 12, 3L, 6000m), (@base, 0, 0L, 500m) }) + { + await DarlingMcpTestData.ExecAsync(connection, ct, + @"INSERT INTO memory_grant_stats (collection_id, collection_time, server_id, server_name, resource_semaphore_id, pool_id, target_memory_mb, max_target_memory_mb, total_memory_mb, available_memory_mb, granted_memory_mb, used_memory_mb, grantee_count, waiter_count, timeout_error_count, forced_grant_count, timeout_error_count_delta, forced_grant_count_delta) +VALUES ($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,$11,$12,$13,$14,$15,$16,$17,$18)", + CollectionIdGenerator.Next(), t, ServerId, ServerName, (short)0, 2, 8000m, 12000m, 8000m, 8000m - granted, granted, granted, 3, waiters, 4L, 2L, timeouts, 0L); + } + + var semaphore = Parse(await DarlingMcpMemoryGrantTools.GetResourceSemaphore(postgres, ServerName, 1, as_of: anchor)); + Assert.Equal(Stamp(@base), semaphore.GetProperty("captured_at").GetString()); + Assert.Equal(300, semaphore.GetProperty("age_seconds").GetInt64()); + var latestRow = Assert.Single(semaphore.GetProperty("grants").EnumerateArray()); + Assert.Equal(0, latestRow.GetProperty("waiter_count").GetInt32()); + var windowRow = Assert.Single(semaphore.GetProperty("window").EnumerateArray()); + Assert.Equal(2, windowRow.GetProperty("snapshots_in_window").GetInt64()); + Assert.Equal(12, windowRow.GetProperty("peak_waiter_count").GetInt64()); + Assert.Equal(Stamp(@base.AddMinutes(-30)), windowRow.GetProperty("peak_waiters_at").GetString()); + Assert.Equal(3, windowRow.GetProperty("timeout_errors_in_window").GetInt64()); + Assert.Equal(6000d, windowRow.GetProperty("peak_granted_memory_mb").GetDouble()); + Assert.Equal(2000d, windowRow.GetProperty("min_available_memory_mb").GetDouble()); + Assert.Equal(Stamp(@base), windowRow.GetProperty("last_snapshot_at").GetString()); + + var grants = Parse(await DarlingMcpMemoryGrantTools.GetMemoryGrants(postgres, ServerName, 1, as_of: anchor)); + Assert.Equal(Stamp(@base), grants.GetProperty("captured_at").GetString()); + Assert.Equal(300, grants.GetProperty("age_seconds").GetInt64()); + var poolWindow = Assert.Single(grants.GetProperty("window").EnumerateArray()); + Assert.Equal(JsonValueKind.Null, poolWindow.GetProperty("resource_semaphore_id").ValueKind); + Assert.Equal(12, poolWindow.GetProperty("peak_waiter_count").GetInt64()); + + /* ── scheduler: one snapshot three hours before the anchor — inside a 4 h search, outside a 1 h one ── */ + var schedulerAt = @base.AddMinutes(5).AddHours(-3); + await DarlingMcpTestData.ExecAsync(connection, ct, + @"INSERT INTO cpu_scheduler_stats (collection_id, collection_time, server_id, server_name, max_workers_count, scheduler_count, cpu_count, total_runnable_tasks_count, total_work_queue_count, total_current_workers_count, avg_runnable_tasks_count, total_active_request_count, total_queued_request_count, total_blocked_task_count, total_active_parallel_thread_count, runnable_percent, worker_thread_exhaustion_warning, runnable_tasks_warning, blocked_tasks_warning, queued_requests_warning, total_physical_memory_kb, available_physical_memory_kb, physical_memory_pressure_warning, total_node_count, nodes_online_count, offline_cpu_count, offline_cpu_warning) +VALUES ($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,$11,$12,$13,$14,$15,$16,$17,$18,$19,$20,$21,$22,$23,$24,$25,$26,$27)", + CollectionIdGenerator.Next(), schedulerAt, ServerId, ServerName, 512, 8, 8, 60, 5L, 100, 7.5m, 40, 12, 2, 20L, 12.5m, false, true, false, true, 65536000L, 32768000L, false, 1, 1, 0, false); + + var scheduler = Parse(await DarlingMcpPlanCacheSchedulerTools.GetCpuSchedulerPressure(postgres, ServerName, 4, as_of: anchor)); + Assert.Equal(Stamp(schedulerAt), scheduler.GetProperty("captured_at").GetString()); + Assert.Equal(3 * 3600, scheduler.GetProperty("age_seconds").GetInt64()); + Assert.StartsWith("CRITICAL", scheduler.GetProperty("pressure_level").GetString(), StringComparison.Ordinal); + Assert.Equal("unavailable", DarlingMcpTestData.StatusOf(await DarlingMcpPlanCacheSchedulerTools.GetCpuSchedulerPressure(postgres, ServerName, 1, as_of: anchor))); + + /* ── config: captured on connect, stamped with that connect ── */ + var connectAt = @base.AddDays(-3); + await DarlingMcpTestData.ExecAsync(connection, ct, + @"INSERT INTO server_config (config_id, capture_time, server_id, server_name, configuration_name, value_configured, value_in_use, is_dynamic, is_advanced) +VALUES ($1,$2,$3,$4,$5,$6,$7,$8,$9)", + CollectionIdGenerator.Next(), connectAt, ServerId, ServerName, "max degree of parallelism", 4L, 4L, true, true); + await DarlingMcpTestData.ExecAsync(connection, ct, + @"INSERT INTO trace_flags (config_id, capture_time, server_id, server_name, trace_flag, status, is_global, is_session) +VALUES ($1,$2,$3,$4,$5,$6,$7,$8)", + CollectionIdGenerator.Next(), connectAt, ServerId, ServerName, 3226, true, true, false); + + Assert.Equal(Stamp(connectAt), Parse(await DarlingMcpConfigTools.GetServerConfig(postgres, ServerName)).GetProperty("captured_at").GetString()); + Assert.Equal(Stamp(connectAt), Parse(await DarlingMcpConfigTools.GetTraceFlags(postgres, ServerName)).GetProperty("captured_at").GetString()); + + /* ── clerks: the newest snapshot's stamp, not the older one's ── */ + foreach (var t in new[] { @base.AddMinutes(-10), @base }) + { + await DarlingMcpTestData.ExecAsync(connection, ct, + @"INSERT INTO memory_clerks (collection_id, collection_time, server_id, server_name, clerk_type, memory_mb) +VALUES ($1,$2,$3,$4,$5,$6)", CollectionIdGenerator.Next(), t, ServerId, ServerName, "MEMORYCLERK_SQLBUFFERPOOL", 40000m); + } + Assert.Equal(Stamp(@base), Parse(await DarlingMcpDataTools.GetMemoryClerks(postgres, ServerName)).GetProperty("captured_at").GetString()); + + /* ── latch: hot earlier, quiet now — LOW severity beside a large window total, and the band says why ── */ + foreach (var (t, delta) in new[] { (@base.AddMinutes(-20), 20000L), (@base, 100L) }) + { + await DarlingMcpTestData.ExecAsync(connection, ct, + @"INSERT INTO latch_stats (collection_id, collection_time, server_id, server_name, latch_class, waiting_requests_count, wait_time_ms, max_wait_time_ms, delta_waiting_requests_count, delta_wait_time_ms, delta_max_wait_time_ms, sample_interval_seconds) +VALUES ($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,$11,$12)", + CollectionIdGenerator.Next(), t, ServerId, ServerName, "ACCESS_METHODS_DATASET_PARENT", 1000L, 20100L, 50L, 100L, delta, 5L, 60); + } + var latch = Assert.Single(Parse(await DarlingMcpLatchSpinlockTools.GetLatchStats(postgres, ServerName, 1, as_of: anchor)).GetProperty("latches").EnumerateArray()); + Assert.Equal(20100, latch.GetProperty("total_delta_wait_time_ms").GetInt64()); + Assert.Equal("LOW", latch.GetProperty("severity").GetString()); + var band = latch.GetProperty("severity_banded_from"); + Assert.Equal(100, band.GetProperty("delta_wait_time_ms").GetInt64()); + Assert.Equal(60d, band.GetProperty("interval_seconds").GetDouble()); + Assert.Equal(Stamp(@base), band.GetProperty("captured_at").GetString()); + + /* ── server summary: three clocks — a stale CPU row under a fresh collection log ── */ + var cpuAt = @base.AddDays(-1); + await DarlingMcpTestData.ExecAsync(connection, ct, + @"INSERT INTO cpu_utilization_stats (collection_id, collection_time, server_id, server_name, sample_time, sqlserver_cpu_utilization, other_process_cpu_utilization) +VALUES ($1,$2,$3,$4,$5,$6,$7)", CollectionIdGenerator.Next(), cpuAt, ServerId, ServerName, cpuAt, 42, 3); + await DarlingMcpTestData.ExecAsync(connection, ct, + @"INSERT INTO memory_stats (collection_id, collection_time, server_id, server_name, total_physical_memory_mb, available_physical_memory_mb, total_server_memory_mb) +VALUES ($1,$2,$3,$4,$5,$6,$7)", CollectionIdGenerator.Next(), @base, ServerId, ServerName, 65536m, 8192m, 40000m); + await DarlingMcpTestData.ExecAsync(connection, ct, + @"INSERT INTO collection_log (log_id, server_id, server_name, collector_name, collection_time, duration_ms, status, rows_collected) +VALUES ($1,$2,$3,$4,$5,120,'SUCCESS',7)", CollectionIdGenerator.Next(), ServerId, ServerName, "memory_stats", @base.AddMinutes(1)); + + var summary = Parse(await DarlingMcpHealthTools.GetServerSummary(postgres, ServerName)); + Assert.Equal(Stamp(cpuAt), summary.GetProperty("cpu_captured_at").GetString()); + Assert.Equal(Stamp(@base), summary.GetProperty("memory_captured_at").GetString()); + Assert.Equal(Stamp(@base.AddMinutes(1)), summary.GetProperty("last_collection").GetString()); + Assert.Equal(DarlingHealthReader.ServerSummaryCountsWindowHours, summary.GetProperty("counts_window_hours").GetInt32()); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, async (cleanup, cleanupCt) => + await DeleteRowsAsync(cleanup, cleanupCt)); + } + } + + private static JsonElement Parse(string json) + { + Assert.False(json.StartsWith("Error during", StringComparison.Ordinal), $"tool returned an error: {json}"); + var root = JsonDocument.Parse(json).RootElement.Clone(); + Assert.False(root.TryGetProperty("status", out _), "expected a data-bearing payload, got a status envelope: " + json); + return root; + } + + /// A seeded naive-UTC instant as the tools emit it: ToString("o") on a Kind=Unspecified + /// value, so no Z. + private static string Stamp(DateTime naiveUtc) => DateTime.SpecifyKind(naiveUtc, DateTimeKind.Unspecified).ToString("o"); + + private static async Task DeleteRowsAsync(NpgsqlConnection connection, System.Threading.CancellationToken ct) + { + var sql = string.Join(" ", Tables.Select(t => $"DELETE FROM {t} WHERE server_id = {ServerId};")) + + $" DELETE FROM servers WHERE server_id = {ServerId};"; + using var cleanup = new NpgsqlCommand(sql, connection); + await cleanup.ExecuteNonQueryAsync(ct); + } +} diff --git a/Darling/Darling.Tests/RepoFileAdoptionTests.cs b/Darling/Darling.Tests/RepoFileAdoptionTests.cs index 3d21571a0..afc6f269c 100644 --- a/Darling/Darling.Tests/RepoFileAdoptionTests.cs +++ b/Darling/Darling.Tests/RepoFileAdoptionTests.cs @@ -116,6 +116,10 @@ public sealed class RepoFileAdoptionTests "FleetCardCollectionStaleNamesItsPopulationTests.cs", "FleetPageAttentionFilterTests.cs", "LockedModeRestoreCoverageTests.cs", + /* #3541 A10: its top-level-key discriminator anchors on the line break BEFORE the key (a per-row + collection_time inside a Select is indented deeper and must not match), and its tool-body slicing + keys on attribute text either side of one. */ + "McpLatestSnapshotStampTests.cs", /* #3541 A3: its Lite-description anchor spans the line break between `Description(` and the string on get_plan_corrections, and its tool-body slicing keys on attribute text either side of one. */ diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingWebEndpoints.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingWebEndpoints.cs index 34c972fd7..9cf5d2486 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingWebEndpoints.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingWebEndpoints.cs @@ -1990,7 +1990,7 @@ private static CatalogRead R(string category, string description, params Catalog ["get_table_index_sizes"] = R(CatObjects, "Per-table/index size breakdown.", PServer()), /* ── plan cache / scheduler (DarlingMcpPlanCacheSchedulerTools) ── */ - ["get_cpu_scheduler_pressure"] = R(CatPlanCache, "CPU scheduler pressure indicators.", PServer()), + ["get_cpu_scheduler_pressure"] = R(CatPlanCache, "CPU scheduler pressure indicators from the newest snapshot within the window.", PServer(), PHours(24), PAsOf()), ["get_plan_cache_bloat"] = R(CatPlanCache, "Plan-cache bloat / single-use plan indicators.", PServer(), PHours(24), PAsOf()), /* ── jobs (DarlingMcpJobTools) ── */ @@ -2696,7 +2696,7 @@ cannot parse exactly rather than silently matching nothing. */ ["get_table_index_sizes"] = (c, pg, an) => DarlingMcpObjectStatsTools.GetTableIndexSizes(pg, Server(c)), /* ── plan cache / scheduler ── */ - ["get_cpu_scheduler_pressure"] = (c, pg, an) => DarlingMcpPlanCacheSchedulerTools.GetCpuSchedulerPressure(pg, Server(c)), + ["get_cpu_scheduler_pressure"] = (c, pg, an) => DarlingMcpPlanCacheSchedulerTools.GetCpuSchedulerPressure(pg, Server(c), Hours(c, 24), as_of: AsOf(c)), ["get_plan_cache_bloat"] = (c, pg, an) => DarlingMcpPlanCacheSchedulerTools.GetPlanCacheBloat(pg, Server(c), Hours(c, 24), as_of: AsOf(c)), /* ── jobs ── */ diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingConfigHistoryReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingConfigHistoryReader.cs index 3ddd5288d..c16183c7d 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingConfigHistoryReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingConfigHistoryReader.cs @@ -200,19 +200,21 @@ FROM v_trace_flags /* ─────────────────────────── database scoped config (latest snapshot) ─────────────────────────── */ /// The latest sys.database_scoped_configurations snapshot — the viewer's - /// DatabaseScopedConfigSql. $1 server_id. + /// DatabaseScopedConfigSql plus the trailing capture_time (#3541 A10: captured on connect, so + /// the tool must be able to say how old "current" is). $1 server_id. public const string DatabaseScopedConfigSql = """ - SELECT database_name, configuration_name, value, value_for_secondary + SELECT database_name, configuration_name, value, value_for_secondary, capture_time FROM v_database_scoped_config WHERE server_id = $1 AND capture_time = (SELECT MAX(capture_time) FROM v_database_scoped_config WHERE server_id = $1) ORDER BY database_name, configuration_name """; - public static async Task> GetLatestDatabaseScopedConfigAsync( + public static async Task> GetLatestDatabaseScopedConfigAsync( NpgsqlDataSource postgres, int serverId, CancellationToken cancellationToken = default) { var rows = new List(); + DateTime? capturedAt = null; await using var command = postgres.CreateCommand(DatabaseScopedConfigSql); command.CommandTimeout = McpCommandDeadlines.ReadSeconds; DarlingMcpReadParameters.AddInt(command, serverId); @@ -224,9 +226,10 @@ public static async Task> GetLatestDatabaseSco reader.IsDBNull(1) ? "" : reader.GetString(1), reader.IsDBNull(2) ? null : reader.GetString(2), reader.IsDBNull(3) ? null : reader.GetString(3))); + capturedAt ??= reader.GetDateTime(4); } - return rows; + return new LatestSnapshot(capturedAt, rows); } /* ─────────────────────────── query store health (latest snapshot) ─────────────────────────── */ @@ -236,17 +239,18 @@ public static async Task> GetLatestDatabaseSco /// scoped-config read above). Unlike the config-family reads this table is HOURLY, not on-connect, /// so "latest" here is at most an hour old on a healthy schedule. $1 server_id. public const string QueryStoreHealthSql = """ - SELECT database_name, actual_state, desired_state, readonly_reason, current_storage_size_mb, max_storage_size_mb, size_based_cleanup_mode, stale_query_threshold_days, max_plans_per_query, interval_length_minutes + SELECT database_name, actual_state, desired_state, readonly_reason, current_storage_size_mb, max_storage_size_mb, size_based_cleanup_mode, stale_query_threshold_days, max_plans_per_query, interval_length_minutes, capture_time FROM v_query_store_health WHERE server_id = $1 AND capture_time = (SELECT MAX(capture_time) FROM v_query_store_health WHERE server_id = $1) ORDER BY database_name """; - public static async Task> GetLatestQueryStoreHealthAsync( + public static async Task> GetLatestQueryStoreHealthAsync( NpgsqlDataSource postgres, int serverId, CancellationToken cancellationToken = default) { var rows = new List(); + DateTime? capturedAt = null; await using var command = postgres.CreateCommand(QueryStoreHealthSql); command.CommandTimeout = McpCommandDeadlines.ReadSeconds; DarlingMcpReadParameters.AddInt(command, serverId); @@ -264,8 +268,9 @@ public static async Task> GetLatestQueryStoreHealt reader.IsDBNull(7) ? 0L : reader.GetInt64(7), reader.IsDBNull(8) ? 0L : reader.GetInt64(8), reader.IsDBNull(9) ? 0L : reader.GetInt64(9))); + capturedAt ??= reader.GetDateTime(10); } - return rows; + return new LatestSnapshot(capturedAt, rows); } } diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingCurrentConfigReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingCurrentConfigReader.cs index dc7a422de..4073e4af0 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingCurrentConfigReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingCurrentConfigReader.cs @@ -37,20 +37,23 @@ public sealed record ServerConfigReadRow( public bool ValuesMatch => ValueConfigured == ValueInUse; } - /// Latest sys.configurations snapshot for one server — the viewer's ServerConfigSql. + /// Latest sys.configurations snapshot for one server — the viewer's ServerConfigSql plus + /// the trailing capture_time (#3541 A10): config is captured ON CONNECT, so the "current" value this + /// read serves can be as old as the last successful connect, and the tool must be able to say so. /// $1 server_id. public const string ServerConfigSql = """ - SELECT configuration_name, value_configured, value_in_use, is_dynamic, is_advanced + SELECT configuration_name, value_configured, value_in_use, is_dynamic, is_advanced, capture_time FROM v_server_config WHERE server_id = $1 AND capture_time = (SELECT MAX(capture_time) FROM v_server_config WHERE server_id = $1) ORDER BY configuration_name """; - public static async Task> GetLatestServerConfigAsync( + public static async Task> GetLatestServerConfigAsync( NpgsqlDataSource postgres, int serverId, CancellationToken cancellationToken = default) { var rows = new List(); + DateTime? capturedAt = null; await using var command = postgres.CreateCommand(ServerConfigSql); command.CommandTimeout = McpCommandDeadlines.ReadSeconds; DarlingMcpReadParameters.AddInt(command, serverId); @@ -63,9 +66,10 @@ public static async Task> GetLatestServerConfigAsync( reader.IsDBNull(2) ? 0 : reader.GetInt64(2), !reader.IsDBNull(3) && reader.GetBoolean(3), !reader.IsDBNull(4) && reader.GetBoolean(4))); + capturedAt ??= reader.GetDateTime(5); } - return rows; + return new LatestSnapshot(capturedAt, rows); } /* ─────────────────────────── database config (sys.databases) ─────────────────────────── */ @@ -80,7 +84,8 @@ public sealed record DatabaseConfigReadRow( bool IsMemoryOptimizedEnabled, bool IsOptimizedLockingOn); /* 28 columns in the viewer's / Lite's exact SELECT order — the reader below maps them by incrementing - ordinal, so this list's order is load-bearing and must stay byte-identical. */ + ordinal, so this list's order is load-bearing and must stay byte-identical. capture_time is APPENDED + as a 29th column (#3541 A10) and read by explicit ordinal 28, so the 28-column mapping is untouched. */ public const string DatabaseConfigSql = """ SELECT database_name, state_desc, compatibility_level, collation_name, recovery_model, is_read_only, is_auto_close_on, is_auto_shrink_on, @@ -89,17 +94,23 @@ public sealed record DatabaseConfigReadRow( is_query_store_on, is_encrypted, is_trustworthy_on, is_db_chaining_on, is_broker_enabled, is_cdc_enabled, is_mixed_page_allocation_on, log_reuse_wait_desc, page_verify_option, target_recovery_time_seconds, delayed_durability, - is_accelerated_database_recovery_on, is_memory_optimized_enabled, is_optimized_locking_on + is_accelerated_database_recovery_on, is_memory_optimized_enabled, is_optimized_locking_on, + capture_time FROM v_database_config WHERE server_id = $1 AND capture_time = (SELECT MAX(capture_time) FROM v_database_config WHERE server_id = $1) ORDER BY database_name """; - public static async Task> GetLatestDatabaseConfigAsync( + /// The ordinal of the appended capture_time column in — one + /// past the 28-column block the incrementing mapping consumes. + private const int DatabaseConfigCaptureTimeOrdinal = 28; + + public static async Task> GetLatestDatabaseConfigAsync( NpgsqlDataSource postgres, int serverId, CancellationToken cancellationToken = default) { var rows = new List(); + DateTime? capturedAt = null; await using var command = postgres.CreateCommand(DatabaseConfigSql); command.CommandTimeout = McpCommandDeadlines.ReadSeconds; DarlingMcpReadParameters.AddInt(command, serverId); @@ -138,29 +149,32 @@ DatabaseConfigSql binds to the same fields — see the byte-identical note above !reader.IsDBNull(++ordinal) && reader.GetBoolean(ordinal), !reader.IsDBNull(++ordinal) && reader.GetBoolean(ordinal), !reader.IsDBNull(++ordinal) && reader.GetBoolean(ordinal))); + capturedAt ??= reader.GetDateTime(DatabaseConfigCaptureTimeOrdinal); } - return rows; + return new LatestSnapshot(capturedAt, rows); } /* ─────────────────────────── trace flags (DBCC TRACESTATUS) ─────────────────────────── */ public sealed record TraceFlagReadRow(int TraceFlag, bool Status, bool IsGlobal, bool IsSession); - /// Latest trace-flags snapshot for one server — the viewer's TraceFlagsSql. $1 server_id. - /// A row exists only while a flag is enabled, so an empty result means no active flags at the last capture. + /// Latest trace-flags snapshot for one server — the viewer's TraceFlagsSql plus the trailing + /// capture_time (#3541 A10). $1 server_id. A row exists only while a flag is enabled, so an empty + /// result means no active flags at the last capture — and, having no row, no stamp either. public const string TraceFlagsSql = """ - SELECT trace_flag, status, is_global, is_session + SELECT trace_flag, status, is_global, is_session, capture_time FROM v_trace_flags WHERE server_id = $1 AND capture_time = (SELECT MAX(capture_time) FROM v_trace_flags WHERE server_id = $1) ORDER BY trace_flag """; - public static async Task> GetLatestTraceFlagsAsync( + public static async Task> GetLatestTraceFlagsAsync( NpgsqlDataSource postgres, int serverId, CancellationToken cancellationToken = default) { var rows = new List(); + DateTime? capturedAt = null; await using var command = postgres.CreateCommand(TraceFlagsSql); command.CommandTimeout = McpCommandDeadlines.ReadSeconds; DarlingMcpReadParameters.AddInt(command, serverId); @@ -172,8 +186,9 @@ public static async Task> GetLatestTraceFlagsAsync( !reader.IsDBNull(1) && reader.GetBoolean(1), !reader.IsDBNull(2) && reader.GetBoolean(2), !reader.IsDBNull(3) && reader.GetBoolean(3))); + capturedAt ??= reader.GetDateTime(4); } - return rows; + return new LatestSnapshot(capturedAt, rows); } } diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingDataReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingDataReader.cs index 633152fcd..d370e1fa3 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingDataReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingDataReader.cs @@ -568,19 +568,22 @@ LIMIT 1 /// /// The latest memory-clerk breakdown — Lite's GetLatestMemoryClerksAsync: every clerk at the /// newest collection, heaviest first. memory_mb is numeric(18,2) → double precision. $1 server_id. + /// collection_time rides along on every row (#3541 A10) so the tool can say WHEN the snapshot it + /// serves was taken — the same statement as the rows, never a second read that could stamp the next one. /// public const string LatestMemoryClerksSql = """ - SELECT clerk_type, CAST(memory_mb AS double precision) + SELECT clerk_type, CAST(memory_mb AS double precision), collection_time FROM v_memory_clerks WHERE server_id = $1 AND collection_time = (SELECT MAX(collection_time) FROM v_memory_clerks WHERE server_id = $1) ORDER BY memory_mb DESC """; - public static async Task> GetLatestMemoryClerksAsync( + public static async Task> GetLatestMemoryClerksAsync( NpgsqlDataSource postgres, int serverId, CancellationToken cancellationToken = default) { var rows = new List(); + DateTime? capturedAt = null; await using var command = postgres.CreateCommand(LatestMemoryClerksSql); command.CommandTimeout = McpCommandDeadlines.ReadSeconds; AddInt(command, serverId); @@ -590,9 +593,10 @@ public static async Task> GetLatestMemoryClerksAsync( rows.Add(new MemoryClerkRow( reader.GetString(0), reader.IsDBNull(1) ? 0 : reader.GetDouble(1))); + capturedAt ??= reader.GetDateTime(2); } - return rows; + return new LatestSnapshot(capturedAt, rows); } /* ─────────────────────────── file I/O ─────────────────────────── */ @@ -600,7 +604,8 @@ public static async Task> GetLatestMemoryClerksAsync( /// /// The latest file-I/O snapshot per database file — Lite's GetLatestFileIoStatsAsync, ordered /// by total stall descending; avg latency (stall/op) is computed by the tool. size_mb is - /// numeric → double precision; the delta columns are bigint. $1 server_id. + /// numeric → double precision; the delta columns are bigint. $1 server_id. collection_time + /// is the trailing column (#3541 A10): the snapshot's stamp, read once and published as captured_at. /// public const string LatestFileIoStatsSql = """ SELECT @@ -615,17 +620,19 @@ public static async Task> GetLatestMemoryClerksAsync( delta_write_bytes, delta_stall_read_ms, delta_stall_write_ms, - sample_interval_seconds + sample_interval_seconds, + collection_time FROM v_file_io_stats WHERE server_id = $1 AND collection_time = (SELECT MAX(collection_time) FROM v_file_io_stats WHERE server_id = $1) ORDER BY (delta_stall_read_ms + delta_stall_write_ms) DESC """; - public static async Task> GetLatestFileIoStatsAsync( + public static async Task> GetLatestFileIoStatsAsync( NpgsqlDataSource postgres, int serverId, CancellationToken cancellationToken = default) { var rows = new List(); + DateTime? capturedAt = null; await using var command = postgres.CreateCommand(LatestFileIoStatsSql); command.CommandTimeout = McpCommandDeadlines.ReadSeconds; AddInt(command, serverId); @@ -645,9 +652,10 @@ public static async Task> GetLatestFileIoStatsAsync( reader.IsDBNull(9) ? 0 : reader.GetInt64(9), reader.IsDBNull(10) ? 0 : reader.GetInt64(10), reader.IsDBNull(11) ? null : reader.GetInt32(11))); + capturedAt ??= reader.GetDateTime(12); } - return rows; + return new LatestSnapshot(capturedAt, rows); } /* ─────────────────────────── tempdb ─────────────────────────── */ @@ -706,24 +714,27 @@ public static async Task> GetTempDbTrendAsync( /// /// The latest perfmon counters — Lite's GetLatestPerfmonStatsAsync: counter_name / - /// instance_name / cntr_value / delta_cntr_value at the newest collection. $1 server_id. + /// instance_name / cntr_value / delta_cntr_value at the newest collection, with that collection's + /// collection_time trailing (#3541 A10, published once as captured_at). $1 server_id. /// public const string LatestPerfmonStatsSql = """ SELECT counter_name, instance_name, cntr_value, - delta_cntr_value + delta_cntr_value, + collection_time FROM v_perfmon_stats WHERE server_id = $1 AND collection_time = (SELECT MAX(collection_time) FROM v_perfmon_stats WHERE server_id = $1) ORDER BY counter_name """; - public static async Task> GetLatestPerfmonStatsAsync( + public static async Task> GetLatestPerfmonStatsAsync( NpgsqlDataSource postgres, int serverId, CancellationToken cancellationToken = default) { var rows = new List(); + DateTime? capturedAt = null; await using var command = postgres.CreateCommand(LatestPerfmonStatsSql); command.CommandTimeout = McpCommandDeadlines.ReadSeconds; AddInt(command, serverId); @@ -735,9 +746,10 @@ public static async Task> GetLatestPerfmonStatsAsync( reader.IsDBNull(1) ? "" : reader.GetString(1), reader.IsDBNull(2) ? 0 : reader.GetInt64(2), reader.IsDBNull(3) ? 0 : reader.GetInt64(3))); + capturedAt ??= reader.GetDateTime(4); } - return rows; + return new LatestSnapshot(capturedAt, rows); } /* ─────────────────────────── top queries ─────────────────────────── */ diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingHealthReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingHealthReader.cs index b40f3c0d6..c7409049c 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingHealthReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingHealthReader.cs @@ -43,6 +43,16 @@ internal static class DarlingHealthReader public sealed record ServerSummaryReadResult( double? CpuPercent, double? MemoryMb, int BlockingCount, int DeadlockCount, DateTime? LastCollectionTime) { + /// The collection_time of the CPU snapshot came from (#3541 A10) + /// — its own clock, distinct from , which is the newest collection of + /// ANY collector for the server. init rather than positional so the positional shape existing + /// callers construct is unchanged. Null when there is no CPU row. + public DateTime? CpuCapturedAt { get; init; } + + /// The collection_time of the memory snapshot came from (#3541 + /// A10). Null when there is no memory row. + public DateTime? MemoryCapturedAt { get; init; } + /// True when the server has no collected data at all (no CPU/memory snapshot and no collection /// log) — the tool surfaces the #1224 "unavailable" miss instead of an all-zero card. public bool HasNoData => @@ -55,15 +65,16 @@ public sealed record ServerSummaryReadResult( /// sample_time as the within-batch tiebreak, and no time predicate because sample_time is /// the monitored server's local wall clock. public const string ServerSummaryCpuSql = @" -SELECT sqlserver_cpu_utilization +SELECT sqlserver_cpu_utilization, collection_time FROM v_cpu_utilization_stats WHERE server_id = $1 ORDER BY collection_time DESC, sample_time DESC LIMIT 1"; - /// Latest total server memory (MB) for one server. $1 server_id. + /// Latest total server memory (MB) for one server, with the snapshot's own collection_time + /// (#3541 A10). $1 server_id. public const string ServerSummaryMemorySql = @" -SELECT CAST(total_server_memory_mb AS double precision) +SELECT CAST(total_server_memory_mb AS double precision), collection_time FROM v_memory_stats WHERE server_id = $1 ORDER BY collection_time DESC @@ -84,18 +95,27 @@ ORDER BY collection_time DESC public const string ServerSummaryLastCollectionSql = @" SELECT MAX(collection_time) FROM v_collection_log WHERE server_id = $1"; + /// The span the blocking and deadlock counts cover, ending at the read's clock — Lite's window. + /// Published by the tool so "recent" has a number. + public const int ServerSummaryCountsWindowHours = 1; + /// /// One server's one-shot health summary — the viewer's GetServerSummaryAsync reduced to the subset /// the same-named Lite tool serves. Blocking / deadlock counts use a one-hour window (Lite's window); CPU - /// and memory take the newest snapshot. + /// and memory take the newest snapshot, each carrying its own collection_time (#3541 A10) — the + /// payload used to publish ONE clock (last_collection, the newest collection of ANY collector) beside + /// two figures it did not stamp, so a CPU row from a collector that died yesterday read as current + /// because the collection log was fresh from the collectors still running. /// public static async Task GetServerSummaryAsync( NpgsqlDataSource postgres, int serverId, CancellationToken cancellationToken = default) { - var windowStart = DateTime.UtcNow.AddHours(-1); + var windowStart = DateTime.UtcNow.AddHours(-ServerSummaryCountsWindowHours); double? cpuPercent = null; + DateTime? cpuCapturedAt = null; double? memoryMb = null; + DateTime? memoryCapturedAt = null; var blockingCount = 0; var deadlockCount = 0; DateTime? lastCollection = null; @@ -108,6 +128,7 @@ public static async Task GetServerSummaryAsync( if (await reader.ReadAsync(cancellationToken)) { cpuPercent = reader.IsDBNull(0) ? null : Convert.ToDouble(reader.GetValue(0)); + cpuCapturedAt = reader.GetDateTime(1); } } @@ -119,6 +140,7 @@ public static async Task GetServerSummaryAsync( if (await reader.ReadAsync(cancellationToken)) { memoryMb = reader.IsDBNull(0) ? null : Convert.ToDouble(reader.GetValue(0)); + memoryCapturedAt = reader.GetDateTime(1); } } @@ -160,7 +182,11 @@ public static async Task GetServerSummaryAsync( } } - return new ServerSummaryReadResult(cpuPercent, memoryMb, blockingCount, deadlockCount, lastCollection); + return new ServerSummaryReadResult(cpuPercent, memoryMb, blockingCount, deadlockCount, lastCollection) + { + CpuCapturedAt = cpuCapturedAt, + MemoryCapturedAt = memoryCapturedAt, + }; } /* ═══════════════════════════ daily summary ═══════════════════════════ */ diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingLatchSpinlockReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingLatchSpinlockReader.cs index 4aab3735d..b9c0cb696 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingLatchSpinlockReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingLatchSpinlockReader.cs @@ -43,10 +43,13 @@ internal static class DarlingLatchSpinlockReader /* ─────────────────────────── result rows ─────────────────────────── */ /// One latch class aggregated over the window: the summed deltas plus the latest interval's - /// per-second rate and last delta wait (the severity input). + /// per-second rate and last delta wait (the severity input). is the + /// span that last delta accrued over (#3541 A10 — the interval the severity band was computed from, published + /// beside the window totals so the two are distinguishable); null when the interval was unknowable. public sealed record LatchStatRow( string LatchClass, long TotalDeltaWaitTimeMs, long TotalDeltaWaitingRequests, - double? WaitsPerSecond, double? WaitMsPerSecond, long LatestDeltaWaitTimeMs, DateTime LatestCollectionTime); + double? WaitsPerSecond, double? WaitMsPerSecond, long LatestDeltaWaitTimeMs, DateTime LatestCollectionTime, + double? LatestIntervalSeconds); /// One spinlock aggregated over the window: the summed deltas plus the latest interval's /// per-second collision/spin rates. @@ -101,7 +104,10 @@ SELECT DISTINCT ON (latch_class) latch_class, delta_wait_time_ms AS latest_delta_wait_time_ms, CASE WHEN interval_seconds > 0 THEN CAST(delta_waiting_requests_count AS double precision) / interval_seconds END AS waits_per_second, - CASE WHEN interval_seconds > 0 THEN CAST(delta_wait_time_ms AS double precision) / interval_seconds END AS wait_ms_per_second + CASE WHEN interval_seconds > 0 THEN CAST(delta_wait_time_ms AS double precision) / interval_seconds END AS wait_ms_per_second, + /* #3541 A10: the span the severity-banded delta accrued over, so the tool can publish the + interval the band came from beside the window totals it does NOT come from. */ + CASE WHEN interval_seconds > 0 THEN CAST(interval_seconds AS double precision) END AS latest_interval_seconds FROM windowed ORDER BY latch_class, collection_time DESC ) @@ -112,7 +118,8 @@ FROM windowed l.waits_per_second, l.wait_ms_per_second, l.latest_delta_wait_time_ms, - a.latest_collection_time + a.latest_collection_time, + l.latest_interval_seconds FROM agg AS a JOIN latest AS l ON l.latch_class = a.latch_class ORDER BY a.total_delta_wait_time_ms DESC @@ -137,7 +144,8 @@ public static async Task> GetLatchStatsTopNAsync( reader.IsDBNull(3) ? null : reader.GetDouble(3), reader.IsDBNull(4) ? null : reader.GetDouble(4), reader.IsDBNull(5) ? 0 : reader.GetInt64(5), - reader.GetDateTime(6))); + reader.GetDateTime(6), + reader.IsDBNull(7) ? null : reader.GetDouble(7))); } return rows; diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpConfigHistoryTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpConfigHistoryTools.cs index f48837065..38a8a9fe9 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpConfigHistoryTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpConfigHistoryTools.cs @@ -187,7 +187,7 @@ public static async Task GetTraceFlagChanges( } } - [McpServerTool(Name = "get_database_scoped_config"), Description("Gets database-scoped configuration settings (sys.database_scoped_configurations). Shows MAXDOP, legacy CE, parameter sniffing, and other per-database settings.")] + [McpServerTool(Name = "get_database_scoped_config"), Description("Gets database-scoped configuration settings (sys.database_scoped_configurations). Shows MAXDOP, legacy CE, parameter sniffing, and other per-database settings. LATEST IS A TIME: captured when the collector connects, not on a schedule - captured_at is the instant these settings are as of.")] public static async Task GetDatabaseScopedConfig( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -198,14 +198,14 @@ public static async Task GetDatabaseScopedConfig( try { - var rows = await DarlingConfigHistoryReader.GetLatestDatabaseScopedConfigAsync(postgres, resolved.ServerId); - if (rows.Count == 0) + var snapshot = await DarlingConfigHistoryReader.GetLatestDatabaseScopedConfigAsync(postgres, resolved.ServerId); + if (snapshot.IsEmpty) return await DarlingEngineCapability.NotCollectedStatusAsync(postgres, resolved.ServerId, resolved.ServerName, "database_scoped_config") ?? McpHelpers.Status( "unavailable", "No database-scoped configuration data available. The config collector may not have run yet."); - IEnumerable filtered = rows; + IEnumerable filtered = snapshot.Rows; if (!string.IsNullOrEmpty(database_name)) filtered = filtered.Where(r => r.DatabaseName.Equals(database_name, StringComparison.OrdinalIgnoreCase)); @@ -225,6 +225,8 @@ public static async Task GetDatabaseScopedConfig( return JsonSerializer.Serialize(new { server = resolved.ServerName, + /* #3541 A10: the connect-time capture these settings are as of. */ + captured_at = snapshot.CapturedAt!.Value.ToString("o"), database_count = grouped.Count, databases = grouped }, McpHelpers.JsonOptions); @@ -235,7 +237,7 @@ public static async Task GetDatabaseScopedConfig( } } - [McpServerTool(Name = "get_query_store_health"), Description("Gets per-database Query Store health (sys.database_query_store_options): actual vs desired state, readonly_reason (decoded), storage used vs cap, cleanup mode and thresholds, and the runtime-stats interval length. The classic silent failure is desired READ_WRITE with actual READ_ONLY after the storage cap hit — check this when Query Store data looks stale or missing. Collected hourly; OFF is recorded as OFF (an absent database means not collected, never off).")] + [McpServerTool(Name = "get_query_store_health"), Description("Gets per-database Query Store health (sys.database_query_store_options): actual vs desired state, readonly_reason (decoded), storage used vs cap, cleanup mode and thresholds, and the runtime-stats interval length. The classic silent failure is desired READ_WRITE with actual READ_ONLY after the storage cap hit — check this when Query Store data looks stale or missing. Collected hourly; OFF is recorded as OFF (an absent database means not collected, never off). LATEST IS A TIME: this is the newest hourly capture, and captured_at is its instant.")] public static async Task GetQueryStoreHealth( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -246,14 +248,14 @@ public static async Task GetQueryStoreHealth( try { - var rows = await DarlingConfigHistoryReader.GetLatestQueryStoreHealthAsync(postgres, resolved.ServerId); - if (rows.Count == 0) + var snapshot = await DarlingConfigHistoryReader.GetLatestQueryStoreHealthAsync(postgres, resolved.ServerId); + if (snapshot.IsEmpty) return await DarlingEngineCapability.NotCollectedStatusAsync(postgres, resolved.ServerId, resolved.ServerName, "query_store_health") ?? McpHelpers.Status( "unavailable", "No Query Store health data available. The query_store_health collector runs hourly (SQL Server 2016+); a server with no rows either predates Query Store or has not completed a cycle yet."); - IEnumerable filtered = rows; + IEnumerable filtered = snapshot.Rows; if (!string.IsNullOrEmpty(database_name)) filtered = filtered.Where(r => r.DatabaseName.Equals(database_name, StringComparison.OrdinalIgnoreCase)); @@ -278,6 +280,7 @@ public static async Task GetQueryStoreHealth( return JsonSerializer.Serialize(new { server = resolved.ServerName, + captured_at = snapshot.CapturedAt!.Value.ToString("o"), database_count = result.Count, databases = result }, McpHelpers.JsonOptions); diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpConfigTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpConfigTools.cs index 5c98d5547..52d09370c 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpConfigTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpConfigTools.cs @@ -32,7 +32,7 @@ namespace PerformanceMonitor.Darling.Service.Mcp; [McpServerToolType] public sealed class DarlingMcpConfigTools { - [McpServerTool(Name = "get_server_config"), Description("Gets the current SQL Server instance configuration (sys.configurations). Shows all sp_configure settings with configured and in-use values. Useful for checking CTFP, MAXDOP, max memory, and other instance-level settings right now (unlike get_server_config_changes, which shows only what changed between connect snapshots).")] + [McpServerTool(Name = "get_server_config"), Description("Gets the current SQL Server instance configuration (sys.configurations). Shows all sp_configure settings with configured and in-use values. Useful for checking CTFP, MAXDOP, max memory, and other instance-level settings right now (unlike get_server_config_changes, which shows only what changed between connect snapshots). LATEST IS A TIME: configuration is captured when the collector CONNECTS, not on a schedule, so 'current' here means 'as of the last capture' - captured_at is that instant, and a value can be days old on a server the monitor has stayed connected to.")] public static async Task GetServerConfig( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null) @@ -42,8 +42,8 @@ public static async Task GetServerConfig( try { - var rows = await DarlingCurrentConfigReader.GetLatestServerConfigAsync(postgres, resolved.ServerId); - if (rows.Count == 0) + var snapshot = await DarlingCurrentConfigReader.GetLatestServerConfigAsync(postgres, resolved.ServerId); + if (snapshot.IsEmpty) return await DarlingEngineCapability.NotCollectedStatusAsync(postgres, resolved.ServerId, resolved.ServerName, "server_config") ?? McpHelpers.Status( "unavailable", @@ -52,8 +52,10 @@ public static async Task GetServerConfig( return JsonSerializer.Serialize(new { server = resolved.ServerName, - setting_count = rows.Count, - settings = rows.Select(r => new + /* #3541 A10: the connect-time capture this "current" configuration is as of. */ + captured_at = snapshot.CapturedAt!.Value.ToString("o"), + setting_count = snapshot.Count, + settings = snapshot.Rows.Select(r => new { name = r.ConfigurationName, value_configured = r.ValueConfigured, @@ -70,7 +72,7 @@ public static async Task GetServerConfig( } } - [McpServerTool(Name = "get_database_config"), Description("Gets database-level configuration for all databases (sys.databases). Shows recovery model, RCSI, auto-shrink, auto-close, Query Store, compatibility level, page verify, and other settings. Critical for identifying misconfigured databases.")] + [McpServerTool(Name = "get_database_config"), Description("Gets database-level configuration for all databases (sys.databases). Shows recovery model, RCSI, auto-shrink, auto-close, Query Store, compatibility level, page verify, and other settings. Critical for identifying misconfigured databases. LATEST IS A TIME: captured when the collector connects, not on a schedule - captured_at is the instant these settings are as of, and a database created or altered since is not reflected until the next connect.")] public static async Task GetDatabaseConfig( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -81,14 +83,14 @@ public static async Task GetDatabaseConfig( try { - var rows = await DarlingCurrentConfigReader.GetLatestDatabaseConfigAsync(postgres, resolved.ServerId); - if (rows.Count == 0) + var snapshot = await DarlingCurrentConfigReader.GetLatestDatabaseConfigAsync(postgres, resolved.ServerId); + if (snapshot.IsEmpty) return await DarlingEngineCapability.NotCollectedStatusAsync(postgres, resolved.ServerId, resolved.ServerName, "database_config") ?? McpHelpers.Status( "unavailable", "No database configuration data available. The config collector may not have run yet."); - IEnumerable filtered = rows; + IEnumerable filtered = snapshot.Rows; if (!string.IsNullOrEmpty(database_name)) filtered = filtered.Where(r => r.DatabaseName.Equals(database_name, StringComparison.OrdinalIgnoreCase)); @@ -119,6 +121,7 @@ public static async Task GetDatabaseConfig( return JsonSerializer.Serialize(new { server = resolved.ServerName, + captured_at = snapshot.CapturedAt!.Value.ToString("o"), database_count = result.Count, databases = result }, McpHelpers.JsonOptions); @@ -129,7 +132,7 @@ public static async Task GetDatabaseConfig( } } - [McpServerTool(Name = "get_trace_flags"), Description("Gets active trace flags on the SQL Server instance. Shows flag number, enabled status, and whether the flag is global or session-scoped.")] + [McpServerTool(Name = "get_trace_flags"), Description("Gets active trace flags on the SQL Server instance. Shows flag number, enabled status, and whether the flag is global or session-scoped. LATEST IS A TIME: captured when the collector connects, not on a schedule - captured_at is the instant these flags are as of; a flag turned on or off since is not reflected until the next connect.")] public static async Task GetTraceFlags( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null) @@ -139,16 +142,17 @@ public static async Task GetTraceFlags( try { - var rows = await DarlingCurrentConfigReader.GetLatestTraceFlagsAsync(postgres, resolved.ServerId); - if (rows.Count == 0) + var snapshot = await DarlingCurrentConfigReader.GetLatestTraceFlagsAsync(postgres, resolved.ServerId); + if (snapshot.IsEmpty) return await DarlingEngineCapability.NotCollectedStatusAsync(postgres, resolved.ServerId, resolved.ServerName, "trace_flags") ?? McpHelpers.Status("empty", "No trace flags found (none enabled, or the config collector has not run yet)."); return JsonSerializer.Serialize(new { server = resolved.ServerName, - trace_flag_count = rows.Count, - trace_flags = rows.Select(r => new + captured_at = snapshot.CapturedAt!.Value.ToString("o"), + trace_flag_count = snapshot.Count, + trace_flags = snapshot.Rows.Select(r => new { trace_flag = r.TraceFlag, enabled = r.Status, diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpDataTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpDataTools.cs index 9e63bc4f8..9a713b697 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpDataTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpDataTools.cs @@ -284,7 +284,7 @@ public static async Task GetWaitTrend( } } - [McpServerTool(Name = "get_memory_stats"), Description("Gets the latest memory statistics snapshot: physical memory, buffer pool size, plan cache size, memory utilization %, and SQL Server memory model. Use this for a quick memory health check; use get_memory_clerks to see detailed breakdown by component.")] + [McpServerTool(Name = "get_memory_stats"), Description("Gets the latest memory statistics snapshot: physical memory, buffer pool size, plan cache size, memory utilization %, and SQL Server memory model. Use this for a quick memory health check; use get_memory_clerks to see detailed breakdown by component. LATEST IS A TIME: this reads one snapshot, not a window, and captured_at is the instant that snapshot was collected - read it before treating any figure as current, because the newest row a store holds can be minutes or days old.")] public static async Task GetMemoryStats( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null) @@ -306,7 +306,8 @@ public static async Task GetMemoryStats( return JsonSerializer.Serialize(new { server = resolved.ServerName, - collection_time = stats.CollectionTime.ToString("o"), + /* #3541 A10: the one stamp every latest-snapshot read publishes, under the one name. */ + captured_at = stats.CollectionTime.ToString("o"), total_physical_memory_mb = stats.TotalPhysicalMemoryMb, available_physical_memory_mb = stats.AvailablePhysicalMemoryMb, memory_utilization_pct = Math.Round(utilization, 1), @@ -324,7 +325,7 @@ public static async Task GetMemoryStats( } } - [McpServerTool(Name = "get_memory_clerks"), Description("Gets the top memory consumers by memory clerk type — shows which SQL Server components are using the most memory.")] + [McpServerTool(Name = "get_memory_clerks"), Description("Gets the top memory consumers by memory clerk type — shows which SQL Server components are using the most memory. LATEST IS A TIME: this reads the newest clerk snapshot, not a window, and captured_at is the instant it was collected.")] public static async Task GetMemoryClerks( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null) @@ -334,9 +335,9 @@ public static async Task GetMemoryClerks( try { - var rows = await DarlingDataReader.GetLatestMemoryClerksAsync(postgres, resolved.ServerId); + var snapshot = await DarlingDataReader.GetLatestMemoryClerksAsync(postgres, resolved.ServerId); - if (rows.Count == 0) + if (snapshot.IsEmpty) /* ONE branch here, deliberately, and it is the reason this read gets no existence probe. The read is "every clerk at MAX(collection_time)", so zero rows back is logically the @@ -350,7 +351,7 @@ with the read by construction and tell the caller nothing it did not already hav "unavailable", $"No memory-clerk snapshot is available for {resolved.ServerName}. This read returns the LATEST snapshot rather than a window, so an empty result is never a quiet period — a live SQL Server always has memory clerks. It means nothing the memory_clerks collector stored is still retained, either because it has not run for this server or because its rows have aged out. Check get_collection_health and get_collection_log for the memory_clerks collector."); - var result = rows.Select(r => new + var result = snapshot.Rows.Select(r => new { clerk_type = r.ClerkType, memory_mb = Math.Round(r.MemoryMb, 2) @@ -359,6 +360,7 @@ with the read by construction and tell the caller nothing it did not already hav return JsonSerializer.Serialize(new { server = resolved.ServerName, + captured_at = snapshot.CapturedAt!.Value.ToString("o"), clerks = result }, McpHelpers.JsonOptions); } @@ -368,7 +370,7 @@ with the read by construction and tell the caller nothing it did not already hav } } - [McpServerTool(Name = "get_file_io_stats"), Description("Gets the latest file I/O statistics per database file: read/write counts, bytes, stall times, and calculated latency. High read latency (>20ms) or write latency (>10ms for data, >2ms for log) often indicates storage bottlenecks. Each row carries sample_interval_seconds, the measured seconds its deltas accrued over; a 0 means no delta was knowable for that file at this collection (first sighting, counter reset, or a gap past the delta policy — typically a restart) and its latencies are null rather than 0.")] + [McpServerTool(Name = "get_file_io_stats"), Description("Gets the latest file I/O statistics per database file: read/write counts, bytes, stall times, and calculated latency. High read latency (>20ms) or write latency (>10ms for data, >2ms for log) often indicates storage bottlenecks. Each row carries sample_interval_seconds, the measured seconds its deltas accrued over; a 0 means no delta was knowable for that file at this collection (first sighting, counter reset, or a gap past the delta policy — typically a restart) and its latencies are null rather than 0. LATEST IS A TIME: this reads the newest file-I/O snapshot, not a window, and captured_at is the instant it was collected; the deltas cover the sample_interval_seconds ending there.")] public static async Task GetFileIoStats( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null) @@ -378,12 +380,12 @@ public static async Task GetFileIoStats( try { - var rows = await DarlingDataReader.GetLatestFileIoStatsAsync(postgres, resolved.ServerId); - if (rows.Count == 0) + var snapshot = await DarlingDataReader.GetLatestFileIoStatsAsync(postgres, resolved.ServerId); + if (snapshot.IsEmpty) return await DarlingEngineCapability.NotCollectedStatusAsync(postgres, resolved.ServerId, resolved.ServerName, "file_io_stats") ?? McpHelpers.Status("unavailable", "No file I/O stats available."); - var result = rows.Select(r => new + var result = snapshot.Rows.Select(r => new { database_name = r.DatabaseName, file_name = r.FileName, @@ -409,6 +411,7 @@ itself is a pre-V127 row that never recorded one. */ return JsonSerializer.Serialize(new { server = resolved.ServerName, + captured_at = snapshot.CapturedAt!.Value.ToString("o"), files = result }, McpHelpers.JsonOptions); } @@ -464,7 +467,7 @@ public static async Task GetTempDbTrend( } } - [McpServerTool(Name = "get_perfmon_stats"), Description("Gets the latest SQL Server performance counter values: batch requests/sec, compilations/sec, deadlocks/sec, and more. Provides throughput context to distinguish a busy server from a sick one. Use counter_name or instance_name to filter results.")] + [McpServerTool(Name = "get_perfmon_stats"), Description("Gets the latest SQL Server performance counter values: batch requests/sec, compilations/sec, deadlocks/sec, and more. Provides throughput context to distinguish a busy server from a sick one. Use counter_name or instance_name to filter results. LATEST IS A TIME: this reads the newest counter snapshot, not a window, and captured_at is the instant it was collected; use get_perfmon_trend for a counter over time.")] public static async Task GetPerfmonStats( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -476,12 +479,12 @@ public static async Task GetPerfmonStats( try { - var rows = await DarlingDataReader.GetLatestPerfmonStatsAsync(postgres, resolved.ServerId); - if (rows.Count == 0) + var snapshot = await DarlingDataReader.GetLatestPerfmonStatsAsync(postgres, resolved.ServerId); + if (snapshot.IsEmpty) return await DarlingEngineCapability.NotCollectedStatusAsync(postgres, resolved.ServerId, resolved.ServerName, "perfmon_stats") ?? McpHelpers.Status("unavailable", "No perfmon stats available."); - IEnumerable filtered = rows; + IEnumerable filtered = snapshot.Rows; if (!string.IsNullOrEmpty(counter_name)) filtered = filtered.Where(r => r.CounterName.Contains(counter_name, StringComparison.OrdinalIgnoreCase)); if (!string.IsNullOrEmpty(instance_name)) @@ -498,6 +501,7 @@ public static async Task GetPerfmonStats( return JsonSerializer.Serialize(new { server = resolved.ServerName, + captured_at = snapshot.CapturedAt!.Value.ToString("o"), counters = result }, McpHelpers.JsonOptions); } diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpHealthTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpHealthTools.cs index 69d072f45..926df0e07 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpHealthTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpHealthTools.cs @@ -45,7 +45,12 @@ namespace PerformanceMonitor.Darling.Service.Mcp; [McpServerToolType] public sealed class DarlingMcpHealthTools { - [McpServerTool(Name = "get_server_summary"), Description("Gets a quick health overview for a SQL Server instance: current CPU %, memory usage, recent blocking count, and deadlock count. Use this for a fast health check before drilling into specific areas.")] + /// get_server_summary's description, VERBATIM Lite's (#3541 A10); the cross-SKU census pins them + /// equal. Names the three clocks the payload carries, because the payload used to carry one. + internal const string ServerSummaryDescription = + "Gets a quick health overview for a SQL Server instance: current CPU %, memory usage, recent blocking count, and deadlock count. Use this for a fast health check before drilling into specific areas. THREE CLOCKS, NAMED: cpu_percent is the newest CPU snapshot and cpu_captured_at is its instant; memory_mb is the newest memory snapshot and memory_captured_at is its instant; last_collection is the newest collection of ANY collector for this server - the store's freshness, NOT the age of the two figures above, which can be far older when their own collectors have stopped. blocking_count and deadlock_count cover the counts_window_hours ending now."; + + [McpServerTool(Name = "get_server_summary"), Description(ServerSummaryDescription)] public static async Task GetServerSummary( NpgsqlDataSource postgres, [Description("Server name or display name. Optional if only one server is configured.")] string? server_name = null) @@ -65,9 +70,15 @@ public static async Task GetServerSummary( { server = resolved.ServerName, cpu_percent = summary.CpuPercent, + /* #3541 A10: each latest figure carries ITS OWN clock. last_collection below is the newest + collection of ANY collector — a live collection log beside a dead CPU collector made a + day-old cpu_percent read as current, because the only stamp on the payload was fresh. */ + cpu_captured_at = summary.CpuCapturedAt?.ToString("o"), memory_mb = summary.MemoryMb, + memory_captured_at = summary.MemoryCapturedAt?.ToString("o"), blocking_count = summary.BlockingCount, deadlock_count = summary.DeadlockCount, + counts_window_hours = DarlingHealthReader.ServerSummaryCountsWindowHours, last_collection = summary.LastCollectionTime?.ToString("o") }, McpHelpers.JsonOptions); } diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs index a52b92dcf..7eff50849 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs @@ -78,6 +78,7 @@ public static string Build(DarlingPeerDirectory.Snapshot peers) - An unparseable `as_of`, or one in the future, is REFUSED with a message rather than quietly answered as "now" — a read that silently reverts to now is indistinguishable from a correct one. - An `as_of` older than anything the store still holds is NOT refused. It returns the read's normal `empty` / `unavailable` status, which means exactly what it says: we looked in the window you named and there was nothing in it. - Tools that take no window at all (latest-snapshot reads like `get_memory_stats`, `get_file_io_stats`, `get_index_usage`, and the configuration reads) do not take `as_of` — they read the newest row, and there is no window to move. + - **Latest is a time.** Every latest-snapshot read publishes `captured_at` — the snapshot's own collection instant — and the anchored ones publish `age_seconds` against the window's end. Read it before treating a "current" figure as current: the newest row a store holds is as old as its collector's last successful run, and the configuration family is captured on CONNECT, so a "current" setting can be days old. Where a latest read takes `hours_back`, its description says which of two things that means: the span SEARCHED for the newest snapshot (`get_latch_stats`, `get_cpu_scheduler_pressure`, `get_plan_cache_bloat` — a snapshot older than that is `unavailable`, not served as current), or a span READ beside the snapshot (`get_resource_semaphore` / `get_memory_grants` return `grants[]`, the newest snapshot, AND `window[]`, the peak / floor / summed-delta aggregate over every snapshot in the hours). `get_server_summary` carries three clocks by name — `cpu_captured_at`, `memory_captured_at`, and `last_collection` (the newest collection of ANY collector, which is the store's freshness and not the age of the two figures). - The analysis family DOES take it (#2506), and the anchor reaches the ENGINE rather than stopping at the tool: `get_analysis_facts` and `analyze_server` re-run fact collection and scoring over the anchored window, and `analyze_server`'s anomaly detection moves with it, so the window is compared against the hour-of-day x day-of-week baseline for the hours it actually covers instead of for the hours you happen to be asking in. `compare_analysis` hangs BOTH windows off the anchor, since `baseline_hours_back` has always been measured from the comparison window's end. `get_analysis_findings` is the odd one and worth reading twice: its window is on ANALYSIS TIME, so anchoring it asks what a scheduled analysis pass was SAYING then, which is a different question from re-analyzing that window now (that is `analyze_server` with the same anchor). - `analyze_server` with an `as_of` is EXPLORATORY and does NOT persist its findings; the result says so in `persisted` / `persistence_note`. A finding row is stamped with the time the analysis RAN, and `get_analysis_findings` and the viewer's Recommendations tab treat the newest `analysis_time` as the server's CURRENT state — so writing a backdated run would make last week's findings today's headline and would inflate the occurrence stats of any live incident sharing a story path. Run it without `as_of` when you want the present analyzed and recorded. - `get_pvs_stats` and `get_fleet_overview` do not take it. Each mixes a latest-snapshot measurement with a windowed one, so anchoring only the windowed half would return a result whose two halves describe different instants. `get_store_metrics` does not take it either: it windows in DAYS over the store's own growth series. @@ -115,11 +116,11 @@ public static string Build(DarlingPeerDirectory.Snapshot peers) | `get_wait_stats` | Top wait types aggregated over the window (wait/signal/resource ms, signal %) | `server_name`, `hours_back` (default 24), `limit` (default 20), `as_of` | | `get_wait_trend` | A single wait type's per-second trend over time | `wait_type` (required), `server_name`, `hours_back` (default 24), `as_of` | | `get_wait_types` | The distinct wait types observed on the server (heaviest first) — pick a `wait_type` for get_wait_trend. An empty result distinguishes a quiet window (`empty`, widen `hours_back`) from a server no wait stats have ever been stored for (`unavailable`) | `server_name`, `hours_back` (default 24), `as_of` | - | `get_memory_stats` | Latest memory snapshot: physical / buffer pool / plan cache / utilization %, memory model | `server_name` | + | `get_memory_stats` | Latest memory snapshot: physical / buffer pool / plan cache / utilization %, memory model; `captured_at` | `server_name` | | `get_memory_clerks` | Latest top memory consumers by clerk type. An empty result is `unavailable`, never a quiet period — a live SQL Server always has clerks, so nothing retained means the collector has not run or its rows aged out | `server_name` | - | `get_file_io_stats` | Latest per-file I/O: reads/writes/bytes/stall and computed read/write latency | `server_name` | + | `get_file_io_stats` | Latest per-file I/O: reads/writes/bytes/stall and computed read/write latency; `captured_at` | `server_name` | | `get_tempdb_trend` | TempDB space over time (user / internal / version store / unallocated) + top consumer | `server_name`, `hours_back` (default 24), `as_of` | - | `get_perfmon_stats` | Latest perfmon counters (value + delta); filter by counter / instance | `server_name`, `counter_name`, `instance_name` | + | `get_perfmon_stats` | Latest perfmon counters (value + delta); filter by counter / instance; `captured_at` | `server_name`, `counter_name`, `instance_name` | | `get_top_queries_by_cpu` | Expensive queries from query stats (plan cache) with query_hash / sql_handle; `cpu_attribution.attributed_cpu_ratio` says how much of the box's measured CPU the returned rows explain | `server_name`, `hours_back` (default 24), `top` (default 20), `database_name`, `parallel_only`, `min_dop`, `as_of` | | `get_top_procedures_by_cpu` | Most expensive stored procedures by total CPU, with the same `cpu_attribution` disclosure | `server_name`, `hours_back` (default 24), `top` (default 20), `database_name`, `as_of` | | `get_query_store_top` | Expensive queries from Query Store with query_id / plan_id (survives restarts) | `server_name`, `hours_back` (default 24), `top` (default 20), `database_name`, `as_of` | @@ -152,11 +153,11 @@ public static string Build(DarlingPeerDirectory.Snapshot peers) | `get_server_config_changes` | sp_configure changes, diffed from config snapshots | `server_name`, `hours_back` (default 168), `as_of` | | `get_database_config_changes` | sys.databases setting changes, diffed from config snapshots | `server_name`, `hours_back` (default 168), `as_of` | | `get_trace_flag_changes` | Trace flags enabled/disabled/modified, diffed from config snapshots | `server_name`, `hours_back` (default 168), `as_of` | - | `get_database_scoped_config` | Latest database-scoped configuration (MAXDOP, legacy CE, ...) | `server_name`, `database_name` | + | `get_database_scoped_config` | Latest database-scoped configuration (MAXDOP, legacy CE, ...) as of `captured_at` (captured on connect) | `server_name`, `database_name` | | `get_query_store_health` | Per-database Query Store health (latest hourly snapshot) — actual vs desired state, readonly_reason decoded, storage vs cap, cleanup thresholds | `server_name`, `database_name` | - | `get_server_config` | CURRENT sys.configurations (latest snapshot) — what CTFP / MAXDOP / max memory are set to now | `server_name` | - | `get_database_config` | CURRENT per-database settings (latest snapshot) — recovery model, RCSI, Query Store, ... | `server_name`, `database_name` | - | `get_trace_flags` | CURRENT active trace flags (latest snapshot) — flag number, enabled, global/session | `server_name` | + | `get_server_config` | CURRENT sys.configurations (latest snapshot) — what CTFP / MAXDOP / max memory are set to as of `captured_at` (captured on connect — can be days old) | `server_name` | + | `get_database_config` | CURRENT per-database settings (latest snapshot) — recovery model, RCSI, Query Store, ... as of `captured_at` (captured on connect) | `server_name`, `database_name` | + | `get_trace_flags` | CURRENT active trace flags (latest snapshot) — flag number, enabled, global/session, as of `captured_at` (captured on connect) | `server_name` | | `get_table_index_sizes` | Largest tables with size + growth (7d/30d/daily) from the latest daily snapshot | `server_name` | | `get_index_usage` | Per-index usage classified Unused / Write-only / Active | `server_name` | | `get_object_locking` | Per-index lock/latch contention, most contended first | `server_name` | @@ -168,13 +169,13 @@ public static string Build(DarlingPeerDirectory.Snapshot peers) | Tool | Purpose | Key Parameters | |------|---------|----------------| - | `get_latch_stats` | Top latch classes by wait time, with per-second rates + severity / description / recommendation | `server_name`, `hours_back` (default 24), `top` (default 10), `as_of` | + | `get_latch_stats` | Top latch classes by wait time (window totals), with per-second rates + severity / description / recommendation; `severity` is banded from the LATEST interval only and `severity_banded_from` names that interval (delta, seconds, `captured_at`) | `server_name`, `hours_back` (default 24), `top` (default 10), `as_of` | | `get_spinlock_stats` | Top spinlocks by collisions, with per-second rates + description | `server_name`, `hours_back` (default 24), `top` (default 10), `as_of` | - | `get_resource_semaphore` | Latest workspace-memory semaphores: target / max-target ceiling vs granted / used, waiter / timeout / forced | `server_name`, `hours_back` (default 24), `as_of` | - | `get_memory_grants` | Latest per-pool grant detail: available / granted / used + waiter / timeout / forced deltas | `server_name`, `hours_back` (default 1), `as_of` | + | `get_resource_semaphore` | Per-semaphore workspace memory vs target/max ceiling: `grants[]` = newest snapshot in the window (`captured_at` / `age_seconds`) AND `window[]` = peak waiters (+ when), peak grant, available floor, summed timeout / forced deltas over EVERY snapshot in the window, per (semaphore, pool) | `server_name`, `hours_back` (default 24), `as_of` | + | `get_memory_grants` | Per-pool grant pressure: `grants[]` = newest snapshot in the window (`captured_at` / `age_seconds`) AND `window[]` = the same peak / floor / summed-delta aggregate per pool | `server_name`, `hours_back` (default 1), `as_of` | | `get_memory_pressure_events` | RING_BUFFER_RESOURCE_MONITOR memory-pressure notifications (process/system indicator scale 0-3+); not on Azure SQL DB | `server_name`, `hours_back` (default 24), `as_of` | - | `get_plan_cache_bloat` | Plan cache single-use vs multi-use composition + bloat_level classification | `server_name`, `hours_back` (default 24), `as_of` | - | `get_cpu_scheduler_pressure` | Latest scheduler snapshot: runnable queue, worker utilization, pressure_level + warnings | `server_name` | + | `get_plan_cache_bloat` | Plan cache single-use vs multi-use composition + bloat_level classification; `captured_at` / `age_seconds` | `server_name`, `hours_back` (default 24; the span SEARCHED for the newest snapshot), `as_of` | + | `get_cpu_scheduler_pressure` | Latest scheduler snapshot within `hours_back` of `as_of`: runnable queue, worker utilization, pressure_level + warnings; `captured_at` / `age_seconds` | `server_name`, `hours_back` (default 24; the span SEARCHED for the newest snapshot), `as_of` | | `get_running_jobs` | Currently running SQL Agent jobs with duration vs historical average / p95 | `server_name` | ### Trend data-read tools @@ -216,7 +217,7 @@ public static string Build(DarlingPeerDirectory.Snapshot peers) | `get_alert_history` | Alerts that fired (metric, value vs threshold, delivery success/failure, muted), newest first; omit server_name for the whole fleet, each row names its server. EXCLUDES operator-dismissed alerts by default and says so (`dismissed_excluded`, `dismissed_excluded_count`); pass `include_dismissed` when reconstructing an incident. Page bounded by `limit`; `truncated` and `oldest_returned_alert_time` say what it reached | `server_name` (optional — all servers if omitted), `hours_back` (default 24), `limit` (default 50), `as_of`, `include_dismissed` (default false) | | `get_alert_settings` | The current alert config the service uses: per-alert enable + thresholds, cooldown, excluded databases, delivery mode, and the scheduled-analysis cadence | none | | `get_mute_rules` | The alert mute rules in force, so a suppressed server is distinguishable from a healthy-quiet one. An empty result distinguishes no rule ever written from rules that exist but have all lapsed, with the configured count in `hints` | `enabled_only` (default true) | - | `get_server_summary` | One-shot per-server health: current CPU %, memory, recent blocking count, recent deadlock count | `server_name` | + | `get_server_summary` | One-shot per-server health: current CPU %, memory, recent blocking count, recent deadlock count; three clocks named (`cpu_captured_at`, `memory_captured_at`, `last_collection` = newest collection of ANY collector) | `server_name` | | `get_daily_summary` | A day's composite health band (Healthy / Warning / Critical) plus the signals behind it (waits, deadlocks, blocking, high CPU, memory pressure, alerts) | `server_name`, `summary_date` (yyyy-MM-dd, default today) | | `get_daily_summary_range` | The SAME rollup across a span of days — one row per collected day, which is the desktop viewer's Performance Calendar month grid. Use it when the question is WHICH day rather than how one day went: scan the bands, then call `get_daily_summary` for the day that stands out. A day with ANY collection appears even when every signal was quiet (Healthy, not missing), so a day absent from the result is a gap in COLLECTION. `as_of` anchors the LAST day of the range, so a past month is `as_of` its last day with `days_back` its length. An empty result distinguishes a range outside this server's history (`empty`) from a server nothing has ever been collected for (`unavailable`) | `server_name`, `days_back` (default 30, max 366), `as_of` | diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpLatchSpinlockTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpLatchSpinlockTools.cs index ed97bf6a5..f0a3a55cb 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpLatchSpinlockTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpLatchSpinlockTools.cs @@ -22,8 +22,9 @@ namespace PerformanceMonitor.Darling.Service.Mcp; /// /// The latch / spinlock contention MCP tools — get_latch_stats, get_spinlock_stats — served over Darling's /// Postgres store. Each tool body mirrors the Dashboard's McpLatchSpinlockTools field-for-field (Lite -/// exposes neither tool, so the Dashboard is the only reference; there is no divergent Lite shape to follow). -/// Reads flow through — STORED reads (no live monitored-server hit), +/// has since ported both names as LATEST-SNAPSHOT reads over its own store — McpLatchSpinlockTools in +/// Lite/Mcp — so the two SKUs share the names but not the shape: Darling aggregates the window, Lite +/// serves the newest snapshot in it). Reads flow through — STORED reads (no live monitored-server hit), /// windowed on hours_back. /// /// @@ -37,7 +38,7 @@ namespace PerformanceMonitor.Darling.Service.Mcp; [McpServerToolType] public sealed class DarlingMcpLatchSpinlockTools { - [McpServerTool(Name = "get_latch_stats"), Description("Gets top latch contention by class. Shows latch waits, wait time, and per-second rates. High LATCH_EX on ACCESS_METHODS_DATASET_PARENT or FGCB_ADD_REMOVE indicates TempDB allocation contention.")] + [McpServerTool(Name = "get_latch_stats"), Description("Gets top latch contention by class. Shows latch waits, wait time, and per-second rates. High LATCH_EX on ACCESS_METHODS_DATASET_PARENT or FGCB_ADD_REMOVE indicates TempDB allocation contention. TWO CLOCKS PER ROW, NAMED: total_delta_* SUM every collection in the window; severity, waits_per_second and wait_ms_per_second are banded/derived from the LATEST interval only - the severity_banded_from block names that interval (its delta wait, the seconds it accrued over, and the collection it ended at, which is also latest_collection_time). A LOW severity beside a large window total is a class that was hot earlier in the window and is quiet now, not a contradiction.")] public static async Task GetLatchStats( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -74,6 +75,15 @@ public static async Task GetLatchStats( waits_per_second = r.WaitsPerSecond is double waits ? Math.Round(waits, 2) : (double?)null, wait_ms_per_second = r.WaitMsPerSecond is double waitMs ? Math.Round(waitMs, 2) : (double?)null, severity = DarlingLatchSpinlockReader.LatchSeverity(r.LatestDeltaWaitTimeMs), + /* #3541 A10: the band above is a function of ONE interval's delta, published beside window + totals it is not a function of. Naming the interval — its delta, its length, its end — is + what lets a reader tell "LOW now, 40 s of waits over the day" from "LOW all day". */ + severity_banded_from = new + { + delta_wait_time_ms = r.LatestDeltaWaitTimeMs, + interval_seconds = r.LatestIntervalSeconds is double seconds ? Math.Round(seconds, 0) : (double?)null, + captured_at = r.LatestCollectionTime.ToString("o") + }, description = DarlingLatchSpinlockReader.LatchDescription(r.LatchClass), recommendation = DarlingLatchSpinlockReader.LatchRecommendation(r.LatchClass), latest_collection_time = r.LatestCollectionTime.ToString("o") diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpMemoryGrantTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpMemoryGrantTools.cs index 0076e0920..6c7d1ae0e 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpMemoryGrantTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpMemoryGrantTools.cs @@ -21,20 +21,30 @@ namespace PerformanceMonitor.Darling.Service.Mcp; /// /// The memory-grant MCP tools — get_resource_semaphore, get_memory_grants — served over Darling's Postgres -/// store, both reading the LATEST memory_grant_stats snapshot through -/// (STORED reads, no live monitored-server hit). The two names are two -/// lenses on the one collector table, so Darling hosts BOTH: get_resource_semaphore is the Dashboard's -/// semaphore/ceiling shape (per resource semaphore, with the workspace-memory target/max-target ceiling); -/// get_memory_grants is Lite's per-pool grant-detail shape. A client familiar with either SKU finds its tool. +/// store through (STORED reads, no live monitored-server hit). The two +/// names are two lenses on the one collector table, so Darling hosts BOTH: get_resource_semaphore is the +/// Dashboard's semaphore/ceiling shape (per resource semaphore, with the workspace-memory target/max-target +/// ceiling); get_memory_grants is Lite's per-pool grant-detail shape. A client familiar with either SKU finds +/// its tool. +/// +/// #3541 A10 — the window is read, and the snapshot says when it was taken. Each tool serves two +/// things under one hours_back / as_of window: grants, the NEWEST snapshot in the window +/// (stamped once as captured_at, with age_seconds against the window's end), and window, +/// the per-semaphore / per-pool aggregate over EVERY snapshot in it — peak waiters and when, peak grant, the +/// available-workspace floor, and the summed timeout / forced-grant deltas. Before this the tools accepted +/// hours_back and read only the latest row, so a three-hour-old grant storm was invisible behind a +/// calm snapshot while the parameter read as a window. Dropping the parameter was the other honest shape; +/// reading the window was chosen because "was there grant pressure in the last N hours" is the question an +/// agent brings to this tool, and the store answers it in one bounded scan. /// [McpServerToolType] public sealed class DarlingMcpMemoryGrantTools { - [McpServerTool(Name = "get_resource_semaphore"), Description("Gets resource semaphore statistics showing granted vs available workspace memory against the target/max-target ceiling, waiter counts, and timeout/forced grant pressure indicators. High waiter counts or rising timeout/forced deltas indicate memory grant pressure affecting query performance. sample_interval_seconds is the measured seconds the two deltas accrued over; it is null with interval_known false when the row is a restart marker (no delta was knowable, so the zero deltas beside it are not 'no timeouts') or predates the column.")] + [McpServerTool(Name = "get_resource_semaphore"), Description("Gets resource semaphore statistics showing granted vs available workspace memory against the target/max-target ceiling, waiter counts, and timeout/forced grant pressure indicators. High waiter counts or rising timeout/forced deltas indicate memory grant pressure affecting query performance. TWO READS UNDER ONE WINDOW: grants[] is the NEWEST snapshot in the window (one row per resource semaphore and pool), stamped once as captured_at with age_seconds against the window's end - it is a moment, not the window. window[] aggregates EVERY snapshot in the window per (resource_semaphore_id, pool_id): peak_waiter_count and peak_waiters_at (the most sessions ever seen waiting for a grant and when), peak_granted_memory_mb, min_available_memory_mb, and timeout_errors_in_window / forced_grants_in_window (the SUM of the per-interval deltas across the window). A calm grants[] beside a window[] with waiters or timeouts is a grant storm that has passed; read window[] first for 'was there pressure', grants[] for 'is there pressure now'. Each grants[] row also carries sample_interval_seconds, the measured seconds its two deltas accrued over; it is null with interval_known false when the row is a restart marker (no delta was knowable, so the zero deltas beside it are not 'no timeouts') or predates the column.")] public static async Task GetResourceSemaphore( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, - [Description("Hours of history. Default 24.")] int hours_back = 24, + [Description("Hours of history. Default 24. window[] aggregates every snapshot in these hours; grants[] is the newest snapshot in them.")] int hours_back = 24, [Description(McpHelpers.AsOfDescription)] string? as_of = null) { var (resolved, error) = await DarlingServerResolver.ResolveOrErrorAsync(postgres, server_name); @@ -46,12 +56,18 @@ public static async Task GetResourceSemaphore( try { var now = windowEnd; + var windowStart = now.AddHours(-hours_back); var rows = await DarlingMemoryGrantReader.GetResourceSemaphoreLatestAsync( - postgres, resolved.ServerId, now.AddHours(-hours_back), now); + postgres, resolved.ServerId, windowStart, now); if (rows.Count == 0) return await DarlingEngineCapability.NotCollectedStatusAsync(postgres, resolved.ServerId, resolved.ServerName, "memory_grant_stats") ?? McpHelpers.Status("unavailable", "No memory grant data available."); + /* Same window bounds as the latest read, so the window's last_snapshot_at IS captured_at and the + two halves describe one span of the same rows (#3541 A10). */ + var window = await DarlingMemoryGrantReader.GetResourceSemaphoreWindowAsync( + postgres, resolved.ServerId, windowStart, now); + var grants = rows.Select(r => new { collection_time = r.CollectionTime.ToString("o"), @@ -81,7 +97,15 @@ a pre-V128 row that never recorded one is null too. interval_known states the on return JsonSerializer.Serialize(new { server = resolved.ServerName, - grants + hours_back, + window_start = windowStart.ToString("o"), + window_end = now.ToString("o"), + /* Every grants[] row shares this stamp by construction (the read is WHERE collection_time = + MAX(...) in the window); it is published once, as the snapshot's own clock. */ + captured_at = rows[0].CollectionTime.ToString("o"), + age_seconds = LatestSnapshotStamp.AgeSeconds(rows[0].CollectionTime, now), + grants, + window = window.Select(WindowShape) }, McpHelpers.JsonOptions); } catch (Exception ex) @@ -90,11 +114,11 @@ a pre-V128 row that never recorded one is null too. interval_known states the on } } - [McpServerTool(Name = "get_memory_grants"), Description("Gets resource semaphore statistics showing granted vs available workspace memory per resource pool, waiter counts, and timeout/forced grant deltas. High waiter counts or rising timeout deltas indicate memory grant pressure affecting query performance.")] + [McpServerTool(Name = "get_memory_grants"), Description("Gets resource semaphore statistics showing granted vs available workspace memory per resource pool, waiter counts, and timeout/forced grant deltas. High waiter counts or rising timeout deltas indicate memory grant pressure affecting query performance. TWO READS UNDER ONE WINDOW: grants[] is the NEWEST snapshot in the window (one row per pool, summed across its semaphores), stamped once as captured_at with age_seconds against the window's end - it is a moment, not the window. window[] aggregates EVERY snapshot in the window per pool: peak_waiter_count and peak_waiters_at (the most sessions ever seen waiting on the pool at one instant and when), peak_granted_memory_mb, min_available_memory_mb, and timeout_errors_in_window / forced_grants_in_window (the SUM of the per-interval deltas across the window). A calm grants[] beside a window[] with waiters or timeouts is a grant storm that has passed; read window[] first for 'was there pressure', grants[] for 'is there pressure now'.")] public static async Task GetMemoryGrants( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, - [Description("Hours of history. Default 1.")] int hours_back = 1, + [Description("Hours of history. Default 1. window[] aggregates every snapshot in these hours; grants[] is the newest snapshot in them.")] int hours_back = 1, [Description(McpHelpers.AsOfDescription)] string? as_of = null) { var (resolved, error) = await DarlingServerResolver.ResolveOrErrorAsync(postgres, server_name); @@ -106,12 +130,16 @@ public static async Task GetMemoryGrants( try { var now = windowEnd; + var windowStart = now.AddHours(-hours_back); var rows = await DarlingMemoryGrantReader.GetMemoryGrantsLatestAsync( - postgres, resolved.ServerId, now.AddHours(-hours_back), now); + postgres, resolved.ServerId, windowStart, now); if (rows.Count == 0) return await DarlingEngineCapability.NotCollectedStatusAsync(postgres, resolved.ServerId, resolved.ServerName, "memory_grant_stats") ?? McpHelpers.Status("unavailable", "No memory grant data available."); + var window = await DarlingMemoryGrantReader.GetMemoryGrantsWindowAsync( + postgres, resolved.ServerId, windowStart, now); + var grants = rows.Select(r => new { collection_time = r.CollectionTime.ToString("o"), @@ -128,7 +156,13 @@ public static async Task GetMemoryGrants( return JsonSerializer.Serialize(new { server = resolved.ServerName, - grants + hours_back, + window_start = windowStart.ToString("o"), + window_end = now.ToString("o"), + captured_at = rows[0].CollectionTime.ToString("o"), + age_seconds = LatestSnapshotStamp.AgeSeconds(rows[0].CollectionTime, now), + grants, + window = window.Select(WindowShape) }, McpHelpers.JsonOptions); } catch (Exception ex) @@ -137,6 +171,27 @@ public static async Task GetMemoryGrants( } } + /// + /// The window half's payload shape, shared by both lenses so the same key set describes a semaphore's + /// window and a pool's window (the pool lens carries a null resource_semaphore_id, which + /// writes rather than drops — the key is present on both so a + /// caller can read one shape). + /// + private static object WindowShape(DarlingMemoryGrantReader.MemoryGrantWindowRow w) => new + { + resource_semaphore_id = w.ResourceSemaphoreId, + pool_id = w.PoolId, + snapshots_in_window = w.SnapshotsInWindow, + first_snapshot_at = w.FirstSnapshotAt.ToString("o"), + last_snapshot_at = w.LastSnapshotAt.ToString("o"), + peak_waiter_count = w.PeakWaiterCount, + peak_waiters_at = w.PeakWaitersAt.ToString("o"), + peak_granted_memory_mb = Math.Round(w.PeakGrantedMemoryMb, 2), + min_available_memory_mb = Math.Round(w.MinAvailableMemoryMb, 2), + timeout_errors_in_window = w.TimeoutErrorsInWindow, + forced_grants_in_window = w.ForcedGrantsInWindow + }; + [McpServerTool(Name = "get_memory_pressure_events"), Description(@"Gets memory pressure notifications from the RING_BUFFER_RESOURCE_MONITOR ring buffer (same source as sp_pressuredetector). Returns RESOURCE_MEMPHYSICAL_LOW, RESOURCE_MEMVIRTUAL_LOW, RESOURCE_MEMPHYSICAL_HIGH, and RESOURCE_MEM_STEADY notifications with indicator values. Indicator scale (applies to both memory_indicators_process and memory_indicators_system): diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPlanCacheSchedulerTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPlanCacheSchedulerTools.cs index a0797d68b..d55299692 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPlanCacheSchedulerTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPlanCacheSchedulerTools.cs @@ -22,19 +22,38 @@ namespace PerformanceMonitor.Darling.Service.Mcp; /// /// The plan-cache + CPU-scheduler snapshot MCP tools — get_plan_cache_bloat, get_cpu_scheduler_pressure — /// served over Darling's Postgres store. Each tool body mirrors the Dashboard's McpDiagnosticTools / -/// McpSchedulerTools field-for-field (Lite exposes neither, so the Dashboard is the only reference). +/// McpSchedulerTools field-for-field; Lite has since ported both names (Lite/Mcp/McpPlanCacheSchedulerTools), +/// and the two SKUs now share one parameter contract (below). /// Reads flow through — STORED reads of the latest snapshot, no /// live monitored-server hit. The Dashboard's bloat_level (#1410) and pressure_level / /// recommendation (#1410) classifications are reproduced from the reporting-view CASE logic. +/// +/// #3541 A10 — one contract on both SKUs, and the snapshot says when. Both tools take +/// (server_name, hours_back, as_of) with the SAME parameter descriptions as Lite's +/// McpPlanCacheSchedulerTools: hours_back is the span SEARCHED for the newest snapshot, not a +/// span aggregated, and the description says so in those words. get_cpu_scheduler_pressure used to take +/// server_name alone here and (server_name, hours_back, as_of) on Lite — the same tool name +/// with two parameter surfaces, on the one tool whose answer is a CRITICAL/HIGH/MEDIUM/NORMAL verdict. Every +/// payload publishes captured_at (the snapshot's own collection_time) and age_seconds +/// against the window's end, so a verdict computed from a stale row cannot pass as current. /// [McpServerToolType] public sealed class DarlingMcpPlanCacheSchedulerTools { - [McpServerTool(Name = "get_plan_cache_bloat"), Description("Gets plan cache composition showing single-use vs multi-use plans, with a bloat-level classification. High single-use plan counts indicate ad-hoc query bloat consuming buffer pool memory. Consider enabling 'optimize for ad hoc workloads'.")] + /// + /// get_cpu_scheduler_pressure's description, VERBATIM the text Lite's twin carries (#3541 A10): the same + /// tool name described two ways on two servers was half of the drift this lane closed, and a shared const + /// cannot be shared across the two assemblies, so the cross-SKU description census pins the two strings + /// equal instead. Change one, change both. + /// + internal const string CpuSchedulerPressureDescription = + "Gets CPU scheduler pressure from the latest snapshot: runnable task queue depth, worker thread utilization, queued/blocked requests, the collector's pressure warning flags, and the banded pressure_level verdict with its recommendation. Shows whether the server has enough worker threads and if tasks are queuing for CPU time. LATEST IS A TIME: this is the newest scheduler snapshot found within hours_back of as_of, not an aggregate over those hours - captured_at is the instant it was collected and age_seconds its distance from the window's end; the verdict is that instant's, so read age_seconds before reading pressure_level as current."; + + [McpServerTool(Name = "get_plan_cache_bloat"), Description("Gets plan cache composition showing single-use vs multi-use plans, with a bloat-level classification. High single-use plan counts indicate ad-hoc query bloat consuming buffer pool memory. Consider enabling 'optimize for ad hoc workloads'. LATEST IS A TIME: this is the newest plan-cache snapshot found within hours_back of as_of, not an aggregate over those hours - captured_at is the instant the snapshot was collected and age_seconds its distance from the window's end.")] public static async Task GetPlanCacheBloat( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, - [Description("Hours of data to analyze. Default 24.")] int hours_back = 24, + [Description("Hours of history to search for the latest snapshot. Default 24.")] int hours_back = 24, [Description(McpHelpers.AsOfDescription)] string? as_of = null) { var (resolved, error) = await DarlingServerResolver.ResolveOrErrorAsync(postgres, server_name); @@ -61,7 +80,8 @@ public static async Task GetPlanCacheBloat( return JsonSerializer.Serialize(new { server = resolved.ServerName, - collection_time = rows[0].CollectionTime.ToString("o"), + captured_at = rows[0].CollectionTime.ToString("o"), + age_seconds = LatestSnapshotStamp.AgeSeconds(rows[0].CollectionTime, now), summary = new { total_plans = totalPlans, @@ -93,20 +113,27 @@ public static async Task GetPlanCacheBloat( } } - [McpServerTool(Name = "get_cpu_scheduler_pressure"), Description("Gets CPU scheduler pressure: runnable task queue depth, worker thread utilization, and pressure warnings. Shows whether the server has enough worker threads and if tasks are queuing for CPU time.")] + [McpServerTool(Name = "get_cpu_scheduler_pressure"), Description(CpuSchedulerPressureDescription)] public static async Task GetCpuSchedulerPressure( NpgsqlDataSource postgres, - [Description("Server name or display name.")] string? server_name = null) + [Description("Server name or display name.")] string? server_name = null, + [Description("Hours of history to search for the latest snapshot. Default 24.")] int hours_back = 24, + [Description(McpHelpers.AsOfDescription)] string? as_of = null) { var (resolved, error) = await DarlingServerResolver.ResolveOrErrorAsync(postgres, server_name); if (error != null) return error; + var validation = McpHelpers.ValidateWindow(hours_back, as_of, out var windowEnd); + if (validation != null) return validation; + try { - var item = await DarlingPlanCacheSchedulerReader.GetCpuSchedulerPressureAsync(postgres, resolved.ServerId); + var now = windowEnd; + var item = await DarlingPlanCacheSchedulerReader.GetCpuSchedulerPressureAsync( + postgres, resolved.ServerId, now.AddHours(-hours_back), now); if (item == null) return await DarlingEngineCapability.NotCollectedStatusAsync(postgres, resolved.ServerId, resolved.ServerName, "cpu_scheduler_stats") - ?? McpHelpers.Status("unavailable", "No CPU scheduler data available. The scheduler collector may not have run yet."); + ?? McpHelpers.Status("unavailable", "No CPU scheduler snapshot in the requested time range. The scheduler collector may not have run yet, or its newest snapshot is older than hours_back."); var workerUtilizationPercent = item.MaxWorkersCount > 0 ? Math.Round(item.TotalCurrentWorkersCount * 100.0 / item.MaxWorkersCount, 2) @@ -119,7 +146,8 @@ public static async Task GetCpuSchedulerPressure( return JsonSerializer.Serialize(new { server = resolved.ServerName, - collection_time = item.CollectionTime.ToString("o"), + captured_at = item.CollectionTime.ToString("o"), + age_seconds = LatestSnapshotStamp.AgeSeconds(item.CollectionTime, now), schedulers = item.SchedulerCount, runnable_tasks = item.TotalRunnableTasksCount, avg_runnable_per_scheduler = item.AvgRunnableTasksCount, diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMemoryGrantReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMemoryGrantReader.cs index 8d8d53aff..f0bf31d62 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMemoryGrantReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMemoryGrantReader.cs @@ -33,6 +33,18 @@ namespace PerformanceMonitor.Darling.Service.Mcp; /// is Lite's pool-detail lens: the sizing + activity SUMMED per pool (Lite's GetMemoryGrantChartDataAsync /// shape). Every SQL string is a public const so Darling.Tests can pin the dialect + columns without a live Postgres. /// +/// +/// #3541 A10: the window is READ, not merely searched. Both tools accept hours_back, and +/// until this change the only thing it did was bound the search for the newest snapshot — a grant storm three +/// hours ago (waiters in the dozens, timeouts climbing) was invisible behind a calm latest row while the +/// parameter read as a window. The *WindowSql reads below aggregate the SAME rows the latest read picks +/// its snapshot from, per semaphore or per pool: the peak waiter count and WHEN it peaked, the peak grant, the +/// floor of available workspace, and the SUM of the per-interval timeout / forced-grant deltas across every +/// snapshot in the window. Summing the deltas is the whole reason the collector stores them; it needs no +/// interval arithmetic (the "naked family" #3540 rung that is still to land stores the interval the deltas +/// accrued over, which is a RATE question, not this one) and a restart's fabricated 0 delta adds 0. The +/// latest snapshot is served BESIDE the window figures, stamped, so a caller can tell "calm now" from "calm +/// all window". /// internal static class DarlingMemoryGrantReader { @@ -58,6 +70,18 @@ public sealed record MemoryGrantRow( DateTime CollectionTime, int PoolId, double AvailableMemoryMb, double GrantedMemoryMb, double UsedMemoryMb, long GranteeCount, long WaiterCount, long TimeoutErrorCountDelta, long ForcedGrantCountDelta); + /// + /// One (resource_semaphore_id, pool_id) — or, for the pool lens, one pool with + /// null — aggregated over EVERY snapshot in the window (#3541 A10). / + /// is the storm detector: the most sessions ever seen waiting for a grant in the + /// window and the snapshot it happened at. The two *InWindow figures are SUMs of the stored per-interval + /// deltas — how many grants timed out / were forced across the whole window, not just the last interval. + /// + public sealed record MemoryGrantWindowRow( + short? ResourceSemaphoreId, int PoolId, long SnapshotsInWindow, DateTime FirstSnapshotAt, DateTime LastSnapshotAt, + long PeakWaiterCount, DateTime PeakWaitersAt, double PeakGrantedMemoryMb, double MinAvailableMemoryMb, + long TimeoutErrorsInWindow, long ForcedGrantsInWindow); + /* ─────────────────────────── resource semaphore (latest snapshot, per semaphore) ─────────────────────────── */ /// @@ -191,6 +215,178 @@ public static async Task> GetMemoryGrantsLatestAsync( return rows; } + /* ─────────────────────────── the window, per semaphore / per pool (#3541 A10) ─────────────────────────── */ + + /// + /// Every snapshot in the window aggregated per (resource_semaphore_id, pool_id) — the window half of + /// get_resource_semaphore. The peak's instant comes from a DISTINCT ON over the same windowed rows + /// (highest waiter_count first, newest first on a tie, so a storm that plateaued reports its latest + /// snapshot); the two *_in_window figures SUM the stored per-interval deltas. MB aggregates CAST to + /// double precision, count aggregates to bigint, like the latest reads. $1 server_id, $2 window start, + /// $3 window end (naive UTC). + /// + public const string ResourceSemaphoreWindowSql = """ + WITH windowed AS + ( + SELECT + collection_time, + resource_semaphore_id, + pool_id, + waiter_count, + granted_memory_mb, + available_memory_mb, + timeout_error_count_delta, + forced_grant_count_delta + FROM v_memory_grant_stats + WHERE server_id = $1 + AND collection_time >= $2 + AND collection_time <= $3 + ), + agg AS + ( + SELECT + resource_semaphore_id, + pool_id, + COUNT(*) AS snapshots_in_window, + MIN(collection_time) AS first_snapshot_at, + MAX(collection_time) AS last_snapshot_at, + CAST(MAX(waiter_count) AS bigint) AS peak_waiter_count, + CAST(MAX(granted_memory_mb) AS double precision) AS peak_granted_memory_mb, + CAST(MIN(available_memory_mb) AS double precision) AS min_available_memory_mb, + CAST(SUM(timeout_error_count_delta) AS bigint) AS timeout_errors_in_window, + CAST(SUM(forced_grant_count_delta) AS bigint) AS forced_grants_in_window + FROM windowed + GROUP BY resource_semaphore_id, pool_id + ), + peak AS + ( + SELECT DISTINCT ON (resource_semaphore_id, pool_id) + resource_semaphore_id, + pool_id, + collection_time AS peak_waiters_at + FROM windowed + ORDER BY resource_semaphore_id, pool_id, waiter_count DESC, collection_time DESC + ) + SELECT + a.resource_semaphore_id, + a.pool_id, + a.snapshots_in_window, + a.first_snapshot_at, + a.last_snapshot_at, + a.peak_waiter_count, + p.peak_waiters_at, + a.peak_granted_memory_mb, + a.min_available_memory_mb, + a.timeout_errors_in_window, + a.forced_grants_in_window + FROM agg AS a + JOIN peak AS p + ON p.resource_semaphore_id = a.resource_semaphore_id + AND p.pool_id = a.pool_id + ORDER BY a.resource_semaphore_id, a.pool_id + """; + + /// + /// Every snapshot in the window aggregated per pool — the window half of get_memory_grants. The pool lens + /// SUMs across a pool's semaphores at each snapshot first (the same per-snapshot SUM + /// serves), THEN takes the window's peak / floor / total over those + /// per-snapshot pool figures, so "peak waiters" is the most sessions waiting on the pool at any one + /// instant, not the largest single semaphore's count. $1 server_id, $2 window start, $3 window end (naive UTC). + /// + public const string MemoryGrantsWindowSql = """ + WITH per_snapshot AS + ( + SELECT + collection_time, + pool_id, + SUM(waiter_count) AS waiter_count, + SUM(granted_memory_mb) AS granted_memory_mb, + SUM(available_memory_mb) AS available_memory_mb, + SUM(timeout_error_count_delta) AS timeout_error_count_delta, + SUM(forced_grant_count_delta) AS forced_grant_count_delta + FROM v_memory_grant_stats + WHERE server_id = $1 + AND collection_time >= $2 + AND collection_time <= $3 + GROUP BY collection_time, pool_id + ), + agg AS + ( + SELECT + pool_id, + COUNT(*) AS snapshots_in_window, + MIN(collection_time) AS first_snapshot_at, + MAX(collection_time) AS last_snapshot_at, + CAST(MAX(waiter_count) AS bigint) AS peak_waiter_count, + CAST(MAX(granted_memory_mb) AS double precision) AS peak_granted_memory_mb, + CAST(MIN(available_memory_mb) AS double precision) AS min_available_memory_mb, + CAST(SUM(timeout_error_count_delta) AS bigint) AS timeout_errors_in_window, + CAST(SUM(forced_grant_count_delta) AS bigint) AS forced_grants_in_window + FROM per_snapshot + GROUP BY pool_id + ), + peak AS + ( + SELECT DISTINCT ON (pool_id) + pool_id, + collection_time AS peak_waiters_at + FROM per_snapshot + ORDER BY pool_id, waiter_count DESC, collection_time DESC + ) + SELECT + CAST(NULL AS smallint) AS resource_semaphore_id, + a.pool_id, + a.snapshots_in_window, + a.first_snapshot_at, + a.last_snapshot_at, + a.peak_waiter_count, + p.peak_waiters_at, + a.peak_granted_memory_mb, + a.min_available_memory_mb, + a.timeout_errors_in_window, + a.forced_grants_in_window + FROM agg AS a + JOIN peak AS p ON p.pool_id = a.pool_id + ORDER BY a.pool_id + """; + + public static Task> GetResourceSemaphoreWindowAsync( + NpgsqlDataSource postgres, int serverId, DateTime startUtc, DateTime endUtc, CancellationToken cancellationToken = default) => + ReadWindowAsync(postgres, ResourceSemaphoreWindowSql, serverId, startUtc, endUtc, cancellationToken); + + public static Task> GetMemoryGrantsWindowAsync( + NpgsqlDataSource postgres, int serverId, DateTime startUtc, DateTime endUtc, CancellationToken cancellationToken = default) => + ReadWindowAsync(postgres, MemoryGrantsWindowSql, serverId, startUtc, endUtc, cancellationToken); + + /// The two window reads project the SAME eleven columns in the same order (the pool lens fills + /// resource_semaphore_id with a typed NULL), so one materialiser serves both. + private static async Task> ReadWindowAsync( + NpgsqlDataSource postgres, string sql, int serverId, DateTime startUtc, DateTime endUtc, CancellationToken cancellationToken) + { + var rows = new List(); + await using var command = postgres.CreateCommand(sql); + command.CommandTimeout = McpCommandDeadlines.ReadSeconds; + DarlingMcpReadParameters.AddWindow(command, serverId, startUtc, endUtc); + await using var reader = await command.ExecuteReaderAsync(cancellationToken); + while (await reader.ReadAsync(cancellationToken)) + { + rows.Add(new MemoryGrantWindowRow( + reader.IsDBNull(0) ? null : reader.GetInt16(0), + reader.IsDBNull(1) ? 0 : reader.GetInt32(1), + reader.GetInt64(2), + reader.GetDateTime(3), + reader.GetDateTime(4), + reader.IsDBNull(5) ? 0 : reader.GetInt64(5), + reader.GetDateTime(6), + reader.IsDBNull(7) ? 0 : reader.GetDouble(7), + reader.IsDBNull(8) ? 0 : reader.GetDouble(8), + reader.IsDBNull(9) ? 0 : reader.GetInt64(9), + reader.IsDBNull(10) ? 0 : reader.GetInt64(10))); + } + + return rows; + } + /* ─────────────────────────── memory pressure events (RING_BUFFER_RESOURCE_MONITOR) ─────────────────────────── */ /// One RING_BUFFER_RESOURCE_MONITOR sample — the sample time plus the SQL Server (process) and OS diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingPlanCacheSchedulerReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingPlanCacheSchedulerReader.cs index 6a988e83c..111f30815 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingPlanCacheSchedulerReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingPlanCacheSchedulerReader.cs @@ -20,7 +20,7 @@ namespace PerformanceMonitor.Darling.Service.Mcp; /// / get_cpu_scheduler_pressure and the viewer's ViewerDataService.PlanCache / /// ViewerDataService.CpuScheduler read. Both are point-in-time snapshot collectors (one/many rows per /// collection, no deltas), so each read is the LATEST snapshot: plan-cache = every (cacheobjtype, objtype) -/// group at the newest collection in the window; cpu-scheduler = the single newest row for the server. STORED +/// group at the newest collection in the window; cpu-scheduler = the single newest row in the window. STORED /// reads, no live monitored-server hit, on the v_plan_cache_stats / v_cpu_scheduler_stats views. /// /// @@ -135,9 +135,14 @@ public static (string Level, string Recommendation) ClassifyPlanCacheBloat(long /* ─────────────────────────── cpu scheduler (latest snapshot) ─────────────────────────── */ /// - /// The single most recent CPU-scheduler snapshot for the server — the Dashboard's - /// get_cpu_scheduler_pressure point-in-time read (the Dashboard tool takes no window, reading - /// report.cpu_scheduler_pressure's TOP 1). $1 server_id. + /// The single most recent CPU-scheduler snapshot for the server IN THE WINDOW — the Dashboard's + /// get_cpu_scheduler_pressure point-in-time read (report.cpu_scheduler_pressure's TOP 1), + /// bounded the way Lite's GetCpuSchedulerSnapshotAsync bounds it (#3541 A10): the Dashboard tool took + /// no window and served the newest row a store ever held, so a server whose scheduler collector died a week + /// ago answered "NORMAL" with a week-old row and nothing in the payload to say so. The window is a SEARCH + /// bound for the newest snapshot, not an aggregate — the tool's hours_back description says exactly + /// that — and the anchor makes "what did the scheduler look like at 03:00 Tuesday" answerable. + /// $1 server_id, $2 window start, $3 window end (naive UTC). /// public const string CpuSchedulerPressureSql = """ SELECT @@ -157,16 +162,18 @@ public static (string Level, string Recommendation) ClassifyPlanCacheBloat(long physical_memory_pressure_warning FROM v_cpu_scheduler_stats WHERE server_id = $1 + AND collection_time >= $2 + AND collection_time <= $3 ORDER BY collection_time DESC LIMIT 1 """; public static async Task GetCpuSchedulerPressureAsync( - NpgsqlDataSource postgres, int serverId, CancellationToken cancellationToken = default) + NpgsqlDataSource postgres, int serverId, DateTime startUtc, DateTime endUtc, CancellationToken cancellationToken = default) { await using var command = postgres.CreateCommand(CpuSchedulerPressureSql); command.CommandTimeout = McpCommandDeadlines.ReadSeconds; - DarlingMcpReadParameters.AddInt(command, serverId); + DarlingMcpReadParameters.AddWindow(command, serverId, startUtc, endUtc); await using var reader = await command.ExecuteReaderAsync(cancellationToken); if (!await reader.ReadAsync(cancellationToken)) { diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/LatestSnapshot.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/LatestSnapshot.cs new file mode 100644 index 000000000..1b5bb9bc7 --- /dev/null +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/LatestSnapshot.cs @@ -0,0 +1,76 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; + +namespace PerformanceMonitor.Darling.Service.Mcp; + +/// +/// The rows of one LATEST-snapshot read together with the instant that snapshot was captured (#3541 A10). +/// +/// Why the stamp travels with the rows rather than on them. A latest read is +/// WHERE collection_time = (SELECT MAX(collection_time) ...): every row it returns shares ONE +/// stamp by construction, so the stamp is a property of the snapshot, not of a row. Putting it on each row +/// record would either widen positional records that tests and callers construct by hand, or default it — +/// and a defaulted DateTime on a row that was never stamped is exactly the shape this type exists to +/// make unrepresentable. The reader reads the stamp off the first row it materialises (the SELECT carries +/// the column) and the tool publishes it once, as captured_at. +/// +/// Why it is not a second read. A separate SELECT MAX(collection_time) can disagree with +/// the rows when a collection lands between the two statements — the rows would be one snapshot and the +/// stamp the next. Carrying the column on the row statement makes the two provably the same instant. +/// +/// is null exactly when is empty: there is no snapshot to +/// stamp. A tool must test (or ) before reading the stamp, which is +/// the same branch it already takes to return its unavailable status. +/// +internal sealed class LatestSnapshot +{ + /// The empty snapshot — no rows, no stamp. + public static readonly LatestSnapshot Empty = new(null, new List()); + + public LatestSnapshot(DateTime? capturedAt, List rows) + { + if (rows.Count > 0 && capturedAt is null) + { + throw new ArgumentException("a latest snapshot with rows must carry the instant it was captured", nameof(capturedAt)); + } + + CapturedAt = capturedAt; + Rows = rows; + } + + /// The snapshot's collection_time / capture_time (naive UTC, as stored) — the ONE + /// instant every row in was captured at. Null when there are no rows. + public DateTime? CapturedAt { get; } + + public List Rows { get; } + + public int Count => Rows.Count; + + public bool IsEmpty => Rows.Count == 0; +} + +/// +/// The one arithmetic every stamped latest read shares (#3541 A10). +/// +internal static class LatestSnapshotStamp +{ + /// + /// Whole seconds from a snapshot's stamp to the window's end — the anchor the caller asked for, never the + /// service clock (AsOfWindowAnchorTests: an anchored tool's only "now" is its as_of, and a tool + /// body that names DateTime.UtcNow fails the census). The store stamps naive UTC and the anchor is + /// Kind=Utc; the subtraction ignores Kind, which is correct here because both are UTC instants. + /// Non-negative by construction for a windowed read, whose rows are bounded by + /// collection_time <= window end; clamped at zero anyway so a sub-second precision difference + /// between a microsecond store stamp and a 100 ns anchor can never publish "-0". + /// + public static long AgeSeconds(DateTime capturedAt, DateTime windowEnd) => + Math.Max(0L, (long)Math.Round((windowEnd - capturedAt).TotalSeconds)); +} diff --git a/Lite.Tests/McpLatestSnapshotStampTests.cs b/Lite.Tests/McpLatestSnapshotStampTests.cs new file mode 100644 index 000000000..d1104959f --- /dev/null +++ b/Lite.Tests/McpLatestSnapshotStampTests.cs @@ -0,0 +1,448 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor Lite. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.ComponentModel; +using System.IO; +using System.Linq; +using System.Reflection; +using System.Text.Json; +using System.Text.RegularExpressions; +using System.Threading.Tasks; +using DuckDB.NET.Data; +using ModelContextProtocol.Server; +using PerformanceMonitorLite.Database; +using PerformanceMonitorLite.Mcp; +using PerformanceMonitorLite.Models; +using PerformanceMonitorLite.Services; +using Xunit; + +namespace PerformanceMonitorLite.Tests; + +/// +/// #3541 A10, Lite half — latest is a time. Every latest-snapshot MCP read publishes captured_at, +/// the snapshot's own collection instant; the anchored ones publish age_seconds against the anchor the +/// caller sent, never the wall clock; a tool whose hours_back is the span SEARCHED for the newest snapshot +/// refuses one older than that; and the two memory-grant tools READ their window beside the snapshot, so a +/// grant storm three hours ago is visible under a calm latest row. Darling's twin +/// (Darling.Tests/McpLatestSnapshotStampTests) holds the cross-SKU census — roster, shapes, descriptions +/// pinned byte-equal — and executes against live Postgres; this file executes the Lite tools against a real +/// DuckDB through the real tool methods, because the stamps live in the SQL and a helper-only test would pass +/// with the column missing from the SELECT. +/// +/// Every age assertion is an equality against a PAST anchor. Rows are seeded two hours back and +/// the tools are called with as_of five minutes after the newest row, so age_seconds is exactly +/// 300 — an anchor of "now" would make every age a race, and a range assertion would pass a tool that read the +/// service clock instead of the anchor, which is the defect AsOfWindowAnchorTests exists to prevent. +/// +public sealed class McpLatestSnapshotStampTests : IClassFixture, IDisposable +{ + private const string ServerName = "TestServer"; + + private readonly string _tempDir; + private readonly DuckDbInitializer _duckDb; + private readonly LocalDataService _dataService; + private readonly ServerManager _serverManager; + private readonly int _serverId; + private long _nextId = -1; + private DuckDBConnection? _seedConn; + + public McpLatestSnapshotStampTests(SharedDuckDbFixture fixture) + { + fixture.ResetData(); + _duckDb = fixture.DuckDb; + + _tempDir = Path.Combine(Path.GetTempPath(), "McpLatestStamp_" + Guid.NewGuid().ToString("N")[..8]); + var configDir = Path.Combine(_tempDir, "config"); + Directory.CreateDirectory(configDir); + + _dataService = new LocalDataService(_duckDb); + _serverManager = new ServerManager(configDir); + + var server = new ServerConnection { ServerName = ServerName, DisplayName = ServerName }; + _serverManager.AddServer(server); + + /* The derived id, not a literal: seeding under a hand-picked number makes every read return + nothing and the "unavailable" half of every pair pass for the wrong reason. */ + _serverId = RemoteCollectorService.GetDeterministicHashCode( + RemoteCollectorService.GetServerNameForStorage(server)); + } + + public void Dispose() + { + _seedConn?.Dispose(); + try { if (Directory.Exists(_tempDir)) Directory.Delete(_tempDir, recursive: true); } + catch (IOException) { /* best-effort cleanup */ } + catch (UnauthorizedAccessException) { /* best-effort cleanup */ } + } + + /* ───────────────────────── the windowed pair: a storm the latest snapshot cannot see ───────────────────────── */ + + [Fact] + public async Task GetResourceSemaphore_ReadsTheWindowBesideTheStampedSnapshot() + { + var (@base, anchor) = Past(); + await SeedGrantAsync(@base.AddMinutes(-30), waiters: 12, timeoutsDelta: 3, grantedMb: 6000); + await SeedGrantAsync(@base, waiters: 0, timeoutsDelta: 0, grantedMb: 500); + + var root = Parse(await McpMemoryTools.GetResourceSemaphore(_dataService, _serverManager, ServerName, 1, anchor)); + + Assert.Equal(Stamp(@base), root.GetProperty("captured_at").GetString()); + Assert.Equal(300, root.GetProperty("age_seconds").GetInt64()); + Assert.Equal(1, root.GetProperty("hours_back").GetInt32()); + + /* grants[] is the calm latest row; window[] is where the storm lives. */ + var latest = Assert.Single(root.GetProperty("grants").EnumerateArray()); + Assert.Equal(0, latest.GetProperty("waiter_count").GetInt32()); + + var window = Assert.Single(root.GetProperty("window").EnumerateArray()); + Assert.Equal(0, window.GetProperty("resource_semaphore_id").GetInt32()); + Assert.Equal(2, window.GetProperty("pool_id").GetInt32()); + Assert.Equal(2, window.GetProperty("snapshots_in_window").GetInt64()); + Assert.Equal(12, window.GetProperty("peak_waiter_count").GetInt64()); + Assert.Equal(Stamp(@base.AddMinutes(-30)), window.GetProperty("peak_waiters_at").GetString()); + Assert.Equal(3, window.GetProperty("timeout_errors_in_window").GetInt64()); + Assert.Equal(0, window.GetProperty("forced_grants_in_window").GetInt64()); + Assert.Equal(6000d, window.GetProperty("peak_granted_memory_mb").GetDouble()); + Assert.Equal(2000d, window.GetProperty("min_available_memory_mb").GetDouble()); + Assert.Equal(Stamp(@base.AddMinutes(-30)), window.GetProperty("first_snapshot_at").GetString()); + /* The window's last snapshot IS the stamped one — one span, one set of rows, two halves. */ + Assert.Equal(root.GetProperty("captured_at").GetString(), window.GetProperty("last_snapshot_at").GetString()); + } + + [Fact] + public async Task GetMemoryGrants_ReadsThePoolWindow_WithANullSemaphoreId() + { + var (@base, anchor) = Past(); + /* Two semaphores on one pool at the storm instant: the pool lens SUMs them first (7 + 5 = 12 waiters at + one instant), so the peak is the pool's, not the larger semaphore's. */ + await SeedGrantAsync(@base.AddMinutes(-30), waiters: 7, timeoutsDelta: 2, grantedMb: 4000, semaphore: 0); + await SeedGrantAsync(@base.AddMinutes(-30), waiters: 5, timeoutsDelta: 1, grantedMb: 2000, semaphore: 1); + await SeedGrantAsync(@base, waiters: 0, timeoutsDelta: 0, grantedMb: 500, semaphore: 0); + await SeedGrantAsync(@base, waiters: 0, timeoutsDelta: 0, grantedMb: 100, semaphore: 1); + + var root = Parse(await McpMemoryTools.GetMemoryGrants(_dataService, _serverManager, ServerName, 1, anchor)); + + Assert.Equal(Stamp(@base), root.GetProperty("captured_at").GetString()); + Assert.Equal(300, root.GetProperty("age_seconds").GetInt64()); + + var latest = Assert.Single(root.GetProperty("grants").EnumerateArray()); + Assert.Equal(600d, latest.GetProperty("granted_memory_mb").GetDouble()); + + var window = Assert.Single(root.GetProperty("window").EnumerateArray()); + Assert.Equal(JsonValueKind.Null, window.GetProperty("resource_semaphore_id").ValueKind); + Assert.Equal(12, window.GetProperty("peak_waiter_count").GetInt64()); + Assert.Equal(3, window.GetProperty("timeout_errors_in_window").GetInt64()); + Assert.Equal(6000d, window.GetProperty("peak_granted_memory_mb").GetDouble()); + Assert.Equal(2, window.GetProperty("snapshots_in_window").GetInt64()); + } + + /* ───────────────────────── the search-bound pair: a snapshot older than the span is refused ───────────────────────── */ + + [Fact] + public async Task GetCpuSchedulerPressure_IsStamped_Aged_Verdicted_AndBoundedByItsSearchSpan() + { + var (@base, anchor) = Past(); + var schedulerAt = @base.AddMinutes(5).AddHours(-3); + await SeedSchedulerAsync(schedulerAt, runnable: 60); + + var root = Parse(await McpPlanCacheSchedulerTools.GetCpuSchedulerPressure(_dataService, _serverManager, ServerName, 4, anchor)); + Assert.Equal(Stamp(schedulerAt), root.GetProperty("captured_at").GetString()); + Assert.Equal(3 * 3600, root.GetProperty("age_seconds").GetInt64()); + /* The verdict Darling always published, from the SHARED banding: 60 runnable > 50. */ + Assert.Equal("CRITICAL - High runnable task queue", root.GetProperty("pressure_level").GetString()); + Assert.Contains("CPU pressure detected", root.GetProperty("recommendation").GetString(), StringComparison.Ordinal); + + /* One hour of search does not reach a three-hour-old row: unavailable, never a stale verdict. */ + var refused = JsonDocument.Parse(await McpPlanCacheSchedulerTools.GetCpuSchedulerPressure(_dataService, _serverManager, ServerName, 1, anchor)).RootElement; + Assert.Equal("unavailable", refused.GetProperty("status").GetString()); + } + + [Fact] + public async Task GetLatchAndSpinlockStats_AreStampedAndAged() + { + var (@base, anchor) = Past(); + await SeedLatchAsync(@base.AddMinutes(-20), 20000); + await SeedLatchAsync(@base, 100); + await SeedSpinlockAsync(@base); + + var latch = Parse(await McpLatchSpinlockTools.GetLatchStats(_dataService, _serverManager, ServerName, 1, anchor)); + Assert.Equal(Stamp(@base), latch.GetProperty("captured_at").GetString()); + Assert.Equal(300, latch.GetProperty("age_seconds").GetInt64()); + /* The snapshot is the newest one: its delta, not the hot earlier one's. */ + Assert.Equal(100, Assert.Single(latch.GetProperty("latches").EnumerateArray()).GetProperty("delta_wait_time_ms").GetInt64()); + + var spin = Parse(await McpLatchSpinlockTools.GetSpinlockStats(_dataService, _serverManager, ServerName, 1, anchor)); + Assert.Equal(Stamp(@base), spin.GetProperty("captured_at").GetString()); + Assert.Equal(300, spin.GetProperty("age_seconds").GetInt64()); + } + + /* ───────────────────────── the stamped family: the newest snapshot's own instant ───────────────────────── */ + + [Fact] + public async Task GetMemoryClerks_FileIo_Perfmon_StampTheNewestSnapshot_NotTheOlderOne() + { + var (@base, _) = Past(); + foreach (var t in new[] { @base.AddMinutes(-10), @base }) + { + await SeedClerkAsync(t); + await SeedFileIoAsync(t); + await SeedPerfmonAsync(t); + } + + Assert.Equal(Stamp(@base), Parse(await McpMemoryTools.GetMemoryClerks(_dataService, _serverManager, ServerName)).GetProperty("captured_at").GetString()); + Assert.Equal(Stamp(@base), Parse(await McpIoTools.GetFileIoStats(_dataService, _serverManager, ServerName)).GetProperty("captured_at").GetString()); + + /* Perfmon: the stamp comes from the unfiltered snapshot, so a filter that matches nothing still says when. */ + var perfmon = Parse(await McpPerfmonTools.GetPerfmonStats(_dataService, _serverManager, ServerName, "no such counter")); + Assert.Equal(Stamp(@base), perfmon.GetProperty("captured_at").GetString()); + Assert.Empty(perfmon.GetProperty("counters").EnumerateArray()); + } + + [Fact] + public async Task ConfigFamily_StampsTheConnectTimeCapture() + { + var (@base, _) = Past(); + var connectAt = @base.AddDays(-3); + await ExecAsync(@" +INSERT INTO server_config (config_id, capture_time, server_id, server_name, configuration_name, value_configured, value_in_use, is_dynamic, is_advanced) +VALUES ($1, $2, $3, $4, 'max degree of parallelism', 4, 4, true, true)", _nextId--, Naive(connectAt), _serverId, ServerName); + await ExecAsync(@" +INSERT INTO trace_flags (config_id, capture_time, server_id, server_name, trace_flag, status, is_global, is_session) +VALUES ($1, $2, $3, $4, 3226, true, true, false)", _nextId--, Naive(connectAt), _serverId, ServerName); + await ExecAsync(@" +INSERT INTO database_scoped_config (config_id, capture_time, server_id, server_name, database_name, configuration_name, value, value_for_secondary) +VALUES ($1, $2, $3, $4, 'AppDb', 'MAXDOP', '4', NULL)", _nextId--, Naive(connectAt), _serverId, ServerName); + + Assert.Equal(Stamp(connectAt), Parse(await McpConfigTools.GetServerConfig(_dataService, _serverManager, ServerName)).GetProperty("captured_at").GetString()); + Assert.Equal(Stamp(connectAt), Parse(await McpConfigTools.GetTraceFlags(_dataService, _serverManager, ServerName)).GetProperty("captured_at").GetString()); + + /* A database_name filter that matches nothing still says when the (empty) answer is as of. */ + var scoped = Parse(await McpConfigTools.GetDatabaseScopedConfig(_dataService, _serverManager, ServerName, "NoSuchDb")); + Assert.Equal(Stamp(connectAt), scoped.GetProperty("captured_at").GetString()); + Assert.Equal(0, scoped.GetProperty("database_count").GetInt32()); + } + + [Fact] + public async Task GetServerSummary_NamesThreeClocks_SoAStaleCpuRowCannotHideUnderAFreshLog() + { + var (@base, _) = Past(); + var cpuAt = @base.AddDays(-1); + await ExecAsync(@" +INSERT INTO cpu_utilization_stats (collection_id, collection_time, server_id, server_name, sample_time, sqlserver_cpu_utilization, other_process_cpu_utilization) +VALUES ($1, $2, $3, $4, $2, 42, 3)", _nextId--, Naive(cpuAt), _serverId, ServerName); + await ExecAsync(@" +INSERT INTO memory_stats (collection_id, collection_time, server_id, server_name, total_physical_memory_mb, available_physical_memory_mb, total_server_memory_mb) +VALUES ($1, $2, $3, $4, 65536, 8192, 40000)", _nextId--, Naive(@base), _serverId, ServerName); + await ExecAsync(@" +INSERT INTO collection_log (log_id, server_id, server_name, collector_name, collection_time, duration_ms, status, error_message, rows_collected, sql_duration_ms, duckdb_duration_ms) +VALUES ($1, $2, $3, 'memory_stats', $4, 120, 'SUCCESS', NULL, 7, 90, 30)", _nextId--, _serverId, ServerName, Naive(@base.AddMinutes(1))); + + var root = Parse(await McpHealthTools.GetServerSummary(_dataService, _serverManager, ServerName)); + Assert.Equal(Stamp(cpuAt), root.GetProperty("cpu_captured_at").GetString()); + Assert.Equal(Stamp(@base), root.GetProperty("memory_captured_at").GetString()); + Assert.Equal(Stamp(@base.AddMinutes(1)), root.GetProperty("last_collection").GetString()); + Assert.Equal(McpHealthTools.ServerSummaryCountsWindowHours, root.GetProperty("counts_window_hours").GetInt32()); + Assert.Equal(42d, root.GetProperty("cpu_percent").GetDouble()); + } + + /* ───────────────────────── the Lite surface, from reflection and source ───────────────────────── */ + + /// The Lite half of the roster Darling's census walks: every one publishes captured_at and + /// describes itself as a time. Listed here so a Lite-only edit fails a Lite test, not only the cross-SKU one. + public static readonly (Type Tools, string ToolName)[] LatestTools = + [ + (typeof(McpMemoryTools), "get_memory_stats"), + (typeof(McpMemoryTools), "get_memory_clerks"), + (typeof(McpMemoryTools), "get_resource_semaphore"), + (typeof(McpMemoryTools), "get_memory_grants"), + (typeof(McpIoTools), "get_file_io_stats"), + (typeof(McpPerfmonTools), "get_perfmon_stats"), + (typeof(McpConfigTools), "get_server_config"), + (typeof(McpConfigTools), "get_database_config"), + (typeof(McpConfigTools), "get_database_scoped_config"), + (typeof(McpConfigTools), "get_query_store_health"), + (typeof(McpConfigTools), "get_trace_flags"), + (typeof(McpPlanCacheSchedulerTools), "get_plan_cache_bloat"), + (typeof(McpPlanCacheSchedulerTools), "get_cpu_scheduler_pressure"), + (typeof(McpLatchSpinlockTools), "get_latch_stats"), + (typeof(McpLatchSpinlockTools), "get_spinlock_stats"), + ]; + + [Fact] + public void EveryLatestTool_SaysLatestIsATime_AndNamesCapturedAt() + { + foreach (var (type, name) in LatestTools) + { + var description = ToolMethod(type, name).GetCustomAttribute()!.Description; + Assert.Contains("captured_at", description, StringComparison.Ordinal); + Assert.True( + description.Contains("LATEST IS A TIME", StringComparison.Ordinal) || description.Contains("TWO READS UNDER ONE WINDOW", StringComparison.Ordinal), + $"{name}: the description never says the read is a moment, not a window"); + } + } + + /// A latest tool that takes hours_back says which of the two honest things it means, in the + /// words Darling's census pins, and takes the anchor to measure age_seconds against. + [Fact] + public void EveryLatestToolWithAWindowParameter_SaysWhatTheWindowMeans_AndTakesTheAnchor() + { + foreach (var (type, name) in LatestTools) + { + var parameters = ToolMethod(type, name).GetParameters(); + var hours = parameters.SingleOrDefault(p => p.Name == "hours_back"); + if (hours is null) + { + Assert.DoesNotContain("as_of", parameters.Select(p => p.Name)); + continue; + } + + var words = hours.GetCustomAttribute()!.Description; + Assert.True( + words.Contains("search for the latest snapshot", StringComparison.Ordinal) + || words.Contains("window[] aggregates every snapshot in these hours", StringComparison.Ordinal), + $"{name}: hours_back is described as neither the search span nor the read window: \"{words}\""); + Assert.Contains("as_of", parameters.Select(p => p.Name)); + } + } + + /// The two window reads, pinned on the dialect: the peak's instant from a DISTINCT ON over + /// the SAME windowed rows, the deltas SUMmed, and no interval arithmetic and no literal cap. + [Theory] + [InlineData(nameof(LocalDataService.ResourceSemaphoreWindowSql))] + [InlineData(nameof(LocalDataService.MemoryGrantsWindowSql))] + public void MemoryGrantWindowReads_AggregateEverySnapshot_AndNameThePeaksInstant(string sqlName) + { + var sql = sqlName == nameof(LocalDataService.ResourceSemaphoreWindowSql) + ? LocalDataService.ResourceSemaphoreWindowSql + : LocalDataService.MemoryGrantsWindowSql; + Assert.Contains("collection_time >= $2", sql, StringComparison.Ordinal); + Assert.Contains("collection_time <= $3", sql, StringComparison.Ordinal); + Assert.Contains("COUNT(*) AS snapshots_in_window", sql, StringComparison.Ordinal); + Assert.Contains("MAX(waiter_count) AS peak_waiter_count", sql, StringComparison.Ordinal); + Assert.Contains("SUM(timeout_error_count_delta) AS timeout_errors_in_window", sql, StringComparison.Ordinal); + Assert.Contains("SELECT DISTINCT ON", sql, StringComparison.Ordinal); + Assert.Contains("waiter_count DESC, collection_time DESC", sql, StringComparison.Ordinal); + Assert.DoesNotMatch(@"\bLIMIT\b", sql); + Assert.DoesNotContain("sample_interval_seconds", sql, StringComparison.Ordinal); + } + + /// Whole seconds from stamp to anchor; never negative; Kind-blind — the twin of Darling's + /// LatestSnapshotStamp.AgeSeconds, held to the same values. + [Fact] + public void AgeSeconds_IsTheWholeSecondDistanceToTheAnchor_AndNeverNegative() + { + var stamp = new DateTime(2026, 9, 18, 12, 0, 0, DateTimeKind.Unspecified); + var anchor = new DateTime(2026, 9, 18, 12, 5, 0, DateTimeKind.Utc); + Assert.Equal(300L, McpLatestSnapshotStamp.AgeSeconds(stamp, anchor)); + Assert.Equal(300L, McpLatestSnapshotStamp.AgeSeconds(stamp, anchor.AddTicks(4_000_000))); + Assert.Equal(301L, McpLatestSnapshotStamp.AgeSeconds(stamp, anchor.AddTicks(6_000_000))); + Assert.Equal(0L, McpLatestSnapshotStamp.AgeSeconds(anchor, stamp)); + Assert.Equal(0L, McpLatestSnapshotStamp.AgeSeconds(stamp, stamp)); + } + + /* ───────────────────────── plumbing ───────────────────────── */ + + /// A base instant two hours back, truncated to the second, and an anchor five minutes after it — + /// the pair every age assertion is an equality against. + private static (DateTime Base, string Anchor) Past() + { + var now = DateTime.UtcNow; + var @base = new DateTime(now.Ticks - (now.Ticks % TimeSpan.TicksPerSecond), DateTimeKind.Utc).AddHours(-2); + return (@base, @base.AddMinutes(5).ToString("o")); + } + + private static MethodInfo ToolMethod(Type type, string toolName) => type + .GetMethods(BindingFlags.Public | BindingFlags.Static) + .Single(m => m.GetCustomAttribute()?.Name == toolName); + + private static JsonElement Parse(string json) + { + var root = JsonDocument.Parse(json).RootElement.Clone(); + Assert.False(root.TryGetProperty("status", out _), "expected a data-bearing payload, got a status envelope: " + json); + return root; + } + + /// What a seeded UTC instant looks like on the payload: the store holds a naive timestamp, and the + /// tools emit it with ToString("o") on a Kind = Unspecified value, so no Z. + private static string Stamp(DateTime utc) => DateTime.SpecifyKind(utc, DateTimeKind.Unspecified).ToString("o"); + + private static DateTime Naive(DateTime utc) => DateTime.SpecifyKind(utc, DateTimeKind.Unspecified); + + private async Task SeedConnectionAsync() + { + if (_seedConn is null) + { + _seedConn = _duckDb.CreateConnection(); + await _seedConn.OpenAsync(); + } + return _seedConn; + } + + private async Task ExecAsync(string sql, params object?[] values) + { + using var readLock = _duckDb.AcquireReadLock(); + var conn = await SeedConnectionAsync(); + using var cmd = conn.CreateCommand(); + cmd.CommandText = sql; + foreach (var v in values) + cmd.Parameters.Add(new DuckDBParameter { Value = v ?? DBNull.Value }); + await cmd.ExecuteNonQueryAsync(); + } + + private Task SeedGrantAsync(DateTime at, int waiters, long timeoutsDelta, double grantedMb, short semaphore = 0, int pool = 2) => ExecAsync(@" +INSERT INTO memory_grant_stats + (collection_id, collection_time, server_id, server_name, resource_semaphore_id, pool_id, + target_memory_mb, max_target_memory_mb, total_memory_mb, available_memory_mb, granted_memory_mb, used_memory_mb, + grantee_count, waiter_count, timeout_error_count, forced_grant_count, timeout_error_count_delta, forced_grant_count_delta) +VALUES ($1, $2, $3, $4, $5, $6, 8000, 12000, 8000, $7, $8, $8, 3, $9, 4, 2, $10, 0)", + _nextId--, Naive(at), _serverId, ServerName, semaphore, pool, 8000 - grantedMb, grantedMb, waiters, timeoutsDelta); + + private Task SeedSchedulerAsync(DateTime at, int runnable) => ExecAsync(@" +INSERT INTO cpu_scheduler_stats + (collection_id, collection_time, server_id, server_name, max_workers_count, scheduler_count, cpu_count, + total_runnable_tasks_count, total_work_queue_count, total_current_workers_count, avg_runnable_tasks_count, + total_active_request_count, total_queued_request_count, total_blocked_task_count, total_active_parallel_thread_count, + runnable_percent, worker_thread_exhaustion_warning, runnable_tasks_warning, blocked_tasks_warning, queued_requests_warning, + total_physical_memory_kb, available_physical_memory_kb, physical_memory_pressure_warning, + total_node_count, nodes_online_count, offline_cpu_count, offline_cpu_warning) +VALUES ($1, $2, $3, $4, 512, 8, 8, $5, 5, 100, 7.5, 40, 12, 2, 20, 12.5, false, true, false, true, 65536000, 32768000, false, 1, 1, 0, false)", + _nextId--, Naive(at), _serverId, ServerName, runnable); + + private Task SeedLatchAsync(DateTime at, long deltaWaitMs) => ExecAsync(@" +INSERT INTO latch_stats + (collection_id, collection_time, server_id, server_name, latch_class, waiting_requests_count, wait_time_ms, max_wait_time_ms, + delta_waiting_requests_count, delta_wait_time_ms, delta_max_wait_time_ms, sample_interval_seconds) +VALUES ($1, $2, $3, $4, 'ACCESS_METHODS_DATASET_PARENT', 1000, 20100, 50, 100, $5, 5, 60)", + _nextId--, Naive(at), _serverId, ServerName, deltaWaitMs); + + private Task SeedSpinlockAsync(DateTime at) => ExecAsync(@" +INSERT INTO spinlock_stats + (collection_id, collection_time, server_id, server_name, spinlock_name, collisions, spins, spins_per_collision, sleep_time, backoffs, + delta_collisions, delta_spins, delta_sleep_time, delta_backoffs) +VALUES ($1, $2, $3, $4, 'LOCK_HASH', 900000, 5000000, 5.5, 100, 200, 400, 2000, 3, 7)", + _nextId--, Naive(at), _serverId, ServerName); + + private Task SeedClerkAsync(DateTime at) => ExecAsync(@" +INSERT INTO memory_clerks (collection_id, collection_time, server_id, server_name, clerk_type, memory_mb) +VALUES ($1, $2, $3, $4, 'MEMORYCLERK_SQLBUFFERPOOL', 40000)", + _nextId--, Naive(at), _serverId, ServerName); + + private Task SeedFileIoAsync(DateTime at) => ExecAsync(@" +INSERT INTO file_io_stats + (collection_id, collection_time, server_id, server_name, database_name, file_name, file_type, physical_name, size_mb, + delta_reads, delta_writes, delta_read_bytes, delta_write_bytes, delta_stall_read_ms, delta_stall_write_ms, sample_interval_seconds) +VALUES ($1, $2, $3, $4, 'AppDb', 'AppDb_data', 'ROWS', 'D:\AppDb.mdf', 100, 10, 5, 81920, 40960, 50, 10, 60)", + _nextId--, Naive(at), _serverId, ServerName); + + private Task SeedPerfmonAsync(DateTime at) => ExecAsync(@" +INSERT INTO perfmon_stats + (collection_id, collection_time, server_id, server_name, object_name, counter_name, instance_name, cntr_value, delta_cntr_value, sample_interval_seconds) +VALUES ($1, $2, $3, $4, 'SQLServer:SQL Statistics', 'Batch Requests/sec', '', 1200000, 600000, 60)", + _nextId--, Naive(at), _serverId, ServerName); +} diff --git a/Lite/Mcp/McpConfigTools.cs b/Lite/Mcp/McpConfigTools.cs index 33bdadd60..65f7584cb 100644 --- a/Lite/Mcp/McpConfigTools.cs +++ b/Lite/Mcp/McpConfigTools.cs @@ -9,7 +9,7 @@ namespace PerformanceMonitorLite.Mcp; [McpServerToolType] public sealed class McpConfigTools { - [McpServerTool(Name = "get_server_config"), Description("Gets the current SQL Server instance configuration (sys.configurations). Shows all sp_configure settings with configured and in-use values. Useful for checking CTFP, MAXDOP, max memory, and other instance-level settings.")] + [McpServerTool(Name = "get_server_config"), Description("Gets the current SQL Server instance configuration (sys.configurations). Shows all sp_configure settings with configured and in-use values. Useful for checking CTFP, MAXDOP, max memory, and other instance-level settings right now (unlike get_server_config_changes, which shows only what changed between connect snapshots). LATEST IS A TIME: configuration is captured when the collector CONNECTS, not on a schedule, so 'current' here means 'as of the last capture' - captured_at is that instant, and a value can be days old on a server the monitor has stayed connected to.")] public static async Task GetServerConfig( LocalDataService dataService, ServerManager serverManager, @@ -30,6 +30,8 @@ public static async Task GetServerConfig( return JsonSerializer.Serialize(new { server = resolved.ServerName, + /* #3541 A10: the connect-time capture this "current" configuration is as of. */ + captured_at = rows[0].CaptureTime.ToString("o"), setting_count = rows.Count, settings = rows.Select(r => new { @@ -48,7 +50,7 @@ public static async Task GetServerConfig( } } - [McpServerTool(Name = "get_database_config"), Description("Gets database-level configuration for all databases (sys.databases). Shows recovery model, RCSI, auto-shrink, auto-close, Query Store, compatibility level, page verify, and other settings. Critical for identifying misconfigured databases.")] + [McpServerTool(Name = "get_database_config"), Description("Gets database-level configuration for all databases (sys.databases). Shows recovery model, RCSI, auto-shrink, auto-close, Query Store, compatibility level, page verify, and other settings. Critical for identifying misconfigured databases. LATEST IS A TIME: captured when the collector connects, not on a schedule - captured_at is the instant these settings are as of, and a database created or altered since is not reflected until the next connect.")] public static async Task GetDatabaseConfig( LocalDataService dataService, ServerManager serverManager, @@ -71,6 +73,9 @@ public static async Task GetDatabaseConfig( if (!string.IsNullOrEmpty(database_name)) filtered = filtered.Where(r => r.DatabaseName.Equals(database_name, StringComparison.OrdinalIgnoreCase)); + /* Taken from the unfiltered snapshot, so a database_name that matches nothing still says when. */ + var capturedAt = rows[0].CaptureTime; + var result = filtered.Select(r => new { database_name = r.DatabaseName, @@ -98,6 +103,7 @@ public static async Task GetDatabaseConfig( return JsonSerializer.Serialize(new { server = resolved.ServerName, + captured_at = capturedAt.ToString("o"), database_count = result.Count, databases = result }, McpHelpers.JsonOptions); @@ -108,7 +114,7 @@ public static async Task GetDatabaseConfig( } } - [McpServerTool(Name = "get_database_scoped_config"), Description("Gets database-scoped configuration settings (sys.database_scoped_configurations). Shows MAXDOP, legacy CE, parameter sniffing, and other per-database settings.")] + [McpServerTool(Name = "get_database_scoped_config"), Description("Gets database-scoped configuration settings (sys.database_scoped_configurations). Shows MAXDOP, legacy CE, parameter sniffing, and other per-database settings. LATEST IS A TIME: captured when the collector connects, not on a schedule - captured_at is the instant these settings are as of.")] public static async Task GetDatabaseScopedConfig( LocalDataService dataService, ServerManager serverManager, @@ -131,6 +137,8 @@ public static async Task GetDatabaseScopedConfig( if (!string.IsNullOrEmpty(database_name)) filtered = filtered.Where(r => r.DatabaseName.Equals(database_name, StringComparison.OrdinalIgnoreCase)); + var capturedAt = rows[0].CaptureTime; + var grouped = filtered .GroupBy(r => r.DatabaseName) .Select(g => new @@ -147,6 +155,7 @@ public static async Task GetDatabaseScopedConfig( return JsonSerializer.Serialize(new { server = resolved.ServerName, + captured_at = capturedAt.ToString("o"), database_count = grouped.Count, databases = grouped }, McpHelpers.JsonOptions); @@ -157,7 +166,7 @@ public static async Task GetDatabaseScopedConfig( } } - [McpServerTool(Name = "get_query_store_health"), Description("Gets per-database Query Store health (sys.database_query_store_options): actual vs desired state, readonly_reason (decoded), storage used vs cap, cleanup mode and thresholds, and the runtime-stats interval length. The classic silent failure is desired READ_WRITE with actual READ_ONLY after the storage cap hit — check this when Query Store data looks stale or missing. Collected hourly; OFF is recorded as OFF (an absent database means not collected, never off).")] + [McpServerTool(Name = "get_query_store_health"), Description("Gets per-database Query Store health (sys.database_query_store_options): actual vs desired state, readonly_reason (decoded), storage used vs cap, cleanup mode and thresholds, and the runtime-stats interval length. The classic silent failure is desired READ_WRITE with actual READ_ONLY after the storage cap hit — check this when Query Store data looks stale or missing. Collected hourly; OFF is recorded as OFF (an absent database means not collected, never off). LATEST IS A TIME: this is the newest hourly capture, and captured_at is its instant.")] public static async Task GetQueryStoreHealth( LocalDataService dataService, ServerManager serverManager, @@ -180,6 +189,8 @@ public static async Task GetQueryStoreHealth( if (!string.IsNullOrEmpty(database_name)) filtered = filtered.Where(r => r.DatabaseName.Equals(database_name, StringComparison.OrdinalIgnoreCase)); + var capturedAt = rows[0].CaptureTime; + var result = filtered.Select(r => new { database_name = r.DatabaseName, @@ -201,6 +212,7 @@ public static async Task GetQueryStoreHealth( return JsonSerializer.Serialize(new { server = resolved.ServerName, + captured_at = capturedAt.ToString("o"), database_count = result.Count, databases = result }, McpHelpers.JsonOptions); @@ -211,7 +223,7 @@ public static async Task GetQueryStoreHealth( } } - [McpServerTool(Name = "get_trace_flags"), Description("Gets active trace flags on the SQL Server instance. Shows flag number, enabled status, and whether the flag is global or session-scoped.")] + [McpServerTool(Name = "get_trace_flags"), Description("Gets active trace flags on the SQL Server instance. Shows flag number, enabled status, and whether the flag is global or session-scoped. LATEST IS A TIME: captured when the collector connects, not on a schedule - captured_at is the instant these flags are as of; a flag turned on or off since is not reflected until the next connect.")] public static async Task GetTraceFlags( LocalDataService dataService, ServerManager serverManager, @@ -230,6 +242,7 @@ public static async Task GetTraceFlags( return JsonSerializer.Serialize(new { server = resolved.ServerName, + captured_at = rows[0].CaptureTime.ToString("o"), trace_flag_count = rows.Count, trace_flags = rows.Select(r => new { diff --git a/Lite/Mcp/McpHealthTools.cs b/Lite/Mcp/McpHealthTools.cs index f6b9a6ac1..a94cb14d3 100644 --- a/Lite/Mcp/McpHealthTools.cs +++ b/Lite/Mcp/McpHealthTools.cs @@ -12,7 +12,17 @@ namespace PerformanceMonitorLite.Mcp; [McpServerToolType] public sealed class McpHealthTools { - [McpServerTool(Name = "get_server_summary"), Description("Gets a quick health overview for a SQL Server instance: current CPU %, memory usage, recent blocking count, and deadlock count. Use this for a fast health check before drilling into specific areas.")] + /// get_server_summary's description, VERBATIM Darling's (#3541 A10); the cross-SKU census pins them + /// equal. Names the three clocks the payload carries, because the payload used to carry one. + internal const string ServerSummaryDescription = + "Gets a quick health overview for a SQL Server instance: current CPU %, memory usage, recent blocking count, and deadlock count. Use this for a fast health check before drilling into specific areas. THREE CLOCKS, NAMED: cpu_percent is the newest CPU snapshot and cpu_captured_at is its instant; memory_mb is the newest memory snapshot and memory_captured_at is its instant; last_collection is the newest collection of ANY collector for this server - the store's freshness, NOT the age of the two figures above, which can be far older when their own collectors have stopped. blocking_count and deadlock_count cover the counts_window_hours ending now."; + + /// The span GetServerSummaryAsync's blocking and deadlock counts cover (its two + /// UtcNow.AddHours(-1) bounds), published so "recent" has a number. Darling's twin is + /// DarlingHealthReader.ServerSummaryCountsWindowHours. + internal const int ServerSummaryCountsWindowHours = 1; + + [McpServerTool(Name = "get_server_summary"), Description(ServerSummaryDescription)] public static async Task GetServerSummary( LocalDataService dataService, ServerManager serverManager, @@ -35,9 +45,15 @@ public static async Task GetServerSummary( { server = resolved.ServerName, cpu_percent = summary.CpuPercent, + /* #3541 A10: each latest figure carries ITS OWN clock. last_collection below is the newest + collection of ANY collector — a live collection log beside a dead CPU collector made a + day-old cpu_percent read as current, because the only stamp on the payload was fresh. */ + cpu_captured_at = summary.CpuCollectionTime?.ToString("o"), memory_mb = summary.MemoryMb, + memory_captured_at = summary.MemoryCollectionTime?.ToString("o"), blocking_count = summary.BlockingCount, deadlock_count = summary.DeadlockCount, + counts_window_hours = ServerSummaryCountsWindowHours, last_collection = summary.LastCollectionTime?.ToString("o") }, McpHelpers.JsonOptions); } diff --git a/Lite/Mcp/McpInstructions.cs b/Lite/Mcp/McpInstructions.cs index c8e58e723..b957c8001 100644 --- a/Lite/Mcp/McpInstructions.cs +++ b/Lite/Mcp/McpInstructions.cs @@ -46,6 +46,7 @@ You are connected to a SQL Server performance monitoring tool via Performance Mo - An unparseable `as_of`, or one in the future, is REFUSED with a message rather than quietly answered as "now" — a read that silently reverts to now is indistinguishable from a correct one. - An `as_of` older than anything the store still holds is NOT refused. It returns the read's normal `empty` / `unavailable` status, which means exactly what it says: we looked in the window you named and there was nothing in it. - Tools that take no window at all (latest-snapshot reads like `get_memory_stats`, `get_file_io_stats`, `get_index_usage`, and the configuration reads) do not take `as_of` — they read the newest row, and there is no window to move. + - **Latest is a time.** Every latest-snapshot read publishes `captured_at` — the snapshot's own collection instant — and the anchored ones publish `age_seconds` against the window's end. Read it before treating a "current" figure as current: the newest row a store holds is as old as its collector's last successful run, and the configuration family is captured on CONNECT, so a "current" setting can be days old. Where a latest read takes `hours_back`, its description says which of two things that means: the span SEARCHED for the newest snapshot (`get_latch_stats`, `get_cpu_scheduler_pressure`, `get_plan_cache_bloat` — a snapshot older than that is `unavailable`, not served as current), or a span READ beside the snapshot (`get_resource_semaphore` / `get_memory_grants` return `grants[]`, the newest snapshot, AND `window[]`, the peak / floor / summed-delta aggregate over every snapshot in the hours). `get_server_summary` carries three clocks by name — `cpu_captured_at`, `memory_captured_at`, and `last_collection` (the newest collection of ANY collector, which is the store's freshness and not the age of the two figures). - The analysis family DOES take it (#2506), and the anchor reaches the ENGINE rather than stopping at the tool: `get_analysis_facts` and `analyze_server` re-run fact collection and scoring over the anchored window, and `analyze_server`'s anomaly detection moves with it, so the window is compared against the hour-of-day x day-of-week baseline for the hours it actually covers instead of for the hours you happen to be asking in. `compare_analysis` hangs BOTH windows off the anchor, since `baseline_hours_back` has always been measured from the comparison window's end. `get_analysis_findings` is the odd one and worth reading twice: its window is on ANALYSIS TIME, so anchoring it asks what a scheduled analysis pass was SAYING then, which is a different question from re-analyzing that window now (that is `analyze_server` with the same anchor). - `analyze_server` with an `as_of` is EXPLORATORY and does NOT persist its findings; the result says so in `persisted` / `persistence_note`. A finding row is stamped with the time the analysis RAN, and `get_analysis_findings` and the viewer's Recommendations tab treat the newest `analysis_time` as the server's CURRENT state — so writing a backdated run would make last week's findings today's headline and would inflate the occurrence stats of any live incident sharing a story path. Run it without `as_of` when you want the present analyzed and recorded. - `get_pvs_stats` does not take it. It mixes a latest-snapshot measurement with a windowed trend, so anchoring only the windowed half would return a result whose two halves describe different instants. @@ -58,7 +59,7 @@ You are connected to a SQL Server performance monitoring tool via Performance Mo | `get_collection_log` | The RAW per-run collection log behind that rollup: one row per collector run with total duration split into time on the monitored server and time on the store, rows collected, status and any error. Reach for it when the rollup reads HEALTHY and collection still looks wrong, or to see what a collector was doing during a specific window. Newest first, or SLOWEST first when `min_duration_ms` is supplied — a duration floor under newest-first ordering cannot reach the tail, so the two are one decision and `order` names which you got. Both filters are applied in SQL BEFORE the cap, so `run_count` and `truncated` describe the MATCHING rows. `hours_back` is the span you asked for; `oldest_returned_collection_time` and `newest_returned_collection_time` bound the PAGE you got (under the default ordering that is also the reach; under a duration floor the page is a cost-ranked sample and its oldest row says nothing about reach), and the cap can make those differ by orders of magnitude — read them before concluding anything from the rows. An empty result distinguishes THREE states: filters that matched nothing (`empty`, and it says nothing about the window as a whole), a quiet window (`empty`, widen it), and a server that has never collected (`unavailable`, collection is not running) | `server_name`, `hours_back`, `limit`, `as_of`, `collector_name`, `min_duration_ms` | | `get_current_waits_trend` | The two Current Waits series over time: waiting-task total wait per wait type per collection, and blocked-session counts per database per collection. `get_waiting_tasks` gives the snapshot and can never say whether now is worse than an hour ago; this is that question. Read the two series together — a wait-type spike with no blocked sessions is a resource wait, the same spike with them is contention. An empty result distinguishes a genuine all-clear (`empty`) from a server the collector has never sampled (`unavailable`), which is NOT an all-clear | `server_name`, `hours_back`, `database_name`, `as_of` | | `get_blocking_stats` | Blocking SEVERITY per minute: blocking duration (event count, total, max, avg wait) and deadlock severity (victim count plus total/max/avg wait across EVERY process in the graphs, not just victims). `get_blocking_trend` and `get_deadlock_trend` say how OFTEN; this says how BAD — ten one-second blocks and one ten-minute block are the same count and a different problem. An empty result distinguishes a genuinely clear window (`empty`) from a server where neither capture path has ever produced a row (`unavailable`), which is NOT a clean bill of health | `server_name`, `hours_back`, `as_of` | - | `get_server_summary` | Quick health overview: CPU %, memory, blocking/deadlock counts | `server_name` | + | `get_server_summary` | Quick health overview: CPU %, memory, blocking/deadlock counts; three clocks named (`cpu_captured_at`, `memory_captured_at`, `last_collection` = newest collection of ANY collector) | `server_name` | | `get_daily_summary` | Daily composite health band + wait/query/deadlock/blocking/CPU/memory/alert rollup for one day | `server_name`, `summary_date` (yyyy-MM-dd, default today) | | `get_daily_summary_range` | The SAME rollup across a span of days — one row per collected day, the Performance Calendar's month grid. Use it when the question is WHICH day rather than how one day went: scan the bands, then call `get_daily_summary` for the day that stands out. A day with ANY collection appears even when every signal was quiet, so a day absent from the result is a gap in COLLECTION. `as_of` anchors the LAST day of the range. An empty result distinguishes a range outside this server's history (`empty`) from a server nothing has ever been collected for (`unavailable`) | `server_name`, `days_back` (default 30, max 366), `as_of` | @@ -74,18 +75,18 @@ You are connected to a SQL Server performance monitoring tool via Performance Mo | Tool | Purpose | Key Parameters | |------|---------|----------------| | `get_cpu_utilization` | SQL Server CPU vs other process CPU over time | `server_name`, `hours_back` (default 4), `as_of` | - | `get_cpu_scheduler_pressure` | Latest scheduler snapshot: runnable queue depth, worker-thread utilization, queued/blocked requests, pressure warnings | `server_name`, `hours_back` (default 24), `as_of` | + | `get_cpu_scheduler_pressure` | Latest scheduler snapshot within `hours_back` of `as_of`: runnable queue depth, worker-thread utilization, queued/blocked requests, pressure warnings, banded `pressure_level` + `recommendation`; `captured_at` / `age_seconds` | `server_name`, `hours_back` (default 24; the span SEARCHED for the newest snapshot), `as_of` | ### Contention Tools | Tool | Purpose | Key Parameters | |------|---------|----------------| - | `get_latch_stats` | Latest latch-contention snapshot by class (waits + last-interval delta waits) | `server_name`, `hours_back` (default 24), `as_of` | - | `get_spinlock_stats` | Latest spinlock-contention snapshot (collisions, spins, backoffs) | `server_name`, `hours_back` (default 24), `as_of` | + | `get_latch_stats` | Latest latch-contention snapshot by class (waits + last-interval delta waits); `captured_at` / `age_seconds` | `server_name`, `hours_back` (default 24; the span SEARCHED for the newest snapshot), `as_of` | + | `get_spinlock_stats` | Latest spinlock-contention snapshot (collisions, spins, backoffs); `captured_at` / `age_seconds` | `server_name`, `hours_back` (default 24; the span SEARCHED for the newest snapshot), `as_of` | ### Plan Cache Tools | Tool | Purpose | Key Parameters | |------|---------|----------------| - | `get_plan_cache_bloat` | Single-use vs multi-use plan composition per cache/object type, with bloat-level classification | `server_name`, `hours_back` (default 24), `as_of` | + | `get_plan_cache_bloat` | Single-use vs multi-use plan composition per cache/object type, with bloat-level classification; `captured_at` / `age_seconds` | `server_name`, `hours_back` (default 24; the span SEARCHED for the newest snapshot), `as_of` | ### Query Performance Tools | Tool | Purpose | Key Parameters | @@ -115,17 +116,17 @@ You are connected to a SQL Server performance monitoring tool via Performance Mo ### Memory Tools | Tool | Purpose | Key Parameters | |------|---------|----------------| - | `get_memory_stats` | Latest memory snapshot: physical, buffer pool, plan cache | `server_name` | + | `get_memory_stats` | Latest memory snapshot: physical, buffer pool, plan cache; `captured_at` | `server_name` | | `get_memory_trend` | Memory usage over time. An empty result distinguishes a quiet window (`empty`, widen `hours_back`) from a server nothing has ever been collected for (`unavailable`, collection is not running) | `server_name`, `hours_back`, `as_of` | | `get_memory_clerks` | Top memory consumers by clerk type. An empty result is `unavailable`, never a quiet period — a live SQL Server always has clerks, so nothing retained means the collector has not run or its rows aged out | `server_name` | - | `get_memory_grants` | Active/recent memory grants (detect grant pressure) | `server_name`, `hours_back` (default 1), `limit`, `as_of` | - | `get_resource_semaphore` | Latest resource-semaphore snapshot: workspace memory vs target/max ceiling, waiter/timeout/forced-grant pressure | `server_name`, `hours_back` (default 24), `as_of` | + | `get_memory_grants` | Per-pool grant pressure: `grants[]` = newest snapshot in the window (`captured_at` / `age_seconds`) AND `window[]` = peak waiters (+ when), peak grant, available floor, summed timeout / forced deltas over EVERY snapshot in the window | `server_name`, `hours_back` (default 1), `as_of` | + | `get_resource_semaphore` | Per-semaphore workspace memory vs target/max ceiling: `grants[]` = newest snapshot in the window (`captured_at` / `age_seconds`) AND `window[]` = the same peak / floor / summed-delta aggregate per (semaphore, pool) | `server_name`, `hours_back` (default 24), `as_of` | | `get_memory_pressure_events` | Ring buffer memory pressure notifications (sp_pressuredetector source) | `server_name`, `hours_back`, `as_of` | ### I/O Tools | Tool | Purpose | Key Parameters | |------|---------|----------------| - | `get_file_io_stats` | Latest file I/O stats per database file with latency | `server_name` | + | `get_file_io_stats` | Latest file I/O stats per database file with latency; `captured_at` | `server_name` | | `get_file_io_trend` | I/O latency trend over time per database. An empty result distinguishes a quiet window (`empty`, widen `hours_back`) from a server nothing has ever been collected for (`unavailable`, collection is not running) | `server_name`, `hours_back`, `as_of` | ### TempDB Tools @@ -143,7 +144,7 @@ You are connected to a SQL Server performance monitoring tool via Performance Mo ### Performance Counter Tools | Tool | Purpose | Key Parameters | |------|---------|----------------| - | `get_perfmon_stats` | Latest perfmon counters (batch requests/sec, etc.) | `server_name`, `counter_name`, `instance_name` | + | `get_perfmon_stats` | Latest perfmon counters (batch requests/sec, etc.); `captured_at` | `server_name`, `counter_name`, `instance_name` | | `get_perfmon_trend` | Time-series for a specific perfmon counter | `counter_name` (required), `server_name`, `hours_back`, `as_of` | ### Alert Tools @@ -161,11 +162,11 @@ You are connected to a SQL Server performance monitoring tool via Performance Mo ### Configuration Tools | Tool | Purpose | Key Parameters | |------|---------|----------------| - | `get_server_config` | sp_configure settings with configured and in-use values | `server_name` | - | `get_database_config` | Database-level settings: RCSI, recovery model, auto-shrink, Query Store, etc. | `server_name`, `database_name` | - | `get_database_scoped_config` | Database-scoped configuration (MAXDOP, legacy CE, parameter sniffing) | `server_name`, `database_name` | + | `get_server_config` | sp_configure settings with configured and in-use values; `captured_at` (captured on connect — can be days old) | `server_name` | + | `get_database_config` | Database-level settings: RCSI, recovery model, auto-shrink, Query Store, etc.; `captured_at` (captured on connect) | `server_name`, `database_name` | + | `get_database_scoped_config` | Database-scoped configuration (MAXDOP, legacy CE, parameter sniffing); `captured_at` (captured on connect) | `server_name`, `database_name` | | `get_query_store_health` | Per-database Query Store health (latest hourly snapshot) — actual vs desired state, readonly_reason decoded, storage vs cap, cleanup thresholds | `server_name`, `database_name` | - | `get_trace_flags` | Active trace flags with global/session scope | `server_name` | + | `get_trace_flags` | Active trace flags with global/session scope; `captured_at` (captured on connect) | `server_name` | | `get_server_config_changes` | sp_configure change history (diff of on-connect snapshots) | `server_name`, `hours_back` (default 168), `as_of` | | `get_database_config_changes` | sys.databases change history (recovery model, RCSI, compat level, etc.) | `server_name`, `hours_back` (default 168), `as_of` | | `get_trace_flag_changes` | Trace flag enable/disable history (diff of on-connect snapshots) | `server_name`, `hours_back` (default 168), `as_of` | diff --git a/Lite/Mcp/McpIoTools.cs b/Lite/Mcp/McpIoTools.cs index 4eb8520c3..036566f65 100644 --- a/Lite/Mcp/McpIoTools.cs +++ b/Lite/Mcp/McpIoTools.cs @@ -9,7 +9,7 @@ namespace PerformanceMonitorLite.Mcp; [McpServerToolType] public sealed class McpIoTools { - [McpServerTool(Name = "get_file_io_stats"), Description("Gets the latest file I/O statistics per database file: read/write counts, bytes, stall times, and calculated latency. High read latency (>20ms) or write latency (>10ms for data, >2ms for log) often indicates storage bottlenecks. Each row carries sample_interval_seconds, the measured seconds its deltas accrued over; a 0 means no delta was knowable for that file at this collection (first sighting, counter reset, or a gap past the delta policy — typically a restart) and its latencies are null rather than 0.")] + [McpServerTool(Name = "get_file_io_stats"), Description("Gets the latest file I/O statistics per database file: read/write counts, bytes, stall times, and calculated latency. High read latency (>20ms) or write latency (>10ms for data, >2ms for log) often indicates storage bottlenecks. Each row carries sample_interval_seconds, the measured seconds its deltas accrued over; a 0 means no delta was knowable for that file at this collection (first sighting, counter reset, or a gap past the delta policy — typically a restart) and its latencies are null rather than 0. LATEST IS A TIME: this reads the newest file-I/O snapshot, not a window, and captured_at is the instant it was collected; the deltas cover the sample_interval_seconds ending there.")] public static async Task GetFileIoStats( LocalDataService dataService, ServerManager serverManager, @@ -52,6 +52,8 @@ tools hand theirs. 0 means no delta on this row was knowable (first sighting, co return JsonSerializer.Serialize(new { server = resolved.ServerName, + /* #3541 A10: every file row shares this stamp (the read is every file at MAX(collection_time)). */ + captured_at = rows[0].CollectionTime.ToString("o"), files = result }, McpHelpers.JsonOptions); } diff --git a/Lite/Mcp/McpLatchSpinlockTools.cs b/Lite/Mcp/McpLatchSpinlockTools.cs index cd2464f35..81836000e 100644 --- a/Lite/Mcp/McpLatchSpinlockTools.cs +++ b/Lite/Mcp/McpLatchSpinlockTools.cs @@ -15,7 +15,7 @@ namespace PerformanceMonitorLite.Mcp; [McpServerToolType] public sealed class McpLatchSpinlockTools { - [McpServerTool(Name = "get_latch_stats"), Description("Gets the latest latch-contention snapshot by latch class: cumulative waiting requests and wait time (with the max single wait) plus the last collection interval's delta waits. High LATCH_EX on ACCESS_METHODS_DATASET_PARENT or a page-latch class indicates allocation/page contention (often TempDB).")] + [McpServerTool(Name = "get_latch_stats"), Description("Gets the latest latch-contention snapshot by latch class: cumulative waiting requests and wait time (with the max single wait) plus the last collection interval's delta waits. High LATCH_EX on ACCESS_METHODS_DATASET_PARENT or a page-latch class indicates allocation/page contention (often TempDB). LATEST IS A TIME: this is the newest snapshot found within hours_back of as_of, not an aggregate over those hours - captured_at is the instant it was collected and age_seconds its distance from the window's end.")] public static async Task GetLatchStats( LocalDataService dataService, ServerManager serverManager, @@ -40,6 +40,10 @@ public static async Task GetLatchStats( { server = resolved.ServerName, hours_back, + /* #3541 A10: hours_back here is the span SEARCHED for the newest snapshot (its description says + so); the snapshot's own clock and its distance from the anchor are what make that honest. */ + captured_at = rows[0].CollectionTime.ToString("o"), + age_seconds = McpLatestSnapshotStamp.AgeSeconds(rows[0].CollectionTime, windowEnd), latch_count = rows.Count, latches = rows.Select(r => new { @@ -61,7 +65,7 @@ public static async Task GetLatchStats( } } - [McpServerTool(Name = "get_spinlock_stats"), Description("Gets the latest spinlock-contention snapshot: cumulative collisions, spins, backoffs and spins-per-collision plus the last collection interval's delta collisions/spins. High spinlock contention is CPU-bound internal contention that does not appear in wait stats.")] + [McpServerTool(Name = "get_spinlock_stats"), Description("Gets the latest spinlock-contention snapshot: cumulative collisions, spins, backoffs and spins-per-collision plus the last collection interval's delta collisions/spins. High spinlock contention is CPU-bound internal contention that does not appear in wait stats. LATEST IS A TIME: this is the newest snapshot found within hours_back of as_of, not an aggregate over those hours - captured_at is the instant it was collected and age_seconds its distance from the window's end.")] public static async Task GetSpinlockStats( LocalDataService dataService, ServerManager serverManager, @@ -86,6 +90,8 @@ public static async Task GetSpinlockStats( { server = resolved.ServerName, hours_back, + captured_at = rows[0].CollectionTime.ToString("o"), + age_seconds = McpLatestSnapshotStamp.AgeSeconds(rows[0].CollectionTime, windowEnd), spinlock_count = rows.Count, spinlocks = rows.Select(r => new { diff --git a/Lite/Mcp/McpLatestSnapshotStamp.cs b/Lite/Mcp/McpLatestSnapshotStamp.cs new file mode 100644 index 000000000..666a463e3 --- /dev/null +++ b/Lite/Mcp/McpLatestSnapshotStamp.cs @@ -0,0 +1,28 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor Lite. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +namespace PerformanceMonitorLite.Mcp; + +/// +/// The one arithmetic every stamped latest-snapshot read shares (#3541 A10) — Lite's twin of Darling's +/// LatestSnapshotStamp; the two must stay in step so age_seconds means the same thing on both SKUs. +/// +internal static class McpLatestSnapshotStamp +{ + /// + /// Whole seconds from a snapshot's stamp to the window's end — the anchor the caller asked for, never the + /// service clock (AsOfWindowAnchorTests: an anchored tool's only "now" is its as_of, and a tool + /// body that names DateTime.UtcNow fails the census). The store stamps naive UTC and the anchor is + /// Kind=Utc; the subtraction ignores Kind, which is correct here because both are UTC instants. + /// Non-negative by construction for a windowed read, whose rows are bounded by + /// collection_time <= window end; clamped at zero anyway so a sub-second precision difference + /// between a microsecond store stamp and a 100 ns anchor can never publish "-0". + /// + public static long AgeSeconds(DateTime capturedAt, DateTime windowEnd) => + Math.Max(0L, (long)Math.Round((windowEnd - capturedAt).TotalSeconds)); +} diff --git a/Lite/Mcp/McpMemoryTools.cs b/Lite/Mcp/McpMemoryTools.cs index 561298853..27f780f43 100644 --- a/Lite/Mcp/McpMemoryTools.cs +++ b/Lite/Mcp/McpMemoryTools.cs @@ -9,7 +9,7 @@ namespace PerformanceMonitorLite.Mcp; [McpServerToolType] public sealed class McpMemoryTools { - [McpServerTool(Name = "get_memory_stats"), Description("Gets the latest memory statistics snapshot: physical memory, buffer pool size, plan cache size, memory utilization %, and SQL Server memory model. Use this for a quick memory health check; use get_memory_clerks to see detailed breakdown by component.")] + [McpServerTool(Name = "get_memory_stats"), Description("Gets the latest memory statistics snapshot: physical memory, buffer pool size, plan cache size, memory utilization %, and SQL Server memory model. Use this for a quick memory health check; use get_memory_clerks to see detailed breakdown by component. LATEST IS A TIME: this reads one snapshot, not a window, and captured_at is the instant that snapshot was collected - read it before treating any figure as current, because the newest row a store holds can be minutes or days old.")] public static async Task GetMemoryStats( LocalDataService dataService, ServerManager serverManager, @@ -30,7 +30,8 @@ public static async Task GetMemoryStats( return JsonSerializer.Serialize(new { server = resolved.ServerName, - collection_time = stats.CollectionTime.ToString("o"), + /* #3541 A10: the one stamp every latest-snapshot read publishes, under the one name. */ + captured_at = stats.CollectionTime.ToString("o"), total_physical_memory_mb = stats.TotalPhysicalMemoryMb, available_physical_memory_mb = stats.AvailablePhysicalMemoryMb, memory_utilization_pct = Math.Round(stats.MemoryUtilizationPercent, 1), @@ -181,7 +182,7 @@ a snapshot exists with nothing granted. */ return aligned; } - [McpServerTool(Name = "get_memory_clerks"), Description("Gets the top memory consumers by memory clerk type — shows which SQL Server components are using the most memory.")] + [McpServerTool(Name = "get_memory_clerks"), Description("Gets the top memory consumers by memory clerk type — shows which SQL Server components are using the most memory. LATEST IS A TIME: this reads the newest clerk snapshot, not a window, and captured_at is the instant it was collected.")] public static async Task GetMemoryClerks( LocalDataService dataService, ServerManager serverManager, @@ -217,6 +218,8 @@ with the read by construction. What the caller needs told is that an empty clerk return JsonSerializer.Serialize(new { server = resolved.ServerName, + /* Every row shares this stamp by construction (the read is every clerk at MAX(collection_time)). */ + captured_at = rows[0].CollectionTime.ToString("o"), clerks = result }, McpHelpers.JsonOptions); } @@ -278,12 +281,12 @@ public static async Task GetMemoryPressureEvents( } } - [McpServerTool(Name = "get_resource_semaphore"), Description("Gets resource semaphore statistics from the latest snapshot: granted vs available workspace memory against the target/max-target ceiling, per resource semaphore, with waiter counts and cumulative + per-interval timeout/forced-grant pressure indicators. High waiter counts or rising timeout/forced deltas indicate memory grant pressure affecting query performance. sample_interval_seconds is the measured seconds the two deltas accrued over; it is null with interval_known false when the row is a restart marker (no delta was knowable, so the zero deltas beside it are not 'no timeouts') or predates the column.")] + [McpServerTool(Name = "get_resource_semaphore"), Description("Gets resource semaphore statistics showing granted vs available workspace memory against the target/max-target ceiling, waiter counts, and timeout/forced grant pressure indicators. High waiter counts or rising timeout/forced deltas indicate memory grant pressure affecting query performance. TWO READS UNDER ONE WINDOW: grants[] is the NEWEST snapshot in the window (one row per resource semaphore and pool), stamped once as captured_at with age_seconds against the window's end - it is a moment, not the window. window[] aggregates EVERY snapshot in the window per (resource_semaphore_id, pool_id): peak_waiter_count and peak_waiters_at (the most sessions ever seen waiting for a grant and when), peak_granted_memory_mb, min_available_memory_mb, and timeout_errors_in_window / forced_grants_in_window (the SUM of the per-interval deltas across the window). A calm grants[] beside a window[] with waiters or timeouts is a grant storm that has passed; read window[] first for 'was there pressure', grants[] for 'is there pressure now'. Each grants[] row also carries sample_interval_seconds, the measured seconds its two deltas accrued over; it is null with interval_known false when the row is a restart marker (no delta was knowable, so the zero deltas beside it are not 'no timeouts') or predates the column.")] public static async Task GetResourceSemaphore( LocalDataService dataService, ServerManager serverManager, [Description("Server name or display name.")] string? server_name = null, - [Description("Hours of history to search for the latest snapshot. Default 24.")] int hours_back = 24, + [Description("Hours of history. Default 24. window[] aggregates every snapshot in these hours; grants[] is the newest snapshot in them.")] int hours_back = 24, [Description(McpHelpers.AsOfDescription)] string? as_of = null) { var (resolved, error) = ServerResolver.ResolveOrError(serverManager, server_name); @@ -301,6 +304,10 @@ public static async Task GetResourceSemaphore( ?? McpHelpers.Status("unavailable", "No memory grant data available."); } + /* #3541 A10: the window half, over the SAME window the latest read searched — its last_snapshot_at + IS captured_at, so the two halves describe one span of the same rows. Same shape as Darling's. */ + var window = await dataService.GetResourceSemaphoreWindowAsync(resolved.ServerId, hours_back, asOfUtc: windowEnd); + var result = rows.Select(r => new { collection_time = r.CollectionTime.ToString("o"), @@ -330,7 +337,13 @@ a pre-v61 row that never recorded one is null too. interval_known states the one return JsonSerializer.Serialize(new { server = resolved.ServerName, - grants = result + hours_back, + window_start = windowEnd.AddHours(-hours_back).ToString("o"), + window_end = windowEnd.ToString("o"), + captured_at = rows[0].CollectionTime.ToString("o"), + age_seconds = McpLatestSnapshotStamp.AgeSeconds(rows[0].CollectionTime, windowEnd), + grants = result, + window = window.Select(WindowShape) }, McpHelpers.JsonOptions); } catch (Exception ex) @@ -339,12 +352,12 @@ a pre-v61 row that never recorded one is null too. interval_known states the one } } - [McpServerTool(Name = "get_memory_grants"), Description("Gets resource semaphore statistics showing granted vs available workspace memory per resource pool, waiter counts, and timeout/forced grant deltas. High waiter counts or rising timeout deltas indicate memory grant pressure affecting query performance.")] + [McpServerTool(Name = "get_memory_grants"), Description("Gets resource semaphore statistics showing granted vs available workspace memory per resource pool, waiter counts, and timeout/forced grant deltas. High waiter counts or rising timeout deltas indicate memory grant pressure affecting query performance. TWO READS UNDER ONE WINDOW: grants[] is the NEWEST snapshot in the window (one row per pool, summed across its semaphores), stamped once as captured_at with age_seconds against the window's end - it is a moment, not the window. window[] aggregates EVERY snapshot in the window per pool: peak_waiter_count and peak_waiters_at (the most sessions ever seen waiting on the pool at one instant and when), peak_granted_memory_mb, min_available_memory_mb, and timeout_errors_in_window / forced_grants_in_window (the SUM of the per-interval deltas across the window). A calm grants[] beside a window[] with waiters or timeouts is a grant storm that has passed; read window[] first for 'was there pressure', grants[] for 'is there pressure now'.")] public static async Task GetMemoryGrants( LocalDataService dataService, ServerManager serverManager, [Description("Server name or display name.")] string? server_name = null, - [Description("Hours of history. Default 1.")] int hours_back = 1, + [Description("Hours of history. Default 1. window[] aggregates every snapshot in these hours; grants[] is the newest snapshot in them.")] int hours_back = 1, [Description(McpHelpers.AsOfDescription)] string? as_of = null) { var (resolved, error) = ServerResolver.ResolveOrError(serverManager, server_name); @@ -362,9 +375,10 @@ public static async Task GetMemoryGrants( ?? McpHelpers.Status("unavailable", "No memory grant data available."); } - /* Return latest snapshot */ + /* grants[] is the latest snapshot in the window; window[] is the whole window (#3541 A10). */ var latestTime = rows.Max(r => r.CollectionTime); var latest = rows.Where(r => r.CollectionTime == latestTime); + var window = await dataService.GetMemoryGrantsWindowAsync(resolved.ServerId, hours_back, asOfUtc: windowEnd); var result = latest.Select(r => new { @@ -382,7 +396,13 @@ public static async Task GetMemoryGrants( return JsonSerializer.Serialize(new { server = resolved.ServerName, - grants = result + hours_back, + window_start = windowEnd.AddHours(-hours_back).ToString("o"), + window_end = windowEnd.ToString("o"), + captured_at = latestTime.ToString("o"), + age_seconds = McpLatestSnapshotStamp.AgeSeconds(latestTime, windowEnd), + grants = result, + window = window.Select(WindowShape) }, McpHelpers.JsonOptions); } catch (Exception ex) @@ -390,4 +410,25 @@ public static async Task GetMemoryGrants( return McpHelpers.FormatError("get_memory_grants", ex); } } + + /// + /// The window half's payload shape, shared by both lenses so the same key set describes a semaphore's + /// window and a pool's window (the pool lens carries a null resource_semaphore_id, which + /// writes rather than drops — the key is present on both so a caller + /// can read one shape). Twin of Darling's DarlingMcpMemoryGrantTools.WindowShape. + /// + private static object WindowShape(MemoryGrantWindowRow w) => new + { + resource_semaphore_id = w.ResourceSemaphoreId, + pool_id = w.PoolId, + snapshots_in_window = w.SnapshotsInWindow, + first_snapshot_at = w.FirstSnapshotAt.ToString("o"), + last_snapshot_at = w.LastSnapshotAt.ToString("o"), + peak_waiter_count = w.PeakWaiterCount, + peak_waiters_at = w.PeakWaitersAt.ToString("o"), + peak_granted_memory_mb = Math.Round(w.PeakGrantedMemoryMb, 2), + min_available_memory_mb = Math.Round(w.MinAvailableMemoryMb, 2), + timeout_errors_in_window = w.TimeoutErrorsInWindow, + forced_grants_in_window = w.ForcedGrantsInWindow + }; } diff --git a/Lite/Mcp/McpPerfmonTools.cs b/Lite/Mcp/McpPerfmonTools.cs index 37c5822a9..b5ad5e7ac 100644 --- a/Lite/Mcp/McpPerfmonTools.cs +++ b/Lite/Mcp/McpPerfmonTools.cs @@ -9,7 +9,7 @@ namespace PerformanceMonitorLite.Mcp; [McpServerToolType] public sealed class McpPerfmonTools { - [McpServerTool(Name = "get_perfmon_stats"), Description("Gets the latest SQL Server performance counter values: batch requests/sec, compilations/sec, deadlocks/sec, and more. Provides throughput context to distinguish a busy server from a sick one. Use counter_name or instance_name to filter results.")] + [McpServerTool(Name = "get_perfmon_stats"), Description("Gets the latest SQL Server performance counter values: batch requests/sec, compilations/sec, deadlocks/sec, and more. Provides throughput context to distinguish a busy server from a sick one. Use counter_name or instance_name to filter results. LATEST IS A TIME: this reads the newest counter snapshot, not a window, and captured_at is the instant it was collected; use get_perfmon_trend for a counter over time.")] public static async Task GetPerfmonStats( LocalDataService dataService, ServerManager serverManager, @@ -46,6 +46,8 @@ public static async Task GetPerfmonStats( return JsonSerializer.Serialize(new { server = resolved.ServerName, + /* #3541 A10: taken from the unfiltered snapshot, so a filter that matches nothing still says when. */ + captured_at = rows[0].CollectionTime.ToString("o"), counters = result }, McpHelpers.JsonOptions); } diff --git a/Lite/Mcp/McpPlanCacheSchedulerTools.cs b/Lite/Mcp/McpPlanCacheSchedulerTools.cs index 3e599f6f0..2e77b638c 100644 --- a/Lite/Mcp/McpPlanCacheSchedulerTools.cs +++ b/Lite/Mcp/McpPlanCacheSchedulerTools.cs @@ -17,7 +17,7 @@ namespace PerformanceMonitorLite.Mcp; [McpServerToolType] public sealed class McpPlanCacheSchedulerTools { - [McpServerTool(Name = "get_plan_cache_bloat"), Description("Gets plan cache composition showing single-use vs multi-use plans per cache/object type, with a bloat-level classification. High single-use plan counts indicate ad-hoc query bloat consuming buffer pool memory. Consider enabling 'optimize for ad hoc workloads' or Forced Parameterization.")] + [McpServerTool(Name = "get_plan_cache_bloat"), Description("Gets plan cache composition showing single-use vs multi-use plans per cache/object type, with a bloat-level classification. High single-use plan counts indicate ad-hoc query bloat consuming buffer pool memory. Consider enabling 'optimize for ad hoc workloads' or Forced Parameterization. LATEST IS A TIME: this is the newest plan-cache snapshot found within hours_back of as_of, not an aggregate over those hours - captured_at is the instant the snapshot was collected and age_seconds its distance from the window's end.")] public static async Task GetPlanCacheBloat( LocalDataService dataService, ServerManager serverManager, @@ -35,7 +35,9 @@ public static async Task GetPlanCacheBloat( var summary = await dataService.GetPlanCacheSummaryAsync(resolved.ServerId, hours_back, asOfUtc: windowEnd); var cacheTypes = await dataService.GetPlanCacheSnapshotAsync(resolved.ServerId, hours_back, asOfUtc: windowEnd); - if (summary.TotalPlans == 0 && cacheTypes.Count == 0) + /* The summary's stamp is null exactly when the window held no snapshot, which is the same state + the (0 plans, no groups) test below names — one branch, so a stamped payload always has rows. */ + if (summary.CollectionTime is not DateTime capturedAt || (summary.TotalPlans == 0 && cacheTypes.Count == 0)) return await McpEngineCapability.NotCollectedStatusAsync(dataService, resolved.ServerId, resolved.ServerName, "plan_cache_stats") ?? McpHelpers.Status("unavailable", "No plan cache statistics available in the requested time range."); @@ -47,6 +49,8 @@ public static async Task GetPlanCacheBloat( return JsonSerializer.Serialize(new { server = resolved.ServerName, + captured_at = capturedAt.ToString("o"), + age_seconds = McpLatestSnapshotStamp.AgeSeconds(capturedAt, windowEnd), summary = new { total_plans = summary.TotalPlans, @@ -77,7 +81,16 @@ public static async Task GetPlanCacheBloat( } } - [McpServerTool(Name = "get_cpu_scheduler_pressure"), Description("Gets CPU scheduler pressure from the latest snapshot: runnable task queue depth, worker thread utilization, queued/blocked requests, and the collector's pressure warning flags. Shows whether the server has enough worker threads and if tasks are queuing for CPU time.")] + /// + /// get_cpu_scheduler_pressure's description, VERBATIM the text Darling's twin carries (#3541 A10): the same + /// tool name described two ways on two servers was half of the drift that lane closed, and a shared const + /// cannot be shared across the two assemblies, so the cross-SKU description census pins the two strings + /// equal instead. Change one, change both. + /// + internal const string CpuSchedulerPressureDescription = + "Gets CPU scheduler pressure from the latest snapshot: runnable task queue depth, worker thread utilization, queued/blocked requests, the collector's pressure warning flags, and the banded pressure_level verdict with its recommendation. Shows whether the server has enough worker threads and if tasks are queuing for CPU time. LATEST IS A TIME: this is the newest scheduler snapshot found within hours_back of as_of, not an aggregate over those hours - captured_at is the instant it was collected and age_seconds its distance from the window's end; the verdict is that instant's, so read age_seconds before reading pressure_level as current."; + + [McpServerTool(Name = "get_cpu_scheduler_pressure"), Description(CpuSchedulerPressureDescription)] public static async Task GetCpuSchedulerPressure( LocalDataService dataService, ServerManager serverManager, @@ -96,16 +109,22 @@ public static async Task GetCpuSchedulerPressure( var item = await dataService.GetCpuSchedulerSnapshotAsync(resolved.ServerId, hours_back, asOfUtc: windowEnd); if (item == null) return await McpEngineCapability.NotCollectedStatusAsync(dataService, resolved.ServerId, resolved.ServerName, "cpu_scheduler_stats") - ?? McpHelpers.Status("unavailable", "No CPU scheduler data available. The scheduler collector may not have run yet."); + ?? McpHelpers.Status("unavailable", "No CPU scheduler snapshot in the requested time range. The scheduler collector may not have run yet, or its newest snapshot is older than hours_back."); var workerUtilizationPercent = item.MaxWorkersCount > 0 ? Math.Round(item.TotalCurrentWorkersCount * 100.0 / item.MaxWorkersCount, 2) : 0; + /* #3541 A10: the verdict Darling's twin has always published, from the SHARED banding + (install/47's report.cpu_scheduler_pressure CASE, the one the CPU Scheduler tab renders) — the + same tool name answered with a verdict on one SKU and without one on the other. */ + var pressure = CpuSchedulerMetrics.ClassifyCpuPressure(item); + return JsonSerializer.Serialize(new { server = resolved.ServerName, - collection_time = item.CollectionTime.ToString("o"), + captured_at = item.CollectionTime.ToString("o"), + age_seconds = McpLatestSnapshotStamp.AgeSeconds(item.CollectionTime, windowEnd), schedulers = item.SchedulerCount, cpu_count = item.CpuCount, runnable_tasks = item.TotalRunnableTasksCount, @@ -118,6 +137,8 @@ public static async Task GetCpuSchedulerPressure( queued_requests = item.TotalQueuedRequestCount, blocked_tasks = item.TotalBlockedTaskCount, system_memory_state = item.SystemMemoryStateDesc, + pressure_level = pressure.Level, + recommendation = pressure.Recommendation, warnings = new { worker_thread_exhaustion = item.WorkerThreadExhaustionWarning, diff --git a/Lite/Services/LocalDataService.Config.cs b/Lite/Services/LocalDataService.Config.cs index f5ff976a2..271d092a4 100644 --- a/Lite/Services/LocalDataService.Config.cs +++ b/Lite/Services/LocalDataService.Config.cs @@ -15,6 +15,11 @@ namespace PerformanceMonitorLite.Services; public partial class LocalDataService { + /* #3541 A10: every latest-snapshot read here carries capture_time on the row (the SAME statement as the + values, never a second MAX() read that could stamp the next capture), so the MCP tools can publish + captured_at. Config is captured ON CONNECT, not on a schedule — the "current" value this family + serves is as old as the last successful connect, and the stamp is what says so. */ + /// /// Gets the latest server configuration snapshot (sys.configurations). /// @@ -23,7 +28,7 @@ public async Task> GetLatestServerConfigAsync(int serverId using var connection = await OpenConnectionAsync(); using var command = connection.CreateCommand(); command.CommandText = @" -SELECT configuration_name, value_configured, value_in_use, is_dynamic, is_advanced +SELECT configuration_name, value_configured, value_in_use, is_dynamic, is_advanced, capture_time FROM v_server_config WHERE server_id = $1 AND capture_time = (SELECT MAX(capture_time) FROM v_server_config WHERE server_id = $1) @@ -41,7 +46,8 @@ FROM v_server_config ValueConfigured = reader.IsDBNull(1) ? 0 : ToInt64(reader.GetValue(1)), ValueInUse = reader.IsDBNull(2) ? 0 : ToInt64(reader.GetValue(2)), IsDynamic = !reader.IsDBNull(3) && reader.GetBoolean(3), - IsAdvanced = !reader.IsDBNull(4) && reader.GetBoolean(4) + IsAdvanced = !reader.IsDBNull(4) && reader.GetBoolean(4), + CaptureTime = reader.GetDateTime(5) }); } @@ -64,7 +70,8 @@ public async Task> GetLatestDatabaseConfigAsync(int serv is_query_store_on, is_encrypted, is_trustworthy_on, is_db_chaining_on, is_broker_enabled, is_cdc_enabled, is_mixed_page_allocation_on, log_reuse_wait_desc, page_verify_option, target_recovery_time_seconds, delayed_durability, - is_accelerated_database_recovery_on, is_memory_optimized_enabled, is_optimized_locking_on + is_accelerated_database_recovery_on, is_memory_optimized_enabled, is_optimized_locking_on, + capture_time FROM v_database_config WHERE server_id = $1 AND capture_time = (SELECT MAX(capture_time) FROM v_database_config WHERE server_id = $1)" + dbClause + @" @@ -109,6 +116,10 @@ FROM v_database_config IsAcceleratedDatabaseRecoveryOn = !reader.IsDBNull(++ordinal) && reader.GetBoolean(ordinal), IsMemoryOptimizedEnabled = !reader.IsDBNull(++ordinal) && reader.GetBoolean(ordinal), IsOptimizedLockingOn = !reader.IsDBNull(++ordinal) && reader.GetBoolean(ordinal), + /* Appended as the 29th column and read one past the 28-column block the incrementing + mapping above consumes, so that mapping — shared byte-for-byte with the Darling + viewer and reader — is untouched. */ + CaptureTime = reader.GetDateTime(++ordinal), }); } @@ -125,7 +136,7 @@ public async Task> GetLatestQueryStoreHealthAsync(int using var command = connection.CreateCommand(); var dbClause = BuildDbInClause(databaseNames, "database_name", 2, out var dbValues); command.CommandText = @" -SELECT database_name, actual_state, desired_state, readonly_reason, current_storage_size_mb, max_storage_size_mb, size_based_cleanup_mode, stale_query_threshold_days, max_plans_per_query, interval_length_minutes +SELECT database_name, actual_state, desired_state, readonly_reason, current_storage_size_mb, max_storage_size_mb, size_based_cleanup_mode, stale_query_threshold_days, max_plans_per_query, interval_length_minutes, capture_time FROM v_query_store_health WHERE server_id = $1 AND capture_time = (SELECT MAX(capture_time) FROM v_query_store_health WHERE server_id = $1)" + dbClause + @" @@ -151,6 +162,7 @@ FROM v_query_store_health StaleQueryThresholdDays = reader.IsDBNull(7) ? 0L : Convert.ToInt64(reader.GetValue(7)), MaxPlansPerQuery = reader.IsDBNull(8) ? 0L : Convert.ToInt64(reader.GetValue(8)), IntervalLengthMinutes = reader.IsDBNull(9) ? 0L : Convert.ToInt64(reader.GetValue(9)), + CaptureTime = reader.GetDateTime(10), }); } @@ -166,7 +178,7 @@ public async Task> GetLatestDatabaseScopedConfigAs using var command = connection.CreateCommand(); var dbClause = BuildDbInClause(databaseNames, "database_name", 2, out var dbValues); command.CommandText = @" -SELECT database_name, configuration_name, value, value_for_secondary +SELECT database_name, configuration_name, value, value_for_secondary, capture_time FROM v_database_scoped_config WHERE server_id = $1 AND capture_time = (SELECT MAX(capture_time) FROM v_database_scoped_config WHERE server_id = $1)" + dbClause + @" @@ -185,7 +197,8 @@ FROM v_database_scoped_config DatabaseName = reader.GetString(0), ConfigurationName = reader.GetString(1), Value = reader.IsDBNull(2) ? "" : reader.GetString(2), - ValueForSecondary = reader.IsDBNull(3) ? "" : reader.GetString(3) + ValueForSecondary = reader.IsDBNull(3) ? "" : reader.GetString(3), + CaptureTime = reader.GetDateTime(4) }); } @@ -200,7 +213,7 @@ public async Task> GetLatestTraceFlagsAsync(int serverId) using var connection = await OpenConnectionAsync(); using var command = connection.CreateCommand(); command.CommandText = @" -SELECT trace_flag, status, is_global, is_session +SELECT trace_flag, status, is_global, is_session, capture_time FROM v_trace_flags WHERE server_id = $1 AND capture_time = (SELECT MAX(capture_time) FROM v_trace_flags WHERE server_id = $1) @@ -217,7 +230,8 @@ FROM v_trace_flags TraceFlag = reader.GetInt32(0), Status = !reader.IsDBNull(1) && reader.GetBoolean(1), IsGlobal = !reader.IsDBNull(2) && reader.GetBoolean(2), - IsSession = !reader.IsDBNull(3) && reader.GetBoolean(3) + IsSession = !reader.IsDBNull(3) && reader.GetBoolean(3), + CaptureTime = reader.GetDateTime(4) }); } @@ -227,6 +241,8 @@ FROM v_trace_flags public class ServerConfigRow { + /// The connect-time capture this row belongs to (#3541 A10); every row of one snapshot shares it. + public DateTime CaptureTime { get; set; } public string ConfigurationName { get; set; } = ""; public long ValueConfigured { get; set; } public long ValueInUse { get; set; } @@ -239,6 +255,8 @@ public class ServerConfigRow public class DatabaseConfigRow { + /// The connect-time capture this row belongs to (#3541 A10). + public DateTime CaptureTime { get; set; } public string DatabaseName { get; set; } = ""; public string StateDesc { get; set; } = ""; public int CompatibilityLevel { get; set; } @@ -300,6 +318,8 @@ public class DatabaseConfigRow /// public class QueryStoreHealthRow { + /// The hourly capture this row belongs to (#3541 A10). + public DateTime CaptureTime { get; set; } public string DatabaseName { get; set; } = ""; public string ActualState { get; set; } = ""; public string DesiredState { get; set; } = ""; @@ -326,6 +346,8 @@ public class QueryStoreHealthRow public class DatabaseScopedConfigRow { + /// The connect-time capture this row belongs to (#3541 A10). + public DateTime CaptureTime { get; set; } public string DatabaseName { get; set; } = ""; public string ConfigurationName { get; set; } = ""; public string Value { get; set; } = ""; @@ -334,6 +356,8 @@ public class DatabaseScopedConfigRow public class TraceFlagRow { + /// The connect-time capture this row belongs to (#3541 A10). + public DateTime CaptureTime { get; set; } public int TraceFlag { get; set; } public bool Status { get; set; } public bool IsGlobal { get; set; } diff --git a/Lite/Services/LocalDataService.FileIo.cs b/Lite/Services/LocalDataService.FileIo.cs index 7ca68045c..aa6defcae 100644 --- a/Lite/Services/LocalDataService.FileIo.cs +++ b/Lite/Services/LocalDataService.FileIo.cs @@ -35,7 +35,8 @@ public async Task> GetLatestFileIoStatsAsync(int serverId) delta_write_bytes, delta_stall_read_ms, delta_stall_write_ms, - sample_interval_seconds + sample_interval_seconds, + collection_time FROM v_file_io_stats WHERE server_id = $1 AND collection_time = (SELECT MAX(collection_time) FROM v_file_io_stats WHERE server_id = $1) @@ -60,7 +61,8 @@ FROM v_file_io_stats DeltaWriteBytes = reader.IsDBNull(8) ? 0 : reader.GetInt64(8), DeltaStallReadMs = reader.IsDBNull(9) ? 0 : reader.GetInt64(9), DeltaStallWriteMs = reader.IsDBNull(10) ? 0 : reader.GetInt64(10), - SampleIntervalSeconds = reader.IsDBNull(11) ? null : reader.GetInt32(11) + SampleIntervalSeconds = reader.IsDBNull(11) ? null : reader.GetInt32(11), + CollectionTime = reader.GetDateTime(12) }); } @@ -313,6 +315,9 @@ public class FileIoThroughputPoint public class FileIoRow { + /// The snapshot this row belongs to (#3541 A10); the deltas cover the + /// ending here. + public DateTime CollectionTime { get; set; } public string DatabaseName { get; set; } = ""; public string FileName { get; set; } = ""; public string FileType { get; set; } = ""; diff --git a/Lite/Services/LocalDataService.LatchSpinlock.cs b/Lite/Services/LocalDataService.LatchSpinlock.cs index b9dbcd916..c34ce0375 100644 --- a/Lite/Services/LocalDataService.LatchSpinlock.cs +++ b/Lite/Services/LocalDataService.LatchSpinlock.cs @@ -124,7 +124,8 @@ FROM v_latch_stats wait_time_ms, max_wait_time_ms, delta_waiting_requests_count, - delta_wait_time_ms + delta_wait_time_ms, + collection_time FROM v_latch_stats WHERE server_id = $1 AND collection_time = (SELECT mx FROM latest) @@ -146,7 +147,8 @@ FROM v_latch_stats WaitTimeMs = reader.IsDBNull(2) ? 0 : reader.GetInt64(2), MaxWaitTimeMs = reader.IsDBNull(3) ? 0 : reader.GetInt64(3), DeltaWaitingRequestsCount = reader.IsDBNull(4) ? 0 : reader.GetInt64(4), - DeltaWaitTimeMs = reader.IsDBNull(5) ? 0 : reader.GetInt64(5) + DeltaWaitTimeMs = reader.IsDBNull(5) ? 0 : reader.GetInt64(5), + CollectionTime = reader.GetDateTime(6) }); } @@ -257,7 +259,8 @@ FROM v_spinlock_stats sleep_time, backoffs, delta_collisions, - delta_spins + delta_spins, + collection_time FROM v_spinlock_stats WHERE server_id = $1 AND collection_time = (SELECT mx FROM latest) @@ -281,7 +284,8 @@ FROM v_spinlock_stats SleepTime = reader.IsDBNull(4) ? 0 : reader.GetInt64(4), Backoffs = reader.IsDBNull(5) ? 0 : reader.GetInt64(5), DeltaCollisions = reader.IsDBNull(6) ? 0 : reader.GetInt64(6), - DeltaSpins = reader.IsDBNull(7) ? 0 : reader.GetInt64(7) + DeltaSpins = reader.IsDBNull(7) ? 0 : reader.GetInt64(7), + CollectionTime = reader.GetDateTime(8) }); } @@ -302,6 +306,8 @@ public class LatchStatsTrendPoint /// deltas for one latch class at the most recent collection in the window. public class LatchStatsSnapshotRow { + /// The snapshot this row belongs to (#3541 A10); every row of one snapshot shares it. + public DateTime CollectionTime { get; set; } public string LatchClass { get; set; } = ""; public long WaitingRequestsCount { get; set; } public long WaitTimeMs { get; set; } @@ -323,6 +329,8 @@ public class SpinlockStatsTrendPoint /// interval's deltas for one spinlock at the most recent collection in the window. public class SpinlockStatsSnapshotRow { + /// The snapshot this row belongs to (#3541 A10); every row of one snapshot shares it. + public DateTime CollectionTime { get; set; } public string SpinlockName { get; set; } = ""; public long Collisions { get; set; } public long Spins { get; set; } diff --git a/Lite/Services/LocalDataService.Memory.cs b/Lite/Services/LocalDataService.Memory.cs index b20508a07..a0741c4ea 100644 --- a/Lite/Services/LocalDataService.Memory.cs +++ b/Lite/Services/LocalDataService.Memory.cs @@ -268,8 +268,10 @@ public async Task> GetLatestMemoryClerksAsync(int serverId) { using var connection = await OpenConnectionAsync(); using var command = connection.CreateCommand(); + /* #3541 A10: collection_time rides on the row (same statement as the values), so get_memory_clerks + can publish captured_at — every clerk of one snapshot shares it by construction. */ command.CommandText = @" -SELECT clerk_type, memory_mb +SELECT clerk_type, memory_mb, collection_time FROM v_memory_clerks WHERE server_id = $1 AND collection_time = (SELECT MAX(collection_time) FROM v_memory_clerks WHERE server_id = $1) @@ -284,7 +286,8 @@ FROM v_memory_clerks items.Add(new MemoryClerkRow { ClerkType = reader.GetString(0), - MemoryMb = reader.IsDBNull(1) ? 0 : ToDouble(reader.GetValue(1)) + MemoryMb = reader.IsDBNull(1) ? 0 : ToDouble(reader.GetValue(1)), + CollectionTime = reader.GetDateTime(2) }); } @@ -321,6 +324,8 @@ public class MemoryTrendPoint public class MemoryClerkRow { + /// The snapshot this clerk row belongs to (#3541 A10); every row of one snapshot shares it. + public DateTime CollectionTime { get; set; } public string ClerkType { get; set; } = ""; public double MemoryMb { get; set; } public string MemoryFormatted => MemoryMb >= 1024 ? $"{MemoryMb / 1024:F1} GB" : $"{MemoryMb:F1} MB"; diff --git a/Lite/Services/LocalDataService.MemoryGrants.cs b/Lite/Services/LocalDataService.MemoryGrants.cs index 9de994e15..35dc82caa 100644 --- a/Lite/Services/LocalDataService.MemoryGrants.cs +++ b/Lite/Services/LocalDataService.MemoryGrants.cs @@ -187,6 +187,217 @@ statement from the 0 the calculator writes when no delta was knowable. */ } return items; } + + /* ─────────────────────────── the window, per semaphore / per pool (#3541 A10) ─────────────────────────── */ + + /// + /// The SQL behind : every memory_grant_stats snapshot in the + /// window aggregated per (resource_semaphore_id, pool_id) — the storm detector get_resource_semaphore + /// serves BESIDE its latest snapshot. Until #3541 A10 the tool accepted hours_back and read only the + /// newest snapshot in it, so a grant storm three hours ago was invisible behind a calm latest row while + /// the parameter read as a window. The peak's instant comes from a DISTINCT ON over the same + /// windowed rows (highest waiter_count first, newest first on a tie, so a storm that plateaued reports its + /// latest snapshot); the two *_in_window figures SUM the stored per-interval deltas — the reason the + /// collector stores them, and no interval arithmetic (a restart's fabricated 0 adds 0). Public so + /// Lite.Tests can pin the dialect without a store. $1 server_id, $2 window start, $3 window end. + /// + public const string ResourceSemaphoreWindowSql = @" +WITH windowed AS +( + SELECT + collection_time, + resource_semaphore_id, + pool_id, + waiter_count, + granted_memory_mb, + available_memory_mb, + timeout_error_count_delta, + forced_grant_count_delta + FROM v_memory_grant_stats + WHERE server_id = $1 + AND collection_time >= $2 + AND collection_time <= $3 +), +agg AS +( + SELECT + resource_semaphore_id, + pool_id, + COUNT(*) AS snapshots_in_window, + MIN(collection_time) AS first_snapshot_at, + MAX(collection_time) AS last_snapshot_at, + MAX(waiter_count) AS peak_waiter_count, + MAX(granted_memory_mb) AS peak_granted_memory_mb, + MIN(available_memory_mb) AS min_available_memory_mb, + SUM(timeout_error_count_delta) AS timeout_errors_in_window, + SUM(forced_grant_count_delta) AS forced_grants_in_window + FROM windowed + GROUP BY resource_semaphore_id, pool_id +), +peak AS +( + SELECT DISTINCT ON (resource_semaphore_id, pool_id) + resource_semaphore_id, + pool_id, + collection_time AS peak_waiters_at + FROM windowed + ORDER BY resource_semaphore_id, pool_id, waiter_count DESC, collection_time DESC +) +SELECT + a.resource_semaphore_id, + a.pool_id, + a.snapshots_in_window, + a.first_snapshot_at, + a.last_snapshot_at, + a.peak_waiter_count, + p.peak_waiters_at, + a.peak_granted_memory_mb, + a.min_available_memory_mb, + a.timeout_errors_in_window, + a.forced_grants_in_window +FROM agg AS a +JOIN peak AS p + ON p.resource_semaphore_id = a.resource_semaphore_id + AND p.pool_id = a.pool_id +ORDER BY a.pool_id, a.resource_semaphore_id"; + + /// + /// The SQL behind : the pool lens of . + /// A pool's semaphores are SUMMED at each snapshot first (the same per-snapshot SUM + /// serves), THEN the window's peak / floor / total is taken over + /// those per-snapshot pool figures, so peak waiters is the most sessions waiting on the pool at one instant, + /// not the largest single semaphore's count. resource_semaphore_id is a typed NULL so both reads + /// project the same eleven columns. $1 server_id, $2 window start, $3 window end. + /// + public const string MemoryGrantsWindowSql = @" +WITH per_snapshot AS +( + SELECT + collection_time, + pool_id, + SUM(waiter_count) AS waiter_count, + SUM(granted_memory_mb) AS granted_memory_mb, + SUM(available_memory_mb) AS available_memory_mb, + SUM(timeout_error_count_delta) AS timeout_error_count_delta, + SUM(forced_grant_count_delta) AS forced_grant_count_delta + FROM v_memory_grant_stats + WHERE server_id = $1 + AND collection_time >= $2 + AND collection_time <= $3 + GROUP BY collection_time, pool_id +), +agg AS +( + SELECT + pool_id, + COUNT(*) AS snapshots_in_window, + MIN(collection_time) AS first_snapshot_at, + MAX(collection_time) AS last_snapshot_at, + MAX(waiter_count) AS peak_waiter_count, + MAX(granted_memory_mb) AS peak_granted_memory_mb, + MIN(available_memory_mb) AS min_available_memory_mb, + SUM(timeout_error_count_delta) AS timeout_errors_in_window, + SUM(forced_grant_count_delta) AS forced_grants_in_window + FROM per_snapshot + GROUP BY pool_id +), +peak AS +( + SELECT DISTINCT ON (pool_id) + pool_id, + collection_time AS peak_waiters_at + FROM per_snapshot + ORDER BY pool_id, waiter_count DESC, collection_time DESC +) +SELECT + CAST(NULL AS INTEGER) AS resource_semaphore_id, + a.pool_id, + a.snapshots_in_window, + a.first_snapshot_at, + a.last_snapshot_at, + a.peak_waiter_count, + p.peak_waiters_at, + a.peak_granted_memory_mb, + a.min_available_memory_mb, + a.timeout_errors_in_window, + a.forced_grants_in_window +FROM agg AS a +JOIN peak AS p ON p.pool_id = a.pool_id +ORDER BY a.pool_id"; + + /// Every snapshot in the window aggregated per (resource_semaphore_id, pool_id) — the window half + /// of get_resource_semaphore (#3541 A10). Same window arguments as + /// , so the window's last_snapshot_at IS the + /// snapshot that read serves. + public Task> GetResourceSemaphoreWindowAsync(int serverId, int hoursBack = 24, DateTime? fromDate = null, DateTime? toDate = null, DateTime? asOfUtc = null) => + ReadMemoryGrantWindowAsync(ResourceSemaphoreWindowSql, "GetResourceSemaphoreWindowAsync", serverId, hoursBack, fromDate, toDate, asOfUtc); + + /// Every snapshot in the window aggregated per pool — the window half of get_memory_grants + /// (#3541 A10). + public Task> GetMemoryGrantsWindowAsync(int serverId, int hoursBack = 24, DateTime? fromDate = null, DateTime? toDate = null, DateTime? asOfUtc = null) => + ReadMemoryGrantWindowAsync(MemoryGrantsWindowSql, "GetMemoryGrantsWindowAsync", serverId, hoursBack, fromDate, toDate, asOfUtc); + + /// The two window reads project the SAME eleven columns in the same order, so one materialiser + /// serves both. DuckDB widens SUM(INTEGER) to HUGEINT and COUNT(*) to BIGINT, so every count goes through + /// ToInt64 and every MB figure through ToDouble, like the readers above. + private async Task> ReadMemoryGrantWindowAsync( + string sql, string queryName, int serverId, int hoursBack, DateTime? fromDate, DateTime? toDate, DateTime? asOfUtc) + { + using var _q = TimeQuery(queryName, "v_memory_grant_stats window aggregate"); + using var connection = await OpenConnectionAsync(); + using var command = connection.CreateCommand(); + + var (startTime, endTime) = GetTimeRange(hoursBack, fromDate, toDate, asOfUtc, SelectedServerTabUtcOffsetMinutes); + + command.CommandText = sql; + command.Parameters.Add(new DuckDBParameter { Value = serverId }); + command.Parameters.Add(new DuckDBParameter { Value = startTime }); + command.Parameters.Add(new DuckDBParameter { Value = endTime }); + + var items = new List(); + using var reader = await command.ExecuteReaderAsync(); + while (await reader.ReadAsync()) + { + items.Add(new MemoryGrantWindowRow + { + ResourceSemaphoreId = reader.IsDBNull(0) ? null : (int)ToInt64(reader.GetValue(0)), + PoolId = reader.IsDBNull(1) ? 0 : (int)ToInt64(reader.GetValue(1)), + SnapshotsInWindow = ToInt64(reader.GetValue(2)), + FirstSnapshotAt = reader.GetDateTime(3), + LastSnapshotAt = reader.GetDateTime(4), + PeakWaiterCount = reader.IsDBNull(5) ? 0 : ToInt64(reader.GetValue(5)), + PeakWaitersAt = reader.GetDateTime(6), + PeakGrantedMemoryMb = reader.IsDBNull(7) ? 0 : ToDouble(reader.GetValue(7)), + MinAvailableMemoryMb = reader.IsDBNull(8) ? 0 : ToDouble(reader.GetValue(8)), + TimeoutErrorsInWindow = reader.IsDBNull(9) ? 0 : ToInt64(reader.GetValue(9)), + ForcedGrantsInWindow = reader.IsDBNull(10) ? 0 : ToInt64(reader.GetValue(10)) + }); + } + return items; + } +} + +/// +/// One (resource_semaphore_id, pool_id) — or, for the pool lens, one pool with +/// null — aggregated over EVERY memory_grant_stats snapshot in the window (#3541 A10). +/// / is the storm detector: the most sessions ever seen waiting for a grant in the +/// window and the snapshot it happened at. The two *InWindow figures are SUMs of the stored per-interval +/// deltas — how many grants timed out / were forced across the whole window, not just the last interval. +/// Twin of Darling's DarlingMemoryGrantReader.MemoryGrantWindowRow. +/// +public class MemoryGrantWindowRow +{ + public int? ResourceSemaphoreId { get; set; } + public int PoolId { get; set; } + public long SnapshotsInWindow { get; set; } + public DateTime FirstSnapshotAt { get; set; } + public DateTime LastSnapshotAt { get; set; } + public long PeakWaiterCount { get; set; } + public DateTime PeakWaitersAt { get; set; } + public double PeakGrantedMemoryMb { get; set; } + public double MinAvailableMemoryMb { get; set; } + public long TimeoutErrorsInWindow { get; set; } + public long ForcedGrantsInWindow { get; set; } } public class MemoryGrantChartPoint diff --git a/Lite/Services/LocalDataService.Overview.cs b/Lite/Services/LocalDataService.Overview.cs index 03871f0c3..7d33f6543 100644 --- a/Lite/Services/LocalDataService.Overview.cs +++ b/Lite/Services/LocalDataService.Overview.cs @@ -27,7 +27,9 @@ public partial class LocalDataService double? cpuPercent = null; double? otherProcessCpuPercent = null; DateTime? cpuSampleTime = null; + DateTime? cpuCollectionTime = null; double? memoryMb = null; + DateTime? memoryCollectionTime = null; int blockingCount = 0; int deadlockCount = 0; DateTime? lastCollection = null; @@ -37,7 +39,7 @@ total non-idle CPU alongside the SQL-only number. */ using (var cmd = connection.CreateCommand()) { cmd.CommandText = @" -SELECT sqlserver_cpu_utilization, other_process_cpu_utilization, sample_time +SELECT sqlserver_cpu_utilization, other_process_cpu_utilization, sample_time, collection_time FROM v_cpu_utilization_stats WHERE server_id = $1 ORDER BY sample_time DESC @@ -54,6 +56,11 @@ ORDER BY sample_time DESC this one is the CPU persistence gate's observation identity and must stay the instant of the CPU reading these percentages came from. */ cpuSampleTime = lastCollection; + /* #3541 A10: the store's UTC clock for the same row — the stamp get_server_summary publishes + as cpu_captured_at. sample_time above is the monitored server's local wall clock and the + gate's identity; collection_time is the instant the monitor stored it, comparable with every + other captured_at on the MCP surface. */ + cpuCollectionTime = reader.IsDBNull(3) ? null : reader.GetDateTime(3); } } @@ -61,7 +68,7 @@ the CPU reading these percentages came from. */ using (var cmd = connection.CreateCommand()) { cmd.CommandText = @" -SELECT total_server_memory_mb +SELECT total_server_memory_mb, collection_time FROM v_memory_stats WHERE server_id = $1 ORDER BY collection_time DESC @@ -71,6 +78,7 @@ ORDER BY collection_time DESC if (await reader.ReadAsync()) { memoryMb = reader.IsDBNull(0) ? null : ToDouble(reader.GetValue(0)); + memoryCollectionTime = reader.GetDateTime(1); } } @@ -124,7 +132,9 @@ FROM v_collection_log CpuPercent = cpuPercent, OtherProcessCpuPercent = otherProcessCpuPercent, CpuSampleTime = cpuSampleTime, + CpuCollectionTime = cpuCollectionTime, MemoryMb = memoryMb, + MemoryCollectionTime = memoryCollectionTime, BlockingCount = blockingCount, DeadlockCount = deadlockCount, LastCollectionTime = lastCollection @@ -284,6 +294,15 @@ public class ServerSummaryItem /// freshness band is computed from. /// public DateTime? CpuSampleTime { get; set; } + + /// The store's collection_time for the CPU row came from (#3541 + /// A10) — the stamp get_server_summary publishes as cpu_captured_at. UTC, comparable with every other + /// captured_at; is the monitored server's local clock and the gate's identity. + public DateTime? CpuCollectionTime { get; set; } + + /// The store's collection_time for the memory row came from (#3541 + /// A10) — get_server_summary's memory_captured_at. + public DateTime? MemoryCollectionTime { get; set; } /// Total non-idle CPU on the host = sql_server + other_process. Tracks closer to OS user+system counters. public double? TotalCpuPercent => CpuPercent.HasValue ? CpuPercent.Value + (OtherProcessCpuPercent ?? 0) : null; diff --git a/Lite/Services/LocalDataService.Perfmon.cs b/Lite/Services/LocalDataService.Perfmon.cs index ff5700fdd..058b90a7f 100644 --- a/Lite/Services/LocalDataService.Perfmon.cs +++ b/Lite/Services/LocalDataService.Perfmon.cs @@ -27,7 +27,8 @@ public async Task> GetLatestPerfmonStatsAsync(int serverId) counter_name, instance_name, cntr_value, - delta_cntr_value + delta_cntr_value, + collection_time FROM v_perfmon_stats WHERE server_id = $1 AND collection_time = (SELECT MAX(collection_time) FROM v_perfmon_stats WHERE server_id = $1) @@ -44,7 +45,8 @@ FROM v_perfmon_stats CounterName = reader.IsDBNull(0) ? "" : reader.GetString(0), InstanceName = reader.IsDBNull(1) ? "" : reader.GetString(1), Value = reader.IsDBNull(2) ? 0 : reader.GetInt64(2), - DeltaValue = reader.IsDBNull(3) ? 0 : reader.GetInt64(3) + DeltaValue = reader.IsDBNull(3) ? 0 : reader.GetInt64(3), + CollectionTime = reader.GetDateTime(4) }); } @@ -189,6 +191,8 @@ AND counter_name IN ({nameParams}) public class PerfmonRow { + /// The snapshot this counter row belongs to (#3541 A10); every row of one snapshot shares it. + public DateTime CollectionTime { get; set; } public string CounterName { get; set; } = ""; public string InstanceName { get; set; } = ""; public long Value { get; set; } diff --git a/Lite/Services/LocalDataService.PlanCache.cs b/Lite/Services/LocalDataService.PlanCache.cs index 977bab943..bf0f57a0c 100644 --- a/Lite/Services/LocalDataService.PlanCache.cs +++ b/Lite/Services/LocalDataService.PlanCache.cs @@ -158,7 +158,8 @@ FROM v_plan_cache_stats SELECT COALESCE(SUM(total_plans), 0) AS total_plans, COALESCE(SUM(single_use_plans), 0) AS single_use_plans, - MIN(oldest_plan_create_time) AS oldest_plan_create_time + MIN(oldest_plan_create_time) AS oldest_plan_create_time, + (SELECT mx FROM latest) AS collection_time FROM v_plan_cache_stats WHERE server_id = $1 AND collection_time = (SELECT mx FROM latest)"; @@ -177,7 +178,10 @@ FROM v_plan_cache_stats { TotalPlans = reader.IsDBNull(0) ? 0 : ToInt64(reader.GetValue(0)), SingleUsePlans = reader.IsDBNull(1) ? 0 : ToInt64(reader.GetValue(1)), - OldestPlanCreateTime = reader.IsDBNull(2) ? null : reader.GetDateTime(2) + OldestPlanCreateTime = reader.IsDBNull(2) ? null : reader.GetDateTime(2), + /* #3541 A10: the snapshot's own stamp, from the same `latest` CTE the totals are keyed on. NULL + (no snapshot in the window) is the (0, 0, null) empty case the summary already returns. */ + CollectionTime = reader.IsDBNull(3) ? null : reader.GetDateTime(3) }; } @@ -238,6 +242,8 @@ public class PlanCacheSnapshotRow /// and the oldest cached plan's create time. public class PlanCacheSummary { + /// The snapshot the totals are of (#3541 A10); null when the window held none. + public DateTime? CollectionTime { get; set; } public long TotalPlans { get; set; } public long SingleUsePlans { get; set; } public DateTime? OldestPlanCreateTime { get; set; } From 5cd6513805359e14ca454e36682a15c79440c1d5 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 16:17:46 -0400 Subject: [PATCH 61/69] PostgreSQL servers get a measured deadlock band from their own counters, their Long-Running Query alert skips maintenance like SQL Server's does, and every one-sided alert says why it is one-sided (#3539 parity) (#3638) The fleet card's deadlock dot on a PostgreSQL target read Unknown while pg_database_stats held the server's own deadlock counter, because the fleet reader took v_deadlocks (the SQL Server extended-event capture) and ServerMetricSources nulled the structural zero. Both fleet surfaces (the service's get_fleet_overview / /api/fleet reader and the viewer's Overview card and totals) now difference pg_stat_database.deadlocks per (server, database) series between consecutive samples, clamp at zero across a statistics reset, sum across databases, and band the result through the SAME deadlock_warn_per_hour / deadlock_critical_per_hour tiers as the SQL Server graph count. Never SUM(deadlocks): the column is a lifetime counter repeated in every sample. FleetDeadlockSource.PostgresTarget now means "counted from the server counter" and is a covered arm; servers_read counts it and postgres_servers is a sub-count. A PostgreSQL target whose pg_database_stats collector is silent or denied reads those arms, exactly as a SQL Server whose deadlocks collector is. The PostgreSQL Long-Running Query read gains the noise opt-outs SQL Server's has had: non-client backends (mirrors session_id > 50), the VACUUM/ANALYZE/REINDEX/CLUSTER statements by command tag (no text is stored, so the tag is the only handle), the dump/restore utilities by application_name on the shared longRunningQueryExcludeBackups switch, and excludedDatabases after the read. CREATE stays reported with its tag. The Darling README gains an engine-coverage table for every built-in alert and band, stating why Blocking Wait Time and Database File Growth are SQL-only, why the PostgreSQL blocking BAND stays Unknown (sampled evidence against tiers measured on engine-recorded reports), that Poison Wait is one shape since #3593, and that slot retention's SQL analogue (log_reuse_wait_desc) is collected only as a load-time snapshot today. --- .../Darling.Tests/DarlingFleetReaderTests.cs | 122 ++++++--- .../DarlingPgReadSqlParsesLiveTests.cs | 10 +- .../DarlingPgSessionStatesReaderTests.cs | 87 +++++++ .../Darling.Tests/DeadlockRateBandTests.cs | 17 +- .../FleetCardPostgresCpuLivePostgresTests.cs | 107 +++++++- .../FleetCardPostgresCpuTests.cs | 1 + .../PgCpuCapacityHeadroomTests.cs | 4 +- .../UnmeasuredMetricsAreNotHealthyTests.cs | 147 +++++++++-- .../Darling.Tests/ViewerFleetRollupTests.cs | 175 +++++++++---- Darling/Darling.Tests/ViewerW2aTests.cs | 20 ++ .../DarlingConfig.cs | 10 +- .../DarlingWorker.cs | 16 +- .../Mcp/DarlingFleetReader.cs | 242 +++++++++++++++--- .../Mcp/DarlingMcpAlertTools.cs | 2 +- .../Mcp/DarlingMcpFleetTools.cs | 12 +- .../wwwroot/js/pages/fleet.js | 12 +- .../DarlingPgSessionStatesReader.cs | 108 +++++++- .../MainWindow.xaml.cs | 7 +- .../ViewerDataService.Fleet.cs | 57 +++-- .../ViewerDataService.Overview.cs | 147 +++++++++-- Darling/README.md | 22 +- .../PostgresAlertEvaluator.cs | 21 +- .../FleetDeadlockCoverage.cs | 135 ++++++---- .../ServerHealthBands.cs | 61 +++-- 24 files changed, 1262 insertions(+), 280 deletions(-) diff --git a/Darling/Darling.Tests/DarlingFleetReaderTests.cs b/Darling/Darling.Tests/DarlingFleetReaderTests.cs index 91370e5ee..77cfe3f09 100644 --- a/Darling/Darling.Tests/DarlingFleetReaderTests.cs +++ b/Darling/Darling.Tests/DarlingFleetReaderTests.cs @@ -193,6 +193,7 @@ public void FleetTagForestSql_SelectsTheWholeHierarchy_OrderedForStableSiblings( [InlineData(nameof(DarlingFleetReader.FleetThreadsSql))] [InlineData(nameof(DarlingFleetReader.FleetBlockingSql))] [InlineData(nameof(DarlingFleetReader.FleetDeadlockSql))] + [InlineData(nameof(DarlingFleetReader.FleetPgDeadlockSql))] [InlineData(nameof(DarlingFleetReader.FleetLastCollectionSql))] [InlineData(nameof(DarlingFleetReader.FleetCollectionHealthSql))] public void EveryFleetSql_IsPgDialect_NoTSql(string constName) @@ -415,19 +416,36 @@ public void ADeadlockReaderThatReadNothing_DoesNotCount_AndNamesItsCause( => Assert.Equal(expected, FleetDeadlockCoverage.ClassifyDeadlockSource(isPostgres: false, band)); /// - /// The issue's own case: a PostgreSQL target is never covered, and its collector's band cannot change - /// that. pg_deadlocks can be perfectly HEALTHY on all fifty targets and this total still counts - /// none of it — the rows are in a different table. That is why PostgreSQL is asked before any band. + /// #3539 reversed #3017's PostgreSQL arm: a PostgreSQL target IS covered when its deadlock-source + /// collector (pg_database_stats) read, on exactly the terms a SQL Server's deadlocks + /// collector is — degraded still counts, silent and denied do not — and the covered arm is + /// PostgresTarget rather than Read only because the instrument differs (a counter + /// difference, not a graph). The pre-#3539 answer, PostgresTarget on the engine alone, would now + /// call a server whose collector never ran "counted". /// [Theory] - [InlineData(CollectorHealthClassifier.Healthy)] - [InlineData(CollectorHealthClassifier.NoPermissions)] - [InlineData(CollectorHealthClassifier.Stopped)] - [InlineData(null)] - public void APostgresTarget_IsNeverCovered_WhateverItsCollectorSays(string? band) - => Assert.Equal( - FleetDeadlockSource.PostgresTarget, - FleetDeadlockCoverage.ClassifyDeadlockSource(isPostgres: true, band)); + [InlineData(CollectorHealthClassifier.Healthy, FleetDeadlockSource.PostgresTarget)] + [InlineData(CollectorHealthClassifier.Warning, FleetDeadlockSource.PostgresTarget)] + [InlineData(CollectorHealthClassifier.Stale, FleetDeadlockSource.PostgresTarget)] + [InlineData(CollectorHealthClassifier.Failing, FleetDeadlockSource.PostgresTarget)] + [InlineData(CollectorHealthClassifier.NoPermissions, FleetDeadlockSource.CollectorDenied)] + [InlineData(CollectorHealthClassifier.Stopped, FleetDeadlockSource.CollectorSilent)] + [InlineData(CollectorHealthClassifier.NeverRun, FleetDeadlockSource.CollectorSilent)] + [InlineData(null, FleetDeadlockSource.CollectorSilent)] + public void APostgresTarget_IsCoveredOnItsOwnCollectorsTerms(string? band, FleetDeadlockSource expected) + => Assert.Equal(expected, FleetDeadlockCoverage.ClassifyDeadlockSource(isPostgres: true, band)); + + /// The one predicate both roll-ups reduce servers_read with: the two covered arms and + /// nothing else. Enumerated over the whole enum so a value added later lands uncovered by default. + [Fact] + public void ExactlyTheTwoCoveredArmsCount() + { + Assert.True(FleetDeadlockCoverage.IsCovered(FleetDeadlockSource.Read)); + Assert.True(FleetDeadlockCoverage.IsCovered(FleetDeadlockSource.PostgresTarget)); + Assert.False(FleetDeadlockCoverage.IsCovered(FleetDeadlockSource.CollectorSilent)); + Assert.False(FleetDeadlockCoverage.IsCovered(FleetDeadlockSource.CollectorDenied)); + Assert.Equal(2, Enum.GetValues().Count(FleetDeadlockCoverage.IsCovered)); + } /// /// A card that sets nothing reads as UNCOVERED, and that is the load-bearing default. DeadlockSource @@ -464,52 +482,69 @@ public void BuildRollup_ReducesCoverageFromTheCards_WithEveryCauseAttributed() { Card(1, band: CollectorHealthClassifier.Healthy), Card(2, band: CollectorHealthClassifier.Failing), - Card(3, isPostgres: true), - Card(4, isPostgres: true), + Card(3, isPostgres: true, band: CollectorHealthClassifier.Healthy), + Card(4, isPostgres: true, band: CollectorHealthClassifier.Stale), Card(5, band: CollectorHealthClassifier.Stopped), Card(6, band: CollectorHealthClassifier.NoPermissions), Card(7, band: null), + /* #3539: a PostgreSQL target whose pg_database_stats collector left no band is SILENT, not + a PostgreSQL bucket entry - the engine no longer answers on its own. */ + Card(8, isPostgres: true, band: null), }, Now, Now.AddHours(-1), Now); var coverage = rollup.DeadlockCoverage; - Assert.Equal(2, coverage.ServersRead); - Assert.Equal(7, coverage.ServersTotal); + /* Four read: two SQL Servers through their deadlocks collector, two PostgreSQL targets through + pg_database_stats (#3539). */ + Assert.Equal(4, coverage.ServersRead); + Assert.Equal(8, coverage.ServersTotal); + /* The PostgreSQL sub-count names the instrument for two of the four read. */ Assert.Equal(2, coverage.PostgresServers); - Assert.Equal(2, coverage.ServersCollectorSilent); // STOPPED + the null band + Assert.Equal(3, coverage.ServersCollectorSilent); // STOPPED + the null band + the bandless PostgreSQL target Assert.Equal(1, coverage.ServersCollectorDenied); /* Every server is accounted for exactly once — an unattributed server would mean coverage that - reports a gap it cannot explain, which is the same shape as a total that reports no denominator. */ + reports a gap it cannot explain, which is the same shape as a total that reports no denominator. + Three terms, not four: postgres_servers is a SUBSET of servers_read since #3539, and a consumer + still summing it in would over-count the fleet by every PostgreSQL target. */ Assert.Equal( coverage.ServersTotal, - coverage.ServersRead + coverage.PostgresServers - + coverage.ServersCollectorSilent + coverage.ServersCollectorDenied); + coverage.ServersRead + coverage.ServersCollectorSilent + coverage.ServersCollectorDenied); + Assert.True(coverage.PostgresServers <= coverage.ServersRead); /* And it agrees with the field the fleet already reported. */ Assert.Equal(rollup.TotalServers, coverage.ServersTotal); } /// - /// The measured case, end to end: a PostgreSQL-only fleet reports total_deadlocks: 0 with zero - /// coverage beside it, and the note sends the reader to the tool that can actually answer. + /// The measured case, end to end (#3539 reversed its direction): a PostgreSQL-only fleet whose + /// pg_database_stats collectors are running reports FULL coverage, its deadlocks summed into + /// total_deadlocks, and the note names the instrument and the tool that has the graphs — the + /// pre-#3539 sentence, "cannot count at all", is gone. /// [Fact] - public void APostgresOnlyFleet_ReportsZeroCoverage_AndNamesTheToolThatCanAnswer() + public void APostgresOnlyFleet_IsCovered_AndTheNoteNamesTheInstrument() { var rollup = DarlingFleetReader.BuildRollup( - new[] { Card(1, isPostgres: true), Card(2, isPostgres: true), Card(3, isPostgres: true) }, + new[] + { + Card(1, isPostgres: true, band: CollectorHealthClassifier.Healthy, deadlockCount: 2), + Card(2, isPostgres: true, band: CollectorHealthClassifier.Healthy), + Card(3, isPostgres: true, band: CollectorHealthClassifier.Failing, deadlockCount: 1), + }, Now, Now.AddHours(-1), Now); - Assert.Equal(0, rollup.TotalDeadlocks); - Assert.Equal(0, rollup.DeadlockCoverage.ServersRead); + Assert.Equal(3, rollup.TotalDeadlocks); + Assert.Equal(3, rollup.DeadlockCoverage.ServersRead); Assert.Equal(3, rollup.DeadlockCoverage.PostgresServers); var note = rollup.DeadlockCoverage.Note; - Assert.Contains("read a deadlock source for 0 of 3", note, StringComparison.Ordinal); + Assert.Contains("read a deadlock source for 3 of 3", note, StringComparison.Ordinal); + Assert.Contains("3 of those are PostgreSQL targets counted from the server's own pg_stat_database.deadlocks counter", note, StringComparison.Ordinal); Assert.Contains("get_pg_deadlocks", note, StringComparison.Ordinal); + Assert.DoesNotContain("cannot count", note, StringComparison.Ordinal); /* The two causes that do not apply are absent, so a reader is not handed three actions when one is called for. */ @@ -517,6 +552,29 @@ called for. */ Assert.DoesNotContain("needs a grant", note, StringComparison.Ordinal); } + /// + /// The cross-server PostgreSQL deadlock read is a per-series counter DIFFERENCE, clamped, summed — + /// pinned on the SQL's text because the alternative, SUM(deadlocks), is a plausible-looking + /// one-liner that returns a lifetime counter multiplied by the sample count (measured: eight million + /// "deadlocks" on a store with none in the window). The live test runs it; this stops a rewrite from + /// quietly reintroducing the sum. + /// + [Fact] + public void ThePostgresDeadlockRead_DifferencesTheCounterPerDatabaseSeries_AndNeverSumsTheColumn() + { + var sql = DarlingFleetReader.FleetPgDeadlockSql; + + Assert.Contains("FROM pg_database_stats", sql, StringComparison.Ordinal); + Assert.Contains("deadlocks - LAG(deadlocks) OVER (PARTITION BY server_id, database_name ORDER BY collection_time)", sql, StringComparison.Ordinal); + Assert.Contains("SUM(GREATEST(raw_delta, 0))", sql, StringComparison.Ordinal); + Assert.Contains("MAX(collection_time) FILTER (WHERE raw_delta > 0)", sql, StringComparison.Ordinal); + Assert.DoesNotContain("SUM(deadlocks)", sql, StringComparison.Ordinal); + /* Windowed on the partitioning column, both bounds, like the SQL Server twin. */ + Assert.Contains("collection_time >= $1", sql, StringComparison.Ordinal); + Assert.Contains("collection_time <= $2", sql, StringComparison.Ordinal); + Assert.Contains("GROUP BY server_id", sql, StringComparison.Ordinal); + } + /// /// The sentence this whole issue turns on. The two windows genuinely diverge — /// total_deadlocks is counted between the caller's bounds (one hour by default), while coverage is @@ -562,7 +620,7 @@ public void TheCoverageSerializes_BesideTheTotal_WithTheCauseAsAName() { var json = JsonSerializer.Serialize( DarlingFleetReader.BuildRollup( - new[] { Card(1, isPostgres: true), Card(2, band: CollectorHealthClassifier.Healthy) }, + new[] { Card(1, isPostgres: true, band: CollectorHealthClassifier.Healthy), Card(2, band: CollectorHealthClassifier.Healthy) }, Now, Now.AddHours(-1), Now), DarlingFleetReader.JsonOptions); @@ -578,12 +636,17 @@ public void TheCoverageSerializes_BesideTheTotal_WithTheCauseAsAName() JsonAssert.Contains("\"deadlock_source\": \"PostgresTarget\"", json); JsonAssert.Contains("\"deadlock_source\": \"Read\"", json); - JsonAssert.Contains("\"servers_read\": 1", json); + /* Both covered (#3539): servers_read counts the PostgreSQL target, and postgres_servers names it + as the counter-read one of the two. */ + JsonAssert.Contains("\"servers_read\": 2", json); + JsonAssert.Contains("\"postgres_servers\": 1", json); } private static readonly DateTime Now = new(2026, 9, 5, 12, 0, 0, DateTimeKind.Unspecified); - private static FleetServerCard Card(int id, bool isPostgres = false, string? band = null) => + /// The ENGINE'S deadlock-source collector band — deadlocks on a SQL Server, + /// pg_database_stats on a PostgreSQL target (#3539); the card carries one field for it. + private static FleetServerCard Card(int id, bool isPostgres = false, string? band = null, int deadlockCount = 0) => new() { ServerId = id, @@ -591,6 +654,7 @@ private static FleetServerCard Card(int id, bool isPostgres = false, string? ban ServerName = "target-" + id.ToString(CultureInfo.InvariantCulture), IsPostgres = isPostgres, DeadlockCollectorBand = band, + DeadlockCount = deadlockCount, }; } diff --git a/Darling/Darling.Tests/DarlingPgReadSqlParsesLiveTests.cs b/Darling/Darling.Tests/DarlingPgReadSqlParsesLiveTests.cs index 986429317..896e767c8 100644 --- a/Darling/Darling.Tests/DarlingPgReadSqlParsesLiveTests.cs +++ b/Darling/Darling.Tests/DarlingPgReadSqlParsesLiveTests.cs @@ -74,8 +74,8 @@ public sealed class DarlingPgReadSqlParsesLiveTests /// explicitly rather than inferred, because every rule for inferring it reads the field's VALUE — which /// is reflection, the very thing the census exists to check independently. /// - /// Two of the three are SQL FRAGMENTS interpolated into a query rather than queries themselves, - /// and the third is the trap that makes a name-shaped rule useless here: + /// All but one are SQL FRAGMENTS interpolated into a query rather than queries themselves, + /// and the one is the trap that makes a name-shaped rule useless here: /// StaleStatisticsChurnRatioSql is spelled like a query and holds "0.2". Keyed per SITE /// rather than per name, so the same name on a different reader is not excused by inheritance from /// this one — and the unused-entry clause below makes a move show up as an edit here rather than as @@ -90,6 +90,12 @@ statement. The query it is spliced into - CoverageEvidenceSql - IS in the parse- population, so the fragment is verified where it is used rather than left unverified. */ "DarlingPgIndexBloatReader.EvidenceStatusList", "DarlingPgServerConfigReader.SessionScopedSources", + /* #3539: the PostgreSQL Long-Running Query read's switchable dump/restore opt-out - one AND + predicate spliced into CurrentLongRunningSessionsSqlTemplate by BuildCurrentLongRunningSessionsSql. + Not a statement; both renderings of the template it is spliced into + (CurrentLongRunningSessionsSql with it, CurrentLongRunningSessionsSqlBackupsIncluded without) ARE + in the parse-checked population, so the fragment is verified where it is used. */ + "DarlingPgSessionStatesReader.BackupUtilitiesFilter", "DarlingPgTableBloatReader.StaleStatisticsChurnRatioSql", }; diff --git a/Darling/Darling.Tests/DarlingPgSessionStatesReaderTests.cs b/Darling/Darling.Tests/DarlingPgSessionStatesReaderTests.cs index 7522e5ba1..14954c1c3 100644 --- a/Darling/Darling.Tests/DarlingPgSessionStatesReaderTests.cs +++ b/Darling/Darling.Tests/DarlingPgSessionStatesReaderTests.cs @@ -7,6 +7,7 @@ */ using System; +using System.Collections.Generic; using System.Linq; using System.Text.RegularExpressions; using PerformanceMonitor.Collectors; @@ -106,6 +107,92 @@ public void LongRunningSql_ExcludesIdleInTransactionSessions() Assert.Contains("s.is_idle_in_transaction = false", LongRunningSql, StringComparison.Ordinal); } + // ── #3539 — the noise opt-outs SQL Server's read has had all along ─────────────────────────── + + /// + /// Non-client backends are out unconditionally — the sibling of SQL Server's session_id > 50. + /// Autovacuum workers and walsenders are the two that reached the alert in the field (an autovacuum of + /// a large table at minute 31, a streaming standby's walsender for as long as it is connected). NULL + /// keeps the row: backend_type is privileged and comes back NULL without pg_monitor, and a + /// filter that dropped NULL would silence the alert on exactly the targets the collector has already + /// flagged as redacted. + /// + [Fact] + public void LongRunningSql_ExcludesNonClientBackends_KeepingRedactedNulls() + { + Assert.Contains("coalesce(s.backend_type, 'client backend') = 'client backend'", LongRunningSql, StringComparison.Ordinal); + } + + /// + /// The four maintenance statements are out unconditionally, by the whitelisted command tag — the only + /// statement handle this table has, because it stores no text. CREATE is deliberately not among + /// them: the tag cannot separate an index build from a CREATE TABLE AS, so a CREATE is reported + /// with its tag rather than dropped on a guess. NULL-safe in the KEEPING direction like the backend + /// filter — a NULL tag is a session the collector could not classify, not a maintenance statement. + /// + [Fact] + public void LongRunningSql_ExcludesMaintenanceStatementsByCommandTag_ButNotCreate() + { + Assert.Contains("coalesce(s.command_tag, '') NOT IN ('VACUUM', 'ANALYZE', 'REINDEX', 'CLUSTER')", LongRunningSql, StringComparison.Ordinal); + Assert.DoesNotContain("'CREATE'", LongRunningSql, StringComparison.Ordinal); + } + + /// + /// The dump/restore utilities ride the SHARED longRunningQueryExcludeBackups switch (SQL Server's + /// BackupsFilter), so the filter is in the text exactly when the switch is on, and the four names + /// are libpq's fallback_application_name for those tools. psql is not in the list — an + /// operator's ad-hoc statement running long IS a long-running query. The default rendering (the constant + /// under the pre-#3539 name) is the switch-on shape, because that is the shipped default. + /// + [Fact] + public void LongRunningSql_DropsDumpAndRestoreUtilities_OnlyWhenTheSharedBackupsSwitchIsOn() + { + var on = DarlingPgSessionStatesReader.BuildCurrentLongRunningSessionsSql(excludeBackups: true); + var off = DarlingPgSessionStatesReader.BuildCurrentLongRunningSessionsSql(excludeBackups: false); + + Assert.Equal( + "AND coalesce(s.application_name, '') NOT IN ('pg_dump', 'pg_dumpall', 'pg_restore', 'pg_basebackup')", + DarlingPgSessionStatesReader.BackupUtilitiesFilter); + Assert.Contains(DarlingPgSessionStatesReader.BackupUtilitiesFilter, on, StringComparison.Ordinal); + Assert.DoesNotContain("application_name", off.Replace("s.application_name,", ""), StringComparison.Ordinal); + Assert.DoesNotContain("'psql'", on, StringComparison.Ordinal); + Assert.Equal(on, LongRunningSql); + + /* The placeholder never reaches the server, on either setting. */ + Assert.DoesNotContain("{0}", on, StringComparison.Ordinal); + Assert.DoesNotContain("{0}", off, StringComparison.Ordinal); + + /* And the unconditional filters are in BOTH renderings - the switch governs only the utilities. */ + foreach (var sql in new[] { on, off }) + { + Assert.Contains("'client backend'", sql, StringComparison.Ordinal); + Assert.Contains("'VACUUM'", sql, StringComparison.Ordinal); + Assert.Contains("s.is_idle_in_transaction = false", sql, StringComparison.Ordinal); + } + } + + /// + /// excludedDatabases is applied after the read with the SQL Server adapter's exact rule: ordinal + /// ignore-case on the name, a row with no database name kept, null/empty list a no-op. Executed here, + /// not just pinned — the rule is pure. + /// + [Fact] + public void ExcludedDatabases_AppliedAfterTheRead_CaseInsensitively_KeepingUnnamedRows() + { + var rows = new List + { + new(1, 101, "Orders", "app", "svc", "SELECT", 3_600_000), + new(2, 102, "billing", "app", "svc", "UPDATE", 2_400_000), + new(3, 103, null, "app", "svc", "(other)", 1_900_000), + }; + + var filtered = DarlingPgSessionStatesReader.FilterExcludedDatabases(rows, new[] { "ORDERS" }); + Assert.Equal(new long[] { 2, 3 }, filtered.Select(r => r.BackendId).ToArray()); + + Assert.Same(rows, DarlingPgSessionStatesReader.FilterExcludedDatabases(rows, null)); + Assert.Same(rows, DarlingPgSessionStatesReader.FilterExcludedDatabases(rows, Array.Empty())); + } + // ── Scoping and parameterisation ───────────────────────────────────────────────────────────── [Fact] diff --git a/Darling/Darling.Tests/DeadlockRateBandTests.cs b/Darling/Darling.Tests/DeadlockRateBandTests.cs index 0c93b9f29..2cbed0f9b 100644 --- a/Darling/Darling.Tests/DeadlockRateBandTests.cs +++ b/Darling/Darling.Tests/DeadlockRateBandTests.cs @@ -237,23 +237,20 @@ public void TheMinimumWindowIsTheSmallestWindowAnySurfaceCanAskFor() /* ─────────────────────── the null (no-source) arm ─────────────────────── */ /// - /// A PostgreSQL target has no SQL-Server deadlock reading, so it bands off none (#3272/#3017) — and that - /// survives the rate change at every window, including the unrateable ones where a COUNT above zero - /// reads Warning. The null arm is tested first in the method for exactly this reason: an absent source - /// must not be reachable by the arm that exists for an absent denominator. + /// A caller with no deadlock source bands off none (#3272) — and that survives the rate change at every + /// window, including the unrateable ones where a COUNT above zero reads Warning. The null arm is tested + /// first in the method for exactly this reason: an absent source must not be reachable by the arm that + /// exists for an absent denominator. Since #3539 neither engine's CARD takes this arm (a PostgreSQL + /// target's count is its own counter difference); the arm stays for the daily classifier's empty cells + /// and for any caller that genuinely reads nothing. /// [Fact] - public void NoDeadlockSourceForTheEngine_StaysUnknown() + public void NoDeadlockSource_StaysUnknown() { foreach (var hours in new[] { 0, 1, 24, 168 }) { Assert.Equal(HealthSeverity.Unknown, Band(null, TimeSpan.FromHours(hours))); } - - Assert.Equal( - HealthSeverity.Unknown, - ServerHealthClassifier.DeadlockSeverity( - ServerMetricSources.DmvSourced(0, isPostgres: true), Hour, DeadlockRateThresholds.Default)); } /* ─────────────────────── the tiers are settable ─────────────────────── */ diff --git a/Darling/Darling.Tests/FleetCardPostgresCpuLivePostgresTests.cs b/Darling/Darling.Tests/FleetCardPostgresCpuLivePostgresTests.cs index 4a70fc60c..b8684e800 100644 --- a/Darling/Darling.Tests/FleetCardPostgresCpuLivePostgresTests.cs +++ b/Darling/Darling.Tests/FleetCardPostgresCpuLivePostgresTests.cs @@ -52,6 +52,23 @@ namespace Darling.Tests; /// nullable columns, where a coalesce anywhere in the read path would turn SQL NULL into a measured 0 and /// claim headroom nobody measured. Neither is visible to an in-memory fixture, which is the whole reason /// this file exists beside the shape pins. +/// +/// And #3539's deadlock arm, on the same fixture. The Aurora target carries a +/// pg_database_stats counter series shaped to defeat every wrong read at once: two databases, one +/// of which steps its lifetime counter 40 → 42 → 42 → 45 (five new deadlocks across one flat interval) and +/// the other of which is RESET mid-window (7 → 7 → 0 → 1: one new deadlock after the reset, and a −7 a +/// naive difference would subtract), plus an out-of-window row far below the series and a second server's +/// series under the same database name. SUM(deadlocks) over the window would answer 184; +/// last-minus-first per database would answer 5 − 6 = −1; an unbounded read would add 37; a read that lost +/// its server partition would fold the other server's 50 in. The right answer is 6, and the card must band +/// it Warning (6/hr is past the shipped 5/hr bar) with the rate published and deadlock_source +/// reading the counter arm because the pg_database_stats collector has a HEALTHY row in the health +/// window. The self-hosted target has the same collector row and a FLAT series (two samples, one +/// difference of zero), so its card is the measured, earned Healthy zero; the silent Aurora target has +/// stats rows and no collector row, so its count is read (50) and its source is CollectorSilent — the +/// engine no longer answers on its own. A PostgreSQL target with NO rows in the window would band Unknown: +/// a difference of nothing is not a zero, which is what keeps #3539 A6's never-collected card measuring +/// nothing, and the unit matrix pins that arm. /// [Collection("live-postgres")] public sealed class FleetCardPostgresCpuLivePostgresTests @@ -148,6 +165,33 @@ await InsertPgCpuAsync(connection, AuroraNoCapacityServerId, AuroraNoCapacityNam /* The SQL Server arm, so "additive only" is asserted against a live read rather than argued. */ await InsertSqlServerCpuAsync(connection, SqlServerServerId, SqlServerName, now.AddMinutes(-1), 30, 4, ct); + /* #3539: the PostgreSQL deadlock arm. The pg_database_stats collector's health row makes the + two targets COVERED; the counter series below is what the count is differenced from. */ + await InsertCollectionLogAsync(connection, AuroraServerId, AuroraName, now.AddSeconds(-30), ct, PgDatabaseStatsCollector.Instance.Name); + await InsertCollectionLogAsync(connection, SelfHostedServerId, SelfHostedName, now.AddSeconds(-30), ct, PgDatabaseStatsCollector.Instance.Name); + /* Database "orders": 40 -> 42 -> 42 -> 45 = 2 + 0 + 3 = 5 new deadlocks. */ + await InsertPgDatabaseStatsAsync(connection, AuroraServerId, AuroraName, "orders", now.AddMinutes(-40), 40, ct); + await InsertPgDatabaseStatsAsync(connection, AuroraServerId, AuroraName, "orders", now.AddMinutes(-30), 42, ct); + await InsertPgDatabaseStatsAsync(connection, AuroraServerId, AuroraName, "orders", now.AddMinutes(-20), 42, ct); + await InsertPgDatabaseStatsAsync(connection, AuroraServerId, AuroraName, "orders", now.AddMinutes(-10), 45, ct); + /* Database "billing": 7 -> 7 -> 0 (reset) -> 1 = 0 + clamp(-7) + 1 = 1 new deadlock. */ + await InsertPgDatabaseStatsAsync(connection, AuroraServerId, AuroraName, "billing", now.AddMinutes(-40), 7, ct); + await InsertPgDatabaseStatsAsync(connection, AuroraServerId, AuroraName, "billing", now.AddMinutes(-30), 7, ct); + await InsertPgDatabaseStatsAsync(connection, AuroraServerId, AuroraName, "billing", now.AddMinutes(-20), 0, ct); + await InsertPgDatabaseStatsAsync(connection, AuroraServerId, AuroraName, "billing", now.AddMinutes(-10), 1, ct); + /* The covered, MEASURED zero: two samples, one flat difference. Without the second sample the + card would honestly read Unknown - a difference needs two samples - so the fixture supplies + it, and asserts Healthy is EARNED rather than defaulted. */ + await InsertPgDatabaseStatsAsync(connection, SelfHostedServerId, SelfHostedName, "app", now.AddMinutes(-20), 12, ct); + await InsertPgDatabaseStatsAsync(connection, SelfHostedServerId, SelfHostedName, "app", now.AddMinutes(-10), 12, ct); + /* Trap: a row OUTSIDE the window with a much lower counter. A read that did not bound its + window on the start would see 3 -> 40 and add 37. */ + await InsertPgDatabaseStatsAsync(connection, AuroraServerId, AuroraName, "orders", now.AddMinutes(-90), 3, ct); + /* Trap: a series on ANOTHER server, so a read that lost its per-server partition would fold + these into the Aurora card. */ + await InsertPgDatabaseStatsAsync(connection, AuroraSilentServerId, AuroraSilentName, "orders", now.AddMinutes(-30), 100, ct); + await InsertPgDatabaseStatsAsync(connection, AuroraSilentServerId, AuroraSilentName, "orders", now.AddMinutes(-20), 150, ct); + var result = await DarlingFleetReader.GetFleetOverviewAsync( postgres, now.AddHours(-1), now, now, cancellationToken: ct); @@ -181,24 +225,42 @@ makes the band's input and the number beside it describe one minute. */ Assert.Null(aurora.TotalThreads); Assert.Equal(HealthSeverity.Unknown, aurora.ThreadsSeverity); - /* #3272, end to end: the three DMV-sourced bands claim nothing for this engine. Worth a live + /* #3272, end to end: the two DMV-sourced bands claim nothing for this engine. Worth a live arm and not only the unit matrix, because the engine gate reads `servers.engine_kind` out of the store — a column this test wrote and the reader round-tripped, which is the one part of the decision no in-memory fixture exercises. */ Assert.Equal(HealthSeverity.Unknown, aurora.MemorySeverity); Assert.Equal(HealthSeverity.Unknown, aurora.BlockingSeverity); - Assert.Equal(HealthSeverity.Unknown, aurora.DeadlockSeverity); - /* The counts and #3017's disclosure are deliberately untouched, so the fleet total and its - coverage denominator still reconcile. */ Assert.Equal(0, aurora.BlockingCount); - Assert.Equal(0, aurora.DeadlockCount); + + /* #3539, end to end: the deadlock count is the per-database counter DIFFERENCE, clamped across + the reset, summed - 5 + 1 = 6 - and not the 184 a SUM(deadlocks) gives, the -1 a per-database + last-minus-first gives, the 43 an unbounded read gives, or the extra 50 a read that lost its + server partition folds in. Banded through the shared tiers over the one-hour window: 6/hr is + past the shipped 5/hr Warning bar. */ + Assert.Equal(6, aurora.DeadlockCount); + Assert.True(aurora.DeadlockMeasured); + Assert.Equal(6.0, aurora.DeadlockRatePerHour); + Assert.Equal(HealthSeverity.Warning, aurora.DeadlockSeverity); + /* The sample that first showed the newest step - both series stepped at -10 min. */ + Assert.Equal(DateTime.SpecifyKind(now.AddMinutes(-10), DateTimeKind.Unspecified).Ticks / TimeSpan.TicksPerSecond, + aurora.DeadlockLastSeen!.Value.Ticks / TimeSpan.TicksPerSecond); + /* Covered through the counter arm, because pg_database_stats has a health row. */ Assert.Equal(FleetDeadlockSource.PostgresTarget, aurora.DeadlockSource); + Assert.Equal(CollectorHealthClassifier.Healthy, aurora.DeadlockCollectorBand); // ── a PostgreSQL target this build collects no instance CPU for ───────────────────────── var selfHosted = seeded.Single(c => c.ServerId == SelfHostedServerId); Assert.Null(selfHosted.TotalCpuPercent); Assert.Equal(HealthSeverity.Unknown, selfHosted.CpuSeverity); Assert.Equal(FleetCpuSource.NoSourceForEngine, selfHosted.CpuSource); + /* #3539: the covered zero - a running pg_database_stats collector and a flat counter across the + window is a measured Healthy, exactly as a quiet SQL Server's is. */ + Assert.Equal(0, selfHosted.DeadlockCount); + Assert.True(selfHosted.DeadlockMeasured); + Assert.Equal(0.0, selfHosted.DeadlockRatePerHour); + Assert.Equal(HealthSeverity.Healthy, selfHosted.DeadlockSeverity); + Assert.Equal(FleetDeadlockSource.PostgresTarget, selfHosted.DeadlockSource); // ── an Aurora target whose ingest has produced nothing CURRENT ────────────────────────── /* Its only row is 90 minutes old, so the bound excludes it. The arm has to be NotCollected and @@ -207,6 +269,14 @@ coverage denominator still reconcile. */ Assert.Null(silent.TotalCpuPercent); Assert.Equal(HealthSeverity.Unknown, silent.CpuSeverity); Assert.Equal(FleetCpuSource.NotCollected, silent.CpuSource); + /* #3539: this target has stats rows (the partition trap, 100 -> 150) but NO pg_database_stats + health row, so the count is read - 50 - and the coverage says its collector is silent + rather than the pre-#3539 "PostgreSQL, cannot count". Both facts on one card: the count + is what the store holds, the source is what the health window knows. */ + Assert.Equal(50, silent.DeadlockCount); + Assert.True(silent.DeadlockMeasured); + Assert.Equal(HealthSeverity.Critical, silent.DeadlockSeverity); + Assert.Equal(FleetDeadlockSource.CollectorSilent, silent.DeadlockSource); // ── an Aurora target with a CURRENT reading and no capacity sample (#3281) ────────────── /* Unknown, never Healthy, and never the Critical the raw reading alone would earn. The source @@ -270,15 +340,36 @@ INSERT INTO servers (server_id, server_name, display_name, is_enabled, sql_engin } private static async Task InsertCollectionLogAsync( - NpgsqlConnection connection, int serverId, string name, DateTime collectionTime, CancellationToken ct) + NpgsqlConnection connection, int serverId, string name, DateTime collectionTime, CancellationToken ct, + string collectorName = "pg_cpu_utilization") { using var command = new NpgsqlCommand(@" INSERT INTO collection_log (log_id, server_id, server_name, collector_name, collection_time, status) -VALUES ($1, $2, $3, 'pg_cpu_utilization', $4, 'SUCCESS')", connection); +VALUES ($1, $2, $3, $5, $4, 'SUCCESS')", connection); command.Parameters.AddWithValue(CollectionIdGenerator.Next()); command.Parameters.AddWithValue(serverId); command.Parameters.AddWithValue(name); command.Parameters.AddWithValue(DateTime.SpecifyKind(collectionTime, DateTimeKind.Unspecified)); + command.Parameters.AddWithValue(collectorName); + await command.ExecuteNonQueryAsync(ct); + } + + /// Seeds one collect.pg_database_stats row carrying only the deadlock counter (#3539) — + /// the other counters are left NULL, which the read must tolerate (it differences one column). + private static async Task InsertPgDatabaseStatsAsync( + NpgsqlConnection connection, int serverId, string name, string databaseName, + DateTime collectionTime, long deadlocks, CancellationToken ct) + { + using var command = new NpgsqlCommand(@" +INSERT INTO pg_database_stats + (collection_id, collection_time, server_id, server_name, database_name, deadlocks) +VALUES ($1, $2, $3, $4, $5, $6)", connection); + command.Parameters.AddWithValue(CollectionIdGenerator.Next()); + command.Parameters.AddWithValue(DateTime.SpecifyKind(collectionTime, DateTimeKind.Unspecified)); + command.Parameters.AddWithValue(serverId); + command.Parameters.AddWithValue(name); + command.Parameters.AddWithValue(databaseName); + command.Parameters.AddWithValue(deadlocks); await command.ExecuteNonQueryAsync(ct); } @@ -332,7 +423,7 @@ private static async Task DeleteSentinelRowsAsync(NpgsqlConnection connection, C { var ids = string.Join(", ", SentinelIds.Select(i => i.ToString(CultureInfo.InvariantCulture))); - foreach (var table in new[] { "pg_cpu_utilization", "cpu_utilization_stats", "collection_log", "servers" }) + foreach (var table in new[] { "pg_database_stats", "pg_cpu_utilization", "cpu_utilization_stats", "collection_log", "servers" }) { using var cleanup = new NpgsqlCommand($"DELETE FROM {table} WHERE server_id IN ({ids});", connection); await cleanup.ExecuteNonQueryAsync(ct); diff --git a/Darling/Darling.Tests/FleetCardPostgresCpuTests.cs b/Darling/Darling.Tests/FleetCardPostgresCpuTests.cs index fac08d8ef..a41baed4a 100644 --- a/Darling/Darling.Tests/FleetCardPostgresCpuTests.cs +++ b/Darling/Darling.Tests/FleetCardPostgresCpuTests.cs @@ -64,6 +64,7 @@ private static FleetServerCard Card( default, default, default, + default, Now.AddSeconds(-30), default, null, diff --git a/Darling/Darling.Tests/PgCpuCapacityHeadroomTests.cs b/Darling/Darling.Tests/PgCpuCapacityHeadroomTests.cs index 87f1ec762..769fe18c2 100644 --- a/Darling/Darling.Tests/PgCpuCapacityHeadroomTests.cs +++ b/Darling/Darling.Tests/PgCpuCapacityHeadroomTests.cs @@ -72,6 +72,7 @@ private static FleetServerCard Card( default, default, default, + default, Now.AddSeconds(-30), default, null, @@ -187,6 +188,7 @@ public void TheRingBufferArmBandsOnItsOwnReading(double sqlCpu, double otherCpu, default, default, default, + default, Now.AddSeconds(-30), default, null, @@ -519,7 +521,7 @@ public void BothReasonLinesNameTheCapacityFigure_InTheSameWords() var sqlServer = DarlingFleetReader.BuildCard( new DarlingFleetReader.FleetServerRow(2, "sql-1", "sql-1", null, MonitoredEngineKind.SqlServer, false), new DarlingFleetReader.CpuRow(60.0, 30.0), - default, default, default, default, default, default, + default, default, default, default, default, default, default, Now.AddSeconds(-30), default, null, Now, TimeSpan.FromHours(1), DeadlockRateThresholds.Default); diff --git a/Darling/Darling.Tests/UnmeasuredMetricsAreNotHealthyTests.cs b/Darling/Darling.Tests/UnmeasuredMetricsAreNotHealthyTests.cs index 4111528c9..b4f911600 100644 --- a/Darling/Darling.Tests/UnmeasuredMetricsAreNotHealthyTests.cs +++ b/Darling/Darling.Tests/UnmeasuredMetricsAreNotHealthyTests.cs @@ -28,6 +28,13 @@ namespace Darling.Tests; /// what and /// already did by taking nullable inputs. /// +/// Deadlocks left the unmeasured set in #3539. A PostgreSQL target's deadlocks are now +/// counted from its own pg_stat_database.deadlocks counter (differenced over the window) and banded +/// through the shared rate tiers, so that row is a MEASUREMENT on both engines and this file asserts it +/// bands — Healthy at zero, Warning and Critical at the tiers — rather than reading Unknown. Memory pressure +/// and blocking stay unmeasured on PostgreSQL for the reasons gives, and +/// those two are what the Unknown pins below are about. +/// /// The invariant this is held to. A MEASURED metric must band exactly as it does today. That is /// the real constraint, and it is not the same as "leave the SQL Server path alone": the fix necessarily /// edits functions every SQL Server surface calls. is @@ -56,13 +63,25 @@ public sealed class UnmeasuredMetricsAreNotHealthyTests /// Collectors banded for this server, none failing — forty by default so /// the card's collectors row is a MEASURED calm reading and the DMV metrics stay this file's only /// variable. Zero is the #3539 A6 shape: nothing banded, nothing measured. + /// The SQL Server extended-event count — what a SQL Server card believes. + /// The PostgreSQL counter-difference count (#3539) — what a PostgreSQL card + /// believes. Both are always handed in so a test can assert the card took the ENGINE'S row and ignored + /// the other engine's structural zero. + /// How many differences that count was summed over — the PostgreSQL + /// arm's measured/not-measured test. Zero by default: a PostgreSQL card is UNMEASURED unless a test + /// says a difference was taken, which is the direction a forgotten argument must fail in. + /// The pg_database_stats collector's band, for the coverage + /// arm. private static FleetServerCard Card( string? engineKind, bool memoryPressure = false, int blocking = 0, long maxBlockingWaitMs = 0, int deadlocks = 0, - int bandedCollectors = 40) => + int bandedCollectors = 40, + int pgDeadlocks = 0, + long pgDeadlockIntervals = 0, + string? pgDeadlockBand = null) => DarlingFleetReader.BuildCard( new DarlingFleetReader.FleetServerRow(1, "t", "t", null, engineKind, false), default, @@ -72,8 +91,9 @@ private static FleetServerCard Card( default, new DarlingFleetReader.BlockingRow(blocking, maxBlockingWaitMs, 0, 0), new DarlingFleetReader.DeadlockRow(deadlocks, deadlocks > 0 ? Now.AddMinutes(-5) : null), + new DarlingFleetReader.PgDeadlockRow(pgDeadlocks, pgDeadlocks > 0 ? Now.AddMinutes(-7) : null, pgDeadlockIntervals), Now.AddSeconds(-30), - new DarlingFleetReader.CollectorCounts(bandedCollectors, 0, bandedCollectors), + new DarlingFleetReader.CollectorCounts(bandedCollectors, 0, bandedCollectors, null, pgDeadlockBand), null, Now, /* #3368: a real one-hour window and the shipped tiers. This file's subject is the @@ -117,8 +137,12 @@ private static ServerSummaryItem ViewerCard( /* ─────────────────────────── the defect ─────────────────────────── */ /// - /// A PostgreSQL card's three DMV-sourced metrics read Unknown, not Healthy. Red against the unfixed - /// classifiers, which had no arm that could return anything else for a zero. + /// A PostgreSQL card's two DMV-sourced metrics read Unknown, not Healthy. Red against the unfixed + /// classifiers, which had no arm that could return anything else for a zero. Deadlocks read Unknown + /// here too, but for #3539's OWN reason rather than #3272's: the helper hands this card no counter + /// difference (zero intervals), and a difference of nothing is not a zero. The measured case — a + /// difference taken, zero deadlocks — is Healthy with a published 0.0/hr, and is asserted beside it so + /// the two readings of the same card cannot be confused. /// [Theory] [InlineData(MonitoredEngineKind.Postgres)] @@ -130,20 +154,26 @@ public void APostgresCard_ClaimsNoHealthForWhatItNeverMeasured(string engineKind Assert.Equal(HealthSeverity.Unknown, card.MemorySeverity); Assert.Equal(HealthSeverity.Unknown, card.BlockingSeverity); Assert.Equal(HealthSeverity.Unknown, card.DeadlockSeverity); - - /* And the RATE says the same thing the severity says. A rate of 0.0/hr on a target with no deadlock - source is a fabricated measurement, and both the card chip and the viewer detail line render this - field on non-null alone - so a structural zero here would contradict the Unknown above on the very - same card. */ Assert.Null(card.DeadlockRatePerHour); + Assert.False(card.DeadlockMeasured); /* Threads already reached Unknown on its own (a null ceiling), and CPU does since #3267. So after this the card makes NO unearned claim on any metric row - which is the property worth asserting, - rather than three separate arms that happen to agree today. */ + rather than separate arms that happen to agree today. */ Assert.Equal(HealthSeverity.Unknown, card.ThreadsSeverity); Assert.DoesNotContain( HealthSeverity.Healthy, new[] { card.MemorySeverity, card.BlockingSeverity, card.DeadlockSeverity, card.ThreadsSeverity, card.CpuSeverity }); + + /* #3539: ONE difference taken and the same zero is a measurement. Healthy is EARNED here - a zero + counter difference over a rateable hour - and the rate is published beside it, because the chip + and the viewer detail render the rate on non-null alone and a band with no rate beside it is the + #3368 defect. */ + var measured = Card(engineKind, pgDeadlockIntervals: 1); + Assert.True(measured.DeadlockMeasured); + Assert.Equal(HealthSeverity.Healthy, measured.DeadlockSeverity); + Assert.Equal(0.0, measured.DeadlockRatePerHour); + Assert.Equal(0, measured.DeadlockCount); } /// The viewer's card, same server, same answer — the #2473 rule. @@ -156,16 +186,24 @@ public void TheViewerCardAgrees(string engineKind) Assert.Equal(HealthSeverity.Unknown, card.MemorySeverity); Assert.Equal(HealthSeverity.Unknown, card.BlockingSeverity); + /* No difference taken (the helper leaves PgDeadlockIntervals at zero), so Unknown - and one + difference makes the same zero Healthy, as on the fleet card. */ Assert.Equal(HealthSeverity.Unknown, card.DeadlockSeverity); Assert.Null(card.DeadlockRatePerHour); + + var measured = ViewerCard(engineKind); + measured.PgDeadlockIntervals = 1; + Assert.Equal(HealthSeverity.Healthy, measured.DeadlockSeverity); + Assert.Equal(0.0, measured.DeadlockRatePerHour); } /// - /// The published COUNTS are deliberately unchanged, and #3017's disclosure still reads the same. The - /// fleet's total_deadlocks is summed from those zeros and deadlock_coverage is what - /// explains it; nulling them would leave that denominator describing nothing. So the band stopped - /// claiming health while the count kept saying what it counted — and the two now AGREE, where before - /// deadlock_source: PostgresTarget sat beside deadlock_severity: Healthy. + /// The published COUNTS are deliberately unchanged, and #3017's disclosure reads the collector state on + /// this engine too since #3539. The fleet's total_deadlocks is summed from the cards and + /// deadlock_coverage is what explains it; nulling the counts would leave that denominator + /// describing nothing. A PostgreSQL card whose pg_database_stats collector left no band is + /// UNCOVERED (silent), exactly as a SQL Server card with no deadlocks band is — the pre-#3539 + /// answer, PostgresTarget on the engine alone, would have called a server nothing read "counted". /// [Fact] public void TheCountsAndTheCoverageDisclosureAreUntouched() @@ -175,7 +213,86 @@ public void TheCountsAndTheCoverageDisclosureAreUntouched() Assert.Equal(0, card.BlockingCount); Assert.Equal(0, card.DeadlockCount); Assert.False(card.HasMemoryPressure); + Assert.Equal(FleetDeadlockSource.CollectorSilent, card.DeadlockSource); + + var counted = Card(MonitoredEngineKind.AuroraPostgres, pgDeadlockBand: CollectorHealthClassifier.Healthy); + Assert.Equal(FleetDeadlockSource.PostgresTarget, counted.DeadlockSource); + } + + /* ─────────────────────────── the PostgreSQL deadlock band (#3539) ─────────────────────────── */ + + /// + /// The PostgreSQL card bands its OWN count through the SAME tiers the SQL Server card uses: over the + /// helper's one-hour window the count is the rate, so 4 is under the shipped 5/hr Warning bar, 5 is + /// Warning, and 20 is the shipped Critical bar. The SQL Server row handed in alongside is a + /// structural zero for this engine and must be IGNORED — a card that summed the two would be right by + /// accident here and wrong the moment either read produced a non-zero for the wrong engine. + /// + [Theory] + [InlineData(0, HealthSeverity.Healthy)] + [InlineData(4, HealthSeverity.Healthy)] + [InlineData(5, HealthSeverity.Warning)] + [InlineData(19, HealthSeverity.Warning)] + [InlineData(20, HealthSeverity.Critical)] + public void APostgresCard_BandsItsCounterDifferenceThroughTheSharedTiers(int pgDeadlocks, HealthSeverity expected) + { + var card = Card(MonitoredEngineKind.Postgres, deadlocks: 999, pgDeadlocks: pgDeadlocks, + pgDeadlockIntervals: 59, pgDeadlockBand: CollectorHealthClassifier.Healthy); + + Assert.Equal(pgDeadlocks, card.DeadlockCount); + Assert.True(card.DeadlockMeasured); + Assert.Equal(expected, card.DeadlockSeverity); + Assert.Equal(pgDeadlocks, card.DeadlockRatePerHour); + Assert.Equal(pgDeadlocks > 0 ? Now.AddMinutes(-7) : null, card.DeadlockLastSeen); Assert.Equal(FleetDeadlockSource.PostgresTarget, card.DeadlockSource); + + /* The overall band follows, so the fleet's worst-first ranking sees a deadlocking PostgreSQL + server the way it sees a deadlocking SQL Server. */ + Assert.Equal(expected == HealthSeverity.Healthy ? FleetHealthBand.Healthy + : expected == HealthSeverity.Warning ? FleetHealthBand.Warning : FleetHealthBand.Critical, card.Band); + } + + /// The mirror image: a SQL Server card takes the extended-event row and ignores the PostgreSQL + /// row, and its collector band is the deadlocks collector's, not pg_database_stats'. + [Fact] + public void ASqlServerCard_IgnoresThePostgresRow() + { + var card = Card(MonitoredEngineKind.SqlServer, deadlocks: 2, pgDeadlocks: 999, pgDeadlockIntervals: 59, pgDeadlockBand: CollectorHealthClassifier.Healthy); + + Assert.Equal(2, card.DeadlockCount); + /* A SQL Server count is a measurement whatever the PostgreSQL row's interval count says - and + whatever its own collector band says, which is #3272's engine-not-collector rule. */ + Assert.True(card.DeadlockMeasured); + Assert.True(Card(MonitoredEngineKind.SqlServer).DeadlockMeasured); + Assert.Equal(Now.AddMinutes(-5), card.DeadlockLastSeen); + Assert.Equal(HealthSeverity.Healthy, card.DeadlockSeverity); + /* No deadlocks-collector band was handed in, so a SQL Server is silent whatever pg_database_stats + says about it. */ + Assert.Equal(FleetDeadlockSource.CollectorSilent, card.DeadlockSource); + } + + /// + /// #3528's label counts the row as measured on a PostgreSQL card now: the fold over the six metric + /// severities finds one more non-Unknown than it did, so "Healthy — N of 6 measured" moves by one on + /// every PostgreSQL card. The re-scoring bundle carries the count as a reading too, so the worst-first + /// rank sees the same measurement the dot does. + /// + [Fact] + public void ThePostgresCardsMeasuredMetricCountIncludesDeadlocks() + { + var card = Card(MonitoredEngineKind.Postgres, pgDeadlocks: 1, pgDeadlockIntervals: 59); + + /* CPU (no source), Threads, Memory and Blocking are Unknown; Deadlocks and Collectors (the + helper's forty) are measured. Written as the two numbers rather than "one more than before" so + a regression that un-measured a different row could not pass by coincidence. */ + Assert.Equal(2, card.MeasuredMetricCount); + Assert.Equal(6, card.MetricCount); + Assert.Equal((long?)1L, card.ToHealthMetricsValue.DeadlockCount); + + /* And with no difference taken the re-band bundle carries null, the way the card banded - so the + worst-first score cannot see a measurement the dot did not. */ + Assert.Null(Card(MonitoredEngineKind.Postgres).ToHealthMetricsValue.DeadlockCount); + Assert.Equal(1, Card(MonitoredEngineKind.Postgres).MeasuredMetricCount); } /* ─────────────────────────── the invariant ─────────────────────────── */ diff --git a/Darling/Darling.Tests/ViewerFleetRollupTests.cs b/Darling/Darling.Tests/ViewerFleetRollupTests.cs index 3c9c4d19b..ea35eb9f7 100644 --- a/Darling/Darling.Tests/ViewerFleetRollupTests.cs +++ b/Darling/Darling.Tests/ViewerFleetRollupTests.cs @@ -60,6 +60,29 @@ public void FleetTotalsSql_DeadlocksAreACrossServerCount_OverTheWindow() Assert.Contains("deadlock_time <= $2", sql, StringComparison.Ordinal); } + /// + /// #3539: the PostgreSQL half of the deadlock total is a per-(server_id, database_name) counter + /// DIFFERENCE, clamped at zero, summed, and windowed on both bounds — added to the graph count so the + /// total reconciles with the sum of the card counts on both engines. Pinned on the text because + /// SUM(deadlocks) is the plausible one-liner that returns a lifetime counter times the sample + /// count, and nothing else in the build would notice. + /// + [Fact] + public void FleetTotalsSql_AddsThePostgresCounterDifferences_NeverTheRawColumn() + { + var sql = ViewerDataService.FleetTotalsSql; + Assert.Contains("FROM pg_database_stats", sql, StringComparison.Ordinal); + Assert.Contains("deadlocks - LAG(deadlocks) OVER (PARTITION BY server_id, database_name ORDER BY collection_time)", sql, StringComparison.Ordinal); + Assert.Contains("SUM(GREATEST(sampled.raw_delta, 0))", sql, StringComparison.Ordinal); + Assert.Contains("collection_time >= $1", sql, StringComparison.Ordinal); + Assert.Contains("collection_time <= $2", sql, StringComparison.Ordinal); + Assert.DoesNotContain("SUM(deadlocks)", sql, StringComparison.Ordinal); + + /* The two halves are ONE column: a second column would let a reader that indexes the deadlock + total positionally pick up only the graph count and silently drop the PostgreSQL half. */ + Assert.Equal(1, CountOccurrences(sql, "AS total_deadlocks")); + } + [Fact] public void FleetTotalsSql_WindowsEverySource_OnBothBounds() { @@ -658,12 +681,13 @@ private static async Task DeleteFleetRowsAsync(NpgsqlConnection connection, Syst } /// -/// The denominator beside the Overview's deadlock total (#3029). FleetTotalsSql's -/// SELECT COUNT(*) FROM v_deadlocks reads the SQL Server extended-event capture and nothing else — -/// a PostgreSQL target's deadlocks go to pg_deadlocks, which nothing joins in — so on a PostgreSQL -/// fleet that total is structurally zero forever. Zero is also exactly what a genuinely quiet SQL Server -/// fleet reports, so the reading that needs no action and the reading that does not cover the fleet had the -/// same character, and the tile could not tell an operator which one they were looking at. +/// The denominator beside the Overview's deadlock total (#3029). A server whose deadlock-source collector +/// is silent or denied contributes a structural zero to FleetTotalsSql's total, and zero is also +/// exactly what a genuinely quiet fleet reports, so the reading that needs no action and the reading that +/// does not cover the fleet had the same character, and the tile could not tell an operator which one they +/// were looking at. Until #3539 every PostgreSQL target was such a zero (the total read v_deadlocks, +/// the SQL Server capture, and nothing else); the total now adds the PostgreSQL server counter's differences, +/// and a PostgreSQL target is covered on its own collector's terms. /// /// Both directions, deliberately. The failure mode of a coverage figure is over-exclusion: a /// denominator that quietly shrinks reads as a smaller fleet, which is a new wrong number rather than a fix. @@ -679,6 +703,10 @@ public sealed class ViewerFleetDeadlockCoverageTests { private static readonly FleetTotals NoTotals = new(); + /// The ENGINE'S deadlock-source collector band (#3539): set on the card's + /// pg_database_stats slot for a PostgreSQL target and on its deadlocks slot otherwise, the + /// way the loader's two bands land — so a PostgreSQL card here is covered on the same terms the + /// production card is. private static ServerSummaryItem Card(int id, bool isPostgres = false, string? band = null) => new() { @@ -686,7 +714,8 @@ private static ServerSummaryItem Card(int id, bool isPostgres = false, string? b DisplayName = "target-" + id.ToString(CultureInfo.InvariantCulture), IsOnline = true, IsPostgres = isPostgres, - DeadlockCollectorBand = band, + DeadlockCollectorBand = isPostgres ? null : band, + PgDeadlockCollectorBand = isPostgres ? band : null, }; // ── The card's own reading of whether its deadlock count read anything ────────────────────────── @@ -718,17 +747,26 @@ public void ADeadlockReaderThatReadNothing_DoesNotCount_AndNamesItsCause(string? => Assert.Equal(expected, Card(1, band: band).DeadlockSource); /// - /// The issue's own case: a PostgreSQL target is never covered, and its collector's band cannot change - /// that. pg_deadlocks can be perfectly HEALTHY on all fifty targets and this total still counts - /// none of it — the rows are in a different table. That is why PostgreSQL is asked before any band. + /// #3539: a PostgreSQL target is covered on its OWN collector's terms — pg_database_stats, whose + /// counter the card differences — and lands on the PostgresTarget arm when that collector read, + /// the silent/denied arms when it did not. The card picks the PostgreSQL band because IsPostgres + /// says so; a deadlocks band on the same card is ignored, because that engine has no such + /// collector and a stray value there must not make it read. /// [Theory] - [InlineData(CollectorHealthClassifier.Healthy)] - [InlineData(CollectorHealthClassifier.NoPermissions)] - [InlineData(CollectorHealthClassifier.Stopped)] - [InlineData(null)] - public void APostgresTarget_IsNeverCovered_WhateverItsCollectorSays(string? band) - => Assert.Equal(FleetDeadlockSource.PostgresTarget, Card(1, isPostgres: true, band: band).DeadlockSource); + [InlineData(CollectorHealthClassifier.Healthy, FleetDeadlockSource.PostgresTarget)] + [InlineData(CollectorHealthClassifier.Failing, FleetDeadlockSource.PostgresTarget)] + [InlineData(CollectorHealthClassifier.NoPermissions, FleetDeadlockSource.CollectorDenied)] + [InlineData(CollectorHealthClassifier.Stopped, FleetDeadlockSource.CollectorSilent)] + [InlineData(null, FleetDeadlockSource.CollectorSilent)] + public void APostgresTarget_IsCoveredOnItsOwnCollectorsTerms(string? band, FleetDeadlockSource expected) + { + Assert.Equal(expected, Card(1, isPostgres: true, band: band).DeadlockSource); + + var strayDeadlocksBand = Card(1, isPostgres: true, band: band); + strayDeadlocksBand.DeadlockCollectorBand = CollectorHealthClassifier.Healthy; + Assert.Equal(expected, strayDeadlocksBand.DeadlockSource); + } /// /// A card that sets nothing reads as UNCOVERED, and that is the load-bearing default. @@ -765,29 +803,32 @@ public void Build_ReducesCoverageFromTheCards_WithEveryCauseAttributed() { Card(1, band: CollectorHealthClassifier.Healthy), Card(2, band: CollectorHealthClassifier.Failing), - Card(3, isPostgres: true), - Card(4, isPostgres: true), + Card(3, isPostgres: true, band: CollectorHealthClassifier.Healthy), + Card(4, isPostgres: true, band: CollectorHealthClassifier.Stale), Card(5, band: CollectorHealthClassifier.Stopped), Card(6, band: CollectorHealthClassifier.NoPermissions), Card(7, band: null), + /* #3539: a PostgreSQL target whose pg_database_stats collector left no band is SILENT. */ + Card(8, isPostgres: true, band: null), }, NoTotals); var coverage = rollup.DeadlockCoverage; - Assert.Equal(2, coverage.ServersRead); - Assert.Equal(7, coverage.ServersTotal); + Assert.Equal(4, coverage.ServersRead); + Assert.Equal(8, coverage.ServersTotal); Assert.Equal(2, coverage.PostgresServers); - Assert.Equal(2, coverage.ServersCollectorSilent); // STOPPED + the null band + Assert.Equal(3, coverage.ServersCollectorSilent); // STOPPED + the null band + the bandless PostgreSQL target Assert.Equal(1, coverage.ServersCollectorDenied); - /* With every registered server loaded, the four causes account for the fleet exactly once — an + /* With every registered server loaded, the causes account for the fleet exactly once — an unattributed server would mean coverage reporting a gap it cannot explain, which is the same - shape as a total reporting no denominator. */ + shape as a total reporting no denominator. Three terms since #3539: postgres_servers is a + SUBSET of servers_read, not a bucket beside it. */ Assert.Equal( coverage.ServersTotal, - coverage.ServersRead + coverage.PostgresServers - + coverage.ServersCollectorSilent + coverage.ServersCollectorDenied); + coverage.ServersRead + coverage.ServersCollectorSilent + coverage.ServersCollectorDenied); + Assert.True(coverage.PostgresServers <= coverage.ServersRead); /* And it agrees with the count the panel already shows beside it. */ Assert.Equal(rollup.TotalServers, coverage.ServersTotal); @@ -810,7 +851,7 @@ public void Build_CoverageDenominator_IsTheRegisteredFleet_AndTheShortfallIsTheU var loaded = new[] { Card(1, band: CollectorHealthClassifier.Healthy), - Card(2, isPostgres: true), + Card(2, isPostgres: true, band: CollectorHealthClassifier.Healthy), }; var rollup = FleetRollup.Build(loaded, NoTotals, totalServerCount: 5); @@ -818,7 +859,8 @@ public void Build_CoverageDenominator_IsTheRegisteredFleet_AndTheShortfallIsTheU Assert.Equal(5, coverage.ServersTotal); Assert.NotEqual(loaded.Length, coverage.ServersTotal); - Assert.Equal(1, coverage.ServersRead); + /* Both loaded cards are covered (#3539); the PostgreSQL one is also named in its sub-count. */ + Assert.Equal(2, coverage.ServersRead); Assert.Equal(1, coverage.PostgresServers); Assert.Equal(3, rollup.UnknownCount); @@ -828,7 +870,7 @@ public void Build_CoverageDenominator_IsTheRegisteredFleet_AndTheShortfallIsTheU Assert.Equal( coverage.ServersTotal, - coverage.ServersRead + coverage.PostgresServers + coverage.ServersCollectorSilent + coverage.ServersRead + coverage.ServersCollectorSilent + coverage.ServersCollectorDenied + rollup.UnknownCount); } @@ -866,13 +908,16 @@ public void Build_WithNoFleetAtAll_HasNothingToQualify() } /// - /// Only counts as read, and the enum cannot grow unnoticed. + /// Only the two covered arms count as read, and the enum cannot grow unnoticed. /// Reflected off the type rather than listed by hand: a pin that enumerates the kinds by hand cannot see /// the set grow, and a source kind added later that landed in the read bucket by default would restore /// exactly the defect this figure exists to fix. If this count moves, whoever moved it decides here. + /// Since #3539 the PostgreSQL arm is one of the covered two — the reducer counts it through the shared + /// , so this file and the service's roll-up cannot + /// disagree about which arms are read. /// [Fact] - public void EverySourceKind_IsProducible_AndOnlyReadCountsAsRead() + public void EverySourceKind_IsProducible_AndOnlyTheCoveredArmsCountAsRead() { var kinds = Enum.GetValues(); Assert.Equal(4, kinds.Length); @@ -882,7 +927,7 @@ public void EverySourceKind_IsProducible_AndOnlyReadCountsAsRead() var producible = new[] { (IsPostgres: false, Band: (string?)CollectorHealthClassifier.Healthy), - (IsPostgres: true, Band: (string?)null), + (IsPostgres: true, Band: (string?)CollectorHealthClassifier.Healthy), (IsPostgres: false, Band: (string?)CollectorHealthClassifier.Stopped), (IsPostgres: false, Band: (string?)CollectorHealthClassifier.NoPermissions), } @@ -892,18 +937,21 @@ public void EverySourceKind_IsProducible_AndOnlyReadCountsAsRead() Assert.Equal(kinds.Length, producible.Count); Assert.All(kinds, k => Assert.Contains(k, producible)); - /* And the reducer counts exactly one of them as read: one card of each kind, ServersRead == 1. */ + /* And the reducer counts exactly the covered two as read: one card of each kind, ServersRead == 2, + with the PostgreSQL one also in its own sub-count. */ var coverage = FleetRollup.ReduceDeadlockCoverage( new[] { Card(1, band: CollectorHealthClassifier.Healthy), - Card(2, isPostgres: true), + Card(2, isPostgres: true, band: CollectorHealthClassifier.Healthy), Card(3, band: CollectorHealthClassifier.Stopped), Card(4, band: CollectorHealthClassifier.NoPermissions), }, registeredTotal: 4); - Assert.Equal(1, coverage.ServersRead); + Assert.Equal(2, coverage.ServersRead); + Assert.Equal(1, coverage.PostgresServers); + Assert.Equal(Enum.GetValues().Count(FleetDeadlockCoverage.IsCovered), coverage.ServersRead); } // ── What the panel actually renders ──────────────────────────────────────────────────────────── @@ -947,20 +995,39 @@ public void TheCoverageLine_TracksCoverage_NotTheDeadlockCount() Assert.Equal("Deadlock coverage: read all 2 servers", rollup.DeadlockCoverageText); } - /// The issue's measured case: a PostgreSQL-only fleet reporting zero, now saying so. + /// The issue's measured case, reversed by #3539: a PostgreSQL-only fleet whose + /// pg_database_stats collectors run is FULLY covered, and the tooltip names the instrument + /// rather than saying the total cannot count it. A PostgreSQL-only fleet whose collectors are all silent + /// still reads "0 of 3" — through the silent cause, which is the one that names an action. [Fact] - public void APostgresOnlyFleet_ReportsZeroCoverage_AndNamesWhereThoseDeadlocksAre() + public void APostgresOnlyFleet_IsCovered_AndNamesTheInstrument() { var rollup = FleetRollup.Build( + new[] + { + Card(1, isPostgres: true, band: CollectorHealthClassifier.Healthy), + Card(2, isPostgres: true, band: CollectorHealthClassifier.Healthy), + Card(3, isPostgres: true, band: CollectorHealthClassifier.Warning), + }, + new FleetTotals { TotalDeadlocks = 4 }); + + Assert.Equal(4, rollup.TotalDeadlocks); + Assert.Equal(3, rollup.DeadlockCoverage.ServersRead); + Assert.Equal(3, rollup.DeadlockCoverage.PostgresServers); + Assert.False(rollup.DeadlockCoverageIsPartial); + Assert.Equal("Deadlock coverage: read all 3 servers", rollup.DeadlockCoverageText); + Assert.Contains("3 servers: " + FleetRollup.DeadlockPostgresCause, rollup.DeadlockCoverageTooltip, StringComparison.Ordinal); + Assert.DoesNotContain("cannot count", rollup.DeadlockCoverageTooltip, StringComparison.Ordinal); + + var silent = FleetRollup.Build( new[] { Card(1, isPostgres: true), Card(2, isPostgres: true), Card(3, isPostgres: true) }, new FleetTotals { TotalDeadlocks = 0 }); - Assert.Equal(0, rollup.TotalDeadlocks); - Assert.Equal(0, rollup.DeadlockCoverage.ServersRead); - Assert.Equal(3, rollup.DeadlockCoverage.PostgresServers); - Assert.True(rollup.DeadlockCoverageIsPartial); - Assert.Equal("Deadlock coverage: read 0 of 3 servers", rollup.DeadlockCoverageText); - Assert.Contains(FleetRollup.DeadlockPostgresCause, rollup.DeadlockCoverageTooltip, StringComparison.Ordinal); + Assert.Equal(0, silent.DeadlockCoverage.ServersRead); + Assert.Equal(0, silent.DeadlockCoverage.PostgresServers); + Assert.Equal(3, silent.DeadlockCoverage.ServersCollectorSilent); + Assert.Equal("Deadlock coverage: read 0 of 3 servers", silent.DeadlockCoverageText); + Assert.Contains("3 servers: " + FleetRollup.DeadlockCollectorSilentCause, silent.DeadlockCoverageTooltip, StringComparison.Ordinal); } /// A one-server fleet says "server", not "servers" — both ways round. @@ -973,7 +1040,7 @@ public void TheCoverageLine_AgreesWithItselfOnNumber() Assert.Equal( "Deadlock coverage: read 0 of 1 server", - FleetRollup.Build(new[] { Card(1, isPostgres: true) }, NoTotals).DeadlockCoverageText); + FleetRollup.Build(new[] { Card(1, band: CollectorHealthClassifier.Stopped) }, NoTotals).DeadlockCoverageText); } /// @@ -1002,8 +1069,11 @@ public void TheTooltip_NamesEachWindow_AndClaimsNeitherForTheOther() /* And the disclaimer that keeps the first from being read as the second. */ Assert.Contains("makes no claim about what was read in the last hour", tooltip, StringComparison.Ordinal); - /* What the total is assembled from — the fact that makes a PostgreSQL zero structural. */ - Assert.Contains("SQL Server extended-event capture and nothing else", tooltip, StringComparison.Ordinal); + /* What the total is assembled from - both engines' instruments since #3539, and the collector + state as the thing that makes a zero structural. */ + Assert.Contains("SQL Server extended-event capture", tooltip, StringComparison.Ordinal); + Assert.Contains("deadlock counter differenced over the window", tooltip, StringComparison.Ordinal); + Assert.DoesNotContain("and nothing else", tooltip, StringComparison.Ordinal); } /// @@ -1013,7 +1083,7 @@ public void TheTooltip_NamesEachWindow_AndClaimsNeitherForTheOther() public void TheTooltip_CarriesOnlyTheCausesThatApply() { var tooltip = FleetRollup.Build( - new[] { Card(1, isPostgres: true), Card(2, band: CollectorHealthClassifier.Healthy) }, + new[] { Card(1, isPostgres: true, band: CollectorHealthClassifier.Healthy), Card(2, band: CollectorHealthClassifier.Healthy) }, NoTotals).DeadlockCoverageTooltip; Assert.Contains("1 server: " + FleetRollup.DeadlockPostgresCause, tooltip, StringComparison.Ordinal); @@ -1034,7 +1104,7 @@ public void TheTooltip_AgreesWithItselfOnNumber_ForEveryCause() var singular = FleetRollup.Build( new[] { - Card(1, isPostgres: true), + Card(1, isPostgres: true, band: CollectorHealthClassifier.Healthy), Card(2, band: CollectorHealthClassifier.Stopped), Card(3, band: CollectorHealthClassifier.NoPermissions), }, @@ -1049,7 +1119,7 @@ public void TheTooltip_AgreesWithItselfOnNumber_ForEveryCause() var plural = FleetRollup.Build( new[] { - Card(1, isPostgres: true), Card(2, isPostgres: true), + Card(1, isPostgres: true, band: CollectorHealthClassifier.Healthy), Card(2, isPostgres: true, band: CollectorHealthClassifier.Healthy), Card(3, band: CollectorHealthClassifier.Stopped), Card(4, band: CollectorHealthClassifier.NeverRun), Card(5, band: CollectorHealthClassifier.NoPermissions), Card(6, band: CollectorHealthClassifier.NoPermissions), }, @@ -1135,9 +1205,10 @@ public void TheSummaryRead_RetainsTheDeadlockCollectorsBand_MatchedFromItsOwnNam /* Anchored on the DECLARATION, which is itself the shape being pinned: the helper hands the band back beside the tallies rather than returning a pair the caller has to re-read the store for. - The tuple grew a Total for #3539 A8d (the share's denominator) — a fourth tally, same shape. */ + The tuple grew a Total for #3539 A8d (the share's denominator) — a fourth tally, same shape — + and the pg_database_stats collector's band for #3539's PostgreSQL deadlock arm, a fifth. */ var start = source.IndexOf( - "private async Task<(int Healthy, int Failing, int Total, string? DeadlockBand)> GetCollectorHealthCountsAsync", + "private async Task<(int Healthy, int Failing, int Total, string? DeadlockBand, string? PgDeadlockBand)> GetCollectorHealthCountsAsync", StringComparison.Ordinal); Assert.True(start > 0, "the collector-health helper does not hand back the deadlock band"); var end = source.IndexOf("private static int? MinutesAgo", start, StringComparison.Ordinal); @@ -1146,11 +1217,15 @@ back beside the tallies rather than returning a pair the caller has to re-read t var body = source[start..end]; Assert.Contains("DeadlocksCollector.Instance.Name", body, StringComparison.Ordinal); + /* #3539: the PostgreSQL deadlock-source collector's band, matched the same way for the same reason. */ + Assert.Contains("PgDatabaseStatsCollector.Instance.Name", body, StringComparison.Ordinal); /* The literal is the shape a careless match takes, and it is what the collector name is TODAY — so this is a real trap rather than a hypothetical one. */ Assert.DoesNotContain("\"deadlocks\"", body, StringComparison.Ordinal); + Assert.DoesNotContain("\"pg_database_stats\"", body, StringComparison.Ordinal); Assert.Equal("deadlocks", DeadlocksCollector.Instance.Name); + Assert.Equal("pg_database_stats", PgDatabaseStatsCollector.Instance.Name); } private static int CountOccurrences(string haystack, string needle) diff --git a/Darling/Darling.Tests/ViewerW2aTests.cs b/Darling/Darling.Tests/ViewerW2aTests.cs index 63cbccf91..75ea36574 100644 --- a/Darling/Darling.Tests/ViewerW2aTests.cs +++ b/Darling/Darling.Tests/ViewerW2aTests.cs @@ -111,6 +111,25 @@ public void SummaryDeadlockSql_CountsOverTheWindow_AndNewestEver() Assert.Contains("MAX(deadlock_time)", sql, StringComparison.Ordinal); } + /// + /// #3539: the PostgreSQL arm of the same card read — the server's own pg_stat_database.deadlocks + /// counter differenced per database over the window, clamped at zero across a reset, summed, with the + /// sample that first showed the newest step as "last". Never SUM(deadlocks): the column is a + /// lifetime counter repeated in every sample. Per-server, so partitioned by database only. + /// + [Fact] + public void SummaryPgDeadlockSql_DifferencesTheCounterPerDatabase_OverTheWindow() + { + var sql = ViewerDataService.ServerSummaryPgDeadlockSql; + Assert.Contains("FROM pg_database_stats", sql, StringComparison.Ordinal); + Assert.Contains("WHERE server_id = $1", sql, StringComparison.Ordinal); + Assert.Contains("collection_time >= $2", sql, StringComparison.Ordinal); + Assert.Contains("deadlocks - LAG(deadlocks) OVER (PARTITION BY database_name ORDER BY collection_time)", sql, StringComparison.Ordinal); + Assert.Contains("SUM(GREATEST(sampled.raw_delta, 0))", sql, StringComparison.Ordinal); + Assert.Contains("MAX(sampled.collection_time) FILTER (WHERE sampled.raw_delta > 0)", sql, StringComparison.Ordinal); + Assert.DoesNotContain("SUM(deadlocks)", sql, StringComparison.Ordinal); + } + [Fact] public void SummaryLastCollectionSql_TakesTheNewestCollectionTime() { @@ -131,6 +150,7 @@ public void SummaryReads_ArePgDialect_PositionalParams_NoBareNow_NoNLiterals() ViewerDataService.ServerSummaryThreadsSql, ViewerDataService.ServerSummaryBlockingSql, ViewerDataService.ServerSummaryDeadlockSql, + ViewerDataService.ServerSummaryPgDeadlockSql, ViewerDataService.ServerSummaryLastCollectionSql, }) { diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingConfig.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingConfig.cs index 464248687..2ae956b4b 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingConfig.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingConfig.cs @@ -699,10 +699,12 @@ public sealed class AlertsConfig /// retention and deadlock-band knobs above already follow. DarlingAlertSettings clamps it on /// read. /// - /// Separate from deliberately — see the V122 rung and - /// the default constant for why the SQL Server figure's justification does not travel to an engine with - /// no deadlock band. The enabled switch IS shared: governs both - /// engines. + /// Separate from deliberately — see the default + /// constant: the two engines count with different instruments (captured graphs vs deadlocks parsed + /// from the server log), and an operator tuning one should not silently move the other. The V122 rung's + /// second reason — that a PostgreSQL server had no deadlock band to agree with — ended with #3539, which + /// bands the PostgreSQL card's own counter difference through the shared tiers. The enabled + /// switch IS shared: governs both engines. [JsonPropertyName("pgDeadlockCountThreshold")] public int PgDeadlockCountThreshold { get; set; } = PostgresAlertEvaluator.DeadlockCountThresholdDefault; diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs index 735139e00..5ae9d5148 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs @@ -4497,6 +4497,17 @@ await NotifyPgResolutionAsync(key, snapshot.ServerName, metricName, "Blocking Cl /// No query-text preview: pg_session_states deliberately stores none (see the collector's /// class remarks), so the message identifies the session by pid/database/command tag instead of the /// statement text SQL Server's equivalent shows. + /// + /// The noise opt-outs ride the SAME switches SQL Server's read takes (#3539): the shared + /// longRunningQueryExcludeBackups drops the dump/restore utilities' sessions, and the shared + /// excludedDatabases list is applied after the read exactly as DarlingAlertReadAdapter + /// applies it. The unconditional ones — non-client backends (autovacuum, walsender), the + /// VACUUM/ANALYZE/REINDEX/CLUSTER statements, idle-in-transaction — live in the read's SQL; see + /// for each one's SQL + /// Server sibling and for why the three remaining SQL Server switches have no honest reading here. What + /// is still reported carries its command_tag on the incident line, so a CREATE that is an + /// index build reads as what it is rather than being dropped on a guess — the annotate-never-suppress + /// posture SQL Server's Agent-job name (#3497) takes on its card. /// private async Task EvaluatePgLongRunningQueryAsync( ServerRuntime runtime, AlertServerSnapshot snapshot, DarlingConfig config, CancellationToken cancellationToken) @@ -4524,7 +4535,10 @@ private async Task EvaluatePgLongRunningQueryAsync( var rows = await DarlingPgSessionStatesReader.GetCurrentLongRunningSessionsAsync( _postgres, runtime.ServerId, thresholdMs: thresholdMinutes * 60_000L, now, - PgLongRunningQueryRecencyMinutes, limit: alertSettings.LongRunningQueryMaxResults, cancellationToken); + PgLongRunningQueryRecencyMinutes, limit: alertSettings.LongRunningQueryMaxResults, + excludeBackups: alertSettings.LongRunningQueryExcludeBackups, + excludedDatabases: alertSettings.ExcludedDatabases, + cancellationToken); readClock.Restart(); var cooldown = TimeSpan.FromMinutes(Math.Max(1, _alertCooldownMinutes)); diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingFleetReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingFleetReader.cs index f51daa77c..3cbb51755 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingFleetReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingFleetReader.cs @@ -264,16 +264,84 @@ GROUP BY server_id /// This view is the SQL Server extended-event capture and nothing else (#3017). /// v_deadlocks is SELECT * FROM deadlocks, and deadlocks is written by exactly one /// collector — DeadlocksCollector, whose TargetTable it is. A PostgreSQL target's deadlocks - /// go to pg_deadlocks instead, there is no v_pg_deadlocks, and nothing joins the two, so - /// this count is structurally zero for a PostgreSQL server no matter how many deadlocks its clusters - /// have. Zero is also what a genuinely quiet SQL Server reports, which is why the total ships with - /// beside it: the reading that needs no action and the reading that - /// does not cover the fleet are otherwise the same character. + /// never reach it, so this count is structurally zero for a PostgreSQL server no matter how many + /// deadlocks its clusters have; is that engine's read (#3539), and + /// takes one or the other by engine. Zero is also what a genuinely quiet SQL + /// Server reports, which is why the total ships with beside it: the + /// reading that needs no action and the reading whose collector is silent are otherwise the same + /// character. public const string FleetDeadlockSql = @" SELECT server_id, COUNT(*) AS cnt, MAX(deadlock_time) AS last_seen FROM v_deadlocks WHERE deadlock_time >= $1 AND deadlock_time <= $2 +GROUP BY server_id"; + + /// Deadlocks in the window per PostgreSQL server (#3539) — the same two columns as + /// , from the engine's own counter rather than a captured graph. $1 window + /// start, $2 window end (both naive UTC). + /// + /// A counter DIFFERENCE, never a sum of the column. pg_database_stats.deadlocks is + /// pg_stat_database.deadlocks stored raw: a lifetime counter per database, sampled every minute, + /// so the same 122 sits in every one of a day's 1,440 rows and SUM(deadlocks) over a window is a + /// number with no meaning (measured on one store: 8,006,912 across 50 servers with zero new deadlocks in + /// the window). What happened IN the window is each consecutive pair's difference, taken per + /// (server_id, database_name) series — the same LAG shape DarlingPgDatabaseReader.PgDatabaseSql + /// uses for get_pg_database_stats, and the two must agree on a server or the fleet card would + /// contradict the tool it sends a reader to. + /// + /// Clamped at zero across a reset. pg_stat_reset() or a crash restart rewinds the + /// counter, and a plain difference goes negative there; GREATEST(…, 0) drops that interval + /// rather than subtracting a lifetime from the window. The interval's real deadlocks (those between the + /// reset and the next sample) survive as the next difference. The tool reports how many intervals + /// clamped; this card does not — a band is not the place for a reset count, and + /// get_pg_database_stats is one call away. + /// + /// Summed across databases because the card is per SERVER and the SQL Server count it + /// sits beside is too: a deadlock graph names the databases involved, but v_deadlocks is counted + /// per server_id, and the band's tiers were measured per server-hour (#3368). The NULL-named + /// shared-relation row PostgreSQL emits is its own series under PARTITION BY (grouping + /// semantics), so it differences correctly and is summed in. + /// + /// last_seen is the sample that first showed the increase — the deadlock happened + /// somewhere in the preceding minute, which is the resolution a per-minute counter has. Bounded to the + /// window like the count, where the SQL Server "last seen" is the newest graph in the window too; the + /// unbounded "ever" the viewer's card shows for SQL Server has no cheap PostgreSQL twin (it would be a + /// scan of the whole counter series for its last step), and "last within the window" is what the + /// count-carrying chip actually renders. + /// + /// intervals is how many differences were taken — the count of consecutive-sample + /// pairs across every series in the window — and it is what makes the count a MEASUREMENT. A difference + /// needs two samples; a server with one row in the window, or none, has had no difference taken, and + /// its cnt of zero is the arithmetic of an empty set, not an observation that nothing deadlocked. + /// This is where the PostgreSQL arm parts from the SQL Server one on purpose: COUNT(*) over an + /// event table with no events IS an observation (the collector looked and found none), where + /// LAG over no samples is undefined. bands only when this is positive, + /// which is also what keeps #3539 A6 intact — an online PostgreSQL target nothing has collected from + /// measures nothing and reads Warning, not "Healthy — 1 of 6 measured". + /// + /// Bounded on collection_time, the hypertable's partitioning column, so chunk exclusion + /// keeps the read to the window's chunk(s): on the one-minute cadence the default hour is ~60 rows per + /// database per server, and the fleet's whole hour is tens of thousands of rows behind the + /// (server_id, collection_time) index — the same order as 's + /// two scans. + public const string FleetPgDeadlockSql = @" +WITH sampled AS +( + SELECT + server_id, + collection_time, + deadlocks - LAG(deadlocks) OVER (PARTITION BY server_id, database_name ORDER BY collection_time) AS raw_delta + FROM pg_database_stats + WHERE collection_time >= $1 + AND collection_time <= $2 +) +SELECT + server_id, + CAST(coalesce(SUM(GREATEST(raw_delta, 0)), 0) AS bigint) AS cnt, + MAX(collection_time) FILTER (WHERE raw_delta > 0) AS last_seen, + CAST(count(raw_delta) AS bigint) AS intervals +FROM sampled GROUP BY server_id"; /// The deadlock health band's two tiers from the singleton settings row (#3368, V120). @@ -399,6 +467,7 @@ public static async Task GetFleetOverviewAsync( var threads = await ReadThreadsAsync(postgres, cancellationToken); var blocking = await ReadBlockingAsync(postgres, windowStartUtc, windowEndUtc, cancellationToken); var deadlocks = await ReadDeadlocksAsync(postgres, windowStartUtc, windowEndUtc, cancellationToken); + var pgDeadlocks = await ReadPgDeadlocksAsync(postgres, windowStartUtc, windowEndUtc, cancellationToken); var lastCollection = await ReadLastCollectionAsync(postgres, now, cancellationToken); var failingCollectors = await ReadFailingCollectorCountsAsync(postgres, now, cancellationToken); var tags = await ReadTagsAsync(postgres, cancellationToken); @@ -422,6 +491,10 @@ and the band counts and the worst-first ranking would be derived from it. */ threads.TryGetValue(server.ServerId, out var t); blocking.TryGetValue(server.ServerId, out var b); deadlocks.TryGetValue(server.ServerId, out var deadlock); + /* A miss leaves default(PgDeadlockRow) - zero count, no last-seen, ZERO intervals - which for + a PostgreSQL target is "no difference was taken" (unmeasured, not quiet), and for a SQL + Server is a row BuildCard never looks at. */ + pgDeadlocks.TryGetValue(server.ServerId, out var pgDeadlock); /* Not `lastCollection.TryGetValue(..., out var lastColl)` — that leaves lastColl as default(DateTime) (0001-01-01) on a miss, and default(DateTime) is NOT null, so it does not hit ClassifyFreshness's NeverCollected branch: it falls through to the @@ -436,7 +509,7 @@ ancient timestamp. */ tags.TryGetValue(server.ServerId, out var serverTags); cards.Add(BuildCard( - server, c, pg, m, mp, t, b, deadlock, lastColl, collectors, serverTags, now, + server, c, pg, m, mp, t, b, deadlock, pgDeadlock, lastColl, collectors, serverTags, now, windowEndUtc - windowStartUtc, deadlockTiers)); } @@ -453,6 +526,13 @@ ancient timestamp. */ /// metric passes default for its row, which is why those types never have to be NAMED in a test /// — they are internal only because CS0051 requires every parameter type of an internal method to be /// at least as accessible as it. + /// The SQL Server extended-event count for the window (); + /// structurally zero for a PostgreSQL target and ignored for one. + /// The PostgreSQL counter-difference count for the window + /// (, #3539), with how many differences it was summed from; + /// structurally empty for a SQL Server and ignored for one. Two parameters rather than one pre-merged + /// row so this method — the step that decides what a card CLAIMS — is where the engine picks, and a + /// test can hand a card BOTH rows and assert which one it believed. internal static FleetServerCard BuildCard( FleetServerRow server, CpuRow cpu, @@ -462,6 +542,7 @@ internal static FleetServerCard BuildCard( ThreadsRow threads, BlockingRow blocking, DeadlockRow deadlock, + PgDeadlockRow pgDeadlock, DateTime? lastCollection, CollectorCounts collectors, List? tags, @@ -469,7 +550,6 @@ internal static FleetServerCard BuildCard( TimeSpan deadlockWindow, DeadlockRateThresholds deadlockTiers) { - var deadlockCount = deadlock.Count; /* Lite's XE-preferred / DMV-fallback, per server: XE when it has any row this window, else the DMV snapshot — both count and worst-wait come from whichever source wins. */ @@ -484,6 +564,20 @@ internal static FleetServerCard BuildCard( rather than beside ClassifyPlatform because the CPU source classification needs it. */ var (isPostgres, isAurora) = ClassifyEngineKind(server.EngineKind); + /* The engine's OWN deadlock reading (#3539): the extended-event graph count for a SQL Server, the + pg_stat_database counter difference for a PostgreSQL target. Chosen by engine rather than summed, + because the other engine's row is a structural zero and a sum would hide which instrument + answered; the card says which through deadlock_source. The collector band travels with the + count for the same reason - coverage asks about the collector that produced THIS number. */ + var deadlockCount = isPostgres ? pgDeadlock.Count : deadlock.Count; + var deadlockLastSeen = isPostgres ? pgDeadlock.LastSeen : deadlock.LastSeen; + var deadlockCollectorBand = isPostgres ? collectors.PgDeadlockBand : collectors.DeadlockBand; + /* Whether the count is a MEASUREMENT. On SQL Server it always is: COUNT(*) over the graph table is + an observation even at zero (#3272's engine-not-collector rule). On PostgreSQL the count is a + counter DIFFERENCE, and a difference of fewer than two samples is not zero deadlocks, it is no + reading - see FleetPgDeadlockSql's intervals paragraph. */ + var deadlockMeasured = !isPostgres || pgDeadlock.Intervals > 0; + /* One expression for "total non-idle host CPU, from whichever collector has it" (#3267), shared with the viewer's card so the two cannot drift on the fallback. cpuPercent stays the SQL-Server-process share and is NOT filled from the PostgreSQL arm: Performance Insights publishes only the host @@ -501,15 +595,24 @@ than leaving a consumer to infer it from which fields are null. */ var hasMemoryPressure = pressure.WaiterCount > 0 || pressure.TimeoutCount > 0 || pressure.ForcedCount > 0; var maxBlockedSeconds = maxBlockingWaitMs / 1000.0; - /* The three DMV-sourced readings, with "not measured" expressed as null for an engine that has no + /* The two DMV-sourced readings, with "not measured" expressed as null for an engine that has no row in the views behind them (#3272). The reads above produced zeros for such a target, and a zero here argued Healthy — a green dot for a metric nothing measured. The published COUNTS are - left exactly as they are: #3017's deadlock_source and the fleet coverage block explain a total - built out of those zeros, and nulling them would make the total's own denominator unreadable. - It is the BAND that stops claiming health. */ + left exactly as they are: the fleet coverage block explains a total built out of those zeros, and + nulling them would make the total's own denominator unreadable. It is the BAND that stops + claiming health. + + Deadlocks left this set in #3539: both engines now have a source behind the reading (the graph + capture or the server counter). The SQL Server arm keeps the engine-not-collector rule + ServerMetricSources states - a silent deadlocks collector reads zero and bands Healthy, with + deadlock_source and the coverage block disclosing the gap. The PostgreSQL arm is gated on the + instrument instead: its count is a difference, and a window with fewer than two samples per + series has had no difference taken, so the band reads Unknown there rather than a Healthy + computed from an empty set. That is also what keeps #3539 A6's card - an online PostgreSQL + target nothing has collected from - measuring nothing. */ var memoryPressureForBand = ServerMetricSources.DmvSourced(hasMemoryPressure, isPostgres); var blockingForBand = ServerMetricSources.DmvSourced(blockingCount, isPostgres); - var deadlocksForBand = ServerMetricSources.DmvSourced(deadlockCount, isPostgres); + int? deadlocksForBand = deadlockMeasured ? deadlockCount : null; var metrics = new ServerHealthMetrics { @@ -609,13 +712,12 @@ deliberately not surfaced. */ BlockingWindow = deadlockWindow, BlockingSeverity = ServerHealthClassifier.BlockingSeverity(blockingForBand, maxBlockedSeconds, deadlockWindow), DeadlockCount = deadlockCount, - DeadlockLastSeen = deadlock.LastSeen, - /* deadlocksForBand, not the raw count, for the same reason DeadlockSeverity below takes it: on a - PostgreSQL target there is no deadlock reading at all and the raw count is a STRUCTURAL zero, - so a rate derived from it publishes 0.0/hr - a measurement nobody took - while the severity on - the same card correctly reads Unknown. The card chip and the viewer detail line both render - this field on nothing but non-null, so the disclosure has to live here. #3017 fixed exactly - this confusion for the count and the band; the rate must not reintroduce it. */ + DeadlockLastSeen = deadlockLastSeen, + DeadlockMeasured = deadlockMeasured, + /* Through deadlocksForBand rather than the raw int so the rate and the severity below are + derived from ONE value: the chip and the viewer detail line render this on non-null alone, + and a rate published for a count the band did not see is the #3017 confusion one field + over. Null exactly when the PostgreSQL arm took no difference (#3539). */ DeadlockRatePerHour = deadlocksForBand.HasValue ? ServerHealthClassifier.DeadlockRatePerHour(deadlocksForBand.Value, deadlockWindow) : null, @@ -623,7 +725,7 @@ the same card correctly reads Unknown. The card chip and the viewer detail line DeadlockRateThresholds = deadlockTiers, DeadlockSeverity = ServerHealthClassifier.DeadlockSeverity( deadlocksForBand, deadlockWindow, deadlockTiers), - DeadlockCollectorBand = collectors.DeadlockBand, + DeadlockCollectorBand = deadlockCollectorBand, TotalThreads = threads.TotalThreads, CurrentWorkers = threads.CurrentWorkers, AvailableThreads = availableThreads, @@ -679,13 +781,20 @@ public static FleetOverviewResult BuildRollup( /* #3017's denominator, reduced from the CARDS for the same reason the totals above are: the coverage figure and the total it qualifies then reconcile by construction rather than by two - queries agreeing. Only Read is counted as read — every other arm, INCLUDING an enum value a - later build adds and this switch has never heard of, lands in the silent bucket. A new source - kind that inflated the read count would restore exactly the defect this exists to fix, where - one that lands in an uncovered bucket merely attributes a real gap imprecisely. */ + queries agreeing. Only the arms FleetDeadlockCoverage.IsCovered names count as read — every + other arm, INCLUDING an enum value a later build adds and this switch has never heard of, + lands in the silent bucket. A new source kind that inflated the read count would restore + exactly the defect this exists to fix, where one that lands in an uncovered bucket merely + attributes a real gap imprecisely. The PostgreSQL arm is covered AND tallied on its own + (#3539): the sub-count names the instrument, the read count names the coverage. */ + if (FleetDeadlockCoverage.IsCovered(card.DeadlockSource)) + { + deadlockSourcesRead++; + } + switch (card.DeadlockSource) { - case FleetDeadlockSource.Read: deadlockSourcesRead++; break; + case FleetDeadlockSource.Read: break; case FleetDeadlockSource.PostgresTarget: deadlockPostgresTargets++; break; case FleetDeadlockSource.CollectorDenied: deadlockCollectorsDenied++; break; default: deadlockCollectorsSilent++; break; @@ -1108,6 +1217,33 @@ private static async Task> ReadDeadlocksAsync( return map; } + /// The PostgreSQL twin of (#3539) — same carrier, same window, + /// the engine's own counter behind it. Read for the whole fleet in one pass like every other per-metric + /// read here; a SQL Server has no row in pg_database_stats and simply does not appear. + private static async Task> ReadPgDeadlocksAsync( + NpgsqlDataSource postgres, DateTime startUtc, DateTime endUtc, CancellationToken cancellationToken) + { + var map = new Dictionary(); + await using var command = postgres.CreateCommand(FleetPgDeadlockSql); + command.CommandTimeout = McpCommandDeadlines.ReadSeconds; + AddTimestamp(command, startUtc); + AddTimestamp(command, endUtc); + await using var reader = await command.ExecuteReaderAsync(cancellationToken); + while (await reader.ReadAsync(cancellationToken)) + { + /* The SUM comes back as bigint; the card's count is an int like the SQL Server arm's. A window + whose clamped deadlock differences overflow int is not a reading this card can render either + way, so saturate rather than wrap - a wrapped count could band a catastrophe Healthy. */ + var count = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); + map[reader.GetInt32(0)] = new PgDeadlockRow( + (int)Math.Min(count, int.MaxValue), + reader.IsDBNull(2) ? null : reader.GetDateTime(2), + reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3))); + } + + return map; + } + private static async Task> ReadLastCollectionAsync(NpgsqlDataSource postgres, DateTime now, CancellationToken cancellationToken) { var map = new Dictionary(); @@ -1160,14 +1296,18 @@ private static async Task> ReadFailingCollector existing.Failing + (status == "FAILING" ? 1 : 0), /* #3539 A8d: every banded row, whatever its band — the share's denominator. */ existing.Total + 1, - /* #3017: the ONE collector whose band the deadlock total's coverage turns on, kept - alongside the Healthy/Failing tallies because it comes out of the same aggregate — no - extra round trip, which is what keeps this reader's fan-out bounded. Named from the + /* #3017: the ONE collector per engine whose band the deadlock total's coverage turns on, + kept alongside the Healthy/Failing tallies because it comes out of the same aggregate — + no extra round trip, which is what keeps this reader's fan-out bounded. Named from the collector rather than as a literal so a rename cannot leave this silently matching - nothing and reporting every server uncovered. */ + nothing and reporting every server uncovered. Both engines' bands are kept on every + server because this aggregate does not know the engine; BuildCard picks. */ string.Equals(health.CollectorName, DeadlocksCollector.Instance.Name, StringComparison.Ordinal) ? status - : existing.DeadlockBand); + : existing.DeadlockBand, + string.Equals(health.CollectorName, PgDatabaseStatsCollector.Instance.Name, StringComparison.Ordinal) + ? status + : existing.PgDeadlockBand); } return counts; @@ -1196,16 +1336,25 @@ internal readonly record struct PgCpuRow( internal readonly record struct ThreadsRow(int? TotalThreads, int? CurrentWorkers, int RunnableTasks, long WorkQueue); internal readonly record struct BlockingRow(int XeCount, long XeMaxWait, int DmvCount, long DmvMaxWait); internal readonly record struct DeadlockRow(int Count, DateTime? LastSeen); + /// The PostgreSQL deadlock reading (#3539): the summed counter differences, the sample that + /// showed the newest step, and — how many differences the sum was taken + /// over. Its default is zero intervals, which reads as unmeasured: a + /// PostgreSQL target that fell out of the read (no rows in the window) must band Unknown, and a + /// struct whose default meant "measured zero" would make that the quiet outcome of a miss. + internal readonly record struct PgDeadlockRow(int Count, DateTime? LastSeen, long Intervals); /// The deadlocks collector's own 7-day band for this server, or null /// when that collector left no row in the health window at all (#3017). Null and /// mean the same thing to a reader and take the same /// action, but they arrive differently: null is the absent GROUP, NEVER_RUN would be a present group with /// no runs in it. + /// The pg_database_stats collector's band, same terms (#3539) — the + /// deadlock-source collector on a PostgreSQL target, where is always + /// null because that engine has no deadlocks collector. /// Every collector banded for this server in the health window, on any band (#3539 /// A8d) — the denominator grades the failing count /// against. Healthy + Failing is NOT it: STALE, WARNING, STOPPED, NO_PERMISSIONS and EXTENSION_MISSING rows /// are all banded collectors that are neither. - internal readonly record struct CollectorCounts(int Healthy, int Failing, int Total, string? DeadlockBand = null); + internal readonly record struct CollectorCounts(int Healthy, int Failing, int Total, string? DeadlockBand = null, string? PgDeadlockBand = null); } /// @@ -1388,8 +1537,26 @@ public sealed class FleetServerCard [JsonIgnore] public TimeSpan BlockingWindow { get; init; } + /// Deadlocks in the card's window, from the engine's own instrument: captured deadlock graphs + /// on SQL Server, the pg_stat_database.deadlocks counter differenced per database and summed on + /// PostgreSQL (#3539). says which, and whether the collector behind it was + /// actually running. [JsonPropertyName("deadlock_count")] public int DeadlockCount { get; init; } + + /// The newest deadlock in the window — the graph's own timestamp on SQL Server; on PostgreSQL + /// the sample that first showed the counter step, so "within the preceding minute". [JsonPropertyName("deadlock_last_seen")] public DateTime? DeadlockLastSeen { get; init; } + + /// Whether is a measurement this card banded on (#3539) — always + /// on SQL Server (a graph count is an observation even at zero), and on PostgreSQL only when at least + /// one counter difference was taken in the window (a difference of fewer than two samples is no + /// reading). Carried, not serialized: deadlock_rate_per_hour is null and + /// deadlock_severity is Unknown exactly when this is false on a rateable window, so the wire + /// already says it; this is for , which must hand the re-band the same + /// null the card banded on. Defaults to false so a card built by a path that did not decide reads + /// unmeasured — the direction that cannot claim health. + [JsonIgnore] + public bool DeadlockMeasured { get; init; } /// Deadlocks per HOUR over the card's window — the figure deadlock_severity banded on /// (#3368), or null when the window was too short to normalise. Published beside the raw count because a /// card that bands on a number it does not show leaves a reader unable to tell which tier was @@ -1411,11 +1578,11 @@ public sealed class FleetServerCard [JsonIgnore] public DeadlockRateThresholds DeadlockRateThresholds { get; init; } - /// This server's deadlocks collector band over the trailing seven days of collection - /// health (#3017) — the fact that explains a of zero. Null when that + /// This server's deadlock-source collector band over the trailing seven days of collection + /// health (#3017) — the fact that explains a of zero. The collector is + /// deadlocks on SQL Server and pg_database_stats on PostgreSQL (#3539). Null when that /// collector left no row in the health window, which is itself the answer rather than the absence of - /// one: nothing was read for this server. A PostgreSQL target has no deadlocks collector at all, - /// so it is null there too and answers on the engine instead. + /// one: nothing was read for this server. [JsonPropertyName("deadlock_collector_band")] public string? DeadlockCollectorBand { get; init; } /// Whether read a deadlock source for this server, and when it did @@ -1488,7 +1655,10 @@ deliberately left as zeros (#3017) — reading them here would hand the ranking MaxBlockedSeconds = MaxBlockingWaitMs / 1000.0, /* #3539 A3: the count's denominator, for the same reason the deadlock window travels below. */ BlockingWindow = BlockingWindow, - DeadlockCount = ServerMetricSources.DmvSourced(DeadlockCount, IsPostgres), + /* The engine's own instrument's count since #3539, handed to the re-band exactly as the card + banded it: measured on every SQL Server card, and on a PostgreSQL card only when a difference + was taken - see DeadlockMeasured. */ + DeadlockCount = DeadlockMeasured ? DeadlockCount : null, /* #3368: the three travel together for the reason the CPU trio above does. Without the window the re-band would have no denominator and the worst-first score would rank every deadlocking server at Warning; without the tiers it would rank them against the shipped pair while the card's own diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpAlertTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpAlertTools.cs index f5afef035..11bfb7c4b 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpAlertTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpAlertTools.cs @@ -491,7 +491,7 @@ and the second one is a mute somebody INTENDED that is no longer in force. "is Critical' reading these tiers replaced. Setting critical BELOW warn is accepted and means every " + "banded rate is Critical. " + "file_growth.rise_mb is megabytes per HOUR averaged over file_growth.lookback_minutes (a rate — the same 10240 is 10 GB/hr on any lookback; the engine scales it to the window), so shortening the lookback does not tighten the rise gate and lengthening it does not loosen it; only the rate does. " + - "Two keys govern the PostgreSQL versions of the two count alerts and are NOT the same numbers as their SQL Server neighbours: deadlocks.pg_count_threshold and blocking.pg_count_threshold, both accepting 1 upward. They sit inside those groups rather than a section of their own so both engines' figures are visible together, but tuning deadlocks.count_threshold does NOT move the PostgreSQL gate and tuning deadlocks.pg_count_threshold does NOT move the SQL Server one. The enabled switch in each group DOES govern both engines. They are separate because the reason to move the SQL Server deadlock figure is agreement with health_bands.deadlock_warn_per_hour, and a PostgreSQL server has no deadlock band at all - its deadlocks are served by get_pg_deadlocks and are structurally absent from the fleet deadlock total - while on the blocking side the SQL Server count is engine-recorded blocked-process reports and the PostgreSQL one is distinct root blockers in a periodic SAMPLE of pg_stat_activity. Both PostgreSQL keys are ignored on a store with no PostgreSQL targets. " + + "Two keys govern the PostgreSQL versions of the two count alerts and are NOT the same numbers as their SQL Server neighbours: deadlocks.pg_count_threshold and blocking.pg_count_threshold, both accepting 1 upward. They sit inside those groups rather than a section of their own so both engines' figures are visible together, but tuning deadlocks.count_threshold does NOT move the PostgreSQL gate and tuning deadlocks.pg_count_threshold does NOT move the SQL Server one. The enabled switch in each group DOES govern both engines. They are separate because the two engines count with different instruments - SQL Server's figure is captured deadlock graphs, the PostgreSQL one is deadlocks parsed from the server log (get_pg_deadlocks) - and an operator tuning one should not silently move the other. Both engines' fleet cards now band deadlocks through the SAME health_bands.deadlock_warn_per_hour tiers (the PostgreSQL card differences the server's own pg_stat_database.deadlocks counter over the window, #3539), so the #3444 move - raising a fire gate to meet the band's Warning bar so a page and an amber dot describe the same server - is available on either knob. On the blocking side the SQL Server count is engine-recorded blocked-process reports and the PostgreSQL one is distinct root blockers in a periodic SAMPLE of pg_stat_activity. Both PostgreSQL keys are ignored on a store with no PostgreSQL targets. " + "The fleet_sweep group is NOT an alert family and the alerts_enabled master switch does not govern it: " + "fleet_sweep.enabled turns the scheduled whole-fleet sweep report on or off, and " + "fleet_sweep.interval_minutes (15\u20131440, default 60) is its cadence. Sweeps deliberately keep running " + diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpFleetTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpFleetTools.cs index c849527fa..e4ab91ff9 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpFleetTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpFleetTools.cs @@ -50,7 +50,17 @@ public sealed class DarlingMcpFleetTools "buffer_pool_mb and the whole threads block are null for " + "the same structural reason and no band is claimed for them — those metrics are SQL Server DMV " + "readings with no PostgreSQL equivalent collected; get_pg_buffer_usage, get_pg_kernel_stats and " + - "get_pg_session_states are the reads that answer the nearest PostgreSQL questions.")] + "get_pg_session_states are the reads that answer the nearest PostgreSQL questions. A PostgreSQL " + + "target's deadlock_count IS measured: it is the server's own pg_stat_database.deadlocks counter, " + + "differenced per database over the window (a statistics reset clamps to zero, never subtracts) and " + + "summed, banded through the same deadlock_warn_per_hour / deadlock_critical_per_hour tiers as SQL " + + "Server's graph count; deadlock_source reads PostgresTarget for it, which since #3539 means COUNTED " + + "from that counter (deadlock_coverage.postgres_servers is a sub-count of servers_read, not a gap), " + + "and get_pg_deadlocks has the parsed deadlock reports themselves. Its blocking_severity stays " + + "Unknown on purpose: PostgreSQL blocking is a once-a-minute SAMPLE of pg_stat_activity, and the " + + "blocking band's count tiers were measured in engine-recorded reports per hour, so a sighting count " + + "through them would band on a denominator they were never measured against — the PostgreSQL " + + "Blocking alert speaks for that condition until a sampled-shape band is measured.")] public static async Task GetFleetOverview( NpgsqlDataSource postgres, [Description("Hours of blocking/deadlock history the per-server cards and fleet totals window over. Default 1.")] int hours_back = 1) diff --git a/Darling/PerformanceMonitor.Darling.Service/wwwroot/js/pages/fleet.js b/Darling/PerformanceMonitor.Darling.Service/wwwroot/js/pages/fleet.js index 2a7cc3481..62aabe6d8 100644 --- a/Darling/PerformanceMonitor.Darling.Service/wwwroot/js/pages/fleet.js +++ b/Darling/PerformanceMonitor.Darling.Service/wwwroot/js/pages/fleet.js @@ -451,10 +451,14 @@ function groupControl() { /* * #3017: the deadlock total's denominator, as a VISIBLE sub-line rather than a tooltip. * - * total_deadlocks comes out of v_deadlocks, which is the SQL Server extended-event capture and nothing else, - * so it is structurally zero on a PostgreSQL fleet — permanently, whatever those clusters do. Zero is also - * exactly what a genuinely quiet SQL Server fleet reports, and the tile could not tell an operator which one - * they were looking at. The API answers that (deadlock_coverage), and this renders it. + * total_deadlocks is each engine's own instrument summed across the fleet — the SQL Server extended-event + * capture, and since #3539 the PostgreSQL server counter differenced over the window — so a server whose + * deadlock-source collector is silent or denied contributes a structural zero. Zero is also exactly what a + * genuinely quiet fleet reports, and the tile could not tell an operator which one they were looking at. + * The API answers that (deadlock_coverage), and this renders it; the note on the tile's title names how many + * of the read servers were counted the counter way (postgres_servers), which is a sub-count of servers_read. + * (Before #3539 a PostgreSQL target was structurally uncounted and this sub-line read "N of M" on any mixed + * fleet; a PostgreSQL fleet whose pg_database_stats collectors run now reads "all".) * * ALWAYS shown when the API reports coverage, including at full coverage, for two reasons. A line that * appeared only on partial coverage would make its ABSENCE the load-bearing signal, which an operator has to diff --git a/Darling/PerformanceMonitor.Darling.Storage/DarlingPgSessionStatesReader.cs b/Darling/PerformanceMonitor.Darling.Storage/DarlingPgSessionStatesReader.cs index e095de1d8..89c9b5425 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/DarlingPgSessionStatesReader.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/DarlingPgSessionStatesReader.cs @@ -8,6 +8,7 @@ using System; using System.Collections.Generic; +using System.Linq; using System.Threading; using System.Threading.Tasks; using Npgsql; @@ -383,9 +384,56 @@ public sealed record LongRunningSessionRow( /// sys.dm_exec_requests — a table of requests actually executing, where an idle session has no /// row at all. /// - /// $1 server_id, $2 threshold (ms), $3 recency floor (naive UTC), $4 row limit. + /// The noise opt-outs, and which SQL Server sibling each mirrors (#3539). SQL Server's + /// CheckLongRunningQueriesAsync reads sys.dm_exec_requests through five switchable noise + /// filters plus an unconditional session_id > 50; this read had none, so autovacuum at + /// minute 31, a nightly pg_dump, or a manual VACUUM on a large relation paged with a mute as + /// the only remedy — and the mute is weaker here than on SQL Server, because this table stores no query + /// text (see the collector) and so a mute rule cannot match a statement. What the row DOES carry is + /// backend_type, application_name and the whitelisted command_tag, which are the + /// three handles below. + /// + /// Non-client backends (unconditional): backend_type <> 'client backend' — + /// autovacuum workers, walsenders (streaming replication, pg_basebackup), logical replication + /// workers, background workers. Mirrors SQL Server's unconditional session_id > 50: a system + /// process is not a query. NULL-safe in the INCLUDING direction — backend_type is in the + /// privileged column set and comes back NULL without pg_monitor, and dropping every row on a + /// redacted target would make the alert silently never fire exactly where the collector has already + /// stamped state_is_redacted. + /// Maintenance statements (unconditional): command_tag in VACUUM, + /// ANALYZE, REINDEX, CLUSTER — a manual vacuum of a large table runs for an hour by + /// design, and the alert asks about QUERIES. SQL Server has no statement-shape sibling because it needs + /// none: its row carries the text and an operator mutes ALTER INDEX by pattern; here the tag is + /// the only handle, so the exclusion has to live in the read. CREATE is deliberately NOT in the + /// list: the tag cannot tell CREATE INDEX CONCURRENTLY (maintenance) from CREATE TABLE AS + /// SELECT (a query), and the honest side of that ambiguity is to report — the incident line shows + /// the tag so a reader can see which it was. + /// Dump and restore utilities (the {0} placeholder, on the SHARED + /// longRunningQueryExcludeBackups switch): application_name in pg_dump, + /// pg_dumpall, pg_restore, pg_basebackup — the names libpq's + /// fallback_application_name gives those tools, so a dump's COPY ... TO STDOUT sessions + /// carry them without operator configuration. Mirrors BackupsFilter (BACKUPTHREAD / + /// BACKUPIO) on the SAME knob, the way longRunningQueryEnabled and the threshold are already + /// shared: "do not page me for backups" is one preference, not one per engine. psql is NOT + /// excluded — an operator's ad-hoc statement running long is precisely a long-running query, and SQL + /// Server does not exclude SSMS either. + /// Idle in transaction (unconditional, pre-existing): its own condition — see the + /// paragraph above. + /// + /// The other three SQL Server switches have no honest PostgreSQL reading and are not faked: + /// sp_server_diagnostics and XE_LIVE_TARGET_TVF name SQL Server internals with no + /// counterpart, and WAITFOR's twin (pg_sleep) is invisible here because the row carries + /// no text and SELECT pg_sleep(...) tags as SELECT. CDC's nearest relative — logical + /// replication workers — is already out through backend_type. excludedDatabases is applied + /// after the read, exactly as the SQL Server adapter applies it. + /// + /// $1 server_id, $2 threshold (ms), $3 recency floor (naive UTC), $4 row limit; {0} is + /// the switchable filter block. A PROPERTY rather than a string field on purpose: the shipped-read + /// parse census (DarlingPgReadSqlParsesLiveTests) parse-checks every static string field on a + /// reader, and a template with a placeholder in it cannot parse. The two RENDERINGS below are the + /// fields, so both texts that can actually reach the store are the ones parse-checked. /// - public const string CurrentLongRunningSessionsSql = """ + public static string CurrentLongRunningSessionsSqlTemplate => """ WITH recent AS ( SELECT max(collection_time) AS latest_capture FROM pg_session_states @@ -406,18 +454,50 @@ JOIN recent AS r WHERE s.server_id = $1 AND s.query_duration_ms >= $2 AND s.is_idle_in_transaction = false + AND coalesce(s.backend_type, 'client backend') = 'client backend' + AND coalesce(s.command_tag, '') NOT IN ('VACUUM', 'ANALYZE', 'REINDEX', 'CLUSTER') + {0} ORDER BY s.query_duration_ms DESC, s.pid LIMIT $4 """; + /// The switchable dump/restore opt-out — the PostgreSQL reading of + /// longRunningQueryExcludeBackups. A constant rather than inline so a test can pin the list and + /// the read can be asserted to include it exactly when the switch is on. + public const string BackupUtilitiesFilter = + "AND coalesce(s.application_name, '') NOT IN ('pg_dump', 'pg_dumpall', 'pg_restore', 'pg_basebackup')"; + + /// The read as it runs with the backups opt-out ON — the shipped default, and what the + /// pre-#3539 constant name meant. Kept under the old name so pins on the shape keep pointing at the + /// text that actually executes on an untouched store. A static readonly field, not a property, + /// so the parse census sees it; is a const, so this + /// initializer cannot read it before it exists. + public static readonly string CurrentLongRunningSessionsSql = BuildCurrentLongRunningSessionsSql(excludeBackups: true); + + /// The read as it runs with the backups opt-out OFF — the other text that can reach the store, + /// held as a field for the same parse-census reason. The host does not read this; it calls + /// with the setting. + public static readonly string CurrentLongRunningSessionsSqlBackupsIncluded = BuildCurrentLongRunningSessionsSql(excludeBackups: false); + + /// Renders for one setting of the shared + /// backups switch. Public so the host's call and a test's pin are the same text. + public static string BuildCurrentLongRunningSessionsSql(bool excludeBackups) => + CurrentLongRunningSessionsSqlTemplate.Replace("{0}", excludeBackups ? BackupUtilitiesFilter : ""); + + /// The shared longRunningQueryExcludeBackups switch — drops the + /// dump/restore utilities' sessions (see the SQL's doc comment). + /// The shared excludedDatabases list, applied after the read + /// case-insensitively exactly as the SQL Server adapter applies it; a row with no database name is + /// kept. Null or empty excludes nothing. public static async Task> GetCurrentLongRunningSessionsAsync( NpgsqlDataSource postgres, int serverId, long thresholdMs, DateTime nowUtc, int recencyMinutes, int limit, + bool excludeBackups, IReadOnlyList? excludedDatabases, CancellationToken cancellationToken = default) { ArgumentNullException.ThrowIfNull(postgres); var rows = new List(); - await using var command = postgres.CreateCommand(CurrentLongRunningSessionsSql); + await using var command = postgres.CreateCommand(BuildCurrentLongRunningSessionsSql(excludeBackups)); command.CommandTimeout = StorageCommandDeadlines.McpReadSeconds; command.Parameters.AddWithValue(serverId); command.Parameters.AddWithValue(thresholdMs); @@ -439,7 +519,27 @@ public static async Task> GetCurrentLongRunningSessi reader.IsDBNull(6) ? -1 : reader.GetInt64(6))); } - return rows; + return FilterExcludedDatabases(rows, excludedDatabases); + } + + /// The excludedDatabases arm, pulled out so it is pinnable without a store: the SQL + /// Server adapter's exact rule (ordinal-ignore-case on the name; a row with no database name is kept, + /// because an exclusion list names databases and a session on none of them is not on an excluded + /// one). Applied AFTER the row limit, as the SQL Server adapter applies it, so an excluded database's + /// sessions can crowd the cap — the same known shape on both engines rather than a quiet divergence + /// where one engine's cap counts excluded rows and the other's does not. + public static List FilterExcludedDatabases( + List rows, IReadOnlyList? excludedDatabases) + { + if (excludedDatabases is not { Count: > 0 }) + { + return rows; + } + + return rows + .Where(r => string.IsNullOrEmpty(r.DatabaseName) + || !excludedDatabases.Any(e => string.Equals(e, r.DatabaseName, StringComparison.OrdinalIgnoreCase))) + .ToList(); } public static async Task GetPgSessionStatesCaptureCountsAsync( diff --git a/Darling/PerformanceMonitor.Darling.Viewer/MainWindow.xaml.cs b/Darling/PerformanceMonitor.Darling.Viewer/MainWindow.xaml.cs index ee2a3c825..88fb7d4da 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/MainWindow.xaml.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/MainWindow.xaml.cs @@ -1502,9 +1502,10 @@ is exactly why DarlingFleetReader.GetFleetOverviewAsync hoists its own copy of t summary.ServerName = server.ServerName; /* #3029: the engine discriminator comes from the REGISTRY row, which already carries it (servers.engine_kind, via ManagedServersSql / ServersSql) — the per-server summary - reads have no engine column and need none. It is what tells the fleet deadlock total's - coverage apart from a quiet SQL Server fleet: v_deadlocks holds the SQL Server - extended-event capture and nothing else. */ + reads have no engine column and need none. It is what selects which deadlock-source + collector's band the fleet deadlock total's coverage reads for this card (#3539): + v_deadlocks holds the SQL Server extended-event capture, pg_database_stats the + PostgreSQL counter, and the summary read carries both bands because it cannot pick. */ summary.IsPostgres = server.IsPostgres; /* #3267: and the Aurora half of the same discriminator, for the CPU row. Both are stamped from the one registry row, so a card cannot end up claiming Aurora-ness the diff --git a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Fleet.cs b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Fleet.cs index 1709ed67e..4e8897cf1 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Fleet.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Fleet.cs @@ -58,8 +58,13 @@ public sealed partial class ViewerDataService /// lines the two sources up per server, and each server contributes its XE count when it has any XE row /// this window else its DMV count (Lite's COALESCE(NULLIF(xe,0), dmv), applied per server) — so /// an AWS RDS server with only DMV snapshots still counts and a server with XE reports is never - /// double-counted. total_deadlocks is a plain cross-server COUNT. $1 window start, $2 window end - /// (both naive UTC). + /// double-counted. total_deadlocks is the SQL Server graph COUNT plus the PostgreSQL counter + /// differences (#3539): the second branch is the SAME LAG-per-(server_id, database_name) + /// shape the per-server card read (ServerSummaryPgDeadlockSql) and the service's fleet reader + /// use — positive differences only, so a statistics reset drops its interval rather than subtracting a + /// lifetime — so the fleet total reconciles with the sum of the card counts on both engines. Never a + /// SUM(deadlocks): the column is a lifetime counter repeated in every sample. $1 window start, + /// $2 window end (both naive UTC). /// public const string FleetTotalsSql = @" SELECT @@ -94,6 +99,17 @@ SELECT COUNT(*) FROM v_deadlocks WHERE deadlock_time >= $1 AND deadlock_time <= $2 + ) + + + ( + SELECT COALESCE(SUM(GREATEST(sampled.raw_delta, 0)), 0) + FROM + ( + SELECT deadlocks - LAG(deadlocks) OVER (PARTITION BY server_id, database_name ORDER BY collection_time) AS raw_delta + FROM pg_database_stats + WHERE collection_time >= $1 + AND collection_time <= $2 + ) AS sampled ) AS total_deadlocks"; /// @@ -308,8 +324,9 @@ public sealed class FleetRollup /// leading sentence of . /// public const string DeadlockSourceNote = - "Deadlocks come from the SQL Server extended-event capture and nothing else, so a server this " - + "total does not cover contributes nothing to it whatever that server's deadlocks do."; + "Deadlocks come from each engine's own instrument - the SQL Server extended-event capture, and on " + + "PostgreSQL the server's deadlock counter differenced over the window - so a server whose " + + "collector is not running contributes nothing to this total whatever that server's deadlocks do."; /// /// The sentence that keeps the two figures from being read as one measurement — the desktop wording of @@ -329,15 +346,18 @@ public sealed class FleetRollup + "counts only the last hour. The two windows differ deliberately, and this coverage figure " + "therefore makes no claim about what was read in the last hour."; - /// What to do about the PostgreSQL arm, appended after its count. Names no tab, because a - /// PostgreSQL target's deadlock grid is reached through that server's own tab rather than from here. + /// How the PostgreSQL arm was counted, appended after its count (#3539). Not an uncovered + /// cause — these servers ARE in the total — but a reader is owed the instrument: a counter difference + /// has no graph to show, and that target's own server tab is where the parsed deadlocks are. Names no + /// tab by name, because a PostgreSQL target's deadlock grid is reached through that server's own tab + /// rather than from here. /// /// Every cause here is a VERB-FREE noun phrase, so one form reads correctly after both "1 server:" /// and "4 servers:" — an "N are ..." shape needs a second string the moment N is one, and the surface /// that forgets it prints "1 are PostgreSQL targets". public const string DeadlockPostgresCause = - "PostgreSQL targets, whose deadlocks this total cannot count at all - collected separately, and " - + "shown on that target's own server tab."; + "PostgreSQL targets, counted from the server's own deadlock counter rather than from captured " + + "deadlock graphs - the parsed deadlocks are on that target's own server tab."; /// What to do about the silent arm, appended after its count. public const string DeadlockCollectorSilentCause = @@ -552,13 +572,15 @@ server uncovered on an empty fleet and nothing at all on a populated one. */ /// registered fleet; a denominator that shrank to whatever loaded this cycle would report a smaller /// fleet than exists, which is a new wrong number in place of the old one rather than a fix. /// - /// Only counts as read — every other arm, INCLUDING - /// an enum value a later build adds and this switch has never heard of, lands in the silent bucket. A - /// new source kind that inflated the read count would restore exactly the defect this exists to fix, - /// where one that lands in an uncovered bucket merely attributes a real gap imprecisely. + /// Only the arms names count as read — every + /// other arm, INCLUDING an enum value a later build adds and this switch has never heard of, lands in + /// the silent bucket. A new source kind that inflated the read count would restore exactly the defect + /// this exists to fix, where one that lands in an uncovered bucket merely attributes a real gap + /// imprecisely. The PostgreSQL arm is covered AND tallied on its own (#3539): the sub-count names the + /// instrument, the read count names the coverage. /// - /// The four causes therefore need not sum to : a registered - /// server with no summary this cycle is classified by none of them, and that shortfall is + /// The three uncovered-or-read causes therefore need not sum to : + /// a registered server with no summary this cycle is classified by none of them, and that shortfall is /// — stated in its own words by and by /// , rather than attributed to a cause it was not measured to /// have. @@ -574,9 +596,14 @@ public static FleetDeadlockCoverage ReduceDeadlockCoverage(IReadOnlyList= $2), (SELECT MAX(deadlock_time) FROM v_deadlocks WHERE server_id = $1)"; + /// + /// The PostgreSQL twin of (#3539): deadlocks in the window from + /// the server's own pg_stat_database.deadlocks counter, plus the sample that first showed the + /// newest increase. Runs for every server like does and needs no + /// engine test for the same reason: a SQL Server has no row in pg_database_stats and the read + /// returns a zero and a NULL, which the caller adds to the extended-event count — one engine's arm is + /// always a structural zero, so the sum is the other engine's reading. + /// + /// A difference per database_name series, clamped at zero, summed — the shape + /// DarlingPgDatabaseReader.PgDatabaseSql uses for get_pg_database_stats and the service's + /// fleet card uses for its twin of this card, so the three cannot disagree on a server. The column is a + /// lifetime counter repeated in every one-minute sample; SUM(deadlocks) over a window is + /// meaningless, and a plain last-minus-first goes negative across pg_stat_reset(). Only the + /// intervals that stepped UP are counted. + /// + /// "Last" is bounded to the window, unlike the SQL Server arm's unbounded + /// MAX(deadlock_time): finding the counter's last step over all history is a scan of the whole + /// series, and the card's "Last: N ago" detail renders only when the window is clear, which for this + /// arm is exactly when there is no step in the window to report. + /// + /// intervals is how many differences were taken, and is what makes the count a + /// measurement — DarlingFleetReader.FleetPgDeadlockSql's reasoning, verbatim: a difference needs + /// two samples, and a zero summed over no differences is the arithmetic of an empty set, not an + /// observation that nothing deadlocked. bands the + /// PostgreSQL arm only when this is positive. $1 server_id, $2 window start (naive UTC). + /// + public const string ServerSummaryPgDeadlockSql = @" +SELECT + CAST(COALESCE(SUM(GREATEST(sampled.raw_delta, 0)), 0) AS bigint) AS cnt, + MAX(sampled.collection_time) FILTER (WHERE sampled.raw_delta > 0) AS last_seen, + CAST(count(sampled.raw_delta) AS bigint) AS intervals +FROM +( + SELECT + collection_time, + deadlocks - LAG(deadlocks) OVER (PARTITION BY database_name ORDER BY collection_time) AS raw_delta + FROM pg_database_stats + WHERE server_id = $1 + AND collection_time >= $2 +) AS sampled"; + /// Newest collection time across all collectors for one server. $1 server_id. public const string ServerSummaryLastCollectionSql = @" SELECT MAX(collection_time) @@ -345,6 +386,8 @@ Unknown rather than Healthy. */ } /* Deadlock count in the last hour + the newest deadlock ever (for "Last: N ago"). */ + DateTime? lastDeadlock = null; + long pgDeadlockIntervals = 0; await using (var command = _dataSource.CreateCommand(ServerSummaryDeadlockSql)) { command.CommandTimeout = ViewerCommandDeadlines.CurrentInteractiveReadSeconds; @@ -354,10 +397,38 @@ Unknown rather than Healthy. */ if (await reader.ReadAsync(cancellationToken)) { deadlockCount = reader.IsDBNull(0) ? 0 : Convert.ToInt32(reader.GetValue(0)); - lastDeadlockMinutesAgo = MinutesAgo(reader.IsDBNull(1) ? null : reader.GetDateTime(1), nowUtc); + lastDeadlock = reader.IsDBNull(1) ? null : reader.GetDateTime(1); + } + } + + /* The PostgreSQL arm of the same reading (#3539), added rather than chosen: this method does not + know the engine (the loader stamps IsPostgres afterwards), and one arm is always a structural + zero, so the sum IS the engine's own count - the same engine-test-free shape ServerSummaryPgCpuSql + takes. The newest of the two "last" instants wins, and on a PostgreSQL target only this arm can + have one. Saturated rather than wrapped into the int the card carries, for the fleet reader's + reason: a wrapped count could band a catastrophe Healthy. */ + await using (var command = _dataSource.CreateCommand(ServerSummaryPgDeadlockSql)) + { + command.CommandTimeout = ViewerCommandDeadlines.CurrentInteractiveReadSeconds; + command.Parameters.Add(new NpgsqlParameter { TypedValue = serverId }); + command.Parameters.Add(new NpgsqlParameter { TypedValue = windowStart }); + await using var reader = await command.ExecuteReaderAsync(cancellationToken); + if (await reader.ReadAsync(cancellationToken)) + { + var pgCount = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); + deadlockCount = (int)Math.Min(deadlockCount + pgCount, int.MaxValue); + var pgLast = reader.IsDBNull(1) ? (DateTime?)null : reader.GetDateTime(1); + if (pgLast.HasValue && (!lastDeadlock.HasValue || pgLast.Value > lastDeadlock.Value)) + { + lastDeadlock = pgLast; + } + + pgDeadlockIntervals = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); } } + lastDeadlockMinutesAgo = MinutesAgo(lastDeadlock, nowUtc); + /* Newest collection time across all collectors — drives the freshness status. */ await using (var command = _dataSource.CreateCommand(ServerSummaryLastCollectionSql)) { @@ -373,7 +444,7 @@ Unknown rather than Healthy. */ /* Collectors row — REUSE the viewer's own 7-day per-collector health banding (the same STALE / FAILING / NEVER_RUN / HEALTHY logic the Collection Health tab renders), mirroring the Dashboard's SUM(CASE health_status = 'HEALTHY' / 'FAILING') over report.collection_health. */ - var (healthyCollectors, failingCollectors, totalCollectors, deadlockBand) = await GetCollectorHealthCountsAsync(serverId, cancellationToken); + var (healthyCollectors, failingCollectors, totalCollectors, deadlockBand, pgDeadlockBand) = await GetCollectorHealthCountsAsync(serverId, cancellationToken); return new ServerSummaryItem @@ -396,6 +467,7 @@ Unknown rather than Healthy. */ LastBlockingMinutesAgo = lastBlockingMinutesAgo, DeadlockCount = deadlockCount, LastDeadlockMinutesAgo = lastDeadlockMinutesAgo, + PgDeadlockIntervals = pgDeadlockIntervals, /* #3368: the count's denominator and the store's tiers, so this card bands on the same rate and the same numbers the service's fleet card does. #3539 A3: the blocking count was read over the same window, and carries it on its own terms. */ @@ -410,6 +482,7 @@ Unknown rather than Healthy. */ FailedCollectorCount = failingCollectors, CollectorCount = totalCollectors, DeadlockCollectorBand = deadlockBand, + PgDeadlockCollectorBand = pgDeadlockBand, LastCollectionTime = lastCollection, }; } @@ -462,9 +535,12 @@ public async Task GetDeadlockRateThresholdsAsync(Cancell /// /// Null when the collector left no row in the window. Matched from /// 's own name rather than a literal, so a rename cannot leave this - /// silently matching nothing and reporting every server uncovered. + /// silently matching nothing and reporting every server uncovered. The PostgreSQL deadlock-source + /// collector's band (, #3539) comes back beside it on the same + /// terms; this method does not know the engine, so the card carries both and + /// picks once IsPostgres is stamped. /// - private async Task<(int Healthy, int Failing, int Total, string? DeadlockBand)> GetCollectorHealthCountsAsync(int serverId, CancellationToken cancellationToken) + private async Task<(int Healthy, int Failing, int Total, string? DeadlockBand, string? PgDeadlockBand)> GetCollectorHealthCountsAsync(int serverId, CancellationToken cancellationToken) { var rows = await GetCollectionHealthAsync(serverId, cancellationToken); var healthy = rows.Count(r => r.HealthStatus == "HEALTHY"); @@ -472,8 +548,11 @@ public async Task GetDeadlockRateThresholdsAsync(Cancell var deadlockBand = rows .FirstOrDefault(r => string.Equals(r.CollectorName, DeadlocksCollector.Instance.Name, StringComparison.Ordinal)) ?.HealthStatus; + var pgDeadlockBand = rows + .FirstOrDefault(r => string.Equals(r.CollectorName, PgDatabaseStatsCollector.Instance.Name, StringComparison.Ordinal)) + ?.HealthStatus; /* #3539 A8d: every banded row is the share's denominator — the service's CollectorCounts.Total. */ - return (healthy, failing, rows.Count, deadlockBand); + return (healthy, failing, rows.Count, deadlockBand, pgDeadlockBand); } /// Whole minutes elapsed from a stored naive-UTC instant to now (UTC), floored at 0, or null @@ -615,16 +694,23 @@ public sealed class ServerSummaryItem /// Minutes since the most recent deadlock ever — the "Last: N ago" deadlock detail. public int? LastDeadlockMinutesAgo { get; set; } + /// How many pg_stat_database.deadlocks counter differences 's + /// PostgreSQL arm was summed over in the window (#3539) — zero on every SQL Server, and zero on a + /// PostgreSQL target with fewer than two samples per series, where the count is not a measurement. + /// Set by the read; the default zero reads as unmeasured, the direction that cannot claim + /// health. + public long PgDeadlockIntervals { get; set; } + /// /// Whether the store says this target is PostgreSQL — stamped by the Overview loader from the registry /// row (DarlingServer.IsPostgres), the way is, because the per-server /// summary reads carry no engine column of their own. /// - /// It is here for : comes out of - /// v_deadlocks, which holds the SQL Server extended-event capture and nothing else, so a - /// PostgreSQL target's zero is structural rather than quiet. Absence is not evidence for either engine, - /// so the default false keeps the SQL Server reading for a row no connect has stamped — the same - /// posture DarlingServer.IsPostgres takes. + /// It is here for and the two DMV-sourced bands: it selects which + /// deadlock-source collector's band the coverage reads (#3539), and it is what tells the memory and + /// blocking bands their zero is structural rather than quiet. Absence is not evidence for either + /// engine, so the default false keeps the SQL Server reading for a row no connect has stamped — the + /// same posture DarlingServer.IsPostgres takes. /// public bool IsPostgres { get; set; } @@ -667,18 +753,28 @@ public sealed class ServerSummaryItem /// public string? DeadlockCollectorBand { get; set; } + /// + /// The pg_database_stats collector's band on the same terms as + /// (#3539) — the deadlock-source collector on a PostgreSQL target, where is + /// the pg_stat_database.deadlocks counter differenced over the window. Both bands ride the card + /// because the summary read does not know the engine; picks by + /// . + /// + public string? PgDeadlockCollectorBand { get; set; } + /// /// Whether read a deadlock source for this server at all, and when it did /// not, which cause (#3029) — the shared , so - /// this card and the service's fleet card cannot disagree about what covers a total. + /// this card and the service's fleet card cannot disagree about what covers a total. The band handed in + /// is the ENGINE'S deadlock-source collector's (#3539). /// - /// DERIVED rather than assigned, so a card built by a path that does not set the two inputs reads + /// DERIVED rather than assigned, so a card built by a path that does not set the inputs reads /// as UNCOVERED rather than sitting at an enum default meaning "read" and inflating the fleet's /// coverage. The unset case is , which is the honest /// reading of a card that makes no claim. /// public FleetDeadlockSource DeadlockSource => - FleetDeadlockCoverage.ClassifyDeadlockSource(IsPostgres, DeadlockCollectorBand); + FleetDeadlockCoverage.ClassifyDeadlockSource(IsPostgres, IsPostgres ? PgDeadlockCollectorBand : DeadlockCollectorBand); /// Worker-thread ceiling (max_workers_count). NULL = no scheduler snapshot (e.g. Azure SQL DB). public int? TotalThreads { get; set; } @@ -941,10 +1037,11 @@ trigger on the dot. */ /// . public bool HasMemoryPressure => MemoryWaiterCount > 0 || MemoryTimeoutCount > 0 || MemoryForcedCount > 0; - /* The three DMV-sourced readings with "not measured" expressed as null (#3272), through the SAME shared + /* The two DMV-sourced readings with "not measured" expressed as null (#3272), through the SAME shared decision the service's fleet card uses so the two cannot disagree about whether this server's zero means anything. The raw counts above and beside stay as they are: they are what the fleet total is - summed from, and #3017's coverage block is what explains that total. */ + summed from, and #3017's coverage block is what explains that total. Deadlocks left this pair in + #3539 - see DeadlockCountForBand. */ /// Resource-semaphore pressure as a BANDABLE reading — null when this target's engine has no /// semaphore to read (every PostgreSQL target). @@ -954,9 +1051,16 @@ decision the service's fleet card uses so the two cannot disagree about whether /// this engine. public int? BlockingCountForBand => ServerMetricSources.DmvSourced(BlockingCount, IsPostgres); - /// Deadlocks as a BANDABLE reading — null when this card reads no deadlock source for this - /// engine. is the same fact named for a reader (#3017). - public int? DeadlockCountForBand => ServerMetricSources.DmvSourced(DeadlockCount, IsPostgres); + /// Deadlocks as a BANDABLE reading (#3539). On a SQL Server card it is the count itself — a + /// graph count is an observation even at zero (#3272's engine-not-collector rule). On a PostgreSQL card + /// the count is the pg_stat_database counter differenced over the window, and it is a measurement + /// only when at least one difference was taken (); with fewer than + /// two samples per series the zero is the arithmetic of an empty set, and the band reads Unknown — + /// which is also what keeps an online PostgreSQL target nothing has collected from measuring nothing + /// (#3539 A6). The read sums both engines' arms because it does not know the engine; one arm is always + /// a structural zero, so the count is the engine's own either way. names + /// the instrument and whether its collector was running (#3017). + public int? DeadlockCountForBand => !IsPostgres || PgDeadlockIntervals > 0 ? DeadlockCount : null; /// Memory band — Critical on any resource-semaphore pressure, else Healthy; no source Unknown. public HealthSeverity MemorySeverity => ServerHealthClassifier.MemorySeverity(MemoryPressureForBand); @@ -1002,9 +1106,10 @@ decision the service's fleet card uses so the two cannot disagree about whether /// Deadlocks per HOUR over — what the band evaluates (#3368), or /// null when the window is too short to normalise. Rendered beside the count so the dot's reason is /// legible. - /* DeadlockCountForBand, not the raw count - see DarlingFleetReader.BuildCard's note. A PostgreSQL - target's raw count is a structural zero, and DeadlockDetail renders this on non-null alone, so the - raw value would show 0.0/hr on a card whose severity says Unknown. */ + /* Through DeadlockCountForBand rather than the raw int so the rate and DeadlockSeverity below derive + from ONE value - DeadlockDetail renders this on non-null alone, and a rate published for a count the + band did not see is the #3017 confusion one field over. Null exactly when the PostgreSQL arm took + no difference (#3539). */ public double? DeadlockRatePerHour => DeadlockCountForBand.HasValue ? ServerHealthClassifier.DeadlockRatePerHour(DeadlockCountForBand.Value, DeadlockWindow) diff --git a/Darling/README.md b/Darling/README.md index fddde95c0..b2074fb88 100644 --- a/Darling/README.md +++ b/Darling/README.md @@ -468,7 +468,27 @@ The shared alert engine's switches and thresholds. Every default mirrors Lite's | `cooldownMinutes` | `5` | Minimum minutes between repeats of the same alert condition (clamped 1–120) | | `excludedDatabases` | `[]` | Excluded from blocking/deadlock/long-running-query **alert evaluation** (collection unaffected) | -Not configurable (hardcoded to Lite's defaults until someone needs a knob): the long-running-query read shape (top 5 results; the five noise filters — sp_server_diagnostics, WAITFOR, backups, misc waits, CDC — all on) and the analysis-finding notification policy (notify at severity >= 1.5, 6-hour per-finding cooldown). +The long-running-query read shape is configurable too, on the same defaults Lite ships: `longRunningQueryMaxResults` (`5`, clamped 1–1000) and the five noise opt-outs `longRunningQueryExcludeSpServerDiagnostics`, `longRunningQueryExcludeWaitFor`, `longRunningQueryExcludeBackups`, `longRunningQueryExcludeMiscWaits`, `longRunningQueryExcludeCdc` (all `true`). On a PostgreSQL target the read honours `longRunningQueryExcludeBackups` (it drops `pg_dump` / `pg_dumpall` / `pg_restore` / `pg_basebackup` sessions by `application_name`) and `excludedDatabases`, and unconditionally skips non-client backends (autovacuum workers, walsenders) and the `VACUUM` / `ANALYZE` / `REINDEX` / `CLUSTER` statements; the other three switches name SQL Server internals with no PostgreSQL reading and are ignored there — see the engine-coverage table below. Not configurable: the analysis-finding notification policy (notify at severity >= 1.5, 6-hour per-finding cooldown). + +#### Which engine each built-in alert and health band covers + +One alert name means one condition on both engines wherever both engines can honestly measure it; where only one engine can, the row says why rather than leaving the gap to be discovered (#3539). The health bands are the fleet card's dots (`get_fleet_overview`, `/api/fleet`, the viewer's Overview). + +| Condition | SQL Server | PostgreSQL | Why one-sided, where it is | +|---|---|---|---| +| **Deadlocks** (alert) | Extended-event graphs, `deadlockCountThreshold` per rolling window | Server-log deadlocks parsed into `pg_deadlocks`, `deadlocks.pg_count_threshold` (V122) | Both. Separate count knobs because the instruments differ (a captured graph vs a parsed log entry) and an operator tuning one should not silently move the other. | +| **Deadlocks** (health band) | `v_deadlocks` graphs per hour | `pg_stat_database.deadlocks` — the server's own lifetime counter, differenced between consecutive samples per database, clamped at zero across a statistics reset, summed across the cluster's databases | Both, through the **same** `health_bands.deadlock_warn_per_hour` / `deadlock_critical_per_hour` tiers: a deadlock per hour is the same quantity whichever engine recorded it. The PostgreSQL arm bands only when at least one difference was taken in the window (a difference of fewer than two samples is not a zero), so a target nothing has collected from reads Unknown rather than Healthy. The fleet total's `deadlock_coverage.postgres_servers` names how many servers were counted the counter way; `get_pg_deadlocks` has the graphs. Never `SUM(deadlocks)` on the store — the column is a lifetime counter repeated in every one-minute sample. | +| **Blocking** (alert) | Blocked-process reports + DMV snapshots, `blockingCountThreshold` | Distinct root blockers in a periodic sample of `pg_stat_activity`, `blocking.pg_count_threshold` (V122) | Both; separate knobs because the denominators differ (engine-recorded reports vs sampled sightings). | +| **Blocking** (health band) | Reports per hour + longest block, the #3596 tiers | **Unknown, by design** | The band's count tiers were measured in *reports per server-hour* on engine-recorded evidence. PostgreSQL blocking is SAMPLED once a minute: a waiter seen in three consecutive captures is one wait observed three times, not three reports, so a sighting count through those tiers would band on a denominator they were never measured against. An honest PostgreSQL band needs its own sampled-shape tiers (share of captures holding a waiter; longest observed wait) measured on that fleet's distribution, which has not been taken. Unknown-with-reason beats a fake Healthy; the alert speaks for PostgreSQL blocking meanwhile. Tiering pending: #3601's `lock_wait` log events (`log_lock_waits` "still waiting" lines, one per wait past `deadlock_timeout`) are the event-grain evidence a report-rate band would read, and the join point for a future PostgreSQL arm. | +| **Blocking Wait Time** | Sum of blocked wait across the latest live DMV snapshot, `blockingWaitSecondsThreshold` | — | SQL-only. The condition is a *live total* of how long sessions are currently blocked, which the DMV snapshot gives exactly. `pg_blocking` is a per-minute sample of `pg_stat_activity`: summing observed waits across one capture is a lower bound on the same quantity taken at an arbitrary instant, and a threshold in seconds against a once-a-minute sample would fire on the sampling grain as much as on the workload. A sampled twin is possible but needs its own stated denominator; not faked under the SQL Server name. | +| **Long-Running Query** | `sys.dm_exec_requests`, five noise opt-outs (sp_server_diagnostics, WAITFOR, backups, misc XE waits, CDC) + `session_id > 50` + `excludedDatabases` | Latest `pg_session_states` capture, same threshold and switch | Both. PostgreSQL opt-outs: non-client backends (unconditional — mirrors `session_id > 50`), `VACUUM`/`ANALYZE`/`REINDEX`/`CLUSTER` by command tag (unconditional — SQL Server needs no statement-shape switch because its row carries the text and a mute rule can match it; `pg_session_states` stores no text, so the tag is the only handle), dump/restore utilities by `application_name` on the shared `longRunningQueryExcludeBackups`, `excludedDatabases` after the read. `CREATE` is reported (with its tag) because the tag cannot tell an index build from `CREATE TABLE AS`. The remaining three SQL Server switches name SQL Server internals; `pg_sleep` (WAITFOR's cousin) is invisible without text. | +| **Poison Wait** | THREADPOOL / RESOURCE_SEMAPHORE / RESOURCE_SEMAPHORE_QUERY_COMPILE | IPC:BtreePage / IPC:BufferIo (Aurora) | Both, **one shape** since #3593: accumulated wait over a rolling ten-minute window, graded WARNING/CRITICAL from the shared `PoisonWaitEvaluator` bars on both engines. The earlier divergence (always-CRITICAL on SQL Server, graded on PostgreSQL, under one mute key) is closed. | +| **High CPU** | Ring buffer, 3-sample persistence gate | Performance Insights, banded on ACU headroom, same gate | Both. | +| **Database File Growth** | Per-file growth over the lookback window vs `FileGrowthRiseMb` | — | SQL-only. It is a SQL Server *file-model* condition: data/log files with explicit growth increments and ceilings, where a rise means a file autogrew. PostgreSQL has no file growth to watch — relations grow a page at a time and the disk-level answer is Volume Free Space. The PostgreSQL relative is *bloat growth* (`pg_table_bloat_stats` hourly and `pg_index_bloat` daily are collected); a "bloat growth" alert is a future item and would not share this name, because it answers a different question. | +| **TempDB Space**, **Volume Free Space**, **Long-Running / Failed Agent Job** | Yes | — | SQL-only by construction (tempdb, host volumes via `sys.dm_os_volume_stats`, SQL Agent). | +| **PostgreSQL Wraparound Risk**, **PostgreSQL Vacuum Horizon Blocked** | — | Yes | PostgreSQL-only by construction: MVCC transaction-id wraparound and the xmin horizon have no SQL Server counterpart. | +| **PostgreSQL Replication Slot Retention** | — | `pg_replication_slots` retained WAL, `wal_status` | The SQL Server analogue is a database whose log cannot truncate — `sys.databases.log_reuse_wait_desc` reading REPLICATION / AVAILABILITY_REPLICA / ACTIVE_TRANSACTION / LOG_BACKUP for consecutive collections. That column IS collected, but only as the load-time `database_config` snapshot (frequency 0), not as a time series, so a persistence-gated "Log Reuse Blocked" condition cannot be evaluated honestly today; it needs the column on a cadenced per-database collector (`database_states` is the natural home), which is a collector change with its own census and pins. Named here so the pair is on record: slot retention on PostgreSQL, log-reuse wait on SQL Server. | +| **Server unreachable**, **Collection stopped**, **Retention Held**, collector-health bands | Yes | Yes | Both — these are about the monitor, not the engine. | **Same-statement pileup** is the one analysis finding evaluated on the COLLECTION cadence rather than the analysis interval, because the shape it describes does not survive a 30-minute wait: N concurrent sessions executing one statement, every copy past 10 seconds, at least one suspended on an IO-class wait, where that statement's own snapshots over the prior 45 minutes were sub-second — the live signature of a shared parameterized plan flipping to an IO-heavy shape, which takes every caller down together and is usually over in about ninety seconds. It is computed from the active-query snapshot collector alone, so it works on deployments with the Query Store readers disabled, and it rides the notification policy above unchanged (severity >= 1.5 pages; the severity scales with concurrent sessions × elapsed, so a brief threshold pileup persists as a visible finding without paging). Episodes of the same statement share one incident id, so a recurrence folds onto the same trail as an occurrence rather than arriving as a fresh surprise — and, like every standing condition, it does not re-page inside the cooldown unless it worsens. Unlike the scheduled pass it has no 24-hour data-span requirement: its evidence is the snapshot plus that statement's own recent history, so it works on a young install. diff --git a/PerformanceMonitor.Alerting/PostgresAlertEvaluator.cs b/PerformanceMonitor.Alerting/PostgresAlertEvaluator.cs index 37373b585..38be849f3 100644 --- a/PerformanceMonitor.Alerting/PostgresAlertEvaluator.cs +++ b/PerformanceMonitor.Alerting/PostgresAlertEvaluator.cs @@ -171,16 +171,17 @@ Server then used (PoisonWaitThresholdMs) is meaningless against them — high-vo /// product opinion this change is revising. /// /// Its own knob rather than SQL Server's deadlocks.count_threshold. The two - /// engines' counts are tuned against different evidence and, decisively, against different - /// SURFACES: the reason to move the SQL Server figure is agreement with - /// health_bands.deadlock_warn_per_hour, and a PostgreSQL server has no deadlock band to agree - /// with — DarlingFleetReader.FleetDeadlockSql reads v_deadlocks, which is - /// structurally zero for a PostgreSQL server, and ServerMetricSources.DmvSourced nulls the - /// reading before it reaches the band. Reusing the key would import a number whose whole - /// justification is agreement with a surface the importing engine does not have. The enabled - /// switch IS shared, matching 's own split: whether the condition is - /// worth alerting on at all is one preference, and the volume at which it is worth a page is - /// not. + /// engines' counts are tuned against different evidence: SQL Server's is captured deadlock GRAPHS, this + /// one is distinct deadlocks parsed from the server log, and an operator tuning one should not silently + /// move the other (#3444). When V122 shipped this column there was a second reason — a PostgreSQL server + /// had no deadlock BAND to agree with, because the fleet card read v_deadlocks and nulled the + /// structural zero — and that reason is gone: since #3539 the PostgreSQL card bands its own + /// pg_stat_database.deadlocks counter difference through the SAME + /// health_bands.deadlock_warn_per_hour tiers. So the #3444 move (raise the fire gate to meet the + /// band's Warning bar, so a page and an amber dot describe the same server) is now available on this + /// knob too; the knob stays separate so making it is a choice. The enabled switch IS shared, + /// matching 's own split: whether the condition is worth alerting on at + /// all is one preference, and the volume at which it is worth a page is not. /// public const int DeadlockCountThresholdDefault = 1; diff --git a/PerformanceMonitor.Common/FleetDeadlockCoverage.cs b/PerformanceMonitor.Common/FleetDeadlockCoverage.cs index 0af187248..a283c335a 100644 --- a/PerformanceMonitor.Common/FleetDeadlockCoverage.cs +++ b/PerformanceMonitor.Common/FleetDeadlockCoverage.cs @@ -17,8 +17,13 @@ namespace PerformanceMonitor.Common; /// (#3017). Serialized as its STRING name (the fleet DTOs' JsonStringEnumConverter), so a consumer /// switches on a word rather than on an ordinal that a reordering would move underneath it. /// -/// The three uncovered arms exist as three rather than as one flag because they take OPPOSITE actions, -/// which is the whole reason a bare tally would not have been enough. +/// Two covered arms and two uncovered ones. The covered pair is split because the two engines count +/// with DIFFERENT INSTRUMENTS (#3539): a SQL Server's count is deadlock GRAPHS captured by an extended +/// event, a PostgreSQL target's is the server's own pg_stat_database.deadlocks counter differenced +/// over the window. Both are the server's deadlocks; a reader comparing counts across the fleet is owed +/// the fact that they were taken two ways. The two uncovered arms exist as two rather than as one flag +/// because they take OPPOSITE actions, which is the whole reason a bare tally would not have been +/// enough. /// public enum FleetDeadlockSource { @@ -27,17 +32,25 @@ public enum FleetDeadlockSource /// STALE or WARNING: those collectors succeeded on some cycles, and their rows are in the total. Read, - /// A PostgreSQL target, whose deadlocks the total cannot count at all: they are stored in - /// pg_deadlocks, which nothing joins into v_deadlocks. Nothing about this server's health - /// changes it and no grant addresses it — get_pg_deadlocks is the read that answers. + /// A PostgreSQL target whose deadlocks ARE in the total, counted from the server's own + /// pg_stat_database.deadlocks counter (#3539): the pg_database_stats collector's per-database + /// series, differenced between consecutive samples, clamped at zero across a statistics reset, and + /// summed across the cluster's databases over the window. COVERED, and bucketed apart from + /// only because the instrument differs: a graph capture can miss an event the + /// server still counted (the ring buffer cycled between polls), and a counter difference has no graph + /// to show — get_pg_deadlocks is the read that has those, from the server log. Before #3539 this + /// arm meant "cannot be counted at all"; the total now includes these servers, and + /// counts them. PostgresTarget, - /// The deadlocks collector is not being invoked — STOPPED, NEVER_RUN, or no row in the - /// health window at all. Silence, not failure: a collector still running and erroring every cycle bands - /// FAILING and counts as . Check the collector (get_collection_health). + /// The engine's deadlock-source collector is not being invoked — STOPPED, NEVER_RUN, or no row + /// in the health window at all. Silence, not failure: a collector still running and erroring every cycle + /// bands FAILING and counts as covered. Check the collector (get_collection_health). The + /// collector is deadlocks on SQL Server and pg_database_stats on PostgreSQL (#3539); the + /// action is the same. CollectorSilent, - /// Every one of the deadlocks collector's attempts was refused for permissions + /// Every one of the engine's deadlock-source collector's attempts was refused for permissions /// (NO_PERMISSIONS). The one cause a grant fixes, which is why it is not folded in with /// despite both meaning nothing was read. CollectorDenied, @@ -45,14 +58,18 @@ public enum FleetDeadlockSource /// /// The denominator beside total_deadlocks (#3017): how many of the fleet's -/// servers that total actually read a deadlock source for, and for the rest, which of the three causes it -/// was. +/// servers that total actually read a deadlock source for, how many of those were read from the PostgreSQL +/// server counter rather than from deadlock graphs (#3539), and for the rest, which of the two uncovered +/// causes it was. /// -/// Why a total needs one at all. v_deadlocks is the SQL Server extended-event capture and -/// only that, so total_deadlocks is structurally zero for a PostgreSQL server — permanently, whatever -/// its clusters do and whatever grants the monitoring role holds. Zero is also exactly what a genuinely quiet -/// SQL Server fleet reports. The reading that needs no action and the reading that does not cover the fleet -/// are the same character, and that is what makes the bare number unusable. +/// Why a total needs one at all. A server whose deadlock-source collector is stopped or +/// permission-denied contributes zero, and zero is also exactly what a genuinely quiet server reports. The +/// reading that needs no action and the reading that does not cover the fleet are the same character, and +/// that is what makes the bare number unusable. Until #3539 the PostgreSQL arm was the largest such gap — +/// v_deadlocks is the SQL Server extended-event capture and only that, so a PostgreSQL target's +/// count was structurally zero whatever its clusters did. Those servers are now counted from +/// pg_stat_database.deadlocks (see ), and the bucket +/// that used to name the gap now names the instrument. /// /// The convention this follows. get_pg_blocking already reports captures_total /// beside captures_with_blocking and a sentence saying what an empty answer does and does not mean, @@ -63,7 +80,8 @@ public enum FleetDeadlockSource /// quiet SQL Server fleet with full coverage is healthy and must keep reading that way. /// /// Why this sits in Common rather than beside either reader. TWO surfaces compute this same -/// fleet total from the same v_deadlocks, out of two assemblies that do not reference each other — +/// fleet total from the same two sources (v_deadlocks and pg_database_stats), out of two +/// assemblies that do not reference each other — /// the service's get_fleet_overview / /api/fleet reader and the WPF viewer's Overview /// roll-up (#3029). The buckets, the enum they reduce from and /// therefore live once, here: a cause added to @@ -76,7 +94,9 @@ public sealed class FleetDeadlockCoverage { /// Servers the total read a deadlock source for — the numerator of the coverage this object /// reports, and NOT a count of servers that had deadlocks (that is - /// total_deadlocks's job). + /// total_deadlocks's job). Both covered arms count here: a SQL Server read through its + /// deadlocks collector AND a PostgreSQL target read through its pg_database_stats counters + /// (#3539), so is a SUBSET of this figure, not a bucket beside it. [JsonPropertyName("servers_read")] public int ServersRead { get; init; } /// Enabled servers in the fleet — the same population as @@ -84,16 +104,21 @@ public sealed class FleetDeadlockCoverage /// numerator and a consumer reading only this object is never one field short of the ratio. [JsonPropertyName("servers_total")] public int ServersTotal { get; init; } - /// Uncovered because the target is PostgreSQL — get_pg_deadlocks is the read that - /// answers for these. + /// Of , how many are PostgreSQL targets counted from the server's own + /// pg_stat_database.deadlocks counter rather than from captured deadlock graphs (#3539). A + /// sub-count of the covered figure, kept because the two instruments differ (see + /// ); get_pg_deadlocks is the read that has the + /// graphs for these. Before #3539 this was an UNCOVERED bucket — a consumer summing the four counts to + /// the total must now sum three (read, silent, denied) instead. [JsonPropertyName("postgres_servers")] public int PostgresServers { get; init; } - /// Uncovered because the deadlocks collector is not being invoked (STOPPED, NEVER_RUN, - /// or no row in the health window) — check the collector. + /// Uncovered because the engine's deadlock-source collector (deadlocks or + /// pg_database_stats) is not being invoked (STOPPED, NEVER_RUN, or no row in the health window) — + /// check the collector. [JsonPropertyName("servers_collector_silent")] public int ServersCollectorSilent { get; init; } - /// Uncovered because every one of the deadlocks collector's attempts was refused for - /// permissions — these need a grant. + /// Uncovered because every one of the engine's deadlock-source collector's attempts was refused + /// for permissions — these need a grant. [JsonPropertyName("servers_collector_denied")] public int ServersCollectorDenied { get; init; } /// @@ -115,10 +140,14 @@ public sealed class FleetDeadlockCoverage + "only between window_start and window_end. The two windows differ deliberately, and this coverage " + "figure therefore makes no claim about what was read inside window_start..window_end."; - /// What to do about the PostgreSQL arm, appended after its count. + /// How the PostgreSQL arm was counted, appended after its count. Not a "what to do" — these + /// servers are in the total — but a reader is owed the instrument, because a counter difference and a + /// graph capture are not the same measurement and get_pg_deadlocks is where the graphs + /// are. public const string PostgresCause = - "are PostgreSQL targets, whose deadlocks this total cannot count at all - they are stored separately " - + "and served by get_pg_deadlocks."; + "of those are PostgreSQL targets counted from the server's own pg_stat_database.deadlocks counter " + + "(differenced per database over the window) rather than from captured deadlock graphs; " + + "get_pg_deadlocks serves the graphs."; /// What to do about the silent arm, appended after its count. public const string CollectorSilentCause = @@ -171,20 +200,24 @@ public string Note /// /// Whether one server's deadlock count — and so the fleet total it sums into — actually read a deadlock - /// source, and when it did not, which of the three causes it was (#3017). The three take OPPOSITE - /// actions, which is why one uncovered tally would not have been enough: a different tool for a - /// PostgreSQL target, a collector to look at for one that is not being invoked, a grant for one that was + /// source, through which instrument, and when it did not, which of the two causes it was (#3017, + /// #3539). The two uncovered causes take OPPOSITE actions, which is why one uncovered tally would not + /// have been enough: a collector to look at for one that is not being invoked, a grant for one that was /// refused. /// - /// PostgreSQL is asked first and answers on its own. It is not a degraded collector, it is - /// the wrong table: every fleet deadlock total reads v_deadlocks, which holds only the SQL - /// Server extended-event capture, and no health band on any collector changes that. A PostgreSQL target - /// whose pg_deadlocks collector is perfectly healthy still contributes nothing here. + /// The band is the ENGINE'S deadlock-source collector's (#3539): deadlocks for a SQL + /// Server, pg_database_stats for a PostgreSQL target, chosen by the caller, which is the one place + /// that knows both the engine and both bands. The silent and denied arms then mean the same thing on + /// both engines and take the same action; only the covered arm splits, into + /// and , because the instrument differs. Before #3539 a + /// PostgreSQL target answered PostgresTarget on the engine alone, whatever its collectors did, + /// because nothing read its deadlocks; now that something does, its collector state matters exactly as a + /// SQL Server's does. /// - /// A null band is uncovered, not covered. Null means the deadlocks collector left no - /// row in the health window — nothing was read for this server, whatever else is true of it — and - /// defaulting an absent band to "read" is precisely how a coverage figure becomes another number nobody - /// can trust. It is folded in with STOPPED and NEVER_RUN because all three take the same action. + /// A null band is uncovered, not covered. Null means the collector left no row in the + /// health window — nothing was read for this server, whatever else is true of it — and defaulting an + /// absent band to "read" is precisely how a coverage figure becomes another number nobody can trust. It + /// is folded in with STOPPED and NEVER_RUN because all three take the same action. /// /// FAILING, STALE and WARNING count as READ, deliberately. Those collectors succeeded on /// some cycles, so their rows really are in the total and a denominator excluding them would understate @@ -193,13 +226,12 @@ public string Note /// failed_collector_count are where a degraded collector shows, and the /// card's own deadlock_collector_band carries the band verbatim. /// + /// Whether the store says this target is PostgreSQL — selects which covered arm + /// a read collector lands on. + /// The 7-day band of the ENGINE'S deadlock-source collector + /// (deadlocks or pg_database_stats), or null when it left no row in the health window. public static FleetDeadlockSource ClassifyDeadlockSource(bool isPostgres, string? deadlockCollectorBand) { - if (isPostgres) - { - return FleetDeadlockSource.PostgresTarget; - } - /* Tested ahead of the general set below rather than as part of it: NO_PERMISSIONS is IN NothingReadBands, so asking the set first would swallow the one cause with a different fix. */ if (string.Equals(deadlockCollectorBand, CollectorHealthClassifier.NoPermissions, StringComparison.Ordinal)) @@ -207,8 +239,19 @@ public static FleetDeadlockSource ClassifyDeadlockSource(bool isPostgres, string return FleetDeadlockSource.CollectorDenied; } - return deadlockCollectorBand is null || CollectorHealthClassifier.ReadNothing(deadlockCollectorBand) - ? FleetDeadlockSource.CollectorSilent - : FleetDeadlockSource.Read; + if (deadlockCollectorBand is null || CollectorHealthClassifier.ReadNothing(deadlockCollectorBand)) + { + return FleetDeadlockSource.CollectorSilent; + } + + return isPostgres ? FleetDeadlockSource.PostgresTarget : FleetDeadlockSource.Read; } + + /// Whether a source arm means the server's deadlocks are IN the total (#3539) — the one + /// predicate both roll-ups reduce with, so the service's and the viewer's + /// denominators cannot disagree about which arms count. Only the two named arms are covered: an enum + /// value a later build adds and this switch has never heard of is NOT covered, which is the default + /// that cannot inflate the figure. + public static bool IsCovered(FleetDeadlockSource source) => + source is FleetDeadlockSource.Read or FleetDeadlockSource.PostgresTarget; } diff --git a/PerformanceMonitor.Common/ServerHealthBands.cs b/PerformanceMonitor.Common/ServerHealthBands.cs index fa69c46af..c9c0087a1 100644 --- a/PerformanceMonitor.Common/ServerHealthBands.cs +++ b/PerformanceMonitor.Common/ServerHealthBands.cs @@ -611,13 +611,34 @@ public readonly record struct ServerHealthMetrics /// engine (#3272) — the ONE place that decision is made, so the service's fleet card and the viewer's /// Overview card cannot disagree about whether a zero means anything. /// - /// Why these three travel together. The memory-pressure, blocking and deadlock rows on a - /// card come from v_memory_grant_stats, v_blocked_process_reports / - /// v_dmv_blocking_snapshots and v_deadlocks — all SQL Server captures, none of which a - /// PostgreSQL target has a single row in. The per-metric reads therefore hand the card zeros, and a zero - /// is indistinguishable from a genuinely calm SQL Server. Threads already escaped this because its - /// ceiling is nullable and CPU escaped it in #3267; these three had no way to say "not measured" at all. - /// + /// Why these travel together. The memory-pressure and blocking rows on a card come from + /// v_memory_grant_stats and v_blocked_process_reports / v_dmv_blocking_snapshots — + /// SQL Server captures, neither of which a PostgreSQL target has a single row in. The per-metric reads + /// therefore hand the card zeros, and a zero is indistinguishable from a genuinely calm SQL Server. + /// Threads already escaped this because its ceiling is nullable and CPU escaped it in #3267; these had + /// no way to say "not measured" at all. Deadlocks were the third member until #3539 gave the PostgreSQL + /// card its own count (the pg_stat_database.deadlocks counter, differenced over the window), at + /// which point the reading has a source on both engines and no longer passes through here. That arm + /// carries its own measured/not-measured test instead — whether at least one difference was taken in + /// the window — because a difference of fewer than two samples is not a zero, where a COUNT(*) + /// over an event table is; the two fleet readers hold that decision beside the read. + /// + /// Blocking stays here on purpose, and the reason is the shape of the evidence, not its + /// absence. A PostgreSQL target's blocking IS collected (pg_blocking, pg_lock_stats), + /// but as per-minute SAMPLES of pg_stat_activity: a waiter seen in three consecutive captures is + /// one wait observed three times, where SQL Server's blocked-process report is one engine-recorded event + /// per threshold crossing. 's count tiers were + /// measured in reports per server-hour on that engine-recorded shape (#3596), and a sampled sighting + /// count fed through them would band on a denominator the tiers were never measured against — three + /// sightings of one 3-minute wait is not three reports. An honest PostgreSQL band needs its own + /// sampled-shape tiers (share of captures holding a waiter, longest observed wait) measured on that + /// fleet, which is a distribution nobody has taken yet; until then Unknown is the reading, and the + /// alert (DarlingWorker.EvaluatePgBlockingAsync, root blockers per rolling window) is the + /// surface that speaks for PostgreSQL blocking. Tiering pending: #3601's lock_wait log-event + /// family (log_lock_waits "still waiting" lines) is the EVENT-grain evidence a report-rate band + /// would read — one line per wait past deadlock_timeout, the shape the SQL Server tiers were + /// measured on — and is the join point for a future PostgreSQL arm here; nothing reads it yet. The + /// Darling README's engine-coverage table states this beside the band. /// /// It names the ENGINE, not the collector state. A SQL Server whose deadlock collector is /// permission-denied also reads zero, and that stays Healthy here on purpose: #3017 routed that case to @@ -629,9 +650,8 @@ public static class ServerMetricSources { /// /// The reading as measured, or null when this target's engine has no source behind it. - /// Generic over the reading's own type because the three metrics are a bool and two - /// ints, and the DECISION is the same for all three — one function rather than three that - /// could drift. + /// Generic over the reading's own type because the metrics are a bool and an int, + /// and the DECISION is the same for both — one function rather than two that could drift. /// /// What the SQL Server metric read produced (a zero, for a target with no rows). /// Whether the store SAYS this target is PostgreSQL. Absence of an engine @@ -915,14 +935,19 @@ public static HealthSeverity BlockingSeverity( /// green dot for an unmeasured metric is the failure 's Unknown arm and /// 's exist to avoid. /// - /// The null arm completes #3017 rather than reversing it. That issue established that a - /// PostgreSQL target's zero is structural — v_deadlocks is the SQL Server extended-event - /// capture and nothing joins pg_deadlocks into it — and gave the CARD - /// plus the fleet total a coverage denominator to say so. It - /// deliberately added no band to the FLEET ROLLUP, so that a quiet, fully-covered SQL Server fleet - /// keeps reading healthy; that reasoning is untouched here. A PostgreSQL target has no - /// SQL-Server-deadlock reading at any rate, so it bands off none. - /// Deadlocks counted in the window, or null where the engine has no + /// The null arm is for a path with no source, and since #3539 neither engine's card is + /// one. #3017 established that a PostgreSQL target's v_deadlocks zero was structural and + /// gave the CARD plus the fleet total a coverage denominator to + /// say so; the card then passed null here and banded Unknown. #3539 feeds that card the engine's own + /// count instead — pg_stat_database.deadlocks, a server-maintained counter differenced per + /// database over the window and summed — through THIS band and THESE tiers, because a deadlock per + /// hour is the same quantity whichever engine recorded it and the tiers were set on the + /// rate, not on the instrument. The null arm remains for the PostgreSQL card whose window held no + /// two samples to difference (a difference of nothing is not a zero), for any caller that genuinely + /// reads no source, and for the daily classifier's day cells before a count exists. #3017's own rule + /// is untouched: no band was added to the FLEET ROLLUP, and a quiet, fully-covered fleet keeps + /// reading healthy. + /// Deadlocks counted in the window, or null where the caller has no /// source behind the reading. A long for 's reason (#3525): /// the shared daily classifier routes its day-scale Deadlocks roll-up through this same band, /// so the calendar, get_daily_summary, the fleet sweep and the Overview card cannot disagree From 3c50b910e9e71222a56b7a8e1053f0dc6cfaa765 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 16:49:40 -0400 Subject: [PATCH 62/69] get_pg_logging_audit judges a PostgreSQL target's logging settings facet by facet from the stored config, so a target with everything off stops looking identical to an instrumented one (#3607) (#3643) * get_pg_logging_audit judges a PostgreSQL target's logging settings facet by facet from the stored config, so a target with everything off stops looking identical to an instrumented one (#3607) Plan-capture readiness established the shape - a facet per setting, the remedy per finding, the hosting flavour's own syntax - and covered only auto_explain's three preconditions. The rest of the logging surface (log_lock_waits, log_temp_files, log_autovacuum_min_duration, log_checkpoints, log_connections / log_disconnections, log_min_duration_statement) had no audit at all, and a target with all of them off answered every counter-based read exactly like a fully instrumented one. This is a READ over pg_server_config's newest snapshot, not a collector: every judged setting is a core GUC the config collector already stores hourly with its value, source, unit and context, so the judgment is a pure function computed on request and the snapshot's collection_time is the audit's as_of. Verdicts are instrumented / partial / off / unknown; partial names what a threshold hides and is the recommended posture for log_min_duration_statement. Consumers are named honestly as PLANNED (#3601/#3602/#3603) beside the counter read that exists today. Hosting flavour comes from rds.* parameters in the same snapshot, which is the one fact the registry's engine token cannot supply. Plan capture's own settings are listed with the readiness facet that owns each, never judged twice. Registered beside the plan tools, dispatched on the web with a Configuration-tab-adjacent panel under readiness, censused in the instructions / README / llms.txt / runbook, ratcheted in the cross-app inventory pin. * The audit speaks #3541 A10's stamp dialect: captured_at on the wire, LATEST IS A TIME in the description, and a Darling-only allowance in the stamp roster dev moved under the lane: #3637 landed McpLatestSnapshotStampTests, whose reader-call sweep caught GetNewestSnapshotAsync and asked for a Shape. The roster is a roster of PAIRS (every entry names its Lite twin's file), so a Darling-only tool gets a DarlingOnlyStamped list held to the same three Stamped assertions, and the new reader's SQL joins the stamp-column theory. Threshold() became a block body so TsqlConventionGuardTests' member scan reads it whole rather than growing KnownTruncatedRanges. * Review round 2: pending_restart reaches the row and the summary, the catch re-checks the engine gate, the ten-minute default is PostgreSQL's own verdict, and the two unknowns are told apart pending_restart was fetched, documented as load-bearing, and dropped before the Facet - on exactly the row whose remedy needs qualifying, because the value judged is the RUNNING one and the file already holds another. It is now on every facet with a restart_note, named at the top as pending_restart_settings the way get_pg_server_config names them, and seeded in the live test. The catch block asks the engine gate before answering error, as the plan tools do. atDefault reads source = 'default' rather than comparing setting text to boot_val - an administrator who writes 600000 into the file has made a choice, which is the anti-pattern DarlingPgServerConfigReader's own comment names. An unparseable value now says it IS in the snapshot rather than that it is not. --- .../DarlingMcpPgLoggingAuditToolsTests.cs | 793 ++++++++++++++++++ .../McpLatestSnapshotStampTests.cs | 57 ++ Darling/Darling.Tests/ServerPageTabsTests.cs | 5 + .../DarlingWebEndpoints.cs | 2 + .../Mcp/DarlingMcpHostService.cs | 9 + .../Mcp/DarlingMcpInstructions.cs | 2 +- .../Mcp/DarlingMcpPgLoggingAuditTools.cs | 181 ++++ .../Mcp/DarlingPgLoggingAudit.cs | 732 ++++++++++++++++ .../wwwroot/js/pages/server-tabs.js | 32 + .../DarlingPgLoggingAuditReader.cs | 126 +++ .../CrossAppMcpToolInventoryPinTests.cs | 13 + README.md | 4 +- docs/postgres-first-target-runbook.md | 9 +- llms.txt | 2 +- 14 files changed, 1961 insertions(+), 6 deletions(-) create mode 100644 Darling/Darling.Tests/DarlingMcpPgLoggingAuditToolsTests.cs create mode 100644 Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgLoggingAuditTools.cs create mode 100644 Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingPgLoggingAudit.cs create mode 100644 Darling/PerformanceMonitor.Darling.Storage/DarlingPgLoggingAuditReader.cs diff --git a/Darling/Darling.Tests/DarlingMcpPgLoggingAuditToolsTests.cs b/Darling/Darling.Tests/DarlingMcpPgLoggingAuditToolsTests.cs new file mode 100644 index 000000000..1d1daf583 --- /dev/null +++ b/Darling/Darling.Tests/DarlingMcpPgLoggingAuditToolsTests.cs @@ -0,0 +1,793 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.ComponentModel; +using System.Linq; +using System.Reflection; +using System.Text.Json; +using System.Threading; +using System.Threading.Tasks; +using ModelContextProtocol.Server; +using Npgsql; +using PerformanceMonitor.Collectors; +using PerformanceMonitor.Common; +using PerformanceMonitor.Darling.Service; +using PerformanceMonitor.Darling.Service.Mcp; +using PerformanceMonitor.Darling.Storage; +using Xunit; + +namespace Darling.Tests; + +/// +/// get_pg_logging_audit (#3607): the judgment, asserted against the shipped +/// over hand-built snapshots, and the wire shape against the shipped +/// projection — the DarlingMcpPgPlanToolsTests arrangement, for the same reason: a guard that rebuilt +/// either would keep passing while the real one drifted. +/// +/// The snapshots are spelled the way pg_settings renders them — integers in the base unit with +/// no suffix, booleans as on/off, the PostgreSQL 18 log_connections list verbatim — +/// because the parse is the judgment's first step and a fixture in some other shape would test a path no +/// server produces. Measured on 18.4 (the rig this lane ran): log_autovacuum_min_duration renders +/// 600000 with unit ms, log_connections stores on, true, 1, +/// all and receipt,authentication each as written. +/// +public sealed class DarlingMcpPgLoggingAuditToolsTests +{ + private static readonly DateTime Stamp = new(2026, 9, 18, 14, 0, 0, DateTimeKind.Utc); + + private static DarlingPgLoggingAuditReader.PgLoggingSettingRow Row( + string name, string? setting, string? unit = null, string context = "sighup", + string source = "default", string? boot = null, DateTime? at = null, bool pendingRestart = false) => + new(name, setting, unit, context, source, boot ?? setting, pendingRestart, at ?? Stamp); + + /// A PostgreSQL 14 with nothing turned on: every judged setting at its pre-15 default. + private static List AllOff() => new() + { + Row("log_min_duration_statement", "-1", "ms", "superuser"), + Row("log_lock_waits", "off", context: "superuser"), + Row("log_temp_files", "-1", "kB", "superuser"), + Row("log_autovacuum_min_duration", "-1", "ms"), + Row("log_checkpoints", "off"), + Row("log_connections", "off", context: "superuser-backend"), + Row("log_disconnections", "off", context: "superuser-backend"), + Row("deadlock_timeout", "1000", "ms", "superuser"), + Row("shared_preload_libraries", "", context: "postmaster"), + Row("log_line_prefix", "%m [%p] "), + Row("lc_messages", "en_US.utf8", context: "superuser", source: "configuration file"), + Row("work_mem", "4096", "kB", "user"), + }; + + /// The 18.4 rig's defaults, as measured: checkpoints on, autovacuum at ten minutes, connections empty. + private static List Pg18Defaults() + { + var rows = AllOff(); + Replace(rows, Row("log_autovacuum_min_duration", "600000", "ms")); + Replace(rows, Row("log_checkpoints", "on")); + Replace(rows, Row("log_connections", "", context: "superuser-backend")); + return rows; + } + + /// Every setting at its recommended value, so the audit has nothing to ask for. + private static List Recommended() + { + var rows = Pg18Defaults(); + Replace(rows, Row("log_min_duration_statement", "1000", "ms", "superuser", "configuration file", "-1")); + Replace(rows, Row("log_lock_waits", "on", context: "superuser", source: "configuration file", boot: "off")); + Replace(rows, Row("log_temp_files", "0", "kB", "superuser", "configuration file", "-1")); + Replace(rows, Row("log_autovacuum_min_duration", "0", "ms", source: "configuration file", boot: "600000")); + Replace(rows, Row("log_connections", "all", context: "superuser-backend", source: "configuration file", boot: "")); + Replace(rows, Row("log_disconnections", "on", context: "superuser-backend", source: "configuration file", boot: "off")); + return rows; + } + + private static void Replace(List rows, DarlingPgLoggingAuditReader.PgLoggingSettingRow row) + { + rows.RemoveAll(r => r.Name == row.Name); + rows.Add(row); + } + + private static DarlingPgLoggingAudit.Facet FacetOf(DarlingPgLoggingAudit.Result audit, string setting) => + Assert.Single(audit.Facets, f => f.Setting == setting); + + /* ───────────────────────── the verdicts ───────────────────────── */ + + /// + /// The issue's own scenario: a target with everything off. Seven facets, seven off, every one + /// named in the summary, and every remedy is the self-hosted form because nothing in the snapshot says + /// otherwise. + /// + [Fact] + public void ATargetWithEverythingOff_IsOffSevenTimes_AndEveryRowNamesItsRemedy() + { + var audit = DarlingPgLoggingAudit.Audit(AllOff()); + + Assert.Equal(DarlingPgLoggingAudit.JudgedSettings, audit.Facets.Select(f => f.Setting).ToArray()); + Assert.All(audit.Facets, f => Assert.Equal(DarlingPgLoggingAudit.Off, f.Verdict)); + Assert.False(audit.Managed); + Assert.Contains("no rds.* parameter", audit.HostingEvidence, StringComparison.Ordinal); + Assert.Equal(Stamp, audit.CapturedAt); + + foreach (var facet in audit.Facets) + { + Assert.StartsWith("ALTER SYSTEM SET " + facet.Setting + " = ", facet.Remedy, StringComparison.Ordinal); + Assert.Contains("SELECT pg_reload_conf();", facet.Remedy, StringComparison.Ordinal); + Assert.DoesNotContain("parameter group", facet.Remedy, StringComparison.Ordinal); + Assert.Null(facet.ScopeNote); + } + } + + /// + /// The three threshold settings, at each of their three states. -1 is off, 0 is + /// everything, a positive value is partial — and the partial row SAYS what is under the line, + /// with the threshold in it, because "partial" alone is the collapse this read exists to avoid. + /// + [Theory] + [InlineData("log_min_duration_statement", "ms", "-1", DarlingPgLoggingAudit.Off)] + [InlineData("log_min_duration_statement", "ms", "0", DarlingPgLoggingAudit.Instrumented)] + [InlineData("log_min_duration_statement", "ms", "1000", DarlingPgLoggingAudit.Partial)] + [InlineData("log_temp_files", "kB", "-1", DarlingPgLoggingAudit.Off)] + [InlineData("log_temp_files", "kB", "0", DarlingPgLoggingAudit.Instrumented)] + [InlineData("log_temp_files", "kB", "10240", DarlingPgLoggingAudit.Partial)] + [InlineData("log_autovacuum_min_duration", "ms", "-1", DarlingPgLoggingAudit.Off)] + [InlineData("log_autovacuum_min_duration", "ms", "0", DarlingPgLoggingAudit.Instrumented)] + [InlineData("log_autovacuum_min_duration", "ms", "300000", DarlingPgLoggingAudit.Partial)] + public void AThresholdSetting_IsOffAtMinusOne_EverythingAtZero_AndPartialAbove( + string setting, string unit, string value, string expected) + { + var rows = AllOff(); + Replace(rows, Row(setting, value, unit, "superuser", "configuration file", "-1")); + + var facet = FacetOf(DarlingPgLoggingAudit.Audit(rows), setting); + + Assert.Equal(expected, facet.Verdict); + Assert.Equal(value, facet.Value); + Assert.Equal(unit, facet.Unit); + + if (expected == DarlingPgLoggingAudit.Partial) + { + Assert.Contains(value, facet.CostNote, StringComparison.Ordinal); + Assert.Contains("write nothing", facet.CostNote, StringComparison.Ordinal); + } + } + + /// + /// partial is the recommended posture for log_min_duration_statement, and the row says + /// so instead of asking for a change. Sending an agent to "fix" a 1000 ms threshold to 0 is the + /// worst outcome this read could produce: 0 logs every statement the server runs, and #2565 measured + /// the capture-everything shape of that mechanism at 31 percent of throughput. So the threshold row's + /// remedy is "no change", its cost_note calls it the recommendation, and the 0 row — instrumented by the + /// verdict's own definition — carries the measured cost and a remedy that moves it to a threshold. + /// + [Fact] + public void AStatementThreshold_IsTheRecommendation_AndZeroCarriesTheMeasuredCost() + { + var rows = AllOff(); + Replace(rows, Row("log_min_duration_statement", "1000", "ms", "superuser", "configuration file", "-1")); + var threshold = FacetOf(DarlingPgLoggingAudit.Audit(rows), "log_min_duration_statement"); + + Assert.Equal(DarlingPgLoggingAudit.Partial, threshold.Verdict); + Assert.StartsWith("No change needed", threshold.Remedy, StringComparison.Ordinal); + Assert.Contains("recommended posture", threshold.CostNote, StringComparison.Ordinal); + + Replace(rows, Row("log_min_duration_statement", "0", "ms", "superuser", "configuration file", "-1")); + var everything = FacetOf(DarlingPgLoggingAudit.Audit(rows), "log_min_duration_statement"); + + Assert.Equal(DarlingPgLoggingAudit.Instrumented, everything.Verdict); + Assert.Contains("31 percent", everything.CostNote, StringComparison.Ordinal); + Assert.Contains("#2565", everything.CostNote, StringComparison.Ordinal); + Assert.StartsWith("ALTER SYSTEM SET log_min_duration_statement = 1000;", everything.Remedy, StringComparison.Ordinal); + } + + /// + /// PostgreSQL 15 moved log_autovacuum_min_duration's default from off to ten minutes, and a server + /// sitting on that default is the shape worth naming: partial, and the cost_note says the default + /// sees only the outlier runs. The same threshold set DELIBERATELY (value differs from boot_val) gets the + /// neutral wording — the audit does not accuse a choice of being a default. + /// + [Fact] + public void TheTenMinuteAutovacuumDefault_IsNamedAsTheDefault_AndADeliberateThresholdIsNot() + { + var onDefault = FacetOf(DarlingPgLoggingAudit.Audit(Pg18Defaults()), "log_autovacuum_min_duration"); + Assert.Equal(DarlingPgLoggingAudit.Partial, onDefault.Verdict); + Assert.Contains("own default since 15", onDefault.CostNote, StringComparison.Ordinal); + Assert.Contains("600000", onDefault.CostNote, StringComparison.Ordinal); + + var rows = Pg18Defaults(); + Replace(rows, Row("log_autovacuum_min_duration", "300000", "ms", source: "configuration file", boot: "600000")); + var deliberate = FacetOf(DarlingPgLoggingAudit.Audit(rows), "log_autovacuum_min_duration"); + Assert.Equal(DarlingPgLoggingAudit.Partial, deliberate.Verdict); + Assert.DoesNotContain("own default since 15", deliberate.CostNote, StringComparison.Ordinal); + Assert.Contains("300000", deliberate.CostNote, StringComparison.Ordinal); + + /* Review's case: an administrator who WRITES 600000 into postgresql.conf has made a choice that + happens to equal the boot value. PostgreSQL says source = 'configuration file', and so does this - + a text comparison against boot_val would have called the choice inaction, the anti-pattern + DarlingPgServerConfigReader.CurrentConfigSql documents avoiding. */ + Replace(rows, Row("log_autovacuum_min_duration", "600000", "ms", source: "configuration file", boot: "600000")); + var explicitDefault = FacetOf(DarlingPgLoggingAudit.Audit(rows), "log_autovacuum_min_duration"); + Assert.Equal(DarlingPgLoggingAudit.Partial, explicitDefault.Verdict); + Assert.DoesNotContain("own default since 15", explicitDefault.CostNote, StringComparison.Ordinal); + } + + /// + /// log_connections across the 17/18 boundary, in the spellings 18.4 stores verbatim. Anything + /// that produces lines is instrumented; the empty string — 18's default and its "off" — is off. + /// + [Theory] + [InlineData("on", true)] + [InlineData("true", true)] + [InlineData("1", true)] + [InlineData("all", true)] + [InlineData("receipt,authentication", true)] + [InlineData("off", false)] + [InlineData("false", false)] + [InlineData("0", false)] + [InlineData("", false)] + public void LogConnections_ReadsBothTheBooleanAndThe18ListForm(string value, bool producesLines) + { + Assert.Equal(producesLines, DarlingPgLoggingAudit.ConnectionLogging(value)); + + var rows = AllOff(); + Replace(rows, Row("log_connections", value, context: "superuser-backend")); + var facet = FacetOf(DarlingPgLoggingAudit.Audit(rows), "log_connections"); + + Assert.Equal(producesLines ? DarlingPgLoggingAudit.Instrumented : DarlingPgLoggingAudit.Off, facet.Verdict); + Assert.Equal(value, facet.Value); + } + + /// + /// A setting missing from the snapshot is unknown — not off, not defaulted, not inferred from the + /// major — and it is kept OUT of the off list, because "off" is an action and "we do not know" is not. + /// + [Fact] + public void AnAbsentSetting_IsUnknown_NotInferred_AndNotCountedAsOff() + { + var rows = AllOff(); + rows.RemoveAll(r => r.Name == "log_lock_waits"); + + var audit = DarlingPgLoggingAudit.Audit(rows); + var facet = FacetOf(audit, "log_lock_waits"); + + Assert.Equal(DarlingPgLoggingAudit.Unknown, facet.Verdict); + Assert.Null(facet.Value); + Assert.Null(facet.Source); + Assert.Contains("not in the stored snapshot", facet.CostNote, StringComparison.Ordinal); + Assert.StartsWith("No remedy is offered", facet.Remedy, StringComparison.Ordinal); + + using var doc = JsonDocument.Parse(DarlingMcpPgLoggingAuditTools.BuildAuditJson("srv", audit)); + var root = doc.RootElement; + Assert.Equal(1, root.GetProperty("unknown_count").GetInt32()); + Assert.Equal(6, root.GetProperty("off_count").GetInt32()); + Assert.Equal(new[] { "log_lock_waits" }, root.GetProperty("unknown_settings").EnumerateArray().Select(e => e.GetString()).ToArray()); + Assert.DoesNotContain("log_lock_waits", root.GetProperty("off_settings").EnumerateArray().Select(e => e.GetString())); + } + + /// + /// A value the parse cannot read is unknown too, rather than a guess in either direction — and + /// its message says the row IS in the snapshot with an unreadable value, which is a different fact from + /// the row being absent and must not be reported as it (review on #3643). Close to unreachable, since + /// pg_settings renders well-formed values, which is exactly why the message is pinned: nothing + /// else would ever exercise it. + /// + [Fact] + public void AnUnparseableValue_IsUnknown_AndSaysItIsInTheSnapshot() + { + var rows = AllOff(); + Replace(rows, Row("log_temp_files", "lots", "kB", "superuser")); + Replace(rows, Row("log_checkpoints", "maybe")); + + var audit = DarlingPgLoggingAudit.Audit(rows); + foreach (var facet in new[] { FacetOf(audit, "log_temp_files"), FacetOf(audit, "log_checkpoints") }) + { + Assert.Equal(DarlingPgLoggingAudit.Unknown, facet.Verdict); + Assert.Contains("IS in the stored snapshot", facet.CostNote, StringComparison.Ordinal); + Assert.Contains("'" + facet.Value + "'", facet.CostNote, StringComparison.Ordinal); + Assert.DoesNotContain("not in the stored snapshot", facet.CostNote, StringComparison.Ordinal); + Assert.Contains("could not read", facet.Remedy, StringComparison.Ordinal); + } + } + + /// + /// pending_restart reaches the row and the summary (review on #3643). It is the one case where the + /// value judged is provably not the value the server will have: the file already holds another and the + /// running server has not restarted, so the remedy is written against a value that changes at the next + /// restart. get_pg_server_config reports it loudly for the same reason; dropping it here would + /// have left the audit's own remedy unqualified on exactly the row where it needs qualifying. + /// + [Fact] + public void APendingRestart_IsCarriedOnTheRow_AndNamedInTheSummary() + { + var rows = AllOff(); + Replace(rows, Row("log_lock_waits", "off", context: "superuser", source: "configuration file", pendingRestart: true)); + + var audit = DarlingPgLoggingAudit.Audit(rows); + var pending = FacetOf(audit, "log_lock_waits"); + Assert.True(pending.PendingRestart); + Assert.Contains("pending_restart is TRUE", pending.RestartNote, StringComparison.Ordinal); + Assert.Contains("RUNNING one", pending.RestartNote, StringComparison.Ordinal); + Assert.Contains("does not carry the file's value", pending.RestartNote, StringComparison.Ordinal); + + var plain = FacetOf(audit, "log_checkpoints"); + Assert.False(plain.PendingRestart); + Assert.Null(plain.RestartNote); + + using var doc = JsonDocument.Parse(DarlingMcpPgLoggingAuditTools.BuildAuditJson("srv", audit)); + var root = doc.RootElement; + Assert.Equal(1, root.GetProperty("pending_restart_count").GetInt32()); + Assert.Equal(new[] { "log_lock_waits" }, root.GetProperty("pending_restart_settings").EnumerateArray().Select(e => e.GetString()).ToArray()); + var row = root.GetProperty("facets").EnumerateArray().Single(f => f.GetProperty("setting").GetString() == "log_lock_waits"); + Assert.True(row.GetProperty("pending_restart").GetBoolean()); + Assert.False(string.IsNullOrWhiteSpace(row.GetProperty("restart_note").GetString())); + Assert.Contains("restart_note", root.GetProperty("note").GetString(), StringComparison.Ordinal); + } + + /// + /// Everything at its recommended value: nothing off, nothing unknown, and every remedy declines to ask + /// for a change — including the two threshold rows whose verdict is partial by design. + /// + [Fact] + public void AFullyInstrumentedTarget_HasNothingOff_AndNoRemedyAsksForAChange() + { + var audit = DarlingPgLoggingAudit.Audit(Recommended()); + + Assert.DoesNotContain(audit.Facets, f => f.Verdict is DarlingPgLoggingAudit.Off or DarlingPgLoggingAudit.Unknown); + Assert.Equal(DarlingPgLoggingAudit.Partial, FacetOf(audit, "log_min_duration_statement").Verdict); + Assert.Equal(6, audit.Facets.Count(f => f.Verdict == DarlingPgLoggingAudit.Instrumented)); + Assert.All(audit.Facets, f => Assert.StartsWith("No change needed", f.Remedy, StringComparison.Ordinal)); + } + + /* ───────────────────────── the remedy per flavour ───────────────────────── */ + + /// + /// One rds.* parameter flips every remedy to the parameter-group form. The registry's + /// engine token cannot make this call — MonitoredEngineKind.Postgres is both self-hosted and RDS + /// for PostgreSQL — and the two need opposite instructions: ALTER SYSTEM is refused on RDS and + /// Aurora, and a parameter group does not exist on a server somebody administers. The evidence is in + /// the same snapshot the audit already reads, and the response names it. + /// + [Fact] + public void AnRdsParameterInTheSnapshot_WordsEveryRemedyForAParameterGroup() + { + var rows = AllOff(); + rows.Add(Row("rds.extensions", "auto_explain,pg_stat_statements,...", context: "postmaster")); + rows.Add(Row("rds.superuser_reserved_connections", "2", context: "postmaster")); + + var audit = DarlingPgLoggingAudit.Audit(rows); + + Assert.True(audit.Managed); + Assert.Contains("2 rds.* parameter(s)", audit.HostingEvidence, StringComparison.Ordinal); + + foreach (var facet in audit.Facets) + { + Assert.StartsWith("Set " + facet.Setting + " = ", facet.Remedy, StringComparison.Ordinal); + Assert.Contains("parameter group", facet.Remedy, StringComparison.Ordinal); + Assert.Contains("ALTER SYSTEM is refused", facet.Remedy, StringComparison.Ordinal); + Assert.Contains("WITHOUT a reboot", facet.Remedy, StringComparison.Ordinal); + Assert.DoesNotContain("pg_reload_conf", facet.Remedy, StringComparison.Ordinal); + } + + /* And the rds.* rows themselves are evidence, not facets. */ + Assert.DoesNotContain(audit.Facets, f => f.Setting.StartsWith("rds.", StringComparison.Ordinal)); + } + + /// + /// The remedy's reload-versus-restart clause comes from the setting's OWN context, not from a table + /// here: superuser-backend settings apply to connections opened after the reload and the remedy + /// says so, a postmaster context (none of the judged settings has one today, so it is forced in + /// the fixture) says restart, and everything else says reload. + /// + [Fact] + public void TheChangeClause_FollowsTheSettingsContext() + { + var rows = AllOff(); + Replace(rows, Row("log_checkpoints", "off", context: "postmaster")); + var audit = DarlingPgLoggingAudit.Audit(rows); + + var perBackend = FacetOf(audit, "log_connections"); + Assert.Equal("reload; applies to connections opened after it", perBackend.ChangeNeeds); + Assert.Contains("new connections take the new one", perBackend.Remedy, StringComparison.Ordinal); + + var forcedStatic = FacetOf(audit, "log_checkpoints"); + Assert.Equal("restart", forcedStatic.ChangeNeeds); + Assert.Contains("RESTART", forcedStatic.Remedy, StringComparison.Ordinal); + + var plain = FacetOf(audit, "log_lock_waits"); + Assert.Equal("reload", plain.ChangeNeeds); + Assert.Contains("a reload, not a restart", plain.Remedy, StringComparison.Ordinal); + + rows.Add(Row("rds.extensions", "x", context: "postmaster")); + var managed = DarlingPgLoggingAudit.Audit(rows); + Assert.Contains("needs a reboot", FacetOf(managed, "log_checkpoints").Remedy, StringComparison.Ordinal); + Assert.Contains("to connections opened after it", FacetOf(managed, "log_connections").Remedy, StringComparison.Ordinal); + } + + /// + /// A per-role or per-database override is the monitoring connection's value, not the server's — the + /// limit the readiness collector states for lc_messages, applied to every GUC here. The row + /// carries a scope_note only when the source says so. + /// + [Fact] + public void ARoleOrDatabaseOverride_GetsAScopeNote_AndAServerSettingDoesNot() + { + var rows = AllOff(); + Replace(rows, Row("log_min_duration_statement", "0", "ms", "superuser", "user", "-1")); + Replace(rows, Row("log_lock_waits", "on", context: "superuser", source: "database", boot: "off")); + Replace(rows, Row("log_temp_files", "0", "kB", "superuser", "configuration file", "-1")); + + var audit = DarlingPgLoggingAudit.Audit(rows); + + Assert.Contains("source is 'user'", FacetOf(audit, "log_min_duration_statement").ScopeNote, StringComparison.Ordinal); + Assert.Contains("source is 'database'", FacetOf(audit, "log_lock_waits").ScopeNote, StringComparison.Ordinal); + Assert.Null(FacetOf(audit, "log_temp_files").ScopeNote); + } + + /// + /// log_lock_waits fires at deadlock_timeout, so its row quotes that setting's value from + /// the same snapshot — and says plainly when the snapshot does not have it, rather than quoting the + /// compiled-in default as if it were this server's. + /// + [Fact] + public void LockWaits_QuotesDeadlockTimeoutFromTheSnapshot_OrSaysItIsMissing() + { + var withTimeout = FacetOf(DarlingPgLoggingAudit.Audit(AllOff()), "log_lock_waits"); + Assert.Contains("deadlock_timeout (1000 ms)", withTimeout.Unlocks, StringComparison.Ordinal); + + var rows = AllOff(); + rows.RemoveAll(r => r.Name == "deadlock_timeout"); + var without = FacetOf(DarlingPgLoggingAudit.Audit(rows), "log_lock_waits"); + Assert.Contains("deadlock_timeout (not in the snapshot)", without.Unlocks, StringComparison.Ordinal); + } + + /* ───────────────────────── the cross-reference to readiness ───────────────────────── */ + + /// + /// Plan capture's own settings are LISTED, with the readiness facet that judges each, and never judged + /// here — no facet row carries them, so there is exactly one verdict on auto_explain in the + /// product and it is get_pg_plan_capture_readiness's. An auto_explain.* GUC missing from + /// the snapshot (the library is not loaded) is a null value, not an absent entry. + /// + [Fact] + public void PlanCaptureSettings_AreListedWithTheirReadinessFacet_AndNeverJudgedHere() + { + var audit = DarlingPgLoggingAudit.Audit(AllOff()); + + Assert.Equal( + new[] { "shared_preload_libraries", "auto_explain.log_min_duration", "log_line_prefix", "lc_messages" }, + audit.JudgedByReadiness.Select(s => s.Setting).ToArray()); + Assert.Equal( + new[] { "library_loaded", "capture_threshold", "plan_attribution", "message_locale" }, + audit.JudgedByReadiness.Select(s => s.ReadinessFacet).ToArray()); + + var autoExplain = Assert.Single(audit.JudgedByReadiness, s => s.Setting == "auto_explain.log_min_duration"); + Assert.Null(autoExplain.Value); + + var locale = Assert.Single(audit.JudgedByReadiness, s => s.Setting == "lc_messages"); + Assert.Equal("en_US.utf8", locale.Value); + Assert.Equal("configuration file", locale.Source); + + Assert.DoesNotContain(audit.Facets, f => f.Setting.StartsWith("auto_explain", StringComparison.Ordinal)); + Assert.DoesNotContain(audit.Facets, f => f.Setting is "shared_preload_libraries" or "log_line_prefix" or "lc_messages"); + + using var doc = JsonDocument.Parse(DarlingMcpPgLoggingAuditTools.BuildAuditJson("srv", audit)); + var block = doc.RootElement.GetProperty("judged_by_readiness"); + Assert.Equal("get_pg_plan_capture_readiness", block.GetProperty("tool").GetString()); + Assert.Contains("NOT judged here", block.GetProperty("note").GetString(), StringComparison.Ordinal); + Assert.Equal(4, block.GetProperty("settings").GetArrayLength()); + } + + /* ───────────────────────── the wire ───────────────────────── */ + + /// + /// The response: the snapshot's time as captured_at (the #3541 A10 spelling every stamped + /// latest read uses), counts that sum to the facet total, the off and unknown settings NAMED, and + /// every facet carrying every prose column — unlocks, consumer, recommended, + /// cost_note, remedy — because the readiness lesson (#3070) was that the remedy column is + /// the one that goes missing on the way to the wire. + /// + [Fact] + public void TheWire_CarriesCapturedAt_TheCounts_TheNamedOffSettings_AndEveryProseColumn() + { + var rows = Pg18Defaults(); + rows.RemoveAll(r => r.Name == "log_disconnections"); + var audit = DarlingPgLoggingAudit.Audit(rows); + + using var doc = JsonDocument.Parse(DarlingMcpPgLoggingAuditTools.BuildAuditJson("srv", audit)); + var root = doc.RootElement; + + Assert.Equal("srv", root.GetProperty("server").GetString()); + Assert.Equal("logging_audit", root.GetProperty("status").GetString()); + Assert.Equal(Stamp.ToString("o"), root.GetProperty("captured_at").GetString()); + Assert.False(root.TryGetProperty("as_of", out _)); + Assert.Contains("not the live server", root.GetProperty("source").GetString(), StringComparison.Ordinal); + Assert.Equal("self-hosted", root.GetProperty("hosting").GetString()); + Assert.False(root.TryGetProperty("hours_back", out _)); + + var total = root.GetProperty("total").GetInt32(); + Assert.Equal(7, total); + Assert.Equal(total, + root.GetProperty("instrumented_count").GetInt32() + root.GetProperty("partial_count").GetInt32() + + root.GetProperty("off_count").GetInt32() + root.GetProperty("unknown_count").GetInt32()); + + /* 18 defaults: checkpoints on (instrumented); autovacuum at ten minutes (partial); statements, lock + waits, temp files and connections off; disconnections removed from the fixture (unknown). */ + Assert.Equal(1, root.GetProperty("instrumented_count").GetInt32()); + Assert.Equal(1, root.GetProperty("partial_count").GetInt32()); + Assert.Equal( + new[] { "log_min_duration_statement", "log_lock_waits", "log_temp_files", "log_connections" }, + root.GetProperty("off_settings").EnumerateArray().Select(e => e.GetString()).ToArray()); + Assert.Equal(new[] { "log_disconnections" }, + root.GetProperty("unknown_settings").EnumerateArray().Select(e => e.GetString()).ToArray()); + + foreach (var facet in root.GetProperty("facets").EnumerateArray()) + { + foreach (var key in new[] { "setting", "verdict", "unlocks", "consumer", "recommended", "cost_note", "remedy" }) + { + Assert.False(string.IsNullOrWhiteSpace(facet.GetProperty(key).GetString()), + $"{facet.GetProperty("setting").GetString()}.{key} reached the wire empty"); + } + + Assert.True(facet.TryGetProperty("value", out _)); + Assert.True(facet.TryGetProperty("source", out _)); + Assert.True(facet.TryGetProperty("scope_note", out _)); + Assert.True(facet.TryGetProperty("pending_restart", out _)); + Assert.True(facet.TryGetProperty("restart_note", out _)); + } + + Assert.Equal(0, root.GetProperty("pending_restart_count").GetInt32()); + } + + /// + /// Every consumer says PLANNED today, and this pin is meant to go red. The issue's sequencing + /// note says the audit earns its keep once #3601's pipeline and its parser families (#3602, #3603) + /// consume the lines, and none of those ships. Each facet says so beside the read that exists today. + /// When a consumer lands, the facet whose lines it reads must stop saying planned — and this assertion + /// is what makes that a deliberate edit rather than stale prose an agent plans against. + /// + [Fact] + public void EveryConsumer_IsHonestlyPlanned_UntilTheLogPipelineShips() + { + var audit = DarlingPgLoggingAudit.Audit(AllOff()); + + Assert.All(audit.Facets, f => Assert.StartsWith("PLANNED", f.Consumer, StringComparison.Ordinal)); + + /* And each names its issue and the read that exists today, so "planned" is a pointer and not a shrug. */ + Assert.Contains("#3601", FacetOf(audit, "log_min_duration_statement").Consumer, StringComparison.Ordinal); + Assert.Contains("get_pg_top_queries", FacetOf(audit, "log_min_duration_statement").Consumer, StringComparison.Ordinal); + Assert.Contains("#3601", FacetOf(audit, "log_lock_waits").Consumer, StringComparison.Ordinal); + Assert.Contains("get_pg_blocking", FacetOf(audit, "log_lock_waits").Consumer, StringComparison.Ordinal); + Assert.Contains("#3602", FacetOf(audit, "log_temp_files").Consumer, StringComparison.Ordinal); + Assert.Contains("#3603", FacetOf(audit, "log_autovacuum_min_duration").Consumer, StringComparison.Ordinal); + Assert.Contains("get_pg_autovacuum_health", FacetOf(audit, "log_autovacuum_min_duration").Consumer, StringComparison.Ordinal); + Assert.Contains("get_pg_write_stats", FacetOf(audit, "log_checkpoints").Consumer, StringComparison.Ordinal); + } + + /// An empty snapshot is the tool's empty/not_collected path, never an audit of nothing. + [Fact] + public void AnEmptySnapshot_IsRefusedByTheJudgment() + { + Assert.Throws(() => DarlingPgLoggingAudit.Audit(Array.Empty())); + } + + /* ───────────────────────── the tool's contract ───────────────────────── */ + + /// + /// The description is what an agent plans against, so the claims it must keep making are pinned: it + /// reads the STORED snapshot and not the live server, it names all seven settings, it defers plan + /// capture's settings to the readiness read, it defines partial, and it says PostgreSQL-only. + /// + [Fact] + public void TheToolDescription_StatesWhatItJudges_AndThatItReadsStoredConfig() + { + var method = typeof(DarlingMcpPgLoggingAuditTools).GetMethod(nameof(DarlingMcpPgLoggingAuditTools.GetPgLoggingAudit))!; + + Assert.Equal("get_pg_logging_audit", method.GetCustomAttribute()!.Name); + var description = method.GetCustomAttribute()!.Description; + + foreach (var setting in DarlingPgLoggingAudit.JudgedSettings) + { + Assert.Contains(setting, description, StringComparison.Ordinal); + } + + Assert.Contains("STORED configuration snapshot", description, StringComparison.Ordinal); + Assert.Contains("never the live server", description, StringComparison.Ordinal); + Assert.Contains("LATEST IS A TIME", description, StringComparison.Ordinal); + Assert.Contains("captured_at", description, StringComparison.Ordinal); + Assert.Contains("get_pg_plan_capture_readiness", description, StringComparison.Ordinal); + Assert.Contains("partial means a THRESHOLD is filtering", description, StringComparison.Ordinal); + Assert.Contains("PLANNED", description, StringComparison.Ordinal); + Assert.Contains("parameter group on RDS/Aurora", description, StringComparison.Ordinal); + Assert.EndsWith("PostgreSQL-only.", description, StringComparison.Ordinal); + + /* A latest-snapshot read: no window and no anchor, the get_pg_server_config convention. The + AsOfWindowAnchorTests pins hold the catalog to the same fact from the other side. */ + var parameters = method.GetParameters().Select(p => p.Name).ToArray(); + Assert.DoesNotContain("hours_back", parameters); + Assert.DoesNotContain("as_of", parameters); + } + + /// + /// Reachable from the web, with a catalog entry that binds only the server — the same fact from the + /// dispatch side. Registration with the MCP host is McpToolTypeRegistrationTests' derived pin. + /// + [Fact] + public void TheRead_IsDispatched_AndItsCatalogEntryBindsOnlyTheServer() + { + Assert.Contains("get_pg_logging_audit", DarlingWebEndpoints.BuildReadDispatch().Keys); + + var descriptor = DarlingWebEndpoints.CatalogDescriptors["get_pg_logging_audit"]; + Assert.Equal(new[] { "server" }, descriptor.Params.Select(p => p.Name).ToArray()); + Assert.Contains("get_pg_plan_capture_readiness", descriptor.Description, StringComparison.Ordinal); + } +} + +/// +/// The audit end to end against a live store (#3607): two servers whose newest pg_server_config +/// snapshots model the two hosting flavours, an OLDER snapshot on each that must lose to the newest, a +/// client-sourced row that must not be read as the server's setting, and a third server with no +/// snapshot at all — the empty path. Executed on this lane's 18.4 rig through the mactest harness +/// before CI. +/// +[Collection("live-postgres")] +public sealed class DarlingMcpPgLoggingAuditLivePostgresTests +{ + private const string SelfHostedName = "darling-pg-logging-audit-self-hosted"; + private const string ManagedName = "darling-pg-logging-audit-managed"; + private const string EmptyName = "darling-pg-logging-audit-empty"; + private static readonly int SelfHostedId = ServerIdHelper.GetDeterministicHashCode(SelfHostedName); + private static readonly int ManagedId = ServerIdHelper.GetDeterministicHashCode(ManagedName); + private static readonly int EmptyId = ServerIdHelper.GetDeterministicHashCode(EmptyName); + + private static string? ConnectionString => Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + + [Fact] + public async Task TheAudit_JudgesTheNewestSnapshot_PerFlavour_AndReportsAnEmptyStoreHonestly() + { + var cs = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(cs), + "Set DARLING_TEST_PG to a Postgres connection string to run the live logging-audit test."); + + var ct = TestContext.Current.CancellationToken; + using var connection = new NpgsqlConnection(cs); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + await DeleteRowsAsync(connection, ct); + + await using var postgres = NpgsqlDataSource.Create(cs!); + var bodySucceeded = false; + + try + { + foreach (var (id, name) in new[] { (SelfHostedId, SelfHostedName), (ManagedId, ManagedName), (EmptyId, EmptyName) }) + { + await DarlingMcpTestData.RegisterServerAsync(connection, id, name, ct); + await DarlingMcpTestData.ExecAsync(connection, ct, + "UPDATE servers SET engine_kind = $2 WHERE server_id = $1", id, MonitoredEngineKind.Postgres); + } + + var older = DarlingMcpTestData.TruncateToSeconds(DateTime.UtcNow).AddHours(-2); + var newest = DarlingMcpTestData.TruncateToSeconds(DateTime.UtcNow).AddMinutes(-20); + + /* Self-hosted: the OLDER snapshot had lock waits on; the newest has it off. The audit must + report the newest, and the client-sourced row saying statements are logged at 0 is the + monitoring session's own SET and must not become the server's setting. */ + await SeedAsync(connection, ct, SelfHostedId, SelfHostedName, older, "log_lock_waits", "on", null, "superuser", "configuration file", "off"); + await SeedAsync(connection, ct, SelfHostedId, SelfHostedName, newest, "log_lock_waits", "off", null, "superuser", "default", "off"); + await SeedAsync(connection, ct, SelfHostedId, SelfHostedName, newest, "log_min_duration_statement", "-1", "ms", "superuser", "default", "-1"); + await SeedAsync(connection, ct, SelfHostedId, SelfHostedName, newest, "log_min_duration_statement", "0", "ms", "superuser", "client", "-1"); + await SeedAsync(connection, ct, SelfHostedId, SelfHostedName, newest, "log_temp_files", "0", "kB", "superuser", "configuration file", "-1"); + await SeedAsync(connection, ct, SelfHostedId, SelfHostedName, newest, "log_autovacuum_min_duration", "600000", "ms", "sighup", "default", "600000"); + await SeedAsync(connection, ct, SelfHostedId, SelfHostedName, newest, "log_checkpoints", "on", null, "sighup", "default", "on"); + await SeedAsync(connection, ct, SelfHostedId, SelfHostedName, newest, "log_connections", "", null, "superuser-backend", "default", ""); + await SeedAsync(connection, ct, SelfHostedId, SelfHostedName, newest, "log_disconnections", "off", null, "superuser-backend", "default", "off"); + await SeedAsync(connection, ct, SelfHostedId, SelfHostedName, newest, "deadlock_timeout", "1000", "ms", "superuser", "default", "1000"); + await SeedAsync(connection, ct, SelfHostedId, SelfHostedName, newest, "lc_messages", "C", null, "superuser", "configuration file", ""); + await SeedAsync(connection, ct, SelfHostedId, SelfHostedName, newest, "log_line_prefix", "%m [%p] %q%u@%d %Q ", null, "sighup", "configuration file", "%m [%p] "); + await SeedAsync(connection, ct, SelfHostedId, SelfHostedName, newest, "shared_preload_libraries", "pg_stat_statements", null, "postmaster", "configuration file", ""); + + /* Managed: rds.* in the snapshot, a per-role override on temp files. */ + await SeedAsync(connection, ct, ManagedId, ManagedName, newest, "rds.extensions", "auto_explain,pg_stat_statements", null, "postmaster", "default", ""); + await SeedAsync(connection, ct, ManagedId, ManagedName, newest, "log_lock_waits", "on", null, "superuser", "configuration file", "off"); + await SeedAsync(connection, ct, ManagedId, ManagedName, newest, "log_min_duration_statement", "1000", "ms", "superuser", "configuration file", "-1"); + await SeedAsync(connection, ct, ManagedId, ManagedName, newest, "log_temp_files", "10240", "kB", "superuser", "user", "-1"); + await SeedAsync(connection, ct, ManagedId, ManagedName, newest, "log_autovacuum_min_duration", "-1", "ms", "sighup", "configuration file", "600000"); + await SeedAsync(connection, ct, ManagedId, ManagedName, newest, "log_checkpoints", "on", null, "sighup", "default", "on"); + await SeedAsync(connection, ct, ManagedId, ManagedName, newest, "log_connections", "on", null, "superuser-backend", "configuration file", ""); + await SeedAsync(connection, ct, ManagedId, ManagedName, newest, "log_disconnections", "on", null, "superuser-backend", "configuration file", "off", pendingRestart: true); + + /* ── self-hosted ── */ + var self = JsonDocument.Parse(await DarlingMcpPgLoggingAuditTools.GetPgLoggingAudit(postgres, SelfHostedName)).RootElement; + + Assert.Equal("logging_audit", self.GetProperty("status").GetString()); + Assert.Equal(SelfHostedName, self.GetProperty("server").GetString()); + Assert.Equal(DateTime.SpecifyKind(newest, DateTimeKind.Utc).ToString("o"), self.GetProperty("captured_at").GetString()); + Assert.Equal("self-hosted", self.GetProperty("hosting").GetString()); + + var selfFacets = self.GetProperty("facets").EnumerateArray().ToDictionary(f => f.GetProperty("setting").GetString()!); + + /* The newest snapshot won: off, not the older snapshot's on. */ + Assert.Equal("off", selfFacets["log_lock_waits"].GetProperty("verdict").GetString()); + Assert.StartsWith("ALTER SYSTEM SET log_lock_waits = on;", selfFacets["log_lock_waits"].GetProperty("remedy").GetString(), StringComparison.Ordinal); + + /* The client-sourced 0 was excluded: the server's own -1 is what was judged. */ + Assert.Equal("off", selfFacets["log_min_duration_statement"].GetProperty("verdict").GetString()); + Assert.Equal("-1", selfFacets["log_min_duration_statement"].GetProperty("value").GetString()); + + Assert.Equal("instrumented", selfFacets["log_temp_files"].GetProperty("verdict").GetString()); + Assert.Equal("partial", selfFacets["log_autovacuum_min_duration"].GetProperty("verdict").GetString()); + Assert.Equal("instrumented", selfFacets["log_checkpoints"].GetProperty("verdict").GetString()); + Assert.Equal("off", selfFacets["log_connections"].GetProperty("verdict").GetString()); + Assert.Equal("off", selfFacets["log_disconnections"].GetProperty("verdict").GetString()); + Assert.Equal(0, self.GetProperty("unknown_count").GetInt32()); + + var readiness = self.GetProperty("judged_by_readiness").GetProperty("settings").EnumerateArray() + .ToDictionary(s => s.GetProperty("setting").GetString()!); + Assert.Equal("C", readiness["lc_messages"].GetProperty("value").GetString()); + Assert.Equal("pg_stat_statements", readiness["shared_preload_libraries"].GetProperty("value").GetString()); + Assert.Equal(JsonValueKind.Null, readiness["auto_explain.log_min_duration"].GetProperty("value").ValueKind); + + /* ── managed ── */ + var managed = JsonDocument.Parse(await DarlingMcpPgLoggingAuditTools.GetPgLoggingAudit(postgres, ManagedName)).RootElement; + + Assert.Equal("managed (RDS/Aurora)", managed.GetProperty("hosting").GetString()); + var managedFacets = managed.GetProperty("facets").EnumerateArray().ToDictionary(f => f.GetProperty("setting").GetString()!); + + Assert.Equal("off", managedFacets["log_autovacuum_min_duration"].GetProperty("verdict").GetString()); + var remedy = managedFacets["log_autovacuum_min_duration"].GetProperty("remedy").GetString()!; + Assert.StartsWith("Set log_autovacuum_min_duration = 0 in the DB parameter group", remedy, StringComparison.Ordinal); + Assert.DoesNotContain("pg_reload_conf", remedy, StringComparison.Ordinal); + + Assert.Equal("partial", managedFacets["log_temp_files"].GetProperty("verdict").GetString()); + Assert.Contains("source is 'user'", managedFacets["log_temp_files"].GetProperty("scope_note").GetString(), StringComparison.Ordinal); + Assert.Equal("partial", managedFacets["log_min_duration_statement"].GetProperty("verdict").GetString()); + Assert.StartsWith("No change needed", managedFacets["log_min_duration_statement"].GetProperty("remedy").GetString(), StringComparison.Ordinal); + Assert.Equal(new[] { "log_autovacuum_min_duration" }, + managed.GetProperty("off_settings").EnumerateArray().Select(e => e.GetString()).ToArray()); + /* deadlock_timeout was not seeded for this server, and the row says so rather than quoting 1000. */ + Assert.Contains("not in the snapshot", managedFacets["log_lock_waits"].GetProperty("unlocks").GetString(), StringComparison.Ordinal); + /* The one pending_restart row travelled from the store to the wire and into the summary. */ + Assert.True(managedFacets["log_disconnections"].GetProperty("pending_restart").GetBoolean()); + Assert.Equal(new[] { "log_disconnections" }, + managed.GetProperty("pending_restart_settings").EnumerateArray().Select(e => e.GetString()).ToArray()); + Assert.False(managedFacets["log_lock_waits"].GetProperty("pending_restart").GetBoolean()); + + /* ── no snapshot at all ── */ + var empty = JsonDocument.Parse(await DarlingMcpPgLoggingAuditTools.GetPgLoggingAudit(postgres, EmptyName)).RootElement; + Assert.Equal("empty", empty.GetProperty("status").GetString()); + Assert.Contains("nothing to audit", empty.GetProperty("message").GetString(), StringComparison.Ordinal); + Assert.Contains("not a verdict", empty.GetProperty("message").GetString(), StringComparison.Ordinal); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, DeleteRowsAsync); + } + } + + private static Task SeedAsync( + NpgsqlConnection connection, CancellationToken ct, int serverId, string serverName, DateTime collectionTime, + string name, string? setting, string? unit, string context, string source, string? bootVal, bool pendingRestart = false) => + DarlingMcpTestData.ExecAsync(connection, ct, @" +INSERT INTO pg_server_config + (collection_id, collection_time, server_id, server_name, name, setting, unit, category, context, vartype, + source, boot_val, reset_val, sourcefile, sourceline, pending_restart, short_desc) +VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11, $12, $13, $14, $15, $16, $17)", + CollectionIdGenerator.Next(), collectionTime, serverId, serverName, name, setting, unit, + "Reporting and Logging / What to Log", context, unit is null ? "bool" : "integer", + source, bootVal, setting, null, 0, pendingRestart, null); + + private static async Task DeleteRowsAsync(NpgsqlConnection connection, CancellationToken ct) + { + await DarlingMcpTestData.ExecAsync(connection, ct, + "DELETE FROM pg_server_config WHERE server_id IN ($1, $2, $3)", SelfHostedId, ManagedId, EmptyId); + await DarlingMcpTestData.ExecAsync(connection, ct, + "DELETE FROM servers WHERE server_id IN ($1, $2, $3)", SelfHostedId, ManagedId, EmptyId); + } +} diff --git a/Darling/Darling.Tests/McpLatestSnapshotStampTests.cs b/Darling/Darling.Tests/McpLatestSnapshotStampTests.cs index 7f3ce34b3..8be610f3c 100644 --- a/Darling/Darling.Tests/McpLatestSnapshotStampTests.cs +++ b/Darling/Darling.Tests/McpLatestSnapshotStampTests.cs @@ -133,6 +133,24 @@ public static readonly (Type Tools, string ToolName, string LiteFile, Shape Shap /// holds the shape; the sweep only needs to know it is accounted for. public static readonly string[] ThreeClockTools = ["get_server_summary"]; + /// + /// Stamped latest reads that exist on ONE SKU. is a roster of PAIRS — every entry + /// names the Lite file its twin lives in and reads both — so a Darling-only + /// tool cannot sit in it without a Lite file to point at. These are held to the Stamped dialect by + /// below with the SAME three assertions the + /// roster's Stamped entries get (a captured_at on the payload, LATEST IS A TIME in the description, + /// neither knob in the signature) — only the Lite half is absent, because the SKU is. + /// + /// get_pg_logging_audit (#3607) judges seven logging GUCs from the newest + /// pg_server_config snapshot; Lite has no PostgreSQL target, so there is no twin for the roster to + /// pair it with. The same architectural boundary CrossAppMcpToolInventoryPinTests records for every + /// get_pg_* read. + /// + public static readonly (Type Tools, string ToolName)[] DarlingOnlyStamped = + [ + (typeof(DarlingMcpPgLoggingAuditTools), "get_pg_logging_audit"), + ]; + /* ───────────────────────── the discriminators ───────────────────────── */ /// The stamp, as a payload key. @@ -201,6 +219,32 @@ public void StampedTools_TakeNoWindowAndNoAnchor_OnBothSkus() } } + /// + /// The Stamped dialect, held on the Darling-only reads with the roster's own three assertions — the stamp + /// on the payload, LATEST IS A TIME in the description, neither knob in the signature. Read from the + /// Darling source only, because there is no Lite half to read. + /// + [Fact] + public void DarlingOnlyStampedTools_KeepTheStampedDialect() + { + Assert.NotEmpty(DarlingOnlyStamped); + + foreach (var (type, name) in DarlingOnlyStamped) + { + var body = ToolBody(ReadRepoFileLf(DarlingFileOf(type).Split('/')), name); + Assert.True(CapturedAtKey.IsMatch(Strip(body)), + $"Darling {name}: no `captured_at =` on the payload — a latest read that never says when it was captured"); + + var description = ToolMethod(type, name).GetCustomAttribute()!.Description; + Assert.Contains("captured_at", description, StringComparison.Ordinal); + Assert.Contains("LATEST IS A TIME", description, StringComparison.Ordinal); + + var parameters = ToolMethod(type, name).GetParameters().Select(p => p.Name).ToArray(); + Assert.DoesNotContain("hours_back", parameters); + Assert.DoesNotContain("as_of", parameters); + } + } + /// /// A SearchBound tool's hours_back says it is the SEARCH span in the same words on both SKUs, the /// tool takes the anchor, and the payload carries the anchored distance — the three things that make a @@ -349,6 +393,7 @@ public void EveryLatestReadTool_IsInTheRoster_OrANamedAllowance_AndEveryAllowanc var residualsSeen = new HashSet(StringComparer.Ordinal); var lookupsSeen = new HashSet(StringComparer.Ordinal); var threeClocksSeen = new HashSet(StringComparer.Ordinal); + var darlingOnlySeen = new HashSet(StringComparer.Ordinal); var examined = 0; foreach (var (file, source) in AllDarlingToolSources()) @@ -405,6 +450,14 @@ public void EveryLatestReadTool_IsInTheRoster_OrANamedAllowance_AndEveryAllowanc continue; } + if (DarlingOnlyStamped.Any(t => t.ToolName == toolName)) + { + Assert.True(CapturedAtKey.IsMatch(body), + $"{file} {toolName}: listed as a Darling-only STAMPED read but publishes no captured_at"); + darlingOnlySeen.Add(toolName); + continue; + } + Assert.Fail($"{file} {toolName}: calls a latest-snapshot reader and is in no list here — give it a Shape in LatestTools (and stamp it) or name it as an allowance with its reason"); } } @@ -418,6 +471,8 @@ widened exemption. 26 latest-reading tool bodies at the time of writing. */ "ThreeClockTools no longer matches what the sweep finds: " + string.Join(", ", ThreeClockTools.Except(threeClocksSeen))); Assert.True(StampedUnderCollectionTime.ToHashSet(StringComparer.Ordinal).SetEquals(allowancesUsed), "StampedUnderCollectionTime no longer matches what the sweep finds: " + string.Join(", ", StampedUnderCollectionTime.Except(allowancesUsed))); + Assert.True(DarlingOnlyStamped.Select(t => t.ToolName).ToHashSet(StringComparer.Ordinal).SetEquals(darlingOnlySeen), + "DarlingOnlyStamped no longer matches what the sweep finds: " + string.Join(", ", DarlingOnlyStamped.Select(t => t.ToolName).Except(darlingOnlySeen))); Assert.True(UnstampedLatestReadsPendingA10.ToHashSet(StringComparer.Ordinal).SetEquals(residualsSeen), "UnstampedLatestReadsPendingA10 no longer matches what the sweep finds: " + string.Join(", ", UnstampedLatestReadsPendingA10.Except(residualsSeen))); } @@ -445,6 +500,7 @@ widened exemption. 26 latest-reading tool bodies at the time of writing. */ [InlineData(nameof(DarlingCurrentConfigReader.TraceFlagsSql), "capture_time")] [InlineData(nameof(DarlingConfigHistoryReader.DatabaseScopedConfigSql), "capture_time")] [InlineData(nameof(DarlingConfigHistoryReader.QueryStoreHealthSql), "capture_time")] + [InlineData(nameof(DarlingPgLoggingAuditReader.NewestSnapshotSql), "collection_time")] public void EveryStampedRead_SelectsItsStampColumn_OnTheRowStatement(string sqlName, string column) { var sql = ReaderSql(sqlName); @@ -581,6 +637,7 @@ public void TheDiscriminators_FlagTheDefectShapes_AndPassTheFixedOnes() nameof(DarlingConfigHistoryReader.QueryStoreHealthSql) => DarlingConfigHistoryReader.QueryStoreHealthSql, nameof(DarlingMemoryGrantReader.ResourceSemaphoreWindowSql) => DarlingMemoryGrantReader.ResourceSemaphoreWindowSql, nameof(DarlingMemoryGrantReader.MemoryGrantsWindowSql) => DarlingMemoryGrantReader.MemoryGrantsWindowSql, + nameof(DarlingPgLoggingAuditReader.NewestSnapshotSql) => DarlingPgLoggingAuditReader.NewestSnapshotSql, _ => throw new ArgumentOutOfRangeException(nameof(sqlName), sqlName, "not a read this census names"), }; diff --git a/Darling/Darling.Tests/ServerPageTabsTests.cs b/Darling/Darling.Tests/ServerPageTabsTests.cs index eb7438233..1ff7a3cb4 100644 --- a/Darling/Darling.Tests/ServerPageTabsTests.cs +++ b/Darling/Darling.Tests/ServerPageTabsTests.cs @@ -95,6 +95,11 @@ public sealed class ServerPageTabsTests ["get_pg_top_queries"] = "pg_statement_stats", ["get_pg_plans"] = "pg_plan_capture", ["get_pg_plan_capture_readiness"] = "pg_plan_capture_readiness", + /* #3607: a READ over the config collector's table, not a collector of its own - the audit is + computed from the newest pg_server_config snapshot, so its not_collected gate names that + collector, which is what this map records. Two reads over one collector is the same shape + as get_pg_server_config / get_pg_server_config_changes above. */ + ["get_pg_logging_audit"] = "pg_server_config", ["get_pg_blocking"] = "pg_blocking", ["get_pg_io_stats"] = "pg_io_stats", ["get_pg_autovacuum_health"] = "pg_autovacuum_stats", diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingWebEndpoints.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingWebEndpoints.cs index 9cf5d2486..28915ad3a 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingWebEndpoints.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingWebEndpoints.cs @@ -1916,6 +1916,7 @@ private static CatalogRead R(string category, string description, params Catalog ["get_pg_top_queries"] = R(CatData, "Top PostgreSQL query shapes by total execution time (Aurora targets).", PServer(), PHours(24), PLimit(20), PAsOf()), ["get_pg_plans"] = R(CatData, "Captured PostgreSQL execution plans, grouped by shape. Plans are redacted at collection.", PServer(), PHours(24), PLimit(10), PText("query_id"), PAsOf()), ["get_pg_plan_capture_readiness"] = R(CatData, "Whether a PostgreSQL target can capture execution plans at all, facet by facet, with the remedy for each step that is not in place. Read this when a plan or target-log read is empty.", PServer(), PHours(24), PLimit(25), PAsOf()), + ["get_pg_logging_audit"] = R(CatData, "Whether a PostgreSQL target's logging settings (log_min_duration_statement, log_lock_waits, log_temp_files, log_autovacuum_min_duration, log_checkpoints, log_connections, log_disconnections) are producing the lines they could, per setting: verdict, what it unlocks, the recommended value with its cost, and the remedy in the hosting flavour's syntax. Judged from the newest stored pg_server_config snapshot, not the live server; plan capture's own settings are listed and pointed at get_pg_plan_capture_readiness.", PServer()), ["get_pg_wraparound_risk"] = R(CatData, "PostgreSQL XID/MultiXact freeze headroom per database.", PServer(), PHours(24), PAsOf()), ["get_pg_xmin_horizon"] = R(CatData, "What is holding back the PostgreSQL xmin horizon, by cause.", PServer(), PHours(24), PAsOf()), ["get_pg_replication_slots"] = R(CatData, "PostgreSQL replication slot health, including whether retained WAL is still growing.", PServer(), PHours(24), PAsOf()), @@ -2615,6 +2616,7 @@ logger is the tool's logger seat — the web host's SERVICE logger when MapAll b cannot parse exactly rather than silently matching nothing. */ ["get_pg_plans"] = (c, pg, an) => DarlingMcpPgPlanTools.GetPgPlans(pg, Server(c), Hours(c, 24), Rows(c, "limit", 10), Str(c, "query_id"), AsOf(c)), ["get_pg_plan_capture_readiness"] = (c, pg, an) => DarlingMcpPgPlanTools.GetPgPlanCaptureReadiness(pg, Server(c), Hours(c, 24), Rows(c, "limit", 25), as_of: AsOf(c)), + ["get_pg_logging_audit"] = (c, pg, an) => DarlingMcpPgLoggingAuditTools.GetPgLoggingAudit(pg, Server(c)), ["get_pg_wraparound_risk"] = (c, pg, an) => DarlingMcpPgWraparoundTools.GetPgWraparoundRisk(pg, Server(c), Hours(c, 24), as_of: AsOf(c)), ["get_pg_xmin_horizon"] = (c, pg, an) => DarlingMcpPgXminTools.GetPgXminHorizon(pg, Server(c), Hours(c, 24), as_of: AsOf(c)), ["get_pg_replication_slots"] = (c, pg, an) => DarlingMcpPgSlotTools.GetPgReplicationSlots(pg, Server(c), Hours(c, 24), as_of: AsOf(c)), diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpHostService.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpHostService.cs index f66fc3af4..e57692fea 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpHostService.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpHostService.cs @@ -625,6 +625,15 @@ pg_statement_stats collector. Carries Aurora's I/O source split and per-statemen target can capture a plan at all, facet by facet with the remedy for each, which is the read somebody needs the moment the plans one comes back empty. */ .WithGeminiCompatibleTools() + /* get_pg_logging_audit (#3607) - the rest of the logging surface, in readiness's shape: + log_lock_waits, log_temp_files, log_autovacuum_min_duration, log_checkpoints, + log_connections / log_disconnections and log_min_duration_statement, each judged from + the stored pg_server_config snapshot with what it unlocks, the recommended value and its + cost, and the remedy in the hosting flavour's syntax. Registered beside the plan tools + because it is the other half of one onboarding question - is this target telling us + everything it could - and lists plan capture's own settings with a pointer to the + readiness read rather than judging them twice. */ + .WithGeminiCompatibleTools() /* get_pg_wraparound_risk — XID/MultiXact freeze headroom, the highest-consequence PostgreSQL signal and one with no SQL Server counterpart. Not Aurora-gated. */ .WithGeminiCompatibleTools() diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs index 7eff50849..f83453499 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs @@ -62,7 +62,7 @@ public static string Build(DarlingPeerDirectory.Snapshot peers) ## Tool Reference - This server exposes 151 tools. 86 are the same names Performance Monitor Lite exposes, spanning diagnostic analysis, plan analysis, data reads at core and diagnostic depth, resource contention + jobs, trends, system-health parse-on-read, alerts + health overview, and the Default Trace. The remaining 65 are unique to Darling: thirty-three are the PostgreSQL reads (Aurora/PostgreSQL targets only Darling's central store can hold), eight are the Custom Views tools (seven manage the saved views — the one view-authoring write surface — and `describe_custom_view_catalog` returns the read-only compose vocabulary those authoring tools draw from), eight are the custom-alert-rule tools (`create_custom_alert_rule` / `update_custom_alert_rule` / `delete_custom_alert_rule` manage the user-authored alert rules, the one alert-authoring write surface, while `get_custom_alert_rule` / `list_custom_alert_rules` read them back, `validate_custom_alert_rule` checks a rule definition against the same compose catalog without saving it, `test_custom_alert_rule` evaluates a rule's metric NOW on each in-scope server and reports whether it would breach without delivering or persisting anything, and `list_custom_alert_templates` lists the built-in starter rule templates to browse and create from), five are alert-tuning write tools (`update_alert_settings` tunes the alert engine's thresholds; `create_mute_rule` / `update_mute_rule` / `delete_mute_rule` / `set_mute_rule_enabled` manage the mute rules, `update_mute_rule` editing a rule in place and `set_mute_rule_enabled` taking one out of force and putting it back — both without destroying it or resetting its creation date) that write only the shared alert configuration in the monitoring store, two are server-onboarding write tools (`add_servers` bulk-adds monitored servers; `remove_server` removes one) that add or remove rows in the monitoring store's monitored-server registry, `get_fleet_overview` and `get_ag_health` are the two cross-server reads of the fleet's CURRENT state only a central store can answer, `get_sweep_reports` is the cross-server read WITH MEMORY — the scheduled fleet sweep's persisted whole-fleet reports (the timeline with each sweep's document embedded, one sweep in full with its would-have-paged ledger, and the watch-item worklist), `get_store_metrics` reads the monitoring store's OWN hourly size/compression/growth series for capacity forecasting, `get_store_log` reads what the monitoring store's OWN PostgreSQL server log recorded as a per-class census with its capture denominator (the self-monitoring that shows the store's half of a client-side symptom), `get_collector_cost` reads the tool's OWN per-collector cost on the monitored servers (the self-monitoring that flags a collector regressing into a hog), `get_collector_stall_probes` reads the out-of-band server-wide wait samples this tool takes while one of its own collectors is stalled mid-read — the only surface here that reports what a monitored instance was doing inside the window the sequential sweep records nothing in — `get_oversized_plan_backlog` reads the backlog of cached plans the capture cap declined and what the out-of-band sweep has since done about each one (the self-monitoring that tells a sweep whose fetch half is working from one that has never once succeeded), and `get_blocking` is Darling's name for the blocked-process-report read that Lite exposes as `get_blocked_process_reports` — a naming difference, not a capability gap. Every data-read tool reads the data the collectors already captured into the store — a stored read, never a live query against the monitored server. + This server exposes 152 tools. 86 are the same names Performance Monitor Lite exposes, spanning diagnostic analysis, plan analysis, data reads at core and diagnostic depth, resource contention + jobs, trends, system-health parse-on-read, alerts + health overview, and the Default Trace. The remaining 66 are unique to Darling: thirty-four are the PostgreSQL reads (Aurora/PostgreSQL targets only Darling's central store can hold; two of them are the target-onboarding pair — `get_pg_plan_capture_readiness` judges plan capture's preconditions facet by facet and `get_pg_logging_audit` judges the rest of the logging surface setting by setting from the stored configuration, each with the remedy per finding in the hosting flavour's syntax — which together answer whether a target is telling us everything it could), eight are the Custom Views tools (seven manage the saved views — the one view-authoring write surface — and `describe_custom_view_catalog` returns the read-only compose vocabulary those authoring tools draw from), eight are the custom-alert-rule tools (`create_custom_alert_rule` / `update_custom_alert_rule` / `delete_custom_alert_rule` manage the user-authored alert rules, the one alert-authoring write surface, while `get_custom_alert_rule` / `list_custom_alert_rules` read them back, `validate_custom_alert_rule` checks a rule definition against the same compose catalog without saving it, `test_custom_alert_rule` evaluates a rule's metric NOW on each in-scope server and reports whether it would breach without delivering or persisting anything, and `list_custom_alert_templates` lists the built-in starter rule templates to browse and create from), five are alert-tuning write tools (`update_alert_settings` tunes the alert engine's thresholds; `create_mute_rule` / `update_mute_rule` / `delete_mute_rule` / `set_mute_rule_enabled` manage the mute rules, `update_mute_rule` editing a rule in place and `set_mute_rule_enabled` taking one out of force and putting it back — both without destroying it or resetting its creation date) that write only the shared alert configuration in the monitoring store, two are server-onboarding write tools (`add_servers` bulk-adds monitored servers; `remove_server` removes one) that add or remove rows in the monitoring store's monitored-server registry, `get_fleet_overview` and `get_ag_health` are the two cross-server reads of the fleet's CURRENT state only a central store can answer, `get_sweep_reports` is the cross-server read WITH MEMORY — the scheduled fleet sweep's persisted whole-fleet reports (the timeline with each sweep's document embedded, one sweep in full with its would-have-paged ledger, and the watch-item worklist), `get_store_metrics` reads the monitoring store's OWN hourly size/compression/growth series for capacity forecasting, `get_store_log` reads what the monitoring store's OWN PostgreSQL server log recorded as a per-class census with its capture denominator (the self-monitoring that shows the store's half of a client-side symptom), `get_collector_cost` reads the tool's OWN per-collector cost on the monitored servers (the self-monitoring that flags a collector regressing into a hog), `get_collector_stall_probes` reads the out-of-band server-wide wait samples this tool takes while one of its own collectors is stalled mid-read — the only surface here that reports what a monitored instance was doing inside the window the sequential sweep records nothing in — `get_oversized_plan_backlog` reads the backlog of cached plans the capture cap declined and what the out-of-band sweep has since done about each one (the self-monitoring that tells a sweep whose fetch half is working from one that has never once succeeded), and `get_blocking` is Darling's name for the blocked-process-report read that Lite exposes as `get_blocked_process_reports` — a naming difference, not a capability gap. Every data-read tool reads the data the collectors already captured into the store — a stored read, never a live query against the monitored server. ### Reading an empty result diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgLoggingAuditTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgLoggingAuditTools.cs new file mode 100644 index 000000000..ccceadba5 --- /dev/null +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgLoggingAuditTools.cs @@ -0,0 +1,181 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.ComponentModel; +using System.Linq; +using System.Text.Json; +using System.Threading.Tasks; +using ModelContextProtocol.Server; +using Npgsql; +using PerformanceMonitor.Common; +using PerformanceMonitor.Darling.Storage; + +namespace PerformanceMonitor.Darling.Service.Mcp; + +/// +/// get_pg_logging_audit (#3607): whether a PostgreSQL target's logging settings are producing the lines +/// they could, setting by setting, with the remedy for each in the hosting flavour's own syntax. +/// +/// The judgment is 's and the rows are +/// 's; this class is the wire. It sits beside +/// get_pg_plan_capture_readiness deliberately: that read judges plan capture's own preconditions +/// and this one judges the rest of the logging surface, and between them target onboarding has one "is this +/// target telling us everything it could" answer, which is the issue's ask. The two do not overlap — +/// the plan-capture settings appear here as observed values with a pointer, never as a second verdict. +/// +/// No hours_back and no as_of PARAMETER, and that is the Stamped latest-read shape +/// (#3541 A10) rather than an omission. Configuration is a state, not a window: the audit is of the +/// NEWEST snapshot, and the response carries that snapshot's collection time as captured_at — the +/// one stamp every row shares, selected on the row statement — so the reader knows how old the state is. A +/// windowed form would answer "no logging configuration" about a server whose hourly collector last ran +/// just outside the window. McpLatestSnapshotStampTests holds the dialect; this tool is its +/// Darling-only allowance because the roster's other half is a Lite file and Lite has no PostgreSQL. +/// +[McpServerToolType] +public sealed class DarlingMcpPgLoggingAuditTools +{ + [McpServerTool(Name = "get_pg_logging_audit"), Description("Audits a PostgreSQL target's LOGGING settings - log_min_duration_statement, log_lock_waits, log_temp_files, log_autovacuum_min_duration, log_checkpoints, log_connections, log_disconnections - and says, per setting, whether it is producing the lines it could, what telemetry those lines unlock, the recommended value WITH its cost, and the remedy in the syntax this server's hosting needs (ALTER SYSTEM plus a reload where the server is yours to administer; a parameter group on RDS/Aurora, decided from rds.* parameters in the stored snapshot rather than guessed). It reads the STORED configuration snapshot pg_server_config already collects hourly, never the live server. LATEST IS A TIME: captured_at is the instant that snapshot was taken, the collector runs hourly, so every value here is 'as of' that stamp and a change made since is not reflected until the next collection. Read it at onboarding and whenever a target-side log read comes back empty: a target with every one of these off looks identical to a fully instrumented one from every counter-based read, and the difference shows up at incident time when the log somebody reaches for holds nothing. Verdicts are instrumented, partial, off or unknown - partial means a THRESHOLD is filtering (statements faster than N ms, temp files under N kB) and the row says what falls below it; for log_min_duration_statement the threshold IS the recommended posture, and its cost_note says so, because 0 logs every statement the server runs. unknown means the setting is not in the snapshot and nothing is inferred. Every facet names the Darling family that would consume its lines and says PLANNED where that consumer does not ship yet (#3601/#3602/#3603) - the counter reads that exist today are named beside it with what they cannot see. Plan capture's own settings (auto_explain, log_line_prefix %Q, lc_messages) are LISTED as observed for completeness but judged by get_pg_plan_capture_readiness, which owns their traps; lc_messages decides whether any of these lines are written in the English the parsers match. PostgreSQL-only.")] + public static async Task GetPgLoggingAudit( + NpgsqlDataSource postgres, + [Description("Server name or display name.")] string? server_name = null) + { + var (resolved, error) = await DarlingServerResolver.ResolveOrErrorAsync(postgres, server_name); + if (error != null) return error; + + try + { + var snapshot = await DarlingPgLoggingAuditReader.GetNewestSnapshotAsync(postgres, resolved.ServerId); + + if (snapshot.Count == 0) + { + /* The same miss vocabulary as get_pg_server_config, because it is the same absence: the + collector this reads never ran here (engine gate), or has not run YET. Neither is an audit + result, and an audit of an empty snapshot would say 'unknown' seven times and look like one. */ + return await DarlingEngineCapability.NotCollectedStatusAsync( + postgres, resolved.ServerId, resolved.ServerName, "pg_server_config") + ?? McpHelpers.Status( + "empty", + $"No configuration snapshot has been collected for {resolved.ServerName} yet, so there " + + "is nothing to audit. pg_server_config runs hourly; a server registered in the last " + + "hour has not reached its first collection. This is not a verdict about the " + + "server's logging - it is the absence of the evidence."); + } + + return BuildAuditJson(resolved.ServerName, DarlingPgLoggingAudit.Audit(snapshot)); + } + catch (Exception ex) + { + /* The engine gate again, inside the catch, the way the plan tools do it: a read that throws on a + store where this collector never runs should still answer not_collected rather than a raw + error, because the gate is the more specific fact and the exception is its symptom. */ + var gated = await DarlingEngineCapability.NotCollectedStatusAsync( + postgres, resolved.ServerId, resolved.ServerName, "pg_server_config"); + if (gated != null) + { + return gated; + } + + return McpHelpers.Status("error", $"Reading the PostgreSQL logging audit failed: {ex.Message}"); + } + } + + /// + /// The response body, split out so the WIRE SHAPE can be asserted without a live store — the reason + /// BuildReadinessJson is separate on the plan tools. + /// + internal static string BuildAuditJson(string serverName, DarlingPgLoggingAudit.Result audit) + { + var facets = audit.Facets; + + /* The counts are over ALL facets, never a page - there is no limit on this read, the facet list is + fixed and small, so the #2629 cap-versus-window trap does not arise and the summary is a fact about + the server. off and unknown are NAMED because they are the actionable ones; partial is not + listed as a to-do because for two settings it is the recommendation. */ + var off = facets.Where(f => f.Verdict == DarlingPgLoggingAudit.Off).Select(f => f.Setting).ToArray(); + var unknown = facets.Where(f => f.Verdict == DarlingPgLoggingAudit.Unknown).Select(f => f.Setting).ToArray(); + /* Named at the top for the reason get_pg_server_config names them: a judged value that the next + restart will change is the one row whose remedy should not be acted on from this response alone. */ + var pendingRestart = facets.Where(f => f.PendingRestart).Select(f => f.Setting).ToArray(); + + return JsonSerializer.Serialize(new + { + server = serverName, + status = "logging_audit", + /* The snapshot's collection time, under the #3541 A10 name every stamped latest read uses. It + is how old this state is, not a window. */ + captured_at = audit.CapturedAt.ToString("o"), + source = "pg_server_config, newest snapshot - stored configuration, not the live server", + hosting = audit.Managed ? "managed (RDS/Aurora)" : "self-hosted", + hosting_evidence = audit.HostingEvidence, + total = facets.Count, + instrumented_count = facets.Count(f => f.Verdict == DarlingPgLoggingAudit.Instrumented), + partial_count = facets.Count(f => f.Verdict == DarlingPgLoggingAudit.Partial), + off_count = off.Length, + unknown_count = unknown.Length, + off_settings = off, + unknown_settings = unknown, + pending_restart_count = pendingRestart.Length, + pending_restart_settings = pendingRestart, + note = "One facet per logging setting, in the order an operator reaches for them - statements, " + + "locks, spills, maintenance, checkpoints, connections - not a causal order; nothing here " + + "gates anything else. verdict describes the LINES: instrumented writes every line the " + + "setting can, partial has a threshold filtering and the row says what falls below it, off " + + "writes nothing, unknown is not in the snapshot and nothing is inferred. partial is the " + + "recommended posture for log_min_duration_statement and can be for log_temp_files - read " + + "cost_note before changing a partial row. A row with pending_restart true is judged on the " + + "RUNNING value while the file already holds another - read restart_note before acting on its " + + "remedy. consumer names the Darling family that reads the " + + "lines and says PLANNED where it does not ship yet; the setting is still worth turning on " + + "first, because the log it fills is the one somebody opens at incident time. remedy is " + + "worded for this server's hosting (see hosting_evidence). judged_by_readiness lists plan " + + "capture's own settings as observed in the same snapshot; " + DarlingPgLoggingAudit.ReadinessTool + + " judges them, and its message_locale facet decides whether ANY of these lines are written " + + "in the English the parsers match.", + facets = facets.Select(f => new + { + setting = f.Setting, + verdict = f.Verdict, + /* Verbatim, so a reader sees what the server said rather than this tool's reading of it. */ + value = f.Value, + unit = f.Unit, + default_value = f.DefaultValue, + source = f.Source, + change_needs = f.ChangeNeeds, + unlocks = f.Unlocks, + consumer = f.Consumer, + recommended = f.Recommended, + cost_note = f.CostNote, + remedy = f.Remedy, + scope_note = f.ScopeNote, + /* The file and the running server disagree: the value above is the running one and the + remedy is written against it. restart_note says what that means and where to look. */ + pending_restart = f.PendingRestart, + restart_note = f.RestartNote, + }), + judged_by_readiness = new + { + tool = DarlingPgLoggingAudit.ReadinessTool, + note = "Shown as observed in this snapshot for completeness and NOT judged here: " + + DarlingPgLoggingAudit.ReadinessTool + " owns each of these with its own trap - a loaded " + + "auto_explain at -1 captures nothing, an auto_explain.* value on a server that never " + + "loaded the module is a placeholder, a log_line_prefix without %Q orphans every plan, " + + "and a translated lc_messages blinds every log read here. A null value means the " + + "setting is not in the snapshot, which for auto_explain.log_min_duration usually means " + + "the library is not loaded.", + settings = audit.JudgedByReadiness.Select(s => new + { + setting = s.Setting, + value = s.Value, + source = s.Source, + readiness_facet = s.ReadinessFacet, + }), + }, + }, McpHelpers.JsonOptions); + } +} diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingPgLoggingAudit.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingPgLoggingAudit.cs new file mode 100644 index 000000000..c0fd2d0a9 --- /dev/null +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingPgLoggingAudit.cs @@ -0,0 +1,732 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Globalization; +using System.Linq; +using PerformanceMonitor.Darling.Storage; + +namespace PerformanceMonitor.Darling.Service.Mcp; + +/// +/// The logging-settings audit for a PostgreSQL target (#3607): which of the server's logging GUCs are +/// producing the lines they can, judged facet by facet from the stored pg_settings snapshot, with what +/// each unlocks, the recommended value AND its cost, and the remedy in the syntax the hosting flavour needs. +/// +/// The failure this fixes is the one plan-capture readiness fixed for auto_explain, over the rest +/// of the log. A target with log_lock_waits, log_temp_files, +/// log_autovacuum_min_duration, log_checkpoints, log_connections, +/// log_disconnections and log_min_duration_statement all off looks IDENTICAL to a fully +/// instrumented one from every read this product has — the same counters, the same sampled views — and +/// the operator learns the difference at incident time, when the log they reach for holds nothing. This +/// answers "is this target telling us everything it could" before that moment, and answers it per setting, +/// because each one has a different cost and a different consumer and a single "logging: partial" would tell +/// nobody what to do. +/// +/// It is a READ, not a collector, and the shape follows from what it judges. +/// PgPlanCaptureReadinessCollector persists its facets because judging them means PROBING the target +/// — whether auto_explain is in shared_preload_libraries, whether an auto_explain.* GUC +/// even exists — and that probe is worth a history. Every setting here is a plain core GUC that +/// PgServerConfigCollector already stores hourly with its value, source, unit and context, so the +/// judgment is a pure function over rows the store holds, computed when asked. No new table, no schema rung, +/// nothing for a second collector to disagree with the first about. The snapshot's own +/// collection_time is the audit's captured_at. +/// +/// Four verdicts, and partial is not a lesser instrumented. instrumented +/// means the setting is producing every line it can; off means none; unknown means the setting +/// is not in the stored snapshot and NOTHING is inferred about it; partial means a THRESHOLD is +/// filtering — statements faster than N ms, temp files under N kB, autovacuum runs shorter than N ms — and +/// what falls below it is stated on the row. For two of the settings the threshold is the RECOMMENDED +/// posture: a log_min_duration_statement of 0 logs every statement the server runs, and #2565 +/// measured the capture-everything shape of that mechanism at 31 percent of throughput. So the summary +/// names the off and unknown settings as the actionable ones, and a partial row's +/// cost_note says whether its threshold is the recommendation or a compromise. +/// +/// Consumers are named honestly, "planned" included. The issue's own sequencing note says this +/// audit earns its keep once the log pipeline (#3601) and its first parser families (#3602 temp files, +/// #3603 autovacuum) consume the lines, and none of those ships yet. Each facet therefore names the Darling +/// family that would read its lines and says planned where that is the truth, beside the read that +/// exists today and what it cannot see. The setting is still worth turning on before the consumer lands: +/// the log it fills is the one somebody opens at incident time, whichever tool reads it. +/// +/// Plan capture's own settings are shown, not re-judged. shared_preload_libraries, +/// auto_explain.log_min_duration, log_line_prefix and lc_messages appear in the same +/// snapshot and are listed for completeness with the readiness facet that owns each, because +/// get_pg_plan_capture_readiness already judges them with the trap each one carries (a loaded library +/// capturing nothing at -1, a placeholder GUC on a server that never loaded the module, a translated +/// message catalogue) and a second judgment here would be a second place for those to drift. +/// lc_messages matters to every row above it — the lines these settings produce are English text the +/// parsers match — which is why it is in the list at all. +/// +/// Hosting flavour comes from the snapshot, not from the registry. The store's engine token +/// (MonitoredEngineKind) separates Aurora from everything else, and "everything else" is both +/// self-hosted PostgreSQL and RDS for PostgreSQL — which need OPPOSITE remedies: ALTER SYSTEM plus a +/// reload on one, a parameter group on the other, where ALTER SYSTEM is refused. Any rds.* GUC +/// in the snapshot (rds.extensions, rds.superuser_reserved_connections, … — RDS and Aurora +/// both carry them) is evidence in the same rows this already reads, and the response says which way it +/// decided and on what. +/// +public static class DarlingPgLoggingAudit +{ + /// The verdict vocabulary on the wire — the header says what each means. + public const string Instrumented = "instrumented"; + public const string Partial = "partial"; + public const string Off = "off"; + public const string Unknown = "unknown"; + + /// The tool that judges the plan-capture settings this audit only lists. + public const string ReadinessTool = "get_pg_plan_capture_readiness"; + + /// The GUC. + /// What the snapshot holds, verbatim, or null when the setting is not in it. + /// The unit pg_settings reports for a numeric setting. + /// The compiled-in default, from the same snapshot. + /// Where the value came from. + /// Reload or restart, from the setting's context. + /// One of the four constants above. + /// What telemetry the setting produces, and what this product has INSTEAD today. + /// The Darling family that reads the lines, marked planned where it does not ship. + /// The value to set. + /// What the recommended value costs, and when to deviate from it. + /// The change, in the syntax this server's hosting flavour needs — or why none is needed. + /// Set when the value came from a per-role or per-database override the monitoring + /// connection resolved, so the server-wide value may differ. + /// The file and the running server disagree about this setting, so the value + /// judged here is the RUNNING one and changes at the next restart. + /// The consequence of , spelled out on the row; + /// null when the two agree. + public sealed record Facet( + string Setting, + string? Value, + string? Unit, + string? DefaultValue, + string? Source, + string? ChangeNeeds, + string Verdict, + string Unlocks, + string Consumer, + string Recommended, + string CostNote, + string Remedy, + string? ScopeNote, + bool PendingRestart, + string? RestartNote); + + /// A plan-capture setting shown as observed, with the readiness facet that judges it. + public sealed record ReadinessSetting(string Setting, string? Value, string? Source, string ReadinessFacet); + + /// The snapshot's collection time — the one stamp every row shares. + /// True when the snapshot carries rds.* parameters. + /// What the flavour decision rested on. + /// One per judged setting, in the order an operator reaches for them. + /// The plan-capture settings, listed not judged. + public sealed record Result( + DateTime CapturedAt, + bool Managed, + string HostingEvidence, + IReadOnlyList Facets, + IReadOnlyList JudgedByReadiness); + + /// + /// The settings this audit judges, in the order they are reported: what somebody reaches for FIRST when + /// a server is slow — the statements — then the locks, the spills, the maintenance, the checkpoints, and + /// the connection churn. There is no causal chain between them (unlike readiness, where the library gates + /// the threshold), so the order is the reader's, and the note on the response says so. + /// + public static readonly IReadOnlyList JudgedSettings = new[] + { + "log_min_duration_statement", + "log_lock_waits", + "log_temp_files", + "log_autovacuum_min_duration", + "log_checkpoints", + "log_connections", + "log_disconnections", + }; + + /// The plan-capture settings listed for completeness, with the readiness facet owning each. + public static readonly IReadOnlyList<(string Setting, string ReadinessFacet)> ReadinessSettings = new[] + { + ("shared_preload_libraries", "library_loaded"), + ("auto_explain.log_min_duration", "capture_threshold"), + ("log_line_prefix", "plan_attribution"), + ("lc_messages", "message_locale"), + }; + + /// + /// Judges one server's newest snapshot. The rows are what + /// returned — every non-session setting + /// at the newest collection_time — and MUST be non-empty; an empty snapshot is the tool's + /// empty/not_collected path, not an audit of nothing. + /// + public static Result Audit(IReadOnlyList snapshot) + { + ArgumentNullException.ThrowIfNull(snapshot); + if (snapshot.Count == 0) + { + throw new ArgumentException("An empty snapshot cannot be audited; report it as not collected.", nameof(snapshot)); + } + + var byName = new Dictionary(StringComparer.Ordinal); + foreach (var row in snapshot) + { + /* First wins. The reader excludes session-scoped sources, so two rows for one name at one + collection_time should not happen; if a store ever holds them, the audit must not throw over + a duplicate the collector wrote. */ + byName.TryAdd(row.Name, row); + } + + var capturedAt = snapshot.Max(r => r.CollectionTime); + + /* The hosting flavour, from evidence in the rows rather than from a registry token that cannot + separate RDS from self-hosted. Counted so the response can say how much evidence there was. */ + var rdsParameters = byName.Keys.Count(n => n.StartsWith("rds.", StringComparison.Ordinal)); + var managed = rdsParameters > 0; + var hostingEvidence = managed + ? $"{rdsParameters} rds.* parameter(s) in the snapshot, which only RDS and Aurora carry - remedies are " + + "worded for a parameter group, where ALTER SYSTEM is refused." + : "no rds.* parameter in the snapshot, so this is not RDS or Aurora - remedies are worded as ALTER " + + "SYSTEM plus a reload. If this server IS managed by a provider that hides its own parameters, " + + "translate each remedy to that provider's parameter surface."; + + var deadlockTimeout = Describe(byName, "deadlock_timeout"); + + var facets = new List(JudgedSettings.Count) + { + MinDurationStatement(byName, managed), + LockWaits(byName, managed, deadlockTimeout), + TempFiles(byName, managed), + AutovacuumMinDuration(byName, managed), + Checkpoints(byName, managed), + Connections(byName, managed), + Disconnections(byName, managed), + }; + + var readiness = ReadinessSettings + .Select(s => byName.TryGetValue(s.Setting, out var row) + ? new ReadinessSetting(s.Setting, row.Setting, row.Source, s.ReadinessFacet) + : new ReadinessSetting(s.Setting, null, null, s.ReadinessFacet)) + .ToList(); + + return new Result(capturedAt, managed, hostingEvidence, facets, readiness); + } + + /* ───────────────────────── the facets ───────────────────────── */ + + private static Facet MinDurationStatement( + IReadOnlyDictionary byName, bool managed) + { + const string Setting = "log_min_duration_statement"; + var row = Find(byName, Setting); + var threshold = Threshold(row); + + /* -1 off, 0 everything, N a threshold. The RECOMMENDED state is the threshold, and the verdict is + still partial - the header says why: partial describes the lines, and this row's cost_note says + the filter is the point. */ + var verdict = row is null ? Unknown + : threshold is null ? Unknown + : threshold < 0 ? Off + : threshold == 0 ? Instrumented + : Partial; + + var cost = verdict switch + { + Instrumented => + "0 logs EVERY statement the server runs, with its text. That is the capture-everything shape " + + "#2565 measured for auto_explain at 31 percent of throughput and 772 MB of log in 20 seconds " + + "(pgbench, 8 clients). The statement log at 0 emits one entry per statement exactly as " + + "auto_explain at 0 does - smaller entries with no plan body, but no fewer of them. Move to a " + + "millisecond threshold; the fast statements are the volume and nobody reads them.", + Partial => + $"Statements faster than {threshold} ms write nothing, which is the intended trade: the slow " + + "ones are the question and the fast ones are the volume. This is the recommended posture, " + + "not a gap. To see a SAMPLE of the faster ones without the volume, log_min_duration_sample " + + "with log_statement_sample_rate is the sampling form; it is a separate setting and not judged " + + "here.", + Off => + "Off costs nothing and records nothing: a 40-second statement cancelled by its client leaves " + + "no trace anywhere but this log line. A threshold sized to the workload - well above the " + + "normal statement time, so that only the outliers write - costs one line per outlier.", + _ => UnknownNote(row), + }; + + return new Facet( + Setting, + row?.Setting, row?.Unit, row?.BootValue, row?.Source, ChangeNeeds(row), + verdict, + Unlocks: "One LOG line per EXECUTION that ran longer than the threshold, carrying the duration and " + + "the statement text. What this product has instead is pg_stat_statements through " + + "get_pg_top_queries: per-SHAPE aggregates that can say a shape averages 200 ms and " + + "never that one execution took 40 seconds at 03:07 - the log line is the only record " + + "of the individual slow execution.", + Consumer: "PLANNED - no Darling family reads statement-duration lines yet; #3601's log pipeline is " + + "where they would land, and the statement text carries literals, so the same redaction " + + "pass plan capture applies before storage is a precondition of storing them at all. " + + "get_pg_top_queries is the aggregate the lines would sharpen. Until then the line is " + + "what an operator finds in the server log at incident time.", + Recommended: "A millisecond threshold sized to the workload, never 0 - 1000 is a common starting " + + "point on an OLTP workload; lower it as the volume proves tolerable. " + + "auto_explain.log_min_duration is the separate threshold for PLANS and is judged by " + + ReadinessTool + ".", + CostNote: cost, + Remedy: Remedy(Setting, "1000", row, managed, verdict, alreadyRight: verdict == Partial), + ScopeNote: ScopeNote(row), + PendingRestart: row?.PendingRestart ?? false, + RestartNote: RestartNote(row)); + } + + private static Facet LockWaits( + IReadOnlyDictionary byName, bool managed, + string deadlockTimeout) + { + const string Setting = "log_lock_waits"; + var row = Find(byName, Setting); + var on = Bool(row); + + var verdict = row is null || on is null ? Unknown : on.Value ? Instrumented : Off; + + return new Facet( + Setting, + row?.Setting, row?.Unit, row?.BootValue, row?.Source, ChangeNeeds(row), + verdict, + Unlocks: $"A LOG line whenever a session waits longer than deadlock_timeout ({deadlockTimeout}) for a " + + "lock, naming the waiting process, the lock it wanted and the statement that wanted it - " + + "the ENGINE-recorded record of lock waits. What this product has instead is get_pg_blocking " + + "from pg_blocking, which SAMPLES pg_locks and pg_stat_activity on a cadence: a wait that " + + "starts and ends between two samples is invisible to it and would be in this line.", + Consumer: "PLANNED - #3601 names lock-wait reports among the families the log pipeline would " + + "classify. get_pg_blocking is the sampled read that exists today, and its own response " + + "says how many samples its 'no blocking' rests on.", + Recommended: "on.", + CostNote: verdict == Unknown ? UnknownNote(row) + : "One line per wait longer than deadlock_timeout - negligible on a workload that is not " + + "already lock-bound, and on one that is, the volume is itself the finding. Do NOT lower " + + "deadlock_timeout to make this fire sooner: that is also how often the deadlock " + + "detector runs, and it is a lock-heavy operation in its own right.", + Remedy: Remedy(Setting, "on", row, managed, verdict, alreadyRight: verdict == Instrumented), + ScopeNote: ScopeNote(row), + PendingRestart: row?.PendingRestart ?? false, + RestartNote: RestartNote(row)); + } + + private static Facet TempFiles( + IReadOnlyDictionary byName, bool managed) + { + const string Setting = "log_temp_files"; + var row = Find(byName, Setting); + var threshold = Threshold(row); + + var verdict = row is null ? Unknown + : threshold is null ? Unknown + : threshold < 0 ? Off + : threshold == 0 ? Instrumented + : Partial; + + var cost = verdict switch + { + Instrumented => + "0 logs EVERY temporary file at the moment it is deleted, including the small ones. On a " + + "workload whose sorts sit just over work_mem that is a line per spill, constantly; if that is " + + "this server, set a kilobyte threshold (10240 is 10 MB) and accept that spills under it are " + + "unseen. On most workloads 0 is cheap and is the recommendation.", + Partial => + $"Temporary files under {threshold} kB write nothing. That is a deliberate trade for a workload " + + "that spills small files constantly; if this server does not, 0 sees everything at little " + + "cost. Per-event attribution works from whatever crosses the line either way.", + Off => + "Off records nothing: the counters say a database spilled 4 GB in an hour and the log says " + + "nothing about which statement did it or when. 0 costs one line per temp file; on a workload " + + "that spills constantly a kilobyte threshold bounds it.", + _ => UnknownNote(row), + }; + + return new Facet( + Setting, + row?.Setting, row?.Unit, row?.BootValue, row?.Source, ChangeNeeds(row), + verdict, + Unlocks: "One LOG line per temporary file, with its size and the statement that wrote it - " + + "per-EVENT spill attribution. What this product has instead is the counter shadow of " + + "that: per-database temp_files / temp_bytes deltas (get_pg_database_stats, " + + "get_pg_database_trend) and per-shape temp_blks_* from pg_stat_statements " + + "(get_pg_top_queries). Between them you can know a database spilled and that a shape " + + "spills - never that THIS execution spilled 4 GB at 03:07, which is the question when a " + + "disk fills.", + Consumer: "PLANNED - #3602, per-event temp spill attribution, is the parser family that would read " + + "these lines; its statement text needs plan capture's redaction pass first. " + + "get_pg_database_stats carries the spill FINDING from the counters today.", + Recommended: "0 (every spill), or a kilobyte threshold on a workload that spills small files constantly.", + CostNote: cost, + Remedy: Remedy(Setting, "0", row, managed, verdict, alreadyRight: verdict is Instrumented or Partial), + ScopeNote: ScopeNote(row), + PendingRestart: row?.PendingRestart ?? false, + RestartNote: RestartNote(row)); + } + + private static Facet AutovacuumMinDuration( + IReadOnlyDictionary byName, bool managed) + { + const string Setting = "log_autovacuum_min_duration"; + var row = Find(byName, Setting); + var threshold = Threshold(row); + + var verdict = row is null ? Unknown + : threshold is null ? Unknown + : threshold < 0 ? Off + : threshold == 0 ? Instrumented + : Partial; + + /* PostgreSQL 15 moved the default from -1 to 600000 (ten minutes). A server sitting on that default + at a positive threshold is the shape worth naming: it sees only the outlier runs. The judgment is + PostgreSQL's OWN - source = 'default' is the server saying nobody set it - and not a text + comparison against boot_val, for the reason DarlingPgServerConfigReader.CurrentConfigSql gives: + an administrator who writes 600000 into postgresql.conf has made a choice that happens to equal + the boot value, and calling that choice a default would attribute it to inaction. */ + var atDefault = row is not null && threshold > 0 + && string.Equals(row.Source, "default", StringComparison.Ordinal); + + var cost = verdict switch + { + Instrumented => + "0 logs every autovacuum and autoanalyze run. The line rate is the rate at which tables get " + + "vacuumed, bounded by autovacuum_max_workers (three by default) each finishing one table before " + + "starting the next - small on most servers, and the cheapest setting on this list for what it " + + "returns. A database with thousands of tiny tables is the exception where a threshold earns " + + "its place.", + Partial => + $"Runs shorter than {threshold} ms write nothing." + + (atDefault + ? " This is PostgreSQL's own default since 15 (ten minutes), and on most tables that is " + + "every run: a cost history that sees only the outliers cannot say what a NORMAL run " + + "costs, which is the baseline the outliers are judged against. 0 is cheap here." + : " Whether that is the right cut depends on what you want the history for: outliers " + + "only, or a baseline of what a normal run costs. 0 is cheap here."), + Off => + "Off records nothing about what any run cost. 0 costs one line per run, and the run rate is " + + "bounded by the worker count, so the volume is small on all but a server with thousands of " + + "tiny tables.", + _ => UnknownNote(row), + }; + + return new Facet( + Setting, + row?.Setting, row?.Unit, row?.BootValue, row?.Source, ChangeNeeds(row), + verdict, + Unlocks: "One LOG line per autovacuum or autoanalyze run that took longer than the threshold: " + + "pages and tuples removed, buffer hits and misses, read and write rates, WAL usage and " + + "elapsed time - what the run COST. What this product has instead is pg_stat_user_tables " + + "through get_pg_autovacuum_health: whether autovacuum ran and when, never what it cost, " + + "so a run that takes 40 minutes of I/O in the business peak reads as healthy there.", + Consumer: "PLANNED - #3603, autovacuum per-run cost, is the parser family that would read these " + + "lines and put run history beside get_pg_autovacuum_health. That read is the " + + "whether-not-what read today.", + Recommended: "0 (every run).", + CostNote: cost, + Remedy: Remedy(Setting, "0", row, managed, verdict, alreadyRight: verdict == Instrumented), + ScopeNote: ScopeNote(row), + PendingRestart: row?.PendingRestart ?? false, + RestartNote: RestartNote(row)); + } + + private static Facet Checkpoints( + IReadOnlyDictionary byName, bool managed) + { + const string Setting = "log_checkpoints"; + var row = Find(byName, Setting); + var on = Bool(row); + + var verdict = row is null || on is null ? Unknown : on.Value ? Instrumented : Off; + + return new Facet( + Setting, + row?.Setting, row?.Unit, row?.BootValue, row?.Source, ChangeNeeds(row), + verdict, + Unlocks: "A LOG line per checkpoint with what it did: buffers written, files synced, write, sync " + + "and total time, WAL distance - the CAUSE side of a checkpoint I/O storm, per " + + "checkpoint. What this product has instead is get_pg_write_stats over the checkpointer " + + "counters: timed versus requested and buffers written across a window, never the " + + "duration of one checkpoint.", + Consumer: "PLANNED - #3601's log pipeline. get_pg_write_stats reads the checkpoint counters today.", + Recommended: "on - PostgreSQL 15 made it the default.", + CostNote: verdict == Unknown ? UnknownNote(row) + : "One or two lines per checkpoint, and a checkpoint happens at most every " + + "checkpoint_timeout (five minutes by default) unless something requests one - " + + "negligible. A server reporting off has it set that way in a file or parameter group, " + + "or is on a major before 15 where off was the default.", + Remedy: Remedy(Setting, "on", row, managed, verdict, alreadyRight: verdict == Instrumented), + ScopeNote: ScopeNote(row), + PendingRestart: row?.PendingRestart ?? false, + RestartNote: RestartNote(row)); + } + + private static Facet Connections( + IReadOnlyDictionary byName, bool managed) + { + const string Setting = "log_connections"; + var row = Find(byName, Setting); + + /* A boolean through PostgreSQL 17 and a STRING LIST in 18 (receipt, authentication, authorization, + setup_durations, all), where on/true/yes/1 still mean all and the empty string means off. Measured + on 18.4: the value is stored verbatim as written - 'on', 'true', '1', 'all', 'receipt,authentication' + - so this reads any non-empty, non-false value as producing lines and shows the value itself. */ + var verdict = row is null ? Unknown + : ConnectionLogging(row.Setting) is bool on ? (on ? Instrumented : Off) + : Unknown; + + return new Facet( + Setting, + row?.Setting, row?.Unit, row?.BootValue, row?.Source, ChangeNeeds(row), + verdict, + Unlocks: "A LOG line per connection as it is received, authenticated and authorized - the only " + + "record of connection CHURN and of who a FATAL authentication failure was. What this " + + "product has instead is pg_stat_database's numbackends, a gauge, and its sessions " + + "counter, a cumulative total: neither says who connected when, or that 400 connections " + + "arrived in the minute before the incident.", + Consumer: "PLANNED - #3601 names connection churn among the log pipeline's families.", + Recommended: "on (PostgreSQL 18: 'all', or a list such as receipt,authentication).", + CostNote: verdict == Unknown ? UnknownNote(row) + : "On a POOLED workload connections are rare and this costs nothing. On an unpooled one - a " + + "connection per request - it is a line per request, and that volume is itself the " + + "finding: the pool that is missing. On 18 the list form narrows the lines to the " + + "stages you want.", + Remedy: Remedy(Setting, "on", row, managed, verdict, alreadyRight: verdict == Instrumented), + ScopeNote: ScopeNote(row), + PendingRestart: row?.PendingRestart ?? false, + RestartNote: RestartNote(row)); + } + + private static Facet Disconnections( + IReadOnlyDictionary byName, bool managed) + { + const string Setting = "log_disconnections"; + var row = Find(byName, Setting); + var on = Bool(row); + + var verdict = row is null || on is null ? Unknown : on.Value ? Instrumented : Off; + + return new Facet( + Setting, + row?.Setting, row?.Unit, row?.BootValue, row?.Source, ChangeNeeds(row), + verdict, + Unlocks: "A LOG line per session end WITH the session's duration - the half that turns connection " + + "lines into session lifetimes, so a churning application shows as thousands of " + + "two-second sessions rather than as a connection count that looks stable.", + Consumer: "PLANNED - #3601, beside log_connections.", + Recommended: "on, together with log_connections.", + CostNote: verdict == Unknown ? UnknownNote(row) + : "The same profile as log_connections: one line per session end, nothing on a pooled " + + "workload, a line per request on an unpooled one.", + Remedy: Remedy(Setting, "on", row, managed, verdict, alreadyRight: verdict == Instrumented), + ScopeNote: ScopeNote(row), + PendingRestart: row?.PendingRestart ?? false, + RestartNote: RestartNote(row)); + } + + /* ───────────────────────── shared pieces ───────────────────────── */ + + private const string NotInSnapshot = + "This setting is not in the stored snapshot, so nothing is claimed about it - not inferred from the " + + "default, not inferred from the major version. get_pg_server_config shows what the snapshot holds."; + + /// + /// The two ways a verdict is unknown, told apart on the row: the setting is not in the snapshot at + /// all, or it is there with a value this audit cannot read as the boolean or integer PostgreSQL renders + /// for it. The second is close to unreachable — pg_settings renders well-formed values — but a + /// message that said "not in the snapshot" about a row that is plainly in it would be false, and the + /// contract this type's header states ("unknown means the setting is not in the stored snapshot") is + /// what the first sentence below keeps true by naming the exception. + /// + private static string UnknownNote(DarlingPgLoggingAuditReader.PgLoggingSettingRow? row) => row is null + ? NotInSnapshot + : $"This setting IS in the stored snapshot but its value '{row.Setting}' is not one this audit can read " + + "as the boolean or integer PostgreSQL renders for it, so nothing is claimed about it. " + + "get_pg_server_config shows the raw row; if this recurs, the collector's rendering has changed " + + "and the audit's parse needs to learn it."; + + private static DarlingPgLoggingAuditReader.PgLoggingSettingRow? Find( + IReadOnlyDictionary byName, string name) => + byName.TryGetValue(name, out var row) ? row : null; + + /// A setting's value with its unit, for prose — 1000 ms — or a plain statement that + /// the snapshot does not have it. + private static string Describe( + IReadOnlyDictionary byName, string name) + { + var row = Find(byName, name); + if (row?.Setting is null) + { + return "not in the snapshot"; + } + + return string.IsNullOrEmpty(row.Unit) ? row.Setting : $"{row.Setting} {row.Unit}"; + } + + /// + /// The integer a threshold GUC holds. pg_settings.setting renders integers in the setting's base + /// unit with no suffix (600000 with unit = ms), so this is a plain parse; anything else is + /// null and the caller says unknown rather than guessing. + /// + private static long? Threshold(DarlingPgLoggingAuditReader.PgLoggingSettingRow? row) + { + /* A block body, not an expression with a property pattern: TsqlConventionGuardTests' member scan + reads a `{ }` pattern as the member's body and stops short, which strands everything after it. */ + var text = row?.Setting; + if (text is null) + { + return null; + } + + return long.TryParse(text.Trim(), NumberStyles.Integer, CultureInfo.InvariantCulture, out var value) + ? value + : null; + } + + /// The spellings PostgreSQL accepts for a boolean GUC, as pg_settings renders them. + private static bool? Bool(DarlingPgLoggingAuditReader.PgLoggingSettingRow? row) => BoolText(row?.Setting); + + private static bool? BoolText(string? text) + { + if (text is null) + { + return null; + } + + switch (text.Trim().ToLowerInvariant()) + { + case "on": + case "true": + case "yes": + case "1": + return true; + case "off": + case "false": + case "no": + case "0": + return false; + default: + return null; + } + } + + /// + /// log_connections across the 17/18 boundary: the boolean spellings first, then the 18 list — + /// empty is off, anything else non-empty is a stage list and produces lines. + /// + internal static bool? ConnectionLogging(string? text) + { + if (text is null) + { + return null; + } + + if (BoolText(text) is bool asBool) + { + return asBool; + } + + return text.Trim().Length > 0; + } + + /// Reload or restart, from the GUC's context — the fact get_pg_server_config already + /// exposes as requires_restart_to_change, worded for a remedy. + private static string? ChangeNeeds(DarlingPgLoggingAuditReader.PgLoggingSettingRow? row) => row?.Context switch + { + null => null, + "postmaster" => "restart", + "superuser-backend" or "backend" => "reload; applies to connections opened after it", + _ => "reload", + }; + + /// + /// A per-role or per-database override resolved on the MONITORING connection is not the server's value — + /// the limit the readiness collector states for lc_messages, and it holds for every GUC here. A + /// log line is written by the backend that raised it, under THAT backend's resolved value. + /// + private static string? ScopeNote(DarlingPgLoggingAuditReader.PgLoggingSettingRow? row) => row?.Source switch + { + "user" or "database" or "database user" => + $"source is '{row.Source}': this value came from a per-role or per-database override that the " + + "monitoring connection resolved, so the server-wide value may differ, and every other backend " + + "logs under its OWN resolved value. get_pg_server_config shows the setting's source; ALTER ROLE " + + "/ ALTER DATABASE ... RESET removes the override.", + _ => null, + }; + + /// + /// pending_restart is the one row where the value judged is provably NOT the value the server will + /// have: postgresql.conf (or ALTER SYSTEM's file) already holds something else and the running server + /// has not restarted. get_pg_server_config reports it loudly for the same reason; here it matters twice + /// over, because the remedy is written against the running value and a restart may deliver the change, + /// or a different one, with no deployment to explain it. Which value the file holds is not in the + /// snapshot - pg_settings does not carry it - so the note says the disagreement exists and where to look, + /// and does not guess the direction. + /// + private static string? RestartNote(DarlingPgLoggingAuditReader.PgLoggingSettingRow? row) + { + /* A block body for the reason Threshold has one: a `{ }` property pattern reads as the member's + body to TsqlConventionGuardTests' scan. */ + if (row is null || !row.PendingRestart) + { + return null; + } + + return "pending_restart is TRUE: the configuration file already holds a different value for this setting " + + "and the running server has not restarted, so the value judged here is the RUNNING one and it " + + "changes at the next restart with no deployment to explain it. pg_settings does not carry the " + + "file's value, so which way it changes is not knowable from here - read the file, or " + + "get_pg_server_config's pending_restart_settings, before acting on this row's remedy."; + } + + /// + /// The change in the hosting flavour's own syntax, or the reason none is needed. alreadyRight is + /// the caller's judgment that the current value is the recommended posture — which for a threshold + /// setting can be a partial verdict — so the remedy does not tell somebody to change a value that + /// is already right. + /// + private static string Remedy( + string setting, string recommendedLiteral, + DarlingPgLoggingAuditReader.PgLoggingSettingRow? row, bool managed, string verdict, bool alreadyRight) + { + if (verdict == Unknown) + { + return row is null + ? "No remedy is offered for a setting the snapshot does not hold: check get_pg_server_config, " + + "and if the collector is running, the next hourly snapshot will carry it." + : "No remedy is offered for a value this audit could not read: get_pg_server_config shows the " + + "raw row, and the remedy depends on what it actually says."; + } + + if (alreadyRight) + { + return "No change needed - the current value is the recommended posture. cost_note says what " + + "it does and does not write."; + } + + var restart = string.Equals(row?.Context, "postmaster", StringComparison.Ordinal); + var perBackend = row?.Context is "superuser-backend" or "backend"; + + if (managed) + { + return $"Set {setting} = {recommendedLiteral} in the DB parameter group and apply it - on Aurora the " + + "CLUSTER parameter group covers every instance and an instance-level group overrides it " + + "per instance. ALTER SYSTEM is refused on RDS and Aurora. " + + (restart + ? "This is a STATIC parameter and needs a reboot." + : "This is a dynamic parameter and applies WITHOUT a reboot" + + (perBackend ? ", to connections opened after it." : ".")); + } + + return $"ALTER SYSTEM SET {setting} = {recommendedLiteral}; SELECT pg_reload_conf(); - " + + (restart + ? "then RESTART: this parameter's context is postmaster and a reload does not apply it." + : "a reload, not a restart" + + (perBackend + ? "; sessions already open keep their value and new connections take the new one." + : ".")); + } +} diff --git a/Darling/PerformanceMonitor.Darling.Service/wwwroot/js/pages/server-tabs.js b/Darling/PerformanceMonitor.Darling.Service/wwwroot/js/pages/server-tabs.js index 80ce815f4..1266aea91 100644 --- a/Darling/PerformanceMonitor.Darling.Service/wwwroot/js/pages/server-tabs.js +++ b/Darling/PerformanceMonitor.Darling.Service/wwwroot/js/pages/server-tabs.js @@ -1612,6 +1612,21 @@ export const POSTGRES_TABS = [ ctx.label + ", the newest reading of each facet in it; in CAUSAL order rather than alphabetically - fix them top to bottom; the remedy is per facet, and on Aurora/RDS it says which changes need a parameter group and a reboot", "No readiness state collected. Unlike the grids around it an empty panel here is never the healthy answer - the collector writes one row per facet on every run whatever it finds - so this means it has not run for this server, or the window is shorter than its hourly cadence." ), + /* #3607: the rest of the logging surface, directly under plan-capture readiness because it is the + other half of the same onboarding question - is this target telling us everything it could. Judged + from the newest stored pg_server_config snapshot rather than collected, so it takes no window; the + read reports the snapshot time as captured_at. Every setting is shown whatever its + verdict, for the reason the readiness grid shows satisfied facets: a list of only the failures + cannot show that a target IS instrumented. */ + table( + "Logging Settings Audit", + "get_pg_logging_audit", + { server }, + "facets", + PG_LOGGING_AUDIT_COLUMNS, + "newest configuration snapshot; verdict describes the LINES each setting writes - partial means a threshold is filtering and is the recommended posture for log_min_duration_statement; the remedy is worded for this server's hosting flavour", + "No configuration snapshot to audit. pg_server_config runs hourly, so a server registered in the last hour has not reached its first collection - this is the absence of evidence, not a verdict about the server's logging." + ), /* Directly UNDER the query shapes, because that is the question it answers (#2539). A statement whose time makes no sense from its row count usually spilled, and pg_stat_database's temp counters are the only evidence of that we collect — the statement stats themselves cannot see it. The deadlock and @@ -3461,6 +3476,23 @@ const PG_PLAN_CAPTURE_READINESS_COLUMNS = [ { key: "last_observed", label: "Last Seen", format: "time", small: true }, ]; +/* Verdict beside the setting, then the value, then the prose columns widest and last - the same reading + order as the readiness grid above it: scan down setting and verdict to find the row that is off, then + read across for what it unlocks, what it costs, and what to type. remedy is the column the panel exists + for and is worded for the server's hosting flavour by the read, not by this file. */ +const PG_LOGGING_AUDIT_COLUMNS = [ + { key: "setting", label: "Setting", mono: true }, + { key: "verdict", label: "Verdict" }, + { key: "value", label: "Value", mono: true }, + { key: "unit", label: "Unit", small: true }, + { key: "source", label: "Source", small: true }, + { key: "change_needs", label: "Change needs", small: true }, + { key: "unlocks", label: "Unlocks", wrap: true }, + { key: "recommended", label: "Recommended", wrap: true }, + { key: "cost_note", label: "Cost", wrap: true }, + { key: "remedy", label: "Remedy", wrap: true }, +]; + const PG_TOP_QUERY_COLUMNS = [ { key: "queryid", label: "Query ID", mono: true }, { key: "calls", label: "Calls", format: "int" }, diff --git a/Darling/PerformanceMonitor.Darling.Storage/DarlingPgLoggingAuditReader.cs b/Darling/PerformanceMonitor.Darling.Storage/DarlingPgLoggingAuditReader.cs new file mode 100644 index 000000000..815fa0966 --- /dev/null +++ b/Darling/PerformanceMonitor.Darling.Storage/DarlingPgLoggingAuditReader.cs @@ -0,0 +1,126 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Threading; +using System.Threading.Tasks; +using Npgsql; + +namespace PerformanceMonitor.Darling.Storage; + +/// +/// The newest stored pg_settings snapshot for one server, as the input to the logging-settings audit +/// (get_pg_logging_audit, #3607). +/// +/// This is a READ over pg_server_config, not a collector. Plan-capture readiness +/// (PgPlanCaptureReadinessCollector, #2564) persists its facets because judging them means probing +/// the target — is the library in shared_preload_libraries, does an auto_explain.* GUC exist +/// at all — and the answer is worth a history of its own. Every setting the logging audit judges is a plain +/// core GUC that PgServerConfigCollector (#2658) already stores hourly with its value, source and +/// context, so a second collector would write the same rows under a second name and the two would drift. +/// The judgment happens at read time over the snapshot that is already there, and the snapshot's +/// collection_time is the audit's captured_at (#3541 A10's stamp, selected on the row statement). +/// +/// Anchored on MAX(collection_time), not on a window — the same reason +/// gives: configuration is a state, and an hours +/// filter would answer "this server has no logging configuration" about a server whose hourly collector +/// last ran just outside it. +/// +/// Session-scoped rows are excluded, spelled the same way the config reader spells them. +/// pg_settings is a per-backend view, so a client-sourced row is the collector's own +/// connection. For THIS read the trap is sharper than for the config listing: the monitoring login could +/// carry SET log_min_duration_statement = 0 in its own session, and an audit that read that row would +/// declare the server instrumented while every other backend logs nothing. The list is inline rather than +/// substituted in so the constant stays SQL that DarlingPgReadSqlParsesLiveTests can parse-check — +/// the config reader's header records why. +/// +/// The whole snapshot travels, not just the settings judged. Two reasons. The audit needs +/// evidence of the HOSTING FLAVOUR to word its remedies — ALTER SYSTEM plus a reload on a server +/// somebody administers, a parameter group on RDS/Aurora — and the store's engine token cannot separate RDS +/// from self-hosted (MonitoredEngineKind.Postgres is both). The presence of any rds.* GUC in +/// the snapshot can, and it is in the data already. And the settings list is owned by the judgment code in +/// the service; filtering here would put the list in two places. A snapshot is a few hundred short rows +/// once an hour per server, so this costs nothing the listing read does not already spend. +/// +public static class DarlingPgLoggingAuditReader +{ + /// The GUC name. + /// Its value as the server rendered it — TEXT, units and all, never cast here. + /// The unit pg_settings reports (ms, kB), or null. + /// When a change takes effect: postmaster needs a restart, everything + /// else a reload at most. + /// Where the value came from — default, configuration file, + /// user, database… — which is what separates the server's setting from an override + /// the monitoring role happens to resolve. + /// The compiled-in default, so "PostgreSQL 15 turned this on" is visible + /// without a table of defaults that would rot at every major. + /// The file and the running server disagree about this one. + /// When the snapshot was taken — the audit's captured_at. + public sealed record PgLoggingSettingRow( + string Name, + string? Setting, + string? Unit, + string? Context, + string? Source, + string? BootValue, + bool PendingRestart, + DateTime CollectionTime); + + /* The same two-step anchor as CurrentConfigSql: the newest collection_time for the server, then every + non-session row at that instant. ORDER BY name so the judgment code's lookups and the test fixtures + meet the rows in one stable order. */ + public const string NewestSnapshotSql = """ + SELECT + c.name, + c.setting, + c.unit, + c.context, + c.source, + c.boot_val, + coalesce(c.pending_restart, false), + c.collection_time + FROM pg_server_config AS c + WHERE c.server_id = $1 + AND c.collection_time = ( + SELECT MAX(collection_time) + FROM pg_server_config + WHERE server_id = $1) + AND c.name IS NOT NULL + AND coalesce(c.source, '') NOT IN ('client', 'session', 'override') + ORDER BY c.name + """; + + public static async Task> GetNewestSnapshotAsync( + NpgsqlDataSource postgres, int serverId, CancellationToken cancellationToken = default) + { + ArgumentNullException.ThrowIfNull(postgres); + + var rows = new List(); + await using var command = postgres.CreateCommand(NewestSnapshotSql); + command.CommandTimeout = StorageCommandDeadlines.McpReadSeconds; + command.Parameters.AddWithValue(serverId); + + await using var reader = await command.ExecuteReaderAsync(cancellationToken); + while (await reader.ReadAsync(cancellationToken)) + { + rows.Add(new PgLoggingSettingRow( + Name: reader.GetString(0), + Setting: reader.IsDBNull(1) ? null : reader.GetString(1), + Unit: reader.IsDBNull(2) ? null : reader.GetString(2), + Context: reader.IsDBNull(3) ? null : reader.GetString(3), + Source: reader.IsDBNull(4) ? null : reader.GetString(4), + BootValue: reader.IsDBNull(5) ? null : reader.GetString(5), + PendingRestart: !reader.IsDBNull(6) && reader.GetBoolean(6), + /* Stored naive-UTC, read back as UTC — the readiness reader's convention. */ + CollectionTime: DateTime.SpecifyKind(reader.GetDateTime(7), DateTimeKind.Utc))); + } + + return rows; + } +} diff --git a/Lite.Tests/CrossAppMcpToolInventoryPinTests.cs b/Lite.Tests/CrossAppMcpToolInventoryPinTests.cs index 05764ef7c..2f7078a1e 100644 --- a/Lite.Tests/CrossAppMcpToolInventoryPinTests.cs +++ b/Lite.Tests/CrossAppMcpToolInventoryPinTests.cs @@ -102,6 +102,19 @@ cannot acquire one (the engine gate never dispatches a PostgreSQL definition the is the closest thing conceptually and it is not close - it is a database-scoped feature with its own health read (get_query_store_health) rather than a set of preload-only server GUCs. */ "get_pg_plan_capture_readiness", + + /* get_pg_logging_audit (#3607) - readiness's facet-and-remedy shape over the rest of the logging + surface (log_lock_waits, log_temp_files, log_autovacuum_min_duration, log_checkpoints, + log_connections / log_disconnections, log_min_duration_statement), judged from the stored + pg_server_config snapshot. Same architectural reason as the entry above: the settings are + PostgreSQL GUCs and the snapshot is a PostgreSQL collector's table Lite never creates. + + The near-twin worth naming is get_server_config on the SQL Server side, and it is not one: that + is a LISTING of sp_configure values, this is a JUDGMENT of seven logging GUCs against what each + unlocks. SQL Server's closest concept - whether the error log and default trace capture a class + of event - has no per-setting audit here either, so there is no Lite twin for this to be missing + from. */ + "get_pg_logging_audit", "get_pg_wraparound_risk", "get_pg_xmin_horizon", "get_pg_replication_slots", diff --git a/README.md b/README.md index 9ab48b740..802c221fe 100644 --- a/README.md +++ b/README.md @@ -226,7 +226,7 @@ Configuration is a single JSON file with no schedule knobs. See the **[Darling o | Alerts (tray + email + webhooks) | Yes | Email + webhooks (headless) | Yes | | Themes | Dark and light | Dark and light | Dark and light | | Portability | Single executable | Portable service + viewer zip | Server-bound | -| MCP server (LLM integration) | Built-in (87 tools) | On request (151 tools) | Built into Dashboard (66 tools) | +| MCP server (LLM integration) | Built-in (87 tools) | On request (152 tools) | Built into Dashboard (66 tools) | --- @@ -362,7 +362,7 @@ claude mcp add --transport http --scope user sql-monitor http://localhost:5151/ ### Available Tools -**Lite** exposes 87 tools; **Darling** exposes 151 (the analysis + data-read surface plus its write tools) on request; the deprecated **Dashboard** exposes 66 (see [deprecated/Dashboard/README.md](deprecated/Dashboard/README.md)). Core tools are shared. +**Lite** exposes 87 tools; **Darling** exposes 152 (the analysis + data-read surface plus its write tools) on request; the deprecated **Dashboard** exposes 66 (see [deprecated/Dashboard/README.md](deprecated/Dashboard/README.md)). Core tools are shared. | Category | Tools | |---|---| diff --git a/docs/postgres-first-target-runbook.md b/docs/postgres-first-target-runbook.md index ee24af607..eba661232 100644 --- a/docs/postgres-first-target-runbook.md +++ b/docs/postgres-first-target-runbook.md @@ -408,7 +408,7 @@ sentence the data can support if both are sampled on the same grain. ## 8. Read it Through MCP — a read per collector, plus the trend, detail and config-diff readers that sit on top of -them; 33 `get_pg_*` tools in all, registered by the same service: +them; 34 `get_pg_*` tools in all, registered by the same service: | Tool | Answers | |---|---| @@ -417,6 +417,7 @@ them; 33 `get_pg_*` tools in all, registered by the same service: | `get_pg_top_queries` | top query shapes by total time, wherever `pg_stat_statements` is installed; Aurora adds the storage-vs-cache I/O split and per-statement peak memory | | `get_pg_plans` | captured execution plans from `auto_explain`, grouped by plan shape, literals redacted before storage | | `get_pg_plan_capture_readiness` | whether the target can capture plans at all, facet by facet in causal order, with the remedy for each unmet step | +| `get_pg_logging_audit` | whether the target's logging settings (`log_lock_waits`, `log_temp_files`, `log_autovacuum_min_duration`, `log_checkpoints`, `log_connections`/`log_disconnections`, `log_min_duration_statement`) are producing the lines they could — per setting: verdict, what it unlocks, the recommended value with its cost, and the remedy in your hosting flavour's syntax; judged from the stored `pg_server_config` snapshot, so it answers once `pg_server_config` has its first snapshot (step 7: 60 min) | | `get_pg_wraparound_risk` | XID and MultiXact freeze headroom — how close to a write outage | | `get_pg_xmin_horizon` | *why* vacuum is reclaiming nothing, attributed to the specific holder | | `get_pg_replication_slots` | slot health, and whether retained WAL is still growing | @@ -685,7 +686,11 @@ sends a new operator away from the thing that would have answered their first we - **Plan capture** shipped: `pg_plan_capture` — #2566 self-hosted via the server log, #2538/#2692 on Aurora/RDS via the log API — with `pg_plan_capture_readiness` naming any missing precondition. The grants are step 1's log-reader and IAM subsections, the cadence trap is in step 7, and `get_pg_plans` / - `get_pg_plan_capture_readiness` are in step 8's table. + `get_pg_plan_capture_readiness` are in step 8's table. Its sibling `get_pg_logging_audit` (#3607) applies + the same facet-and-remedy shape to the rest of the logging surface — lock waits, temp files, autovacuum + runs, checkpoints, connection churn, slow statements — so the two together are the onboarding answer to + "is this target telling us everything it could"; read both before concluding anything from an empty + target-side log. - **Blocking chains** shipped: `pg_blocking` (step 7 — a sample, not an event log), `get_pg_blocking` (step 8) and the `Blocking Detected` alert with its own count knob (step 9). The [blocking design note](postgres-blocking-design-note.md) records what building it changed. diff --git a/llms.txt b/llms.txt index bd0c415ff..daba28359 100644 --- a/llms.txt +++ b/llms.txt @@ -1,6 +1,6 @@ # SQL Server Performance Monitor -> Free, open-source SQL Server performance monitoring tool by Erik Darling (Darling Data, LLC). 41 T-SQL collectors, real-time alerts, graphical execution plan viewer, and a built-in MCP server with 87-151 tools for AI-powered analysis. Two current editions: Lite (standalone desktop app with DuckDB storage) and Darling (headless 24/7 service with a central PostgreSQL/TimescaleDB store plus a detached viewer); the older Full/Dashboard edition is deprecated. Replaces expensive commercial tools like SentryOne and Solarwinds DPA. MIT licensed. +> Free, open-source SQL Server performance monitoring tool by Erik Darling (Darling Data, LLC). 41 T-SQL collectors, real-time alerts, graphical execution plan viewer, and a built-in MCP server with 87-152 tools for AI-powered analysis. Two current editions: Lite (standalone desktop app with DuckDB storage) and Darling (headless 24/7 service with a central PostgreSQL/TimescaleDB store plus a detached viewer); the older Full/Dashboard edition is deprecated. Replaces expensive commercial tools like SentryOne and Solarwinds DPA. MIT licensed. - Supports SQL Server 2016-2025, Azure SQL Managed Instance, AWS RDS for SQL Server, and Azure SQL Database (Lite only) - Monitors wait stats, query performance, blocking chains, deadlocks, memory grants, file I/O, tempdb, CPU, perfmon counters, and more From f68180217d9c3a31a3d36833556dea7e49b94122 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 17:00:05 -0400 Subject: [PATCH 63/69] The daily summary stops painting purged months green, and every MCP filter is part of the query: parallel_only/min_dop/blocking_only cut before the page, a negative window is refused, an unknown source names the accepted set (#3541 A9/A13) (#3641) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * The daily summary stops painting purged months green, and every MCP filter is part of the query (#3541 A9/A13) A9: the daily aggregate's day spine outlives its signals (collection log 60 d, alert log 90 d, signals 30 d) and COALESCEs each purged signal to a measured zero, so a year-long range banded the purged months Healthy. Both SKUs' readers now compute the store's retention horizon (Darling: the shortest EFFECTIVE fleet retention among the seven signal collectors and the two log constants; Lite: RetentionService.ArchiveRetentionMonths) on the reader's wall clock, the SQL projects how many signal sources still hold rows for the day, and the shared DailySummaryRetention.StateFor judges each day: purged (past horizon, no signal rows -> No Data, never Healthy), past_horizon (rows survive, verdict withheld), no_run_record, collected. The range tool publishes retention_horizon / days_before_horizon / purged_day_count / collected_day_count and each row its data_state + data_note; the single-day tool answers a purged day with the unavailable envelope. summary_date is TryParseExact yyyy-MM-dd on both SKUs (McpHelpers.ParseSummaryDate). A13: parallel_only / min_dop are a HAVING floor on the grouped population before the CPU ranking and the cap (both SKUs, both Darling grouping reads), with filter_applied on the payload and a filtered-miss sentence instead of "no query stats". get_active_queries pushes database_name and blocking_only into the SQL, counts the filtered population with COUNT(*) OVER (), pages at limit + 1 (snapshots_returned / truncated / bounds / order), never strips a head blocker some row in the same capture names (is_head_blocker), and names why a victim's blocker is absent (blocker_not_shown: not_captured / filtered / past_page). The three uncapped reads refuse a non-positive hours_back through McpHelpers.ValidateUncappedWindow instead of Math.Abs. The get_analysis_facts source filter is validated against FactScorer.KnownSources (15, pinned to every collector literal) and refused with the whole set. Census: get_active_queries joins McpPageContractTests' paged dialect on both SKUs; new A13 pins (no .Where after the read, predicates before ORDER BY/LIMIT, no Math.Abs on a parameter anywhere, the three uncapped reads on the shared validator); live PostgreSQL round-trips for the filter, head-blocker, purged/past-horizon/override arms; Lite DuckDB twins. * xUnit2009: Assert.EndsWith for the daily SQL's trailing ORDER BY pin * A day inside retention with no run record keeps its band; the displaced range-read summary goes back to its member; the two fixed-date Lite fixtures follow the horizon CI's first execution caught three things the Mac harness could not. (1) PerformanceCalendarDataTests has always pinned an alert-only day (no collection-log run) as Warning — the alert is real and inside retention every zero is a measurement — so no_run_record is now a DISCLOSURE (error share has no denominator) that keeps the band, and only the two past-horizon states withhold it; both SKUs' ToSignals, the descriptions, the instructions rows and the pins say so. (2) The same fixture's fixed July 2026 month would have drifted past the three-month horizon within weeks; it is now the calendar month two months before the current one. (3) DocCommentHygiene: the DailySummaryRangeReadResult insertion had displaced GetDailySummaryRangeAsync's summary block. Plus the FindingsRetentionHorizonPinTests literal pin follows the promoted RetentionService.ArchiveRetentionMonths. --- Darling/Darling.Tests/DailyHealthBandTests.cs | 100 +++++ .../DarlingMcpHealthToolsTests.cs | 176 ++++++++- .../Darling.Tests/FactSourceRegistryTests.cs | 133 +++++++ .../McpFilterSemanticsLivePostgresTests.cs | 369 ++++++++++++++++++ Darling/Darling.Tests/McpPageContractTests.cs | 167 +++++++- .../Mcp/DarlingDataReader.cs | 30 +- .../Mcp/DarlingHealthReader.cs | 144 ++++++- .../Mcp/DarlingMcpDataTools.cs | 106 +++-- .../Mcp/DarlingMcpHealthTools.cs | 57 ++- .../Mcp/DarlingMcpInstructions.cs | 10 +- .../Mcp/DarlingMcpReadParameters.cs | 4 + .../Mcp/DarlingMcpSessionTools.cs | 91 ++++- .../Mcp/DarlingMcpTools.cs | 16 +- .../Mcp/DarlingSessionReader.cs | 187 +++++++-- .../DailySummarySql.cs | 20 +- Lite.Tests/DailyHealthBandTests.cs | 100 +++++ .../FindingsRetentionHorizonPinTests.cs | 5 +- Lite.Tests/McpMissMessageParityPinTests.cs | 10 + Lite.Tests/McpPageContractTests.cs | 207 ++++++++++ Lite.Tests/PerformanceCalendarDataTests.cs | 13 +- Lite/Mcp/McpAnalysisTools.cs | 15 +- Lite/Mcp/McpHealthTools.cs | 105 +++-- Lite/Mcp/McpInstructions.cs | 10 +- Lite/Mcp/McpQueryTools.cs | 35 +- Lite/Mcp/McpSessionTools.cs | 85 +++- Lite/Services/CollectionBackgroundService.cs | 2 +- Lite/Services/LocalDataService.Blocking.cs | 196 ++++++++++ .../Services/LocalDataService.DailySummary.cs | 85 +++- Lite/Services/LocalDataService.QueryStats.cs | 20 +- Lite/Services/RetentionService.cs | 15 +- PerformanceMonitor.Analysis/FactScorer.cs | 25 ++ .../DailySummaryDataState.cs | 161 ++++++++ PerformanceMonitor.Common/Mcp/McpHelpers.cs | 100 +++++ 33 files changed, 2582 insertions(+), 217 deletions(-) create mode 100644 Darling/Darling.Tests/FactSourceRegistryTests.cs create mode 100644 Darling/Darling.Tests/McpFilterSemanticsLivePostgresTests.cs create mode 100644 PerformanceMonitor.Common/DailySummaryDataState.cs diff --git a/Darling/Darling.Tests/DailyHealthBandTests.cs b/Darling/Darling.Tests/DailyHealthBandTests.cs index 9b290070b..77a9d11a8 100644 --- a/Darling/Darling.Tests/DailyHealthBandTests.cs +++ b/Darling/Darling.Tests/DailyHealthBandTests.cs @@ -589,3 +589,103 @@ public void TodayCell_MinutesOld_FallsToTheUnrateableArm() DailyHealthBandCalculator.Classify(Signals(deadlocks: 1, window: window))); } } + + +/// +/// #3541 A9: the one decision both SKUs' daily-summary readers make about a returned day — is it a +/// measurement, or the shape retention left behind — pinned identically here and in the twin project. +/// +/// The defect: the daily aggregate's spine is a UNION over nine sources aging out at different +/// horizons, each COALESCEd to zero, so a day between the shortest horizon (the signals' 30 days) and the +/// longest (the alert log's 90) kept its spine row while every signal the band reads was gone — and zeros +/// band Healthy. The state is decided from two inputs, the day and the horizon; the run count only decides +/// between the two INSIDE-retention states. The horizon test comes first, because runs > 0 alone was +/// half the fix and called the whole second month collected. +/// +public class DailySummaryRetentionTests +{ + private static readonly DateTime Horizon = new(2026, 8, 19); + + [Theory] + [InlineData("2026-08-18", 1_440, 0, DailySummaryDataState.Purged)] /* the day before: a surviving run record does not rescue it */ + [InlineData("2026-08-18", 0, 0, DailySummaryDataState.Purged)] /* nor does the absence of one change the verdict */ + [InlineData("2026-07-01", 5, 0, DailySummaryDataState.Purged)] + [InlineData("2026-08-18", 1_440, 7, DailySummaryDataState.PastHorizon)] /* signal rows still there: the purge has not reached it */ + [InlineData("2026-08-18", 0, 1, DailySummaryDataState.PastHorizon)] /* even one source present withholds "purged" */ + [InlineData("2026-08-19", 1, 0, DailySummaryDataState.Collected)] /* the horizon day itself is held */ + [InlineData("2026-09-01", 1_440, 7, DailySummaryDataState.Collected)] + [InlineData("2026-09-01", 1_440, 0, DailySummaryDataState.Collected)] /* inside retention a quiet day needs no signal rows to be collected */ + [InlineData("2026-09-01", 0, 3, DailySummaryDataState.NoRunRecord)] /* inside retention, signals but no run recorded — a disclosure, the band stands */ + public void TheState_IsDecidedByTheHorizonAndPresenceFirst_ThenByTheRunCount(string day, long runs, int present, DailySummaryDataState expected) + { + var date = DateTime.ParseExact(day, "yyyy-MM-dd", System.Globalization.CultureInfo.InvariantCulture); + Assert.Equal(expected, DailySummaryRetention.StateFor(date, runs, present, Horizon)); + /* A time-of-day on either side changes nothing: the decision is on DATES. */ + Assert.Equal(expected, DailySummaryRetention.StateFor(date.AddHours(23), runs, present, Horizon.AddHours(5))); + } + + /// The horizon is the cutoff instant's DATE — see + /// for why that is exact on the TimescaleDB path and at most a partial day generous on the DELETE path. + [Fact] + public void TheHorizon_IsTheCutoffsDate() + { + var now = new DateTime(2026, 9, 18, 14, 30, 0, DateTimeKind.Utc); + Assert.Equal(new DateTime(2026, 8, 19), DailySummaryRetention.HorizonFor(now, 30)); + Assert.Equal(new DateTime(2026, 9, 17), DailySummaryRetention.HorizonFor(now, 1)); + Assert.Equal(DateTimeKind.Utc, DailySummaryRetention.HorizonFor(now, 30).Kind); + Assert.Throws(() => DailySummaryRetention.HorizonFor(now, 0)); + } + + [Fact] + public void TheVocabulary_IsOneWordPerState_AndTheNoteSaysWhatTheZerosAre() + { + Assert.Equal("collected", DailySummaryRetention.Label(DailySummaryDataState.Collected)); + Assert.Equal("purged", DailySummaryRetention.Label(DailySummaryDataState.Purged)); + Assert.Equal("past_horizon", DailySummaryRetention.Label(DailySummaryDataState.PastHorizon)); + Assert.Equal("no_run_record", DailySummaryRetention.Label(DailySummaryDataState.NoRunRecord)); + + Assert.Null(DailySummaryRetention.Note(DailySummaryDataState.Collected, Horizon)); + var purged = DailySummaryRetention.Note(DailySummaryDataState.Purged, Horizon)!; + Assert.StartsWith("PURGED", purged, StringComparison.Ordinal); + Assert.Contains("2026-08-19", purged, StringComparison.Ordinal); + Assert.Contains("absences, not measurements", purged, StringComparison.Ordinal); + var past = DailySummaryRetention.Note(DailySummaryDataState.PastHorizon, Horizon, 3)!; + Assert.StartsWith("PAST HORIZON", past, StringComparison.Ordinal); + Assert.Contains("3 of 7 signal sources", past, StringComparison.Ordinal); + Assert.Equal(7, DailySummaryRetention.SignalSourceCount); + var noRun = DailySummaryRetention.Note(DailySummaryDataState.NoRunRecord, Horizon)!; + Assert.StartsWith("NO RUN RECORD", noRun, StringComparison.Ordinal); + Assert.Contains("the band stands", noRun, StringComparison.Ordinal); + } + + /// The band's own contract, end to end: a signals projection with HasData folded from a + /// past-horizon state is No Data even under a Critical count — the calendar's grey, never green or red. + /// The two inside-retention states keep the band, because inside retention a zero is a measurement. + [Fact] + public void APastHorizonDay_BandsNoData_WhateverItsCounts_AndAnInsideRetentionDayKeepsItsBand() + { + foreach (var state in new[] { DailySummaryDataState.Collected, DailySummaryDataState.NoRunRecord }) + { + var signals = new DailyHealthSignals + { + HasData = state is not (DailySummaryDataState.Purged or DailySummaryDataState.PastHorizon), + Deadlocks = 480, + CollectionRuns = 1_440, + Window = TimeSpan.FromDays(1), + }; + Assert.Equal(DailyHealthBand.Critical, DailyHealthBandCalculator.Classify(signals)); + } + + foreach (var state in new[] { DailySummaryDataState.Purged, DailySummaryDataState.PastHorizon }) + { + var signals = new DailyHealthSignals + { + HasData = state is not (DailySummaryDataState.Purged or DailySummaryDataState.PastHorizon), + Deadlocks = 480, + CollectionRuns = 1_440, + Window = TimeSpan.FromDays(1), + }; + Assert.Equal(DailyHealthBand.NoData, DailyHealthBandCalculator.Classify(signals)); + } + } +} diff --git a/Darling/Darling.Tests/DarlingMcpHealthToolsTests.cs b/Darling/Darling.Tests/DarlingMcpHealthToolsTests.cs index 9f365134f..583122703 100644 --- a/Darling/Darling.Tests/DarlingMcpHealthToolsTests.cs +++ b/Darling/Darling.Tests/DarlingMcpHealthToolsTests.cs @@ -126,6 +126,25 @@ public void DailySummarySql_DayBucketed_AllSources() Assert.Contains("FROM v_memory_pressure_events", sql, StringComparison.Ordinal); Assert.Contains("FROM config_alert_log", sql, StringComparison.Ordinal); Assert.Contains("day_spine", sql, StringComparison.Ordinal); + + /* #3541 A9: the presence count is the LAST projection, after collection_runs, so the thirteen positional + reads before it stay put — and it counts the seven signal joins, never the collection log or the + alert log, whose survival is the reason a day can outlive its signals. The same pin is written for + Lite's copy in DailySummaryCpuBarPinTests' neighbourhood by construction: both SQLs are read by the + SAME ordinal (13) and judged by the SAME DailySummaryRetention.StateFor. */ + Assert.True(sql.IndexOf("AS collection_runs", StringComparison.Ordinal) < sql.IndexOf("AS signal_sources_present", StringComparison.Ordinal), + "signal_sources_present must trail collection_runs so the positional reads before it stay put"); + Assert.Equal(DailySummaryRetention.SignalSourceCount, System.Text.RegularExpressions.Regex.Matches(sql, @"CASE WHEN (\w+)\.d IS NULL THEN 0 ELSE 1 END").Count); + Assert.DoesNotContain("CASE WHEN cl.d IS NULL", sql, StringComparison.Ordinal); + Assert.DoesNotContain("CASE WHEN al.d IS NULL", sql, StringComparison.Ordinal); + Assert.EndsWith("ORDER BY s.d", sql.TrimEnd(), StringComparison.Ordinal); + + /* And Lite's copy carries the same arm, in the same position, read at the same ordinal. */ + var lite = RepoFile.ReadRepoFile("Lite", "Services", "LocalDataService.DailySummary.cs"); + Assert.True(lite.IndexOf("AS collection_runs", StringComparison.Ordinal) < lite.IndexOf("AS signal_sources_present", StringComparison.Ordinal)); + Assert.Equal(DailySummaryRetention.SignalSourceCount, System.Text.RegularExpressions.Regex.Matches(lite, @"CASE WHEN (\w+)\.d IS NULL THEN 0 ELSE 1 END").Count); + Assert.Contains("SignalSourcesPresent = reader.IsDBNull(13)", lite, StringComparison.Ordinal); + Assert.Contains("SignalSourcesPresent = reader.IsDBNull(13)", RepoFile.ReadRepoFile("Darling", "PerformanceMonitor.Darling.Service", "Mcp", "DarlingHealthReader.cs"), StringComparison.Ordinal); } [Theory] @@ -183,6 +202,146 @@ day under raised tiers is not Critical. */ Assert.Equal("No Data", noData.OverallHealth); } + /* ---------------- #3541 A9: retention ghosts (no live PG) ---------------- */ + + /// + /// The row the reader stamps Purged bands No Data whatever the spine still holds for it. This is + /// the defect in one row: HasData: true (a spine row exists — the run record outlives the signals + /// by 30 days), CollectionRuns non-zero, every signal a COALESCEd zero — and before the state + /// existed that banded Healthy. The same row judged Collected is the Healthy it always was, so the state + /// is the ONLY thing that moved the verdict. + /// + [Fact] + public void DailySummaryRow_PurgedOrPastHorizon_IsNoData_NeverHealthy() + { + var date = new DateTime(2026, 7, 9, 0, 0, 0, DateTimeKind.Unspecified); + var shell = new Reader.DailySummaryReadRow(date, 0m, "", 0, 0, 0, 0, 0, 0, 0, 0, 0, HasData: true) { CollectionRuns = 1_440 }; + + Assert.Equal(DailyHealthBand.Healthy, shell.HealthBand); + Assert.Equal(DailyHealthBand.NoData, (shell with { DataState = DailySummaryDataState.Purged }).HealthBand); + Assert.Equal("No Data", (shell with { DataState = DailySummaryDataState.Purged }).OverallHealth); + Assert.Equal(DailyHealthBand.NoData, (shell with { DataState = DailySummaryDataState.PastHorizon }).HealthBand); + /* Inside retention a zero is a measurement: a day with no run record keeps its band (an alert-only + day is Warning, as PerformanceCalendarDataTests has always pinned on Lite), with the caveat on the row. */ + Assert.Equal(DailyHealthBand.Healthy, (shell with { DataState = DailySummaryDataState.NoRunRecord, CollectionRuns = 0 }).HealthBand); + Assert.Equal(DailyHealthBand.Warning, (shell with { DataState = DailySummaryDataState.NoRunRecord, CollectionRuns = 0, AlertCount = 1 }).HealthBand); + + /* A purged day with a real, surviving alert is STILL No Data: the composite band needs every input, + and Warning-on-alerts-alone would understate a day whose deadlocks are gone. */ + var withAlert = shell with { AlertCount = 3, DataState = DailySummaryDataState.Purged }; + Assert.Equal(DailyHealthBand.NoData, withAlert.HealthBand); + Assert.False(withAlert.ToSignals().HasData); + } + + /// + /// The horizon is the SHORTEST effective retention among the sources — on a default store the signal + /// collectors' shared 30 (), never the collection + /// log's 60 or the alert log's 90, which are precisely the horizons that let a spine row outlive its + /// signals. A fleet override on one signal collector moves it; raising every collector past the log + /// leaves the log as the floor. + /// + [Fact] + public void ShortestSignalRetention_IsTheSignalsDefault_AndFollowsFleetOverrides() + { + var none = System.Array.Empty(); + Assert.Equal(PerformanceMonitor.Darling.Service.DarlingRetention.DataRetentionBaseDays, Reader.ShortestSignalRetentionDays(none)); + Assert.Equal(30, PerformanceMonitor.Darling.Service.DarlingRetention.DataRetentionBaseDays); + Assert.True(Reader.ShortestSignalRetentionDays(none) < PerformanceMonitor.Darling.Service.DarlingRetention.CollectionLogRetentionDays); + Assert.True(Reader.ShortestSignalRetentionDays(none) < PerformanceMonitor.Darling.Service.DarlingRetention.AlertHistoryRetentionDays); + + /* Every signal collector the aggregate reads has a schedule entry — the resolver indexes by name. */ + foreach (var collector in Reader.DailySummarySignalCollectors) + Assert.True(CollectorScheduleDefaults.All.ContainsKey(collector), $"{collector} has no CollectorScheduleDefaults entry"); + + var shortened = new[] { new PerformanceMonitor.Darling.Service.ScheduleOverride(null, "deadlocks", null, 10, true) }; + Assert.Equal(10, Reader.ShortestSignalRetentionDays(shortened)); + + /* A PER-SERVER override does not move a shared-table purge, so it does not move the horizon. */ + var perServer = new[] { new PerformanceMonitor.Darling.Service.ScheduleOverride(42, "deadlocks", null, 10, true) }; + Assert.Equal(30, Reader.ShortestSignalRetentionDays(perServer)); + + /* cpu_utilization is floored at the baseline window exactly as the purge floors it. */ + var cpuShort = new[] { new PerformanceMonitor.Darling.Service.ScheduleOverride(null, "cpu_utilization", null, 5, true) }; + Assert.Equal(30, Reader.ShortestSignalRetentionDays(cpuShort)); + + var lengthened = Reader.DailySummarySignalCollectors + .Select(c => new PerformanceMonitor.Darling.Service.ScheduleOverride(null, c, null, 365, true)).ToArray(); + Assert.Equal(PerformanceMonitor.Darling.Service.DarlingRetention.CollectionLogRetentionDays, Reader.ShortestSignalRetentionDays(lengthened)); + } + + /// + /// The fleet-override read names the same table and the same fleet predicate the purge's resolver uses, + /// so the horizon this tool publishes is the horizon the purge enforces. + /// + [Fact] + public void FleetRetentionOverridesSql_ReadsTheFleetRows_OfTheSignalCollectors() + { + var sql = Reader.FleetRetentionOverridesSql; + Assert.Contains("FROM config_collector_schedules", sql, StringComparison.Ordinal); + Assert.Contains("server_id IS NULL", sql, StringComparison.Ordinal); + Assert.Contains("collector_name = ANY($1)", sql, StringComparison.Ordinal); + Assert.Contains("retention_days IS NOT NULL", sql, StringComparison.Ordinal); + } + + /// + /// The descriptions carry the vocabulary an agent will branch on: the horizon field, the count of days + /// before it, the purged state, and the promise that such a day is never Healthy. + /// + [Fact] + public void DailySummaryDescriptions_NameTheHorizon_ThePurgedState_AndTheExactDateFormat() + { + var range = ToolMethods().Single(m => m.GetCustomAttribute()!.Name == "get_daily_summary_range"); + var rangeText = range.GetCustomAttribute()!.Description; + Assert.Contains("retention_horizon", rangeText, StringComparison.Ordinal); + Assert.Contains("days_before_horizon", rangeText, StringComparison.Ordinal); + Assert.Contains("data_state=purged", rangeText, StringComparison.Ordinal); + Assert.Contains("NEVER Healthy", rangeText, StringComparison.Ordinal); + + var single = ToolMethods().Single(m => m.GetCustomAttribute()!.Name == "get_daily_summary"); + Assert.Contains("data_state=purged", single.GetCustomAttribute()!.Description, StringComparison.Ordinal); + var date = single.GetParameters().Single(p => p.Name == "summary_date"); + Assert.Contains("yyyy-MM-dd ONLY", date.GetCustomAttribute()!.Description, StringComparison.Ordinal); + } + + /// + /// summary_date is EXACT ISO-8601 on both SKUs' tools (through the shared parser): the spelling + /// the description promised parses, the ambiguous 01/02/2026 the general parser used to accept + /// as 2 January is refused, and the refusal names the one accepted form. + /// + [Theory] + [InlineData("2026-07-09", true)] + [InlineData(" 2026-07-09 ", true)] + [InlineData("01/02/2026", false)] + [InlineData("07/09/2026", false)] + [InlineData("2026-7-9", false)] + [InlineData("2026-07-09T00:00:00Z", false)] + [InlineData("July 9, 2026", false)] + public void SummaryDate_IsParsedExactly_OrRefusedNamingTheFormat(string input, bool accepted) + { + var error = McpHelpers.ParseSummaryDate(input, out var date); + if (accepted) + { + Assert.Null(error); + Assert.Equal(new DateTime(2026, 7, 9), date!.Value); + Assert.Equal(DateTimeKind.Utc, date.Value.Kind); + } + else + { + Assert.Null(date); + Assert.StartsWith($"Invalid summary_date value '{input}'", error, StringComparison.Ordinal); + Assert.Contains("yyyy-MM-dd", error, StringComparison.Ordinal); + } + } + + [Fact] + public void SummaryDate_Absent_MeansToday_ResolvedByTheReader() + { + Assert.Null(McpHelpers.ParseSummaryDate(null, out var none)); + Assert.Null(none); + Assert.Null(McpHelpers.ParseSummaryDate(" ", out var blank)); + Assert.Null(blank); + } + /* ---------------- advertised MCP schema ---------------- */ private static System.Collections.Generic.List BuildToolSchemas() @@ -281,12 +440,19 @@ await DarlingMcpTestData.ExecAsync(connection, ct, postgres, ServerName, boundary.ToString("yyyy-MM-dd", CultureInfo.InvariantCulture)); DarlingMcpTestData.AssertEnvelope(onItsOwnDay, ServerName, "overall_health"); - /* #3525: deadlocks band as a per-hour RATE through the card band's tiers now, so one deadlock - across a 24h day (0.04/hr, far below the measured Warning tier of 5/hr) is a Healthy day — - the old any-deadlock-is-Critical reading is gone by design. The row's VISIBILITY to the - explicit-date read is what this test pins, so assert the evidence and the honest band. */ + /* The row's VISIBILITY to the explicit-date read is what this test pins, so assert the evidence + first. The band it carries changed twice on purpose: #3525 made one deadlock across a 24h day + (0.04/hr) Healthy rather than Critical, and #3541 A9 then withheld the verdict altogether for + THIS row — a fixed day two months back is before the store's 30-day retention horizon, and a + deadlock row surviving there means the purge has not reached the day (data_state + past_horizon: the row is real, the zeros beside it may not be, so No Data rather than a green + cell). A day with no run record INSIDE the horizon reads no_run_record and keeps its band. */ Assert.Contains("\"deadlock_count\":1", onItsOwnDay, StringComparison.Ordinal); - Assert.Contains("Healthy", onItsOwnDay, StringComparison.Ordinal); + var judged = JsonDocument.Parse(onItsOwnDay).RootElement; + Assert.Equal("past_horizon", judged.GetProperty("data_state").GetString()); + Assert.Equal("No Data", judged.GetProperty("overall_health").GetString()); + Assert.Contains("1 of 7 signal sources", judged.GetProperty("data_note").GetString(), StringComparison.Ordinal); + Assert.DoesNotContain("Healthy", onItsOwnDay, StringComparison.Ordinal); Assert.Contains("2026-07-20", onItsOwnDay, StringComparison.Ordinal); /* The bug: the same rows are invisible to an implicit "today", which is what the sibling test diff --git a/Darling/Darling.Tests/FactSourceRegistryTests.cs b/Darling/Darling.Tests/FactSourceRegistryTests.cs new file mode 100644 index 000000000..b294d110c --- /dev/null +++ b/Darling/Darling.Tests/FactSourceRegistryTests.cs @@ -0,0 +1,133 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.ComponentModel; +using System.IO; +using System.Linq; +using System.Reflection; +using System.Text.RegularExpressions; +using ModelContextProtocol.Server; +using PerformanceMonitor.Analysis; +using PerformanceMonitor.Common; +using PerformanceMonitor.Darling.Service.Mcp; +using Xunit; + +namespace Darling.Tests; + +/// +/// #3541 A13: get_analysis_facts' source filter documented four of the engine's fifteen sources +/// and applied an unknown one as an equality filter, so a caller who typed any of the other eleven read +/// [] as "no facts of that kind" for a value that could never have matched. The accepted set is now +/// , published in both SKUs' descriptions and enforced by refusal. +/// +/// A registry is only the truth if nothing can emit a source it does not list, so the register is +/// pinned three ways: against every Source = "..." literal in the three fact-collector assemblies +/// (exact set equality — a new source that lands in a collector without landing here fails HERE, not in an +/// agent's empty result), against the scorer's own switch arms (a subset — two sources carry context and +/// are deliberately not scored), and against the two tools' descriptions and refusals. +/// +public sealed class FactSourceRegistryTests +{ + /// A source literal wherever it is stamped: Source = "waits" on a fact, or the one named + /// constant (AnalysisContext.FactSource = "coverage") the coverage fact is stamped from. + private static readonly Regex SourceLiteral = new(@"Source\s*=\s*""([a-z_]+)""", RegexOptions.Compiled); + + /// The three assemblies whose collectors stamp Fact.Source. The frozen Dashboard is not + /// swept: it is on bug-fix support and its collectors are a copy of Lite's. + private static readonly string[] CollectorDirectories = + { + "PerformanceMonitor.Analysis", + "Lite/Analysis", + "Darling/PerformanceMonitor.Darling.Analysis", + }; + + [Fact] + public void TheRegistry_IsExactlyTheSourcesTheCollectorsEmit() + { + var emitted = new SortedSet(StringComparer.Ordinal); + var filesSeen = 0; + foreach (var directory in CollectorDirectories) + { + var root = RepoFile.PathTo(directory); + foreach (var file in Directory.EnumerateFiles(root, "*.cs", SearchOption.AllDirectories)) + { + filesSeen++; + foreach (Match m in SourceLiteral.Matches(File.ReadAllText(file))) + emitted.Add(m.Groups[1].Value); + } + } + + Assert.True(filesSeen >= 30, $"only {filesSeen} collector sources were swept; the directories have moved"); + Assert.Equal(emitted.ToArray(), FactScorer.KnownSources.ToArray()); + } + + [Fact] + public void TheRegistry_IsSorted_AndLowercaseSnakeCase() + { + Assert.Equal(FactScorer.KnownSources.Order(StringComparer.Ordinal).ToArray(), FactScorer.KnownSources.ToArray()); + Assert.Equal(FactScorer.KnownSources.Distinct(StringComparer.Ordinal).Count(), FactScorer.KnownSources.Count); + Assert.All(FactScorer.KnownSources, s => Assert.Matches("^[a-z_]+$", s)); + /* The count the campaign wrote down was fourteen; coverage (#3538) made fifteen. A moved count is a + moved contract, and the description on both SKUs spells the list out. */ + Assert.Equal(15, FactScorer.KnownSources.Count); + } + + /// Every source the scorer's Layer-1 switch scores is registered. The reverse is deliberately + /// NOT asserted: coverage and sessions are emitted as context with base severity 0. + [Fact] + public void EveryScoredSource_IsRegistered() + { + var scorer = RepoFile.ReadRepoFile("PerformanceMonitor.Analysis", "FactScorer.cs"); + var switchStart = scorer.IndexOf("fact.BaseSeverity = fact.Source switch", StringComparison.Ordinal); + Assert.True(switchStart >= 0, "the scorer's source switch has moved"); + var switchEnd = scorer.IndexOf("};", switchStart, StringComparison.Ordinal); + var arms = Regex.Matches(scorer[switchStart..switchEnd], @"""([a-z_]+)""\s*=>").Select(m => m.Groups[1].Value).ToArray(); + Assert.True(arms.Length >= 10, "the switch-arm scan found too few arms"); + Assert.All(arms, arm => Assert.Contains(arm, FactScorer.KnownSources)); + } + + /// + /// Both SKUs' source descriptions spell out the registry verbatim, in its order — an attribute + /// argument must be a constant, so the pin is what ties the constant to the list. + /// + [Fact] + public void BothDescriptions_SpellOutTheRegistry() + { + var expected = string.Join(", ", FactScorer.KnownSources); + + var darling = typeof(DarlingMcpTools).GetMethods(BindingFlags.Public | BindingFlags.Static) + .Single(m => m.GetCustomAttribute()?.Name == "get_analysis_facts") + .GetParameters().Single(p => p.Name == "source") + .GetCustomAttribute()!.Description; + Assert.Contains(expected, darling, StringComparison.Ordinal); + Assert.Contains("refused otherwise", darling, StringComparison.Ordinal); + + /* Lite's, from source: this project does not reference the desktop app. */ + var lite = RepoFile.ReadRepoFile("Lite", "Mcp", "McpAnalysisTools.cs"); + Assert.Contains(expected, lite, StringComparison.Ordinal); + Assert.Contains("FactSourceFilterDescription", lite, StringComparison.Ordinal); + Assert.Contains("McpHelpers.ValidateChoice(source, FactScorer.KnownSources, \"source\")", lite, StringComparison.Ordinal); + Assert.DoesNotContain("Filter to a specific source category: waits, blocking, config, memory", lite, StringComparison.Ordinal); + } + + [Fact] + public void AnUnknownSource_IsRefusedWithTheWholeSet_AndAKnownOneInAnyCasePasses() + { + var refusal = McpHelpers.ValidateChoice("perfmon", FactScorer.KnownSources, "source"); + Assert.NotNull(refusal); + Assert.StartsWith("Invalid source value 'perfmon'", refusal, StringComparison.Ordinal); + Assert.Contains(string.Join(", ", FactScorer.KnownSources), refusal, StringComparison.Ordinal); + + Assert.Null(McpHelpers.ValidateChoice("waits", FactScorer.KnownSources, "source")); + Assert.Null(McpHelpers.ValidateChoice("Bad_Actor", FactScorer.KnownSources, "source")); + Assert.Null(McpHelpers.ValidateChoice(null, FactScorer.KnownSources, "source")); + Assert.Null(McpHelpers.ValidateChoice(" ", FactScorer.KnownSources, "source")); + } +} diff --git a/Darling/Darling.Tests/McpFilterSemanticsLivePostgresTests.cs b/Darling/Darling.Tests/McpFilterSemanticsLivePostgresTests.cs new file mode 100644 index 000000000..6a1f17614 --- /dev/null +++ b/Darling/Darling.Tests/McpFilterSemanticsLivePostgresTests.cs @@ -0,0 +1,369 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Linq; +using System.Text.Json; +using System.Threading; +using System.Threading.Tasks; +using Npgsql; +using PerformanceMonitor.Collectors; +using PerformanceMonitor.Common; +using PerformanceMonitor.Darling.Service; +using PerformanceMonitor.Darling.Service.Mcp; +using PerformanceMonitor.Darling.Storage; +using Xunit; + +namespace Darling.Tests; + +/// +/// #3541 A13 / A9 against live PostgreSQL: the properties only the SQL can prove. +/// +/// A filter is part of the query (A13). get_top_queries_by_cpu applied +/// parallel_only / min_dop in C# over the top-N page the SQL had already cut, so a box whose +/// hottest N plans were all serial answered an EMPTY page under parallel_only while the window held a +/// parallel plan just past the cut. The fixture is exactly that shape: three serial groups hotter than one +/// parallel group, read at top = 2. The old code returned nothing; the fixed read must return the +/// parallel group, and the unfiltered read at the same cap must still return the two hottest serial ones — +/// the pair is what separates "filter in the query" from "filter the page". +/// +/// get_active_queries read the whole window, filtered in C#, published the pre-filter +/// rows.Count as total_snapshots, and its WAITFOR trim dropped the head blocker its victims +/// pointed at. One capture is seeded with the three blocker situations the tool now names: a victim whose +/// blocker is a WAITFOR shell in the same capture (kept, flagged), a victim whose blocker was never captured +/// (an idle open transaction), and a victim in one database whose blocker is in another (present unfiltered, +/// filtered under database_name). The count beside the page must be the FILTERED population's, +/// and truncation must be observed on it. +/// +/// Retention ghosts (A9). A collector run 45 days back sits inside the collection log's 60-day +/// horizon and outside the signals' 30-day one — the exact stretch that used to band Healthy on COALESCEd +/// zeros. It must come back purged / NoData with the horizon stated; today's run must stay +/// collected / Healthy beside it; the single-day read of the purged day must refuse a verdict; +/// and a fleet-wide retention override on ONE signal collector must move the horizon, because the horizon is +/// the store's effective retention rather than the shipped default. +/// +[Collection("live-postgres")] +public sealed class McpFilterSemanticsLivePostgresTests +{ + private const string ServerName = "darling-mcp-filter-semantics-e2e"; + private static readonly int ServerId = ServerIdHelper.GetDeterministicHashCode(ServerName); + private const string Db = "StackOverflow"; + private const string OtherDb = "AdventureWorks"; + private static string? ConnectionString => Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + + [Fact] + public async Task ParallelFilter_RanksTheFilteredPopulation_NotTheFilteredPage() + { + var cs = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(cs), "Set DARLING_TEST_PG to a Postgres connection string to run the live filter-semantics test."); + + var ct = TestContext.Current.CancellationToken; + using var connection = new NpgsqlConnection(cs); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + await DeleteRowsAsync(connection, ct); + await using var postgres = NpgsqlDataSource.Create(cs!); + + var bodySucceeded = false; + try + { + await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); + var now = DarlingMcpTestData.TruncateToSeconds(DateTime.UtcNow).AddMinutes(-2); + + /* Three serial groups, hottest first, then ONE parallel group cooler than all three. */ + await PlantQueryAsync(connection, ct, now, "0xSERIAL1", "SELECT 1", cpuUs: 900_000L, maxDop: 1); + await PlantQueryAsync(connection, ct, now, "0xSERIAL2", "SELECT 2", cpuUs: 800_000L, maxDop: 1); + await PlantQueryAsync(connection, ct, now, "0xSERIAL3", "SELECT 3", cpuUs: 700_000L, maxDop: 1); + await PlantQueryAsync(connection, ct, now, "0xPARALLEL", "SELECT 4", cpuUs: 100_000L, maxDop: 8); + + /* Unfiltered at top = 2: the two hottest, both serial, and no filter stated. */ + var unfiltered = JsonDocument.Parse(await DarlingMcpDataTools.GetTopQueriesByCpu(postgres, ServerName, 1, 2)).RootElement; + Assert.Equal(new[] { "0xSERIAL1", "0xSERIAL2" }, Hashes(unfiltered)); + Assert.Equal(JsonValueKind.Null, unfiltered.GetProperty("filter_applied").ValueKind); + + /* parallel_only at top = 2: the OLD code cut the two serial rows and then filtered them away — + an empty page over a window that holds a parallel plan. The fixed read ranks the parallel + population and returns it. */ + var parallel = JsonDocument.Parse(await DarlingMcpDataTools.GetTopQueriesByCpu(postgres, ServerName, 1, 2, parallel_only: true)).RootElement; + Assert.Equal(new[] { "0xPARALLEL" }, Hashes(parallel)); + Assert.Contains("max_dop >= 2", parallel.GetProperty("filter_applied").GetString(), StringComparison.Ordinal); + Assert.True(parallel.GetProperty("queries")[0].GetProperty("is_parallel").GetBoolean()); + + /* min_dop above the seeded DOP: an empty FILTERED page is the window's answer, not a collection miss. */ + var tooHigh = JsonDocument.Parse(await DarlingMcpDataTools.GetTopQueriesByCpu(postgres, ServerName, 1, 2, min_dop: 16)).RootElement; + Assert.Equal("empty", tooHigh.GetProperty("status").GetString()); + Assert.Contains("max_dop >= 16", tooHigh.GetProperty("message").GetString(), StringComparison.Ordinal); + Assert.Contains("applied in SQL over the whole window", tooHigh.GetProperty("message").GetString(), StringComparison.Ordinal); + + /* The rollup read carries the same floor. */ + var rolled = JsonDocument.Parse(await DarlingMcpDataTools.GetTopQueriesByCpu(postgres, ServerName, 1, 2, parallel_only: true, group_by: "host_object")).RootElement; + Assert.Equal(new[] { "0xPARALLEL" }, Hashes(rolled)); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, async (cleanup, cleanupCt) => + await DeleteRowsAsync(cleanup, cleanupCt)); + } + } + + [Fact] + public async Task ActiveQueries_FiltersInTheQuery_KeepsHeadBlockers_AndNamesAbsentOnes() + { + var cs = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(cs), "Set DARLING_TEST_PG to a Postgres connection string to run the live filter-semantics test."); + + var ct = TestContext.Current.CancellationToken; + using var connection = new NpgsqlConnection(cs); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + await DeleteRowsAsync(connection, ct); + await using var postgres = NpgsqlDataSource.Create(cs!); + + var bodySucceeded = false; + try + { + await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); + var t = DarlingMcpTestData.TruncateToSeconds(DateTime.UtcNow).AddMinutes(-2); + + /* One capture: + 55 (Db) blocked by 60 — a WAITFOR shell in the same capture: the classic head blocker. + 60 (Db) WAITFOR DELAY, blocking 55. The old read stripped it. + 56 (Db) blocked by 61 — 61 never captured (idle open transaction). + 57 (OtherDb) blocked by 62 — 62 captured, in Db. + 62 (Db) running, blocking 57. + 70 (Db) running, not involved in blocking. + A SECOND capture two minutes earlier holds an unrelated session 60 so a cross-capture match + would wrongly resurrect it as a head blocker: same id, different capture, must NOT be kept. */ + await PlantSnapshotAsync(connection, ct, t, 55, Db, "UPDATE Posts SET Score = 1", blockingSessionId: 60, cpuMs: 500); + await PlantSnapshotAsync(connection, ct, t, 60, Db, "WAITFOR DELAY '00:05'", blockingSessionId: 0, cpuMs: 1); + await PlantSnapshotAsync(connection, ct, t, 56, Db, "DELETE FROM Votes", blockingSessionId: 61, cpuMs: 400); + await PlantSnapshotAsync(connection, ct, t, 57, OtherDb, "SELECT * FROM Sales", blockingSessionId: 62, cpuMs: 300); + await PlantSnapshotAsync(connection, ct, t, 62, Db, "UPDATE Users SET Reputation = 0", blockingSessionId: 0, cpuMs: 900); + await PlantSnapshotAsync(connection, ct, t, 70, Db, "SELECT COUNT(*) FROM Comments", blockingSessionId: 0, cpuMs: 200); + await PlantSnapshotAsync(connection, ct, t.AddMinutes(-2), 60, Db, "WAITFOR DELAY '00:05'", blockingSessionId: 0, cpuMs: 1); + + /* Unfiltered: the WAITFOR head blocker is on the page, flagged; the stale WAITFOR in the other + capture is not; the never-captured blocker is named as such; the cross-database blocker is present. */ + var all = JsonDocument.Parse(await DarlingMcpSessionTools.GetActiveQueries(postgres, ServerName, 1, limit: 50)).RootElement; + var rows = all.GetProperty("queries").EnumerateArray().ToArray(); + Assert.Equal(6, rows.Length); + Assert.Equal(6, all.GetProperty("total_snapshots").GetInt64()); + Assert.Equal(6, all.GetProperty("snapshots_returned").GetInt32()); + Assert.False(all.GetProperty("truncated").GetBoolean()); + Assert.Equal("collection_time_desc", all.GetProperty("order").GetString()); + + var head = Assert.Single(rows, r => r.GetProperty("session_id").GetInt32() == 60); + Assert.True(head.GetProperty("is_head_blocker").GetBoolean()); + Assert.StartsWith("WAITFOR", head.GetProperty("query_text").GetString(), StringComparison.Ordinal); + + Assert.Equal(JsonValueKind.Null, Row(rows, 55).GetProperty("blocker_not_shown").ValueKind); + Assert.Equal("not_captured", Row(rows, 56).GetProperty("blocker_not_shown").GetString()); + Assert.Equal(JsonValueKind.Null, Row(rows, 57).GetProperty("blocker_not_shown").ValueKind); + Assert.Equal(JsonValueKind.Null, Row(rows, 70).GetProperty("is_head_blocker").ValueKind); + + /* database_name filter, IN the query: the population is OtherDb's one victim, its blocker is + in Db and therefore filtered — the caller asked for that database, and the row says so. */ + var other = JsonDocument.Parse(await DarlingMcpSessionTools.GetActiveQueries(postgres, ServerName, 1, OtherDb)).RootElement; + Assert.Equal(1, other.GetProperty("total_snapshots").GetInt64()); + Assert.Equal("filtered", Assert.Single(other.GetProperty("queries").EnumerateArray()).GetProperty("blocker_not_shown").GetString()); + Assert.Equal(OtherDb, other.GetProperty("filters_applied").GetProperty("database_name").GetString()); + + /* blocking_only, IN the query: victims 55/56/57 + heads 60/62 = 5; 70 is out. The count is the + blocking population's, not the window's. */ + var blocking = JsonDocument.Parse(await DarlingMcpSessionTools.GetActiveQueries(postgres, ServerName, 1, blocking_only: true)).RootElement; + Assert.Equal(5, blocking.GetProperty("total_snapshots").GetInt64()); + Assert.DoesNotContain(blocking.GetProperty("queries").EnumerateArray(), r => r.GetProperty("session_id").GetInt32() == 70); + + /* Truncation is observed on the FILTERED population: limit = 4 over 5 blocking rows is truncated, + limit = 5 is not, and the total stays the population's under both. The page is CPU-descending + within the capture, so the WAITFOR head (1 ms) falls past a 4-row page and its victim says so. */ + var cut = JsonDocument.Parse(await DarlingMcpSessionTools.GetActiveQueries(postgres, ServerName, 1, blocking_only: true, limit: 4)).RootElement; + Assert.True(cut.GetProperty("truncated").GetBoolean()); + Assert.Equal(4, cut.GetProperty("snapshots_returned").GetInt32()); + Assert.Equal(5, cut.GetProperty("total_snapshots").GetInt64()); + Assert.Equal("past_page", Row(cut.GetProperty("queries").EnumerateArray().ToArray(), 55).GetProperty("blocker_not_shown").GetString()); + var whole = JsonDocument.Parse(await DarlingMcpSessionTools.GetActiveQueries(postgres, ServerName, 1, blocking_only: true, limit: 5)).RootElement; + Assert.False(whole.GetProperty("truncated").GetBoolean()); + + /* A filtered miss names the filter rather than calling the window empty. */ + var miss = JsonDocument.Parse(await DarlingMcpSessionTools.GetActiveQueries(postgres, ServerName, 1, "NoSuchDb")).RootElement; + Assert.Equal("empty", miss.GetProperty("status").GetString()); + Assert.Contains("database_name 'NoSuchDb'", miss.GetProperty("message").GetString(), StringComparison.Ordinal); + + /* A13's third item, on the same fixture: the uncapped reads refuse a negative span. */ + Assert.StartsWith("Invalid hours_back value '-24'", await DarlingMcpDataTools.GetCollectionLog(postgres, ServerName, -24), StringComparison.Ordinal); + Assert.StartsWith("Invalid hours_back value '0'", await DarlingMcpDataTools.GetCurrentWaitsTrend(postgres, ServerName, 0), StringComparison.Ordinal); + Assert.StartsWith("Invalid hours_back value '-1'", await DarlingMcpDataTools.GetBlockingStats(postgres, ServerName, -1), StringComparison.Ordinal); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, async (cleanup, cleanupCt) => + await DeleteRowsAsync(cleanup, cleanupCt)); + } + } + + [Fact] + public async Task DailySummary_StopsPaintingPurgedDaysGreen_AndPublishesTheHorizon() + { + var cs = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(cs), "Set DARLING_TEST_PG to a Postgres connection string to run the live retention-ghost test."); + + var ct = TestContext.Current.CancellationToken; + using var connection = new NpgsqlConnection(cs); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + await DeleteRowsAsync(connection, ct); + await using var postgres = NpgsqlDataSource.Create(cs!); + + var bodySucceeded = false; + try + { + await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); + var today = DateTime.UtcNow.Date; + /* Inside the collection log's 60-day horizon, outside the signals' 30 — the ghost stretch. */ + var ghostDay = today.AddDays(-45); + /* Inside every default horizon; becomes a ghost once ONE signal's retention is overridden to 10. */ + var nearDay = today.AddDays(-20); + var expectedHorizon = DailySummaryRetention.HorizonFor(DateTime.UtcNow, DarlingRetention.DataRetentionBaseDays); + + await SeedRunAsync(connection, ct, DateTime.UtcNow.AddMinutes(-2), "SUCCESS"); + await SeedRunAsync(connection, ct, ghostDay.AddHours(12), "SUCCESS"); + await SeedRunAsync(connection, ct, nearDay.AddHours(12), "SUCCESS"); + + var range = JsonDocument.Parse(await DarlingMcpHealthTools.GetDailySummaryRange(postgres, ServerName, 60)).RootElement; + Assert.Equal(expectedHorizon.ToString("yyyy-MM-dd"), range.GetProperty("retention_horizon").GetString()); + Assert.Equal(3, range.GetProperty("day_count").GetInt32()); + Assert.Equal(1, range.GetProperty("days_before_horizon").GetInt32()); + Assert.Equal(1, range.GetProperty("purged_day_count").GetInt32()); + Assert.Equal(2, range.GetProperty("collected_day_count").GetInt32()); + + var days = range.GetProperty("days").EnumerateArray().ToArray(); + var ghost = Assert.Single(days, d => d.GetProperty("summary_date").GetString() == ghostDay.ToString("yyyy-MM-dd")); + /* The day the run record kept on the spine: runs = 1, every signal a COALESCEd zero — and it is + NOT Healthy. */ + Assert.Equal(1, ghost.GetProperty("collection_runs").GetInt64()); + Assert.Equal("purged", ghost.GetProperty("data_state").GetString()); + Assert.Equal("NoData", ghost.GetProperty("health_band").GetString()); + Assert.Equal("No Data", ghost.GetProperty("overall_health").GetString()); + Assert.Contains("PURGED", ghost.GetProperty("data_note").GetString(), StringComparison.Ordinal); + Assert.Contains(expectedHorizon.ToString("yyyy-MM-dd"), ghost.GetProperty("data_note").GetString(), StringComparison.Ordinal); + + var live = Assert.Single(days, d => d.GetProperty("summary_date").GetString() == today.ToString("yyyy-MM-dd")); + Assert.Equal("collected", live.GetProperty("data_state").GetString()); + Assert.Equal("Healthy", live.GetProperty("overall_health").GetString()); + Assert.Equal(JsonValueKind.Null, live.GetProperty("data_note").ValueKind); + + /* The single-day read of the ghost refuses a verdict, in the miss vocabulary's own word for it. */ + var single = JsonDocument.Parse(await DarlingMcpHealthTools.GetDailySummary(postgres, ServerName, ghostDay.ToString("yyyy-MM-dd"))).RootElement; + Assert.Equal("unavailable", single.GetProperty("status").GetString()); + Assert.Contains("retention_horizon", single.GetProperty("message").GetString(), StringComparison.Ordinal); + Assert.Equal("purged", single.GetProperty("hints").GetProperty("data_state").GetString()); + Assert.Equal(1, single.GetProperty("hints").GetProperty("collection_runs").GetInt64()); + + /* And of a collected day, the verdict with its state. */ + var todayRow = JsonDocument.Parse(await DarlingMcpHealthTools.GetDailySummary(postgres, ServerName)).RootElement; + Assert.Equal("collected", todayRow.GetProperty("data_state").GetString()); + Assert.Equal(expectedHorizon.ToString("yyyy-MM-dd"), todayRow.GetProperty("retention_horizon").GetString()); + + /* summary_date is exact ISO-8601: the ambiguous spelling is refused, not guessed at. */ + Assert.StartsWith("Invalid summary_date value '01/02/2026'", await DarlingMcpHealthTools.GetDailySummary(postgres, ServerName, "01/02/2026"), StringComparison.Ordinal); + + /* A signal row surviving before the horizon (the purge has not reached it — gate-held, paused, + or simply not yet run) turns the shell into past_horizon: the row is real, the verdict is + still withheld, and the day is NOT called purged because that would be false. */ + await DarlingMcpTestData.ExecAsync(connection, ct, + @"INSERT INTO deadlocks (deadlock_id, collection_time, server_id, server_name, deadlock_time, victim_process_id, victim_sql_text, deadlock_graph_xml) +VALUES ($1,$2,$3,$4,$5,$6,$7,$8)", + CollectionIdGenerator.Next(), DarlingMcpTestData.Naive(ghostDay.AddHours(6)), ServerId, ServerName, + DarlingMcpTestData.Naive(ghostDay.AddHours(6)), "process1", "DELETE FROM Posts", ""); + var survived = JsonDocument.Parse(await DarlingMcpHealthTools.GetDailySummaryRange(postgres, ServerName, 60)).RootElement; + Assert.Equal(1, survived.GetProperty("days_before_horizon").GetInt32()); + Assert.Equal(0, survived.GetProperty("purged_day_count").GetInt32()); + var pastHorizon = Assert.Single(survived.GetProperty("days").EnumerateArray(), d => d.GetProperty("summary_date").GetString() == ghostDay.ToString("yyyy-MM-dd")); + Assert.Equal("past_horizon", pastHorizon.GetProperty("data_state").GetString()); + Assert.Equal("NoData", pastHorizon.GetProperty("health_band").GetString()); + Assert.Equal(1, pastHorizon.GetProperty("deadlock_count").GetInt64()); + Assert.Contains("1 of 7 signal sources", pastHorizon.GetProperty("data_note").GetString(), StringComparison.Ordinal); + /* The single-day read of a past-horizon day is a DATA payload (the rows are there), not unavailable. */ + var singlePast = JsonDocument.Parse(await DarlingMcpHealthTools.GetDailySummary(postgres, ServerName, ghostDay.ToString("yyyy-MM-dd"))).RootElement; + Assert.Equal("past_horizon", singlePast.GetProperty("data_state").GetString()); + Assert.Equal("No Data", singlePast.GetProperty("overall_health").GetString()); + await DarlingMcpTestData.ExecAsync(connection, ct, $"DELETE FROM deadlocks WHERE server_id = {ServerId}"); + + /* The horizon is the store's EFFECTIVE retention: a fleet-wide override shortening one signal + collector to 10 days moves it, and the 20-day-old run becomes a ghost too. */ + await DarlingMcpTestData.ExecAsync(connection, ct, + "INSERT INTO config_collector_schedules (server_id, collector_name, retention_days, enabled) VALUES (NULL, 'deadlocks', 10, TRUE)"); + var shortened = JsonDocument.Parse(await DarlingMcpHealthTools.GetDailySummaryRange(postgres, ServerName, 60)).RootElement; + Assert.Equal(DailySummaryRetention.HorizonFor(DateTime.UtcNow, 10).ToString("yyyy-MM-dd"), shortened.GetProperty("retention_horizon").GetString()); + Assert.Equal(2, shortened.GetProperty("days_before_horizon").GetInt32()); + var near = Assert.Single(shortened.GetProperty("days").EnumerateArray(), d => d.GetProperty("summary_date").GetString() == nearDay.ToString("yyyy-MM-dd")); + Assert.Equal("purged", near.GetProperty("data_state").GetString()); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, async (cleanup, cleanupCt) => + await DeleteRowsAsync(cleanup, cleanupCt)); + } + } + + private static string[] Hashes(JsonElement root) => + root.GetProperty("queries").EnumerateArray().Select(q => q.GetProperty("query_hash").GetString()!).ToArray(); + + private static JsonElement Row(JsonElement[] rows, int sessionId) => + Assert.Single(rows, r => r.GetProperty("session_id").GetInt32() == sessionId); + + private static async Task PlantQueryAsync( + NpgsqlConnection connection, CancellationToken ct, DateTime at, string queryHash, string queryText, long cpuUs, int maxDop) + { + var digest = System.Security.Cryptography.SHA256.HashData(System.Text.Encoding.UTF8.GetBytes(queryText)); + await DarlingMcpTestData.ExecAsync(connection, ct, + @"INSERT INTO query_stats (collection_id, collection_time, server_id, server_name, database_name, + query_hash, query_plan_hash, sql_handle, plan_handle, query_text, + query_text_digest, delta_execution_count, delta_worker_time, + delta_elapsed_time, delta_logical_reads, min_dop, max_dop) + VALUES ($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,$11,$12,$13,$14,$15,$16,$17)", + CollectionIdGenerator.Next(), at, ServerId, ServerName, Db, + queryHash, "0xPLANHASH", "0xSQLH" + queryHash, "0xPLANH", queryText, + digest, 10L, cpuUs, cpuUs, 100L, 1, maxDop); + } + + private static Task PlantSnapshotAsync( + NpgsqlConnection connection, CancellationToken ct, DateTime at, int sessionId, string database, string text, int blockingSessionId, long cpuMs) => + DarlingMcpTestData.ExecAsync(connection, ct, + @"INSERT INTO query_snapshots (collection_id, collection_time, server_id, server_name, session_id, database_name, query_text, status, blocking_session_id, wait_type, cpu_time_ms, total_elapsed_time_ms, request_id) +VALUES ($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,$11,$12,$13)", + CollectionIdGenerator.Next(), at, ServerId, ServerName, sessionId, database, text, + blockingSessionId > 0 ? "suspended" : "running", blockingSessionId, blockingSessionId > 0 ? "LCK_M_X" : null, cpuMs, cpuMs * 2, 0); + + private static Task SeedRunAsync(NpgsqlConnection connection, CancellationToken ct, DateTime collectionTimeUtc, string status) => + DarlingMcpTestData.ExecAsync(connection, ct, @" +INSERT INTO collection_log + (log_id, server_id, server_name, collector_name, collection_time, + duration_ms, status, error_message, rows_collected, sql_duration_ms, duckdb_duration_ms) +VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11)", + CollectionIdGenerator.Next(), ServerId, ServerName, "wait_stats", + DarlingMcpTestData.Naive(collectionTimeUtc), 100, status, null, 10, 80, 20); + + private static async Task DeleteRowsAsync(NpgsqlConnection connection, CancellationToken ct) + { + var sql = string.Join(" ", new[] { "query_stats", "query_snapshots", "collection_log", "deadlocks" } + .Select(tbl => $"DELETE FROM {tbl} WHERE server_id = {ServerId};")); + sql += " DELETE FROM config_collector_schedules WHERE server_id IS NULL AND collector_name = 'deadlocks' AND retention_days = 10;"; + sql += $" DELETE FROM servers WHERE server_id = {ServerId};"; + sql += $" DELETE FROM config_monitored_servers WHERE server_id = {ServerId};"; + using var cleanup = new NpgsqlCommand(sql, connection); + await cleanup.ExecuteNonQueryAsync(ct); + } +} diff --git a/Darling/Darling.Tests/McpPageContractTests.cs b/Darling/Darling.Tests/McpPageContractTests.cs index faac2c02e..bcc624b63 100644 --- a/Darling/Darling.Tests/McpPageContractTests.cs +++ b/Darling/Darling.Tests/McpPageContractTests.cs @@ -83,15 +83,21 @@ public static readonly (Type Tools, string ToolName, string LiteFile, string Lit (typeof(DarlingMcpPlanCorrectionTools), "get_plan_corrections", "Lite/Mcp/McpPlanCorrectionTools.cs", "get_plan_corrections"), (typeof(DarlingMcpSessionTools), "get_waiting_tasks", "Lite/Mcp/McpWaitTools.cs", "get_waiting_tasks"), (typeof(DarlingMcpDataTools), "get_wait_stats", "Lite/Mcp/McpWaitTools.cs", "get_wait_stats"), + /* #3541 A13: get_active_queries joined the dialect when its filters moved into the SQL — it now pages + the FILTERED population at limit + 1, publishes snapshots_returned / truncated / the page's bounds, + and its total_snapshots is the filtered population's COUNT(*) OVER () rather than rows.Count of an + unfiltered window read. */ + (typeof(DarlingMcpSessionTools), "get_active_queries", "Lite/Mcp/McpSessionTools.cs", "get_active_queries"), ]; /// /// The source span of every paged tool on both SKUs — the census walks these BODIES, not whole files, /// for the total_* = .Count and >= limit shapes. Whole files would sweep in tools with a - /// different contract: get_active_queries publishes an UNBOUNDED window count beside shown - /// (a real total, and #3541 A13's filter semantics are its own item), and get_mute_rules' - /// total_count counts a whole set. Neither is a hidden cap, and a census that flagged them would be - /// asserting a rule the finding did not state. + /// different contract: get_mute_rules' total_count counts a whole set, which is not a hidden + /// cap, and a census that flagged it would be asserting a rule the finding did not state. + /// (get_active_queries used to be the other exclusion, for publishing an unbounded window count + /// beside its page; #3541 A13 made that count the filtered population's, computed in SQL, so it is a + /// member now.) /// private static IEnumerable<(string Label, string Body)> PagedToolBodies() { @@ -118,6 +124,7 @@ private static readonly (string Name, string Sql)[] PagedReads = (nameof(DarlingPlanCorrectionReader.PlanCorrectionsSql), DarlingPlanCorrectionReader.PlanCorrectionsSql), (nameof(DarlingSessionReader.WaitingTasksSql), DarlingSessionReader.WaitingTasksSql), (nameof(DarlingDataReader.WaitStatsSql), DarlingDataReader.WaitStatsSql), + (nameof(DarlingSessionReader.ActiveQueriesSql), DarlingSessionReader.ActiveQueriesSql), ]; /* ───────────────────────── the discriminators ───────────────────────── */ @@ -562,6 +569,158 @@ public void TheA7Discriminators_FlagTheDefectShapes_AndPassTheFixedOnes() Assert.DoesNotMatch(WindowTotalColumn, "GREATEST(reads - LAG(reads) OVER series, 0) AS d_reads"); } + /* ───────────────────────── #3541 A13: a filter is part of the query ───────────────────────── */ + + /// + /// The tools whose filters ran in C# AFTER the read — over a page the SQL had already cut, or over a + /// whole-window read whose count was then published beside the filtered page. Each body, on both SKUs, + /// must now hand every filter to its reader and never .Where( the rows between the read and the + /// emit: a filter applied after the cut makes the page the filtered remainder of an unfiltered top-N, + /// which can be EMPTY while the window holds matches, and a count taken before the filter is a total of + /// a different population from the rows beside it. + /// + public static readonly (Type Tools, string ToolName, string LiteFile, string LiteToolName)[] FilterInQueryTools = + [ + (typeof(DarlingMcpDataTools), "get_top_queries_by_cpu", "Lite/Mcp/McpQueryTools.cs", "get_top_queries_by_cpu"), + (typeof(DarlingMcpSessionTools), "get_active_queries", "Lite/Mcp/McpSessionTools.cs", "get_active_queries"), + ]; + + /// A LINQ filter over the rows a reader returned — the shape that puts the cut before the filter. + private static readonly Regex PostReadWhere = new(@"\.Where\(", RegexOptions.Compiled); + + /// A parameter read as its absolute value — the shape that answers a negative window with a + /// positive one and says nothing. + private static readonly Regex AbsOfParameter = new(@"\bMath\.Abs\(\s*(hours_back|hoursBack|days_back|daysBack|limit|top)\b", RegexOptions.Compiled); + + [Fact] + public void NoFilteredTool_FiltersItsRowsAfterTheRead_OnEitherSku() + { + foreach (var (type, darlingName, liteFile, liteName) in FilterInQueryTools) + { + foreach (var (label, body) in new[] + { + ($"Darling {darlingName}", ToolBody(ReadRepoFileLf(DarlingFileOf(type).Split('/')), darlingName)), + ($"Lite {liteName}", ToolBody(ReadRepoFileLf(liteFile.Split('/')), liteName)), + }) + { + var text = Strip(body); + var hit = PostReadWhere.Match(text); + Assert.False(hit.Success, + $"{label}: `{hit.Value}` filters the rows in C# after the read — push the predicate into the SQL so the page is the top-N of the filtered population, and the count beside it counts that population"); + /* And the filter's presence is STATED on the payload, so a stored result says what shaped it. */ + Assert.True( + text.Contains("filter_applied", StringComparison.Ordinal) || text.Contains("filters_applied", StringComparison.Ordinal), + $"{label}: the payload never names the filter that shaped its population"); + } + } + } + + /// + /// The reader statements behind them carry the predicates: the parallelism floor as a HAVING term on the + /// grouped population BEFORE the CPU ordering and the cap, and the session read's two filters as WHERE + /// terms with the population counted on the same statement above a parameterised LIMIT. + /// + [Fact] + public void TheFilteredReads_CarryTheirPredicates_BeforeTheOrderingAndTheCap() + { + foreach (var (name, sql) in new[] + { + (nameof(DarlingDataReader.TopQueriesSql), DarlingDataReader.TopQueriesSql), + (nameof(DarlingDataReader.TopQueriesByHostObjectSql), DarlingDataReader.TopQueriesByHostObjectSql), + }) + { + var floor = sql.IndexOf("COALESCE(MAX(max_dop), 0) >= $6", StringComparison.Ordinal); + var order = sql.IndexOf("ORDER BY SUM(delta_worker_time) DESC", StringComparison.Ordinal); + Assert.True(floor >= 0, $"{name}: the parallelism floor is not in the statement"); + Assert.True(order > floor, $"{name}: the parallelism floor sits after the ranking, so it filters a ranked page rather than ranking a filtered population"); + } + + var active = DarlingSessionReader.ActiveQueriesSql; + Assert.Contains("($5::text IS NULL OR w.database_name = $5)", active, StringComparison.Ordinal); + Assert.Contains("(NOT $6::boolean OR w.blocking_session_id > 0 OR h.session_id IS NOT NULL)", active, StringComparison.Ordinal); + Assert.Contains("COUNT(*) OVER () AS population_count", active, StringComparison.Ordinal); + /* The head-blocker keep: a WAITFOR row stays when a row in the SAME capture names it. */ + Assert.Contains("(w.query_text NOT LIKE 'WAITFOR%' OR h.session_id IS NOT NULL)", active, StringComparison.Ordinal); + Assert.Contains("h.collection_time = w.collection_time", active, StringComparison.Ordinal); + Assert.True(active.IndexOf("COUNT(*) OVER ()", StringComparison.Ordinal) < active.IndexOf("LIMIT $4", StringComparison.Ordinal), + "the population count must be computed above the cap, or it counts the page"); + } + + /// + /// No tool on either SKU reads a parameter as its absolute value. Three did (hours_back on the + /// uncapped reads), and a caller who sent -24 was answered about the last 24 hours with nothing to + /// say the sign had flipped. Every tool body on both SKUs is swept, comments stripped, because the fix's + /// own comments name the shape. + /// + [Fact] + public void NoTool_ReadsAParameterAsItsAbsoluteValue_OnEitherSku() + { + var examined = 0; + var offenders = new List(); + foreach (var (file, source) in AllMcpToolSources()) + { + var marks = Regex.Matches(source, @"\[McpServerTool\(Name = ""([a-z_0-9]+)"""); + for (var i = 0; i < marks.Count; i++) + { + var end = i + 1 < marks.Count ? marks[i + 1].Index : source.Length; + var body = Strip(source[marks[i].Index..end]); + examined++; + var hit = AbsOfParameter.Match(body); + if (hit.Success) + { + offenders.Add($"{file} {marks[i].Groups[1].Value}: {hit.Value}"); + } + } + } + + Assert.True(examined >= 150, $"only {examined} tool bodies were examined across both SKUs; the marker has stopped matching"); + Assert.True(offenders.Count == 0, + "these tools read a parameter as its absolute value — refuse the negative instead: " + string.Join("; ", offenders)); + } + + /// The three uncapped reads route their span through the shared refusal, on both SKUs. + [Theory] + [InlineData("Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpDataTools.cs", "get_collection_log")] + [InlineData("Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpDataTools.cs", "get_current_waits_trend")] + [InlineData("Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpDataTools.cs", "get_blocking_stats")] + [InlineData("Lite/Mcp/McpHealthTools.cs", "get_collection_log")] + [InlineData("Lite/Mcp/McpHealthTools.cs", "get_current_waits_trend")] + [InlineData("Lite/Mcp/McpHealthTools.cs", "get_blocking_stats")] + public void TheUncappedReads_RefuseANonPositiveSpan_ThroughTheSharedValidator(string file, string toolName) + { + var body = Strip(ToolBody(ReadRepoFileLf(file.Split('/')), toolName)); + Assert.Contains("McpHelpers.ValidateUncappedWindow(hours_back, as_of, out var windowEnd)", body, StringComparison.Ordinal); + Assert.DoesNotContain("McpHelpers.ResolveAsOf(", body, StringComparison.Ordinal); + } + + /// The A13 matchers, witnessed against the defect as it shipped and the fix as it landed. + [Fact] + public void TheA13Discriminators_FlagTheDefectShapes_AndPassTheFixedOnes() + { + /* The defect, verbatim from the shipped queries tool: the page is cut in SQL, then filtered. */ + const string defectFilter = """ + var rows = await DarlingDataReader.GetTopQueriesByCpuAsync(postgres, resolved.ServerId, now.AddHours(-hours_back), now, top, database_name); + var filtered = rows + .Where(r => !(parallel_only || min_dop > 1) || (r.MaxDop > 1 && r.MaxDop >= (min_dop > 1 ? min_dop : 2))) + .ToList(); + """; + Assert.Matches(PostReadWhere, defectFilter); + /* The fix: the floor is an argument to the read. */ + const string fixedFilter = """ + var minMaxDop = min_dop > 1 ? min_dop : parallel_only ? 2 : 0; + var rows = await DarlingDataReader.GetTopQueriesByCpuAsync(postgres, resolved.ServerId, now.AddHours(-hours_back), now, top, database_name, rollUpByHostObject: rollUp, minMaxDop: minMaxDop); + var result = rows.Select(r => new { max_dop = r.MaxDop }); + """; + Assert.DoesNotMatch(PostReadWhere, fixedFilter); + + /* The defect, verbatim from the three uncapped reads. */ + Assert.Matches(AbsOfParameter, " var start = end.AddHours(-Math.Abs(hours_back));"); + Assert.Matches(AbsOfParameter, " var hours = Math.Abs(hours_back);"); + Assert.DoesNotMatch(AbsOfParameter, " var start = end.AddHours(-hours_back);"); + /* A genuine absolute value of a MEASUREMENT is not a parameter flip and must pass. */ + Assert.DoesNotMatch(AbsOfParameter, " var drift = Math.Abs(observed - expected);"); + } + /// Every MCP tool source on both SKUs, LF-normalised, for the cross-SKU sweep. Through /// so a worktree checkout resolves the same root every other pin uses. private static IEnumerable<(string File, string Source)> AllMcpToolSources() diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingDataReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingDataReader.cs index d370e1fa3..ccb1d6a33 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingDataReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingDataReader.cs @@ -762,7 +762,8 @@ public static async Task> GetLatestPerfmonStatsAsync( /// 5 to drop WAITFOR shells via the latest-text LATERAL, cap at top. Summed bigints CAST back to bigint /// for the typed reader. The aggregate reads the base query_stats table (it projects no text); /// the text LATERAL reads v_query_stats, which resolves the #1767 payload dimension — the plan - /// tools read it the same way. $1 server_id, $2/$3 window (naive UTC), $4 top. + /// tools read it the same way. $1 server_id, $2/$3 window (naive UTC), $4 top, $5 database filter (NULL = all), + /// $6 lifetime max_dop floor (0 = no parallelism filter; #3541 A13). /// public const string TopQueriesSql = """ WITH ranked AS ( @@ -804,7 +805,16 @@ FROM query_stats (each proc-hosted statement groups under its own host object), while ad-hoc rows carry NULL and keep collapsing into one group per hash exactly as before. */ GROUP BY database_name, query_hash, host_object_name - HAVING SUM(delta_execution_count) > 0 OR SUM(delta_elapsed_time) > 0 + HAVING (SUM(delta_execution_count) > 0 OR SUM(delta_elapsed_time) > 0) + /* #3541 A13: the parallelism filter is part of the QUERY, applied to the grouped population + BEFORE the CPU ranking and the cap. It used to run in C# over the returned top-N page, so + parallel_only=true on a box whose twenty hottest plans were serial answered an empty page while + the window held parallel plans further down — and the engine's own CXPACKET advice sends agents + to exactly that call. $6 is the group's lifetime max_dop floor: 0 admits every group (the + unfiltered read, byte-identical in result to before), 2 is parallel_only, min_dop is itself. The + COALESCE keeps a group whose max_dop was never captured (NULL) out of a filtered page, which is + what the C# arm did too (null read as 0, and 0 > 1 is false). */ + AND COALESCE(MAX(max_dop), 0) >= $6 ORDER BY SUM(delta_worker_time) DESC LIMIT $4 + 5 ) @@ -915,7 +925,11 @@ FROM query_stats without that arm every unrelated ad-hoc statement in a database would pool into one row. */ GROUP BY database_name, host_object_name, CASE WHEN host_object_name IS NULL THEN query_hash END - HAVING SUM(delta_execution_count) > 0 OR SUM(delta_elapsed_time) > 0 + HAVING (SUM(delta_execution_count) > 0 OR SUM(delta_elapsed_time) > 0) + /* #3541 A13: same in-query parallelism floor as TopQueriesSql — see its note. Under the rollup + the group's max_dop is the max across every fragment, so a procedure whose dynamic SQL went + parallel in ANY fragment passes parallel_only, which is the question being asked. */ + AND COALESCE(MAX(max_dop), 0) >= $6 ORDER BY SUM(delta_worker_time) DESC LIMIT $4 + 5 ) @@ -963,9 +977,16 @@ ORDER BY r.total_cpu_us DESC LIMIT $4 """; + /// + /// The top-N groups by CPU, ranked over the population that passes every filter. + /// is the lifetime max_dop floor a group must reach to be ranked at all (#3541 A13): 0 for no + /// parallelism filter, 2 for parallel_only, the caller's min_dop otherwise — see + /// 's HAVING note. The filter is IN the statement so the page is the top-N of the + /// filtered population, not the filtered remainder of an unfiltered top-N. + /// public static async Task> GetTopQueriesByCpuAsync( NpgsqlDataSource postgres, int serverId, DateTime startUtc, DateTime endUtc, int top, string? databaseName, - bool rollUpByHostObject = false, CancellationToken cancellationToken = default) + bool rollUpByHostObject = false, int minMaxDop = 0, CancellationToken cancellationToken = default) { var rows = new List(); /* #2235: same parameters, same columns, different GROUP BY — see TopQueriesByHostObjectSql. */ @@ -974,6 +995,7 @@ public static async Task> GetTopQueriesByCpuAsync( AddWindow(command, serverId, startUtc, endUtc); AddInt(command, top); AddNullableText(command, databaseName); + AddInt(command, minMaxDop); await using var reader = await command.ExecuteReaderAsync(cancellationToken); while (await reader.ReadAsync(cancellationToken)) { diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingHealthReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingHealthReader.cs index c7409049c..8e17a3ea5 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingHealthReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingHealthReader.cs @@ -12,8 +12,8 @@ using System.Threading; using System.Threading.Tasks; using Npgsql; +using PerformanceMonitor.Analysis.Baselines; using PerformanceMonitor.Common; - using PerformanceMonitor.Darling.Storage; namespace PerformanceMonitor.Darling.Service.Mcp; @@ -215,9 +215,31 @@ public sealed record DailySummaryReadRow( /// trailing collection_runs column. public long CollectionRuns { get; init; } + /// + /// Whether this row's counts are a measurement or the shape retention left behind (#3541 A9) — see + /// . Stamped by the range reader from the day, the run count and the + /// store's retention horizon; the default is Collected so a row constructed without a reader (the + /// tests' hand-built rows, the fleet sweep's) bands as it always did. + /// + public DailySummaryDataState DataState { get; init; } = DailySummaryDataState.Collected; + + /// The horizon was judged against, carried so the single-day tool can + /// publish it beside a purged verdict; null on a row nobody judged. + public DateTime? RetentionHorizon { get; init; } + + /// How many of the seven per-signal sources hold at least one row for the day (#3541 A9) — + /// the aggregate's trailing signal_sources_present column, the fact that tells a purged shell + /// from a day the purge has not reached. + public int SignalSourcesPresent { get; init; } + public DailyHealthSignals ToSignals() => new() { - HasData = HasData, + /* #3541 A9: a purged or past-horizon day is a NoData day to the band, whatever the spine still + holds for it — the COALESCEd zeros it carries may be absences, and measured-zero-Healthy was the + lie. HasData alone said "a spine row exists", which the collection log's longer horizon made + true for a whole second month of purged signals. Inside retention (Collected, NoRunRecord) a + zero IS a measurement and the band stands. */ + HasData = HasData && DataState is not (DailySummaryDataState.Purged or DailySummaryDataState.PastHorizon), Deadlocks = DeadlockCount, CollectionErrors = CollectionErrors, CollectionRuns = CollectionRuns, @@ -250,14 +272,97 @@ have not happened yet (review finding on #3525). The fleet sweep does NOT read t /// public const string DailySummaryRangeSql = DailySummarySql.RangeSql; + /// The range read's rows plus the horizon they were judged against (#3541 A9). + /// One row per day the spine holds, oldest first, each stamped with its . + /// The oldest UTC day every signal source still holds — . + /// The retention (days) the horizon was computed from: the shortest effective horizon among the sources. + public sealed record DailySummaryRangeReadResult(List Rows, DateTime RetentionHorizon, int ShortestRetentionDays); + + /// + /// The collectors whose tables the daily aggregate reads as SIGNALS, by their schedule names — the + /// sources whose retention decides the horizon (#3541 A9). The collection log and the alert log are the + /// other two spine members; they are constants on and are folded in by + /// . query_stats is included even though old windows route + /// its CTE to a rollup with its own longer retention: the horizon is a floor over EVERY signal, and the + /// rollup keeps only the query count, not the band's inputs. + /// + internal static readonly string[] DailySummarySignalCollectors = + { + "wait_stats", "query_stats", "deadlocks", "blocked_process_report", "dmv_blocking_snapshot", + "cpu_utilization", "memory_pressure_events", + }; + + /// + /// The FLEET-WIDE retention overrides (server_id NULL) for the signal collectors — the same rows + /// StoreConfigProvider.ResolveFleetRetentionDays layers over CollectorScheduleDefaults for + /// the purge itself, so the horizon this reader publishes is the horizon the purge actually enforces + /// rather than the shipped default. A per-server override cannot apply to a shared-table purge, which is + /// why only fleet rows are read. $1 the collector names. + /// + public const string FleetRetentionOverridesSql = """ + SELECT collector_name, retention_days + FROM config_collector_schedules + WHERE server_id IS NULL + AND retention_days IS NOT NULL + AND collector_name = ANY($1) + """; + + /// + /// The shortest effective retention among the daily aggregate's sources, in days — the number the + /// horizon is measured back from. Pure: is the collector → + /// retention_days map the store holds (empty on an untouched store). + /// + /// The signal collectors resolve through + /// so an operator-shortened or -lengthened retention moves the horizon with it, with the same floor the + /// purge applies to the two baseline-serving raw tables ( for + /// cpu_utilization). The collection log and the alert log are folded in at their constants; on a + /// default store they are the LONGER horizons (60 and 90 days), which is exactly why a spine row can + /// outlive its signals and why the shortest one is the horizon. + /// + internal static int ShortestSignalRetentionDays(IReadOnlyList fleetOverrides) + { + var shortest = Math.Min(DarlingRetention.CollectionLogRetentionDays, DarlingRetention.AlertHistoryRetentionDays); + foreach (var collector in DailySummarySignalCollectors) + { + var days = StoreConfigProvider.ResolveFleetRetentionDays(collector, fleetOverrides); + if (DarlingRetention.BaselineServingRawCollectors.Contains(collector)) + { + days = Math.Max(days, BaselineMath.BaselineWindowDays); + } + + shortest = Math.Min(shortest, days); + } + + return Math.Max(1, shortest); + } + + private static async Task> ReadFleetRetentionOverridesAsync( + NpgsqlDataSource postgres, CancellationToken cancellationToken) + { + var overrides = new List(); + await using var command = postgres.CreateCommand(FleetRetentionOverridesSql); + command.CommandTimeout = McpCommandDeadlines.ReadSeconds; + command.Parameters.Add(new NpgsqlParameter { TypedValue = DailySummarySignalCollectors }); + await using var reader = await command.ExecuteReaderAsync(cancellationToken); + while (await reader.ReadAsync(cancellationToken)) + { + overrides.Add(new ScheduleOverride(null, reader.GetString(0), null, reader.GetInt32(1), true, null)); + } + + return overrides; + } + /// One per collected day in the half-open [fromDate, toDate) /// window (the viewer's GetDailySummaryRangeAsync). /// /// #1661: routes to the same retention tier the viewer's calendar does. This matters beyond /// correctness — the calendar and this MCP tool answer the same question, so if only one routed they would /// report different query counts for the same day and there would be no way to tell which was right. + /// + /// #3541 A9: returns the rows AND the retention horizon they were judged against — see + /// and the horizon note in the body. /// - public static async Task> GetDailySummaryRangeAsync( + public static async Task GetDailySummaryRangeAsync( NpgsqlDataSource postgres, int serverId, DateTime fromDate, DateTime toDate, DateTime? referenceUtc = null, CancellationToken cancellationToken = default) { @@ -279,6 +384,16 @@ runs at human/model cadence and these are two small lookups. */ old pair and the rest on the new one. */ var rateTiers = await ReadDeadlockRateThresholdsAsync(postgres, cancellationToken); + /* #3541 A9: the retention horizon, from the store's effective retention and the READER's wall clock. + The clock is deliberately NOT the caller's anchor — a purge is a wall-clock event and a backdated + as_of cannot un-purge a table; anchoring the horizon to as_of would let "as_of 25 days ago, + days_back 30" paint the purged stretch green again, which is the defect. The anchor still governs + the WINDOW (fromDate/toDate above) and the still-forming day's clamp (ReferenceUtc below); the + horizon is a property of the store. DailySummaryRetention.HorizonFor documents the date arithmetic. */ + var fleetOverrides = await ReadFleetRetentionOverridesAsync(postgres, cancellationToken); + var shortestRetentionDays = ShortestSignalRetentionDays(fleetOverrides); + var horizon = DailySummaryRetention.HorizonFor(DateTime.UtcNow, shortestRetentionDays); + var results = new List(); await using var command = postgres.CreateCommand(DailySummarySql.RangeSqlFor(tier)); command.CommandTimeout = McpCommandDeadlines.ReadSeconds; @@ -289,14 +404,17 @@ runs at human/model cadence and these are two small lookups. */ await using var reader = await command.ExecuteReaderAsync(cancellationToken); while (await reader.ReadAsync(cancellationToken)) { - results.Add(ReadDailySummaryRow(reader) with + var row = ReadDailySummaryRow(reader); + results.Add(row with { RateTiers = rateTiers, ReferenceUtc = referenceUtc ?? DateTime.UtcNow, + DataState = DailySummaryRetention.StateFor(row.SummaryDate, row.CollectionRuns, row.SignalSourcesPresent, horizon), + RetentionHorizon = horizon, }); } - return results; + return new DailySummaryRangeReadResult(results, horizon, shortestRetentionDays); } /// The deadlock band's tiers from the store's singleton settings row (#3368, V120), or the @@ -363,10 +481,16 @@ public static async Task GetDailySummaryAsync( NpgsqlDataSource postgres, int serverId, DateTime? summaryDate = null, CancellationToken cancellationToken = default) { var targetDate = summaryDate?.Date ?? DateTime.UtcNow.Date; - var rows = await GetDailySummaryRangeAsync(postgres, serverId, targetDate, targetDate.AddDays(1), cancellationToken: cancellationToken); - return rows.Count > 0 - ? rows[0] - : new DailySummaryReadRow(targetDate, 0m, "", 0, 0, 0, 0, 0, 0, 0, 0, 0, HasData: false); + var range = await GetDailySummaryRangeAsync(postgres, serverId, targetDate, targetDate.AddDays(1), cancellationToken: cancellationToken); + return range.Rows.Count > 0 + ? range.Rows[0] + : new DailySummaryReadRow(targetDate, 0m, "", 0, 0, 0, 0, 0, 0, 0, 0, 0, HasData: false) + { + /* A day the spine does not hold at all is not "collected" either: before the horizon it is + purged like any other, inside it simply without a run record — so the single-day tool can say which. */ + DataState = DailySummaryRetention.StateFor(targetDate, 0, 0, range.RetentionHorizon), + RetentionHorizon = range.RetentionHorizon, + }; } private static DailySummaryReadRow ReadDailySummaryRow(DbDataReader reader) => new( @@ -387,5 +511,7 @@ public static async Task GetDailySummaryAsync( /* #3539 A2: the trailing collection_runs column, appended after peak_block_wait_ms so the eleven positional reads above stay where they were. */ CollectionRuns = reader.IsDBNull(12) ? 0L : Convert.ToInt64(reader.GetValue(12)), + /* #3541 A9: the signal-presence count, after collection_runs. */ + SignalSourcesPresent = reader.IsDBNull(13) ? 0 : Convert.ToInt32(reader.GetValue(13)), }; } diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpDataTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpDataTools.cs index 9a713b697..72d51a31d 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpDataTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpDataTools.cs @@ -513,7 +513,7 @@ public static async Task GetPerfmonStats( /* ═══════════════════════════ query performance ═══════════════════════════ */ - [McpServerTool(Name = "get_top_queries_by_cpu"), Description("Gets expensive queries from sys.dm_exec_query_stats (plan cache). Best for: currently cached queries with detailed per-execution stats, DOP, spills, and query_hash for trending. Returns query_hash, query_plan_hash, sql_handle, plan_handle, and host_object (the hosting procedure/function for proc-hosted statements, null for ad-hoc) — groups key on (database, query_hash, host_object), so INSERT...EXEC callers in different procedures report separately with their own text. distinct_texts counts statement texts merged into a group (>1 = ad-hoc literal variants or pre-upgrade history; query_text is one representative, 0 means only rows predating the text dimension). Set group_by='host_object' to roll all of a procedure's statements into one row — necessary when dynamic SQL with per-value literals fragments one statement across many hashes, which no top-N-by-hash ranking can surface. Supports database and parallelism filtering. min/max_cpu_ms and min/max_elapsed_ms are LIFETIME extremes for the plan's time in cache (same semantics as max_dop), not windowed — totals and avgs are windowed deltas; rows where an extreme provably predates the window carry extremes_note. Also returns cpu_attribution: the returned rows' summed CPU-seconds against the SQL process's measured CPU-seconds for the window (avg cpu_utilization % x core count x window) - attributed_cpu_ratio says how much of the box the ranking explains; when the CPU series or core count is missing, or covers too little of the window, the ratio is omitted rather than invented.")] + [McpServerTool(Name = "get_top_queries_by_cpu"), Description("Gets expensive queries from sys.dm_exec_query_stats (plan cache). Best for: currently cached queries with detailed per-execution stats, DOP, spills, and query_hash for trending. Returns query_hash, query_plan_hash, sql_handle, plan_handle, and host_object (the hosting procedure/function for proc-hosted statements, null for ad-hoc) — groups key on (database, query_hash, host_object), so INSERT...EXEC callers in different procedures report separately with their own text. distinct_texts counts statement texts merged into a group (>1 = ad-hoc literal variants or pre-upgrade history; query_text is one representative, 0 means only rows predating the text dimension). Set group_by='host_object' to roll all of a procedure's statements into one row — necessary when dynamic SQL with per-value literals fragments one statement across many hashes, which no top-N-by-hash ranking can surface. Supports database and parallelism filtering; every filter is applied IN the query before the ranking and the cap, so the page is the top-N of the FILTERED population (filter_applied names the parallelism floor in force, null when none), and an empty page under parallel_only/min_dop is the window's answer rather than a page artefact. min/max_cpu_ms and min/max_elapsed_ms are LIFETIME extremes for the plan's time in cache (same semantics as max_dop), not windowed — totals and avgs are windowed deltas; rows where an extreme provably predates the window carry extremes_note. Also returns cpu_attribution: the returned rows' summed CPU-seconds against the SQL process's measured CPU-seconds for the window (avg cpu_utilization % x core count x window) - attributed_cpu_ratio says how much of the box the ranking explains; when the CPU series or core count is missing, or covers too little of the window, the ratio is omitted rather than invented.")] public static async Task GetTopQueriesByCpu( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -543,18 +543,39 @@ the exact wrong conclusion this option exists to prevent. */ validation = McpHelpers.ValidateTop(top, "top"); if (validation != null) return validation; + /* #3541 A13: the parallelism filter goes INTO the read as a lifetime max_dop floor on the grouped + population, applied before the CPU ranking and the cap (see TopQueriesSql's HAVING note). It used + to be a .Where over the returned top-N page: parallel_only=true on a box whose twenty hottest plans + were serial came back EMPTY while the window held parallel plans, and the engine's own CXPACKET + advice steers agents to exactly that call. The floor is 2 for parallel_only (the smallest DOP that + is parallel), min_dop when the caller set one above that, 0 (admit all) otherwise; min_dop implies + parallel filtering, as its description has always said. */ + var minMaxDop = min_dop > 1 ? min_dop : parallel_only ? 2 : 0; + var filterApplied = minMaxDop > 0 + ? $"lifetime max_dop >= {minMaxDop} (applied in SQL before the top-{top} ranking; the page is the top-{top} of the parallel population)" + : null; + try { var now = windowEnd; var rows = await DarlingDataReader.GetTopQueriesByCpuAsync( - postgres, resolved.ServerId, now.AddHours(-hours_back), now, top, database_name, rollUpByHostObject: rollUp); + postgres, resolved.ServerId, now.AddHours(-hours_back), now, top, database_name, rollUpByHostObject: rollUp, minMaxDop: minMaxDop); if (rows.Count == 0) + { + /* A filtered miss is not a collection miss: with the floor in the query, an empty page under + parallel_only means the window held no group whose plan ever ran parallel, and saying + "no query stats available" for that would send the caller to collection health. */ + if (minMaxDop > 0) + { + return McpHelpers.Status( + "empty", + $"No query-stats group on {resolved.ServerName} in the last {hours_back} hour(s) has a cached plan with lifetime max_dop >= {minMaxDop}. The filter was applied in SQL over the whole window, so this is the window's answer rather than a page artefact — drop parallel_only / min_dop to see the unfiltered ranking, or confirm current parallelism with analyze_query_plan.", + new { filter_applied = filterApplied }); + } + return await DarlingEngineCapability.NotCollectedStatusAsync(postgres, resolved.ServerId, resolved.ServerName, "query_stats") ?? McpHelpers.Status("unavailable", "No query stats available for the specified time range."); - - var filtered = rows - .Where(r => !(parallel_only || min_dop > 1) || (r.MaxDop > 1 && r.MaxDop >= (min_dop > 1 ? min_dop : 2))) - .ToList(); + } /* #2320: what fraction of the box's measured CPU the RETURNED rows explain — numerator is the caller-visible ranking (post top-N, post filters), denominator is measured, and the @@ -566,12 +587,12 @@ ratio is omitted rather than invented when a denominator piece is missing. The t var cpuAggregate = await cpuAggregateTask; var properties = await propertiesTask; var attribution = CpuAttribution.Compute( - filtered.Sum(r => r.TotalCpuUs) / 1_000_000.0, + rows.Sum(r => r.TotalCpuUs) / 1_000_000.0, now.AddHours(-hours_back), now, cpuAggregate.SampleCount, cpuAggregate.FirstSample, cpuAggregate.LastSample, cpuAggregate.AvgSqlCpuPercent, properties?.CpuCount ?? 0); - var result = filtered.Select(r => new + var result = rows.Select(r => new { database_name = r.DatabaseName, query_hash = r.QueryHash, @@ -629,6 +650,8 @@ ratio is omitted rather than invented when a denominator piece is missing. The t /* #2235: echoed so a stored or pasted payload cannot be misread as the other grouping — the two answer different questions and the rows look alike. */ group_by = rollUp ? "host_object" : "query_hash", + /* #3541 A13: the filter that shaped the population, stated on the payload; null when none. */ + filter_applied = filterApplied, cpu_attribution = new { ranked_cpu_seconds = attribution.RankedCpuSeconds, @@ -1388,7 +1411,7 @@ internal static (string State, string Message) FleetMaintenanceLogMiss( public static async Task GetCollectionLog( NpgsqlDataSource postgres, [Description("Server name or display name, or the reserved name (fleet) for the fleet-maintenance run-records.")] string? server_name = null, - [Description("Hours of history. Default 24.")] int hours_back = 24, + [Description("Hours of history. Default 24. No upper bound (this read exists to look further back than the 168-hour reads allow); a negative or zero value is refused rather than read as its absolute value.")] int hours_back = 24, [Description("Maximum rows to return. Default 200. Applied AFTER the two filters, so it caps the matching rows rather than the window.")] int limit = 200, [Description(McpHelpers.AsOfDescription)] string? as_of = null, /* @@ -1421,18 +1444,19 @@ every existing one meaning what it already meant. var invalidFloor = McpHelpers.ValidateMinMs(min_duration_ms, "min_duration_ms"); if (invalidFloor != null) return invalidFloor; - /* ResolveAsOf here, deliberately NOT ValidateWindow. These three reads have never capped - hours_back -- they Math.Abs() it and window on the result -- so routing them through the - shared validator would impose the 168-hour ceiling every other read carries, and take reach - away from exactly the read whose premise is looking FURTHER back than the default. The anchor - is validated because it is new; the span keeps the behaviour callers already have. */ - var anchorError = McpHelpers.ResolveAsOf(as_of, out var windowEnd); + /* ValidateUncappedWindow, deliberately NOT ValidateWindow. These three reads have never capped + hours_back, so routing them through the shared validator would impose the 168-hour ceiling every + other read carries and take reach away from exactly the read whose premise is looking FURTHER back + than the default. What they no longer do is Math.Abs() a negative span (#3541 A13): a window that + ends before it starts is a caller error, and flipping the sign answered a different question with + nothing to say so. Refused, like every other unusable parameter here. */ + var anchorError = McpHelpers.ValidateUncappedWindow(hours_back, as_of, out var windowEnd); if (anchorError != null) return anchorError; try { var end = windowEnd; - var start = end.AddHours(-Math.Abs(hours_back)); + var start = end.AddHours(-hours_back); /* Over-fetch by one so truncation is OBSERVED rather than inferred. Comparing count to the cap cannot tell a window holding exactly `limit` runs from one holding more, and this @@ -1485,7 +1509,7 @@ FleetMaintenanceLogMiss. The server branches below are untouched. */ if (resolved.ServerId == DarlingObservability.FleetServerId) { var (state, text) = FleetMaintenanceLogMiss( - everCollected, collector_name, min_duration_ms, Math.Abs(hours_back)); + everCollected, collector_name, min_duration_ms, hours_back); return McpHelpers.Status(state, text); } @@ -1501,12 +1525,12 @@ FleetMaintenanceLogMiss. The server branches below are untouched. */ { return McpHelpers.Status( "empty", - $"No collector runs on {resolved.ServerName} in the last {Math.Abs(hours_back)} hour(s) matched {McpHelpers.DescribeCollectionLogFilters(collector_name, min_duration_ms)}. This says nothing about the window as a whole — the filters were applied, so unfiltered runs may well exist. Drop them to see what the window holds, and check collector_name against the names get_collection_health lists, since it is matched exactly."); + $"No collector runs on {resolved.ServerName} in the last {hours_back} hour(s) matched {McpHelpers.DescribeCollectionLogFilters(collector_name, min_duration_ms)}. This says nothing about the window as a whole — the filters were applied, so unfiltered runs may well exist. Drop them to see what the window holds, and check collector_name against the names get_collection_health lists, since it is matched exactly."); } return McpHelpers.Status( "empty", - $"No collector runs recorded for {resolved.ServerName} in the last {Math.Abs(hours_back)} hour(s). This server HAS collected before, so this window is genuinely quiet rather than broken — widen hours_back to find the most recent runs."); + $"No collector runs recorded for {resolved.ServerName} in the last {hours_back} hour(s). This server HAS collected before, so this window is genuinely quiet rather than broken — widen hours_back to find the most recent runs."); } var result = rows.Select(r => new @@ -1633,7 +1657,7 @@ emits its two sub-lines. Emitted raw rather than pre-divided into ms-per-id -- t server = resolved.ServerName, /* The span REQUESTED. Kept under its shipped name, and no longer the only span reported -- see the two timestamps below. */ - hours_back = Math.Abs(hours_back), + hours_back = hours_back, run_count = rows.Count, /* Observed by the over-fetch above, not inferred from the row count. */ truncated, @@ -1692,25 +1716,26 @@ description is the only other place that coupling is written down. */ public static async Task GetCurrentWaitsTrend( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, - [Description("Hours of history. Default 4.")] int hours_back = 4, + [Description("Hours of history. Default 4. No upper bound (this read exists to look further back than the 168-hour reads allow); a negative or zero value is refused rather than read as its absolute value.")] int hours_back = 4, [Description("Limit the blocked-session series to one database. Omit for all databases.")] string? database_name = null, [Description(McpHelpers.AsOfDescription)] string? as_of = null) { var (resolved, error) = await DarlingServerResolver.ResolveOrErrorAsync(postgres, server_name); if (error != null) return error; - /* ResolveAsOf here, deliberately NOT ValidateWindow. These three reads have never capped - hours_back -- they Math.Abs() it and window on the result -- so routing them through the - shared validator would impose the 168-hour ceiling every other read carries, and take reach - away from exactly the read whose premise is looking FURTHER back than the default. The anchor - is validated because it is new; the span keeps the behaviour callers already have. */ - var anchorError = McpHelpers.ResolveAsOf(as_of, out var windowEnd); + /* ValidateUncappedWindow, deliberately NOT ValidateWindow. These three reads have never capped + hours_back, so routing them through the shared validator would impose the 168-hour ceiling every + other read carries and take reach away from exactly the read whose premise is looking FURTHER back + than the default. What they no longer do is Math.Abs() a negative span (#3541 A13): a window that + ends before it starts is a caller error, and flipping the sign answered a different question with + nothing to say so. Refused, like every other unusable parameter here. */ + var anchorError = McpHelpers.ValidateUncappedWindow(hours_back, as_of, out var windowEnd); if (anchorError != null) return anchorError; try { var end = windowEnd; - var start = end.AddHours(-Math.Abs(hours_back)); + var start = end.AddHours(-hours_back); var waits = await DarlingDataReader.GetWaitingTaskTrendAsync(postgres, resolved.ServerId, start, end); var blocked = await DarlingDataReader.GetBlockedSessionTrendAsync( @@ -1733,7 +1758,7 @@ waiting_tasks collector never ran. A caller told all-clear stops looking. return everCollected ? McpHelpers.Status( "empty", - $"Nothing was waiting on {resolved.ServerName} in the last {Math.Abs(hours_back)} hour(s). The collector HAS sampled this server, so this is a genuine all-clear for the window rather than missing data.") + $"Nothing was waiting on {resolved.ServerName} in the last {hours_back} hour(s). The collector HAS sampled this server, so this is a genuine all-clear for the window rather than missing data.") : McpHelpers.Status( "unavailable", $"No waiting-task samples have EVER been recorded for {resolved.ServerName}, so this is NOT an all-clear — there is nothing to read. Check that collection is running for this server before concluding it was quiet."); @@ -1742,7 +1767,7 @@ waiting_tasks collector never ran. A caller told all-clear stops looking. return JsonSerializer.Serialize(new { server = resolved.ServerName, - hours_back = Math.Abs(hours_back), + hours_back = hours_back, database_name, /* Two series in one payload because they are read together: a wait-type spike with no @@ -1773,24 +1798,25 @@ across two tools a caller can fetch one and draw the wrong conclusion. public static async Task GetBlockingStats( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, - [Description("Hours of history. Default 24.")] int hours_back = 24, + [Description("Hours of history. Default 24. No upper bound (this read exists to look further back than the 168-hour reads allow); a negative or zero value is refused rather than read as its absolute value.")] int hours_back = 24, [Description(McpHelpers.AsOfDescription)] string? as_of = null) { var (resolved, error) = await DarlingServerResolver.ResolveOrErrorAsync(postgres, server_name); if (error != null) return error; - /* ResolveAsOf here, deliberately NOT ValidateWindow. These three reads have never capped - hours_back -- they Math.Abs() it and window on the result -- so routing them through the - shared validator would impose the 168-hour ceiling every other read carries, and take reach - away from exactly the read whose premise is looking FURTHER back than the default. The anchor - is validated because it is new; the span keeps the behaviour callers already have. */ - var anchorError = McpHelpers.ResolveAsOf(as_of, out var windowEnd); + /* ValidateUncappedWindow, deliberately NOT ValidateWindow. These three reads have never capped + hours_back, so routing them through the shared validator would impose the 168-hour ceiling every + other read carries and take reach away from exactly the read whose premise is looking FURTHER back + than the default. What they no longer do is Math.Abs() a negative span (#3541 A13): a window that + ends before it starts is a caller error, and flipping the sign answered a different question with + nothing to say so. Refused, like every other unusable parameter here. */ + var anchorError = McpHelpers.ValidateUncappedWindow(hours_back, as_of, out var windowEnd); if (anchorError != null) return anchorError; try { var end = windowEnd; - var start = end.AddHours(-Math.Abs(hours_back)); + var start = end.AddHours(-hours_back); var blocking = await DarlingDataReader.GetBlockingDurationStatsAsync(postgres, resolved.ServerId, start, end); @@ -1824,7 +1850,7 @@ await DarlingBlockingTrendReader.HasAnyBlockingCollectorRunAsync(postgres, resol return everRan ? McpHelpers.Status( "empty", - $"No blocking or deadlocks recorded for {resolved.ServerName} in the last {Math.Abs(hours_back)} hour(s). The blocking collectors HAVE run successfully for this server, so the window is genuinely clear rather than blind.") + $"No blocking or deadlocks recorded for {resolved.ServerName} in the last {hours_back} hour(s). The blocking collectors HAVE run successfully for this server, so the window is genuinely clear rather than blind.") : McpHelpers.Status( "unavailable", $"The blocking collectors have NEVER run successfully for {resolved.ServerName}, so this is NOT a clean bill of health — nothing looked. Blocked-process reports need the XE session running, or the DMV blocking snapshot collector enabled; check those before concluding this server does not block."); @@ -1833,7 +1859,7 @@ await DarlingBlockingTrendReader.HasAnyBlockingCollectorRunAsync(postgres, resol return JsonSerializer.Serialize(new { server = resolved.ServerName, - hours_back = Math.Abs(hours_back), + hours_back = hours_back, /* Severity, not counts. get_blocking_trend already answers how OFTEN; ten one-second blocks and one ten-minute block share a count and are different problems. diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpHealthTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpHealthTools.cs index 926df0e07..b77c18670 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpHealthTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpHealthTools.cs @@ -88,26 +88,42 @@ collection of ANY collector — a live collection log beside a dead CPU collecto } } - [McpServerTool(Name = "get_daily_summary"), Description("Gets a daily health summary: overall composite health band (Healthy/Warning/Critical), total wait time, top wait type, unique query count, deadlocks, blocking events, memory pressure (and severe memory pressure), high-CPU samples, collection errors, and actionable alert count for one day. Use this for a quick overview to decide which areas need investigation.")] + [McpServerTool(Name = "get_daily_summary"), Description("Gets a daily health summary: overall composite health band (Healthy/Warning/Critical), total wait time, top wait type, unique query count, deadlocks, blocking events, memory pressure (and severe memory pressure), high-CPU samples, collection errors, and actionable alert count for one day. Use this for a quick overview to decide which areas need investigation. A day before the store's retention_horizon (the oldest day the shortest-lived signal table still holds) returns status=unavailable with data_state=purged rather than a health band: its per-signal counts would be COALESCEd zeros, not measurements, and a zero is only a measurement inside retention. A returned day carries data_state=collected (a verdict), past_horizon (before the horizon but some signal table still holds rows — the purge has not reached it; No Data, non-zero counts real) or no_run_record (inside retention, no collector run recorded — banded on the counts as read, which are measurements there; the collection-error share has no denominator).")] public static async Task GetDailySummary( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, - [Description("Summary date (yyyy-MM-dd), interpreted as a UTC day. Default is today.")] string? summary_date = null) + [Description("Summary date, ISO-8601 yyyy-MM-dd ONLY (e.g. 2026-07-09), interpreted as a UTC day; any other spelling is refused rather than guessed at. Default is today.")] string? summary_date = null) { var (resolved, error) = await DarlingServerResolver.ResolveOrErrorAsync(postgres, server_name); if (error != null) return error; - DateTime? date = null; - if (!string.IsNullOrEmpty(summary_date)) - { - if (!DateTime.TryParse(summary_date, System.Globalization.CultureInfo.InvariantCulture, System.Globalization.DateTimeStyles.None, out var parsed)) - return $"Invalid date format '{summary_date}'. Use yyyy-MM-dd format (e.g., 2026-07-09)."; - date = parsed; - } + /* #3541 A9: exact ISO-8601, refused otherwise — McpHelpers.ParseSummaryDate says why the general + parse this replaced was the wrong tool in a file that already held as_of's strict allowlist. */ + var dateError = McpHelpers.ParseSummaryDate(summary_date, out var date); + if (dateError != null) return dateError; try { var row = await DarlingHealthReader.GetDailySummaryAsync(postgres, resolved.ServerId, date); + + /* #3541 A9: a day before the retention horizon is "unavailable" in the miss vocabulary's own + sense — it existed and is not retrievable now — and it is told apart from a never-collected + day because the two send a caller to different places (nowhere useful, versus collection + health). What the spine still holds for it rides in the hints, named for what it is. */ + if (row.DataState == DailySummaryDataState.Purged) + return McpHelpers.Status( + "unavailable", + $"{row.SummaryDate:yyyy-MM-dd} is before {resolved.ServerName}'s retention_horizon ({row.RetentionHorizon:yyyy-MM-dd}): the per-signal tables the health band reads (deadlocks, blocking, CPU, memory, waits) have been purged for that day, so no health verdict is possible and the counts would be zeros by construction, not by measurement. Longer-lived sources may still record the day — collection_runs and alert_count below are real where non-zero.", + new + { + summary_date = row.SummaryDate.ToString("yyyy-MM-dd"), + overall_health = row.OverallHealth, + data_state = DailySummaryRetention.Label(row.DataState), + retention_horizon = row.RetentionHorizon?.ToString("yyyy-MM-dd"), + collection_runs = row.CollectionRuns, + alert_count = row.AlertCount, + }); + if (!row.HasData) return McpHelpers.Status( "empty", @@ -120,6 +136,11 @@ public static async Task GetDailySummary( summary_date = row.SummaryDate.ToString("yyyy-MM-dd"), overall_health = row.OverallHealth, health_band = row.HealthBand.ToString(), + /* #3541 A9: collected, past_horizon or no_run_record here (purged returned above); the note + says what the zeros are on a non-collected day, null on a collected one. */ + data_state = DailySummaryRetention.Label(row.DataState), + data_note = row.RetentionHorizon is { } horizon ? DailySummaryRetention.Note(row.DataState, horizon, row.SignalSourcesPresent) : null, + retention_horizon = row.RetentionHorizon?.ToString("yyyy-MM-dd"), total_wait_time_sec = row.TotalWaitTimeSec, top_wait_type = row.TopWaitType, unique_queries = row.UniqueQueries, @@ -145,7 +166,7 @@ public static async Task GetDailySummary( } } - [McpServerTool(Name = "get_daily_summary_range"), Description("Gets the daily health summary for a SPAN of days rather than one: one row per collected day, each with its composite health band (Healthy/Warning/Critical), total wait time, top wait type, unique query count, deadlocks, blocking events with the peak block wait, high-CPU samples, memory pressure, collection errors and actionable alert count. This is what the desktop viewer's Performance Calendar month grid draws, and it is the read to use when the question is WHICH day rather than how one day went — scan the bands, then call get_daily_summary for the day that stands out. A day on which anything at all was collected appears here even if every signal was quiet (that day is Healthy, not missing), so a gap in the returned days is a gap in COLLECTION.")] + [McpServerTool(Name = "get_daily_summary_range"), Description("Gets the daily health summary for a SPAN of days rather than one: one row per collected day, each with its composite health band (Healthy/Warning/Critical), total wait time, top wait type, unique query count, deadlocks, blocking events with the peak block wait, high-CPU samples, memory pressure, collection errors and actionable alert count. This is what the desktop viewer's Performance Calendar month grid draws, and it is the read to use when the question is WHICH day rather than how one day went — scan the bands, then call get_daily_summary for the day that stands out. A day on which anything at all was collected appears here even if every signal was quiet (that day is Healthy, not missing), so a gap in the returned days is a gap in COLLECTION — INSIDE RETENTION. The per-signal tables age out at the store's shortest retention while the collection log and alert log live longer, so retention_horizon is the oldest day every signal can still answer for; a returned day before it carries data_state=purged (no signal table holds it) or past_horizon (some still do — the purge has not reached it), health_band=NoData and a data_note, NEVER Healthy — a purged day's zeros are absences, and days_before_horizon counts both kinds. A day inside retention with no collector run recorded is data_state=no_run_record — it keeps its band (an alert-only day is Warning), with the caveat that the error share has no denominator. Purged and past_horizon rows carry no verdict.")] public static async Task GetDailySummaryRange( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -181,8 +202,9 @@ means the anchor day is the last one included rather than the first one excluded /* The anchor is also the clock the still-forming day's window clamps against (#3525 review): a backdated as_of must clamp its own "today" against ITSELF, not the process clock. */ - var rows = await DarlingHealthReader.GetDailySummaryRangeAsync( + var range = await DarlingHealthReader.GetDailySummaryRangeAsync( postgres, resolved.ServerId, fromDate, toDate, referenceUtc: windowEnd); + var rows = range.Rows; if (rows.Count == 0) { @@ -218,14 +240,23 @@ edge table it cannot report a healthy server as uncollected. otherwise tell which days they were given from the days they got. */ from_date = fromDate.ToString("yyyy-MM-dd"), to_date = lastDay.ToString("yyyy-MM-dd"), - /* Days WITH data, not days in the span. The two differ exactly where collection has a hole, - and that difference is the most useful thing on this payload. */ + /* Days the spine holds, not days in the span. The two differ exactly where collection has a + hole, and that difference is the most useful thing on this payload — read it together with + days_before_horizon, because a held day before the horizon is a shell, not a collected day. */ day_count = rows.Count, + /* #3541 A9: the store's horizon (reader clock, effective retention — see the reader) and how + many returned days fall before it. */ + retention_horizon = range.RetentionHorizon.ToString("yyyy-MM-dd"), + days_before_horizon = rows.Count(row => row.DataState is DailySummaryDataState.Purged or DailySummaryDataState.PastHorizon), + purged_day_count = rows.Count(row => row.DataState == DailySummaryDataState.Purged), + collected_day_count = rows.Count(row => row.DataState == DailySummaryDataState.Collected), days = rows.Select(row => new { summary_date = row.SummaryDate.ToString("yyyy-MM-dd"), overall_health = row.OverallHealth, health_band = row.HealthBand.ToString(), + data_state = DailySummaryRetention.Label(row.DataState), + data_note = DailySummaryRetention.Note(row.DataState, range.RetentionHorizon, row.SignalSourcesPresent), total_wait_time_sec = row.TotalWaitTimeSec, top_wait_type = row.TopWaitType, unique_queries = row.UniqueQueries, diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs index f83453499..483b1b2ea 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs @@ -88,7 +88,7 @@ public static string Build(DarlingPeerDirectory.Snapshot peers) | Tool | Purpose | Key Parameters | |------|---------|----------------| | `analyze_server` | Runs the inference engine: scores facts, traverses relationship graph, returns evidence-backed findings with severity and recommended next tools. Each finding's `confidence` is an EVIDENCE score (0.20 for the symptom alone, more as the root fact's amplifier checks match and the chain deepens — a lone uncorroborated symptom is 0.20, never 1.0) and `confidence_basis` says in words what it rests on; rank by `severity` for impact and read `confidence` as how much of the engine's own corroboration showed up. A remediable finding also carries `remediation_command` — the full copy-paste T-SQL remediation (identical to the viewer card), with a two-sided risk-disclosure header on destructive changes; advisory only, never executed. Force-plan findings additionally carry `structured_remediation`: the verdict as machine-readable fields (eligible + named blockers), evidence, and split force/unforce/verify SQL. With `as_of` it analyzes a PAST window — anomaly baseline included — and is EXPLORATORY: the findings come back in full but are NOT persisted, which `persisted` / `persistence_note` state on every result | `server_name`, `hours_back` (default 4), `as_of` | - | `get_analysis_facts` | Exposes raw scored facts from the collect+score pipeline — every observation the engine sees with base severity, amplifiers, and metadata (an ANOMALY_* fact's `baseline_confidence` is the baseline's trustworthiness, not a finding's `confidence`) | `server_name`, `hours_back` (default 4), `source` (filter), `min_severity`, `as_of` | + | `get_analysis_facts` | Exposes raw scored facts from the collect+score pipeline — every observation the engine sees with base severity, amplifiers, and metadata (an ANOMALY_* fact's `baseline_confidence` is the baseline's trustworthiness, not a finding's `confidence`) | `server_name`, `hours_back` (default 4), `source` (one of the engine's 15 registered sources — anomaly, bad_actor, blocking, config, coverage, cpu, database_config, disk, io, jobs, memory, queries, sessions, tempdb, waits; an unknown value is refused with the set), `min_severity`, `as_of` | | `compare_analysis` | Compares two time periods (e.g., peak vs off-peak, yesterday vs today, the windows around a change), banding each fact worse / better / stable by how far its VALUE moved on the server's own scale — in the stored per-server baseline's robust sigma where one exists (`delta_sigma`, `band_source` `baseline`), otherwise only when the value moved at least a quarter of the larger side AND registers a quarter of the way up its own severity ladder (`band_source` `absolute`); `band_rules` states the rules on every payload. Rows are grouped into physical-cause `families` (one I/O stall is one family row), and `BAD_ACTOR_` appearances are `plan_cache_churn`, not new or resolved issues. A verdict is a DIFFERENCE, not an experiment: same-hour-yesterday at N=1 vs N=1 cannot show that a change caused anything. When NEITHER window produced facts the result is `unavailable` rather than an all-zero comparison, because "nothing to compare" is not "nothing changed"; when only ONE window is empty the payload carries a `caveat` saying so; a partly collected side flags every verdict row with `coverage_caveat`. `baseline_hours_back` is measured from the comparison window's END, so `as_of` moves BOTH windows together | `server_name`, `hours_back` (default 4), `baseline_hours_back` (default 28), `as_of` | | `audit_config` | Edition-aware configuration audit: evaluates CTFP, MAXDOP, max memory, and max worker threads against best practices | `server_name` | | `get_analysis_findings` | Retrieves persisted findings from previous analysis runs (each with `confidence_basis`: rows persisted before `confidence` measured corroboration are labelled `path-shape (pre-#3538)` — under that formula a lone symptom read 1.0, so do not read those as corroborated) (the service also analyzes on its own schedule, every 30 minutes per server), deduplicated to one entry per diagnostic chain (`story_path_hash` + `incident_id`): the latest occurrence plus `occurrences`/`first_seen`/`last_seen`/`peak_severity` spanning the window; each remediable finding carries `remediation_command` — the full copy-paste T-SQL remediation (identical to the viewer card), rendered from the persisted action, advisory only and never executed; force-plan findings additionally carry `structured_remediation` (verdict + evidence + split artifacts, machine-readable). Its window is on ANALYSIS TIME, so `as_of` asks what analysis was saying then rather than re-analyzing that window now | `server_name`, `hours_back` (default 24), `as_of` | @@ -121,7 +121,7 @@ public static string Build(DarlingPeerDirectory.Snapshot peers) | `get_file_io_stats` | Latest per-file I/O: reads/writes/bytes/stall and computed read/write latency; `captured_at` | `server_name` | | `get_tempdb_trend` | TempDB space over time (user / internal / version store / unallocated) + top consumer | `server_name`, `hours_back` (default 24), `as_of` | | `get_perfmon_stats` | Latest perfmon counters (value + delta); filter by counter / instance; `captured_at` | `server_name`, `counter_name`, `instance_name` | - | `get_top_queries_by_cpu` | Expensive queries from query stats (plan cache) with query_hash / sql_handle; `cpu_attribution.attributed_cpu_ratio` says how much of the box's measured CPU the returned rows explain | `server_name`, `hours_back` (default 24), `top` (default 20), `database_name`, `parallel_only`, `min_dop`, `as_of` | + | `get_top_queries_by_cpu` | Expensive queries from query stats (plan cache) with query_hash / sql_handle; `cpu_attribution.attributed_cpu_ratio` says how much of the box's measured CPU the returned rows explain. `parallel_only` / `min_dop` are applied IN the query before the ranking and the cap (`filter_applied` names the floor), so the page is the top-N of the parallel population | `server_name`, `hours_back` (default 24), `top` (default 20), `database_name`, `parallel_only`, `min_dop`, `as_of` | | `get_top_procedures_by_cpu` | Most expensive stored procedures by total CPU, with the same `cpu_attribution` disclosure | `server_name`, `hours_back` (default 24), `top` (default 20), `database_name`, `as_of` | | `get_query_store_top` | Expensive queries from Query Store with query_id / plan_id (survives restarts) | `server_name`, `hours_back` (default 24), `top` (default 20), `database_name`, `as_of` | | `get_query_heatmap` | The desktop viewer's Query Heatmap as a TABLE: how many DISTINCT queries fell into each (time bin x log-magnitude bucket) cell, with the most-executed query in each cell. The only query read with a TIME axis — `get_top_queries_by_cpu` ranks a whole window and cannot show that the window had a quiet half and a bad half, which is the first question about an incident that has already ended. Bins are **5 minutes** wide by default because that is exactly what the desktop viewer uses, so both surfaces draw the same picture; raise `bucket_minutes` to cover a longer window in fewer cells (it is the lever to reach for before the cap). The seven magnitude buckets are the viewer's, in the metric's own unit, and the labels come back with the result. `limit` caps CELLS, and truncation drops the OLDEST bins rather than the least interesting cells — `first_time_bin` / `last_time_bin` say which slice came back. Zero cells is THREE states and the read says which: never collected (`unavailable`, nobody looked), nothing collected in the window (`empty`, widen it), or collected and genuinely idle — captures exist and every one recorded zero executions (`empty`) | `server_name`, `hours_back` (default 24), `metric` (default `duration`), `database_name`, `bucket_minutes` (default 5), `limit` (default 500), `as_of` | @@ -148,7 +148,7 @@ public static string Build(DarlingPeerDirectory.Snapshot peers) | `get_deadlock_trend` | Per-minute deadlock counts over time. An empty result distinguishes a genuine all-clear (`empty`, with the collector run counts in `hints` so you can see how many captures the window actually holds) from a window no collector covered (`unavailable`), which is NOT an all-clear | `server_name`, `hours_back` (default 24), `as_of` | | `get_lock_wait_trend` | Every LCK% wait type's wait milliseconds per SECOND at each collection — the aggregate lock-wait lane. The two trends above count incidents and `get_wait_trend` charts ONE named wait type; this is the whole lock family as a rate, which is what shows lock pressure rising when no single type dominates. Rate rather than raw delta, so it compares across servers on different cadences. An empty result distinguishes a genuinely quiet window (`empty`, widen `hours_back`) from a server no wait stats have ever been stored for (`unavailable`), which is NOT a report of a server without lock contention | `server_name`, `hours_back` (default 24), `as_of` | | `get_session_stats` | Latest per-application connection/session counts (running/sleeping/dormant) + resource totals | `server_name` | - | `get_active_queries` | Captured running-query snapshots over the window (waits, CPU, blocking, grants) | `server_name`, `hours_back` (default 1), `database_name`, `blocking_only`, `limit` (default 50), `as_of` | + | `get_active_queries` | Captured running-query snapshots over the window (waits, CPU, blocking, grants). `database_name` / `blocking_only` are applied IN the query; `total_snapshots` counts the filtered population, `snapshots_returned` the page, `truncated` says the population held more. Head blockers a victim names are never stripped (`is_head_blocker`); a victim whose blocker is absent says why in `blocker_not_shown` (`not_captured` / `filtered` / `past_page`) | `server_name`, `hours_back` (default 1), `database_name`, `blocking_only`, `limit` (default 50), `as_of` | | `get_waiting_tasks` | Individual waiting tasks captured at collection time, newest capture first. Page bounded by `limit`; `truncated` and `oldest_returned_collection_time` say what it reached | `server_name`, `hours_back` (default 1), `limit` (default 30), `as_of` | | `get_server_config_changes` | sp_configure changes, diffed from config snapshots | `server_name`, `hours_back` (default 168), `as_of` | | `get_database_config_changes` | sys.databases setting changes, diffed from config snapshots | `server_name`, `hours_back` (default 168), `as_of` | @@ -218,8 +218,8 @@ public static string Build(DarlingPeerDirectory.Snapshot peers) | `get_alert_settings` | The current alert config the service uses: per-alert enable + thresholds, cooldown, excluded databases, delivery mode, and the scheduled-analysis cadence | none | | `get_mute_rules` | The alert mute rules in force, so a suppressed server is distinguishable from a healthy-quiet one. An empty result distinguishes no rule ever written from rules that exist but have all lapsed, with the configured count in `hints` | `enabled_only` (default true) | | `get_server_summary` | One-shot per-server health: current CPU %, memory, recent blocking count, recent deadlock count; three clocks named (`cpu_captured_at`, `memory_captured_at`, `last_collection` = newest collection of ANY collector) | `server_name` | - | `get_daily_summary` | A day's composite health band (Healthy / Warning / Critical) plus the signals behind it (waits, deadlocks, blocking, high CPU, memory pressure, alerts) | `server_name`, `summary_date` (yyyy-MM-dd, default today) | - | `get_daily_summary_range` | The SAME rollup across a span of days — one row per collected day, which is the desktop viewer's Performance Calendar month grid. Use it when the question is WHICH day rather than how one day went: scan the bands, then call `get_daily_summary` for the day that stands out. A day with ANY collection appears even when every signal was quiet (Healthy, not missing), so a day absent from the result is a gap in COLLECTION. `as_of` anchors the LAST day of the range, so a past month is `as_of` its last day with `days_back` its length. An empty result distinguishes a range outside this server's history (`empty`) from a server nothing has ever been collected for (`unavailable`) | `server_name`, `days_back` (default 30, max 366), `as_of` | + | `get_daily_summary` | A day's composite health band (Healthy / Warning / Critical) plus the signals behind it (waits, deadlocks, blocking, high CPU, memory pressure, alerts). A day before the store's `retention_horizon` is `unavailable` with `data_state: purged` — never a band, because its zeros are absences | `server_name`, `summary_date` (yyyy-MM-dd ONLY, refused otherwise; default today) | + | `get_daily_summary_range` | The SAME rollup across a span of days — one row per collected day, which is the desktop viewer's Performance Calendar month grid. Use it when the question is WHICH day rather than how one day went: scan the bands, then call `get_daily_summary` for the day that stands out. A day with ANY collection appears even when every signal was quiet (Healthy, not missing), so a day absent from the result is a gap in COLLECTION — inside retention. The payload publishes `retention_horizon` (the oldest day every signal table still holds) and `days_before_horizon`; a returned day before the horizon is `data_state: purged` (no signal table holds it) or `past_horizon` (some still do) and `NoData`, NEVER Healthy, because a purged day's per-signal zeros are absences left by the purge while the longer-lived collection log still names the day. Purged and past-horizon rows carry no verdict; `no_run_record` (inside retention, no collector run recorded) keeps its band with a caveat. `as_of` anchors the LAST day of the range, so a past month is `as_of` its last day with `days_back` its length. An empty result distinguishes a range outside this server's history (`empty`) from a server nothing has ever been collected for (`unavailable`) | `server_name`, `days_back` (default 30, max 366), `as_of` | **Tuning the alerting (write).** Five Darling-only tools change the shared alert configuration the service delivers on — the SAME config `get_alert_settings` / `get_mute_rules` read, and the same the Viewer's Settings window writes. They are the only alert writes here; none touches a monitored server or the collected data, and a change hot-reloads into the running service within one collection sweep. diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpReadParameters.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpReadParameters.cs index 367e07fc1..641a729aa 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpReadParameters.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpReadParameters.cs @@ -33,6 +33,10 @@ public static void AddText(NpgsqlCommand command, string value) => public static void AddNullableText(NpgsqlCommand command, string? value) => command.Parameters.Add(new NpgsqlParameter { NpgsqlDbType = NpgsqlDbType.Text, Value = (object?)value ?? DBNull.Value }); + /// Binds a boolean flag — a read's NOT $N::boolean OR ... "filter off" branch (#3541 A13). + public static void AddBoolean(NpgsqlCommand command, bool value) => + command.Parameters.Add(new NpgsqlParameter { TypedValue = value }); + /// Binds a naive-UTC timestamp (Kind=Unspecified → the store's timestamp columns). public static void AddTimestamp(NpgsqlCommand command, DateTime value) => command.Parameters.Add(new NpgsqlParameter { TypedValue = DateTime.SpecifyKind(value, DateTimeKind.Unspecified) }); diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpSessionTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpSessionTools.cs index 9f8495ec5..660eedbb9 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpSessionTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpSessionTools.cs @@ -87,14 +87,14 @@ public static async Task GetSessionStats( } } - [McpServerTool(Name = "get_active_queries"), Description("Gets active query snapshots captured from sys.dm_exec_requests. Shows what queries were running at each collection point: session ID, query text, wait type, CPU time, elapsed time, blocking info, DOP, and memory grants. Use hours_back to look at a specific time window — critical for finding what was running during a CPU spike or blocking event.")] + [McpServerTool(Name = "get_active_queries"), Description("Gets active query snapshots captured from sys.dm_exec_requests. Shows what queries were running at each collection point: session ID, query text, wait type, CPU time, elapsed time, blocking info, DOP, and memory grants. Use hours_back to look at a specific time window — critical for finding what was running during a CPU spike or blocking event. EVERY FILTER IS PART OF THE QUERY: database_name and blocking_only are applied in SQL before the page is cut, total_snapshots is the count of snapshot rows in the window that pass your filters, snapshots_returned is how many you got, and truncated says the filtered population held more than limit — raise limit or narrow hours_back when it is true (NEWEST CAPTURE FIRST, highest CPU first within a capture; oldest_returned_collection_time / newest_returned_collection_time bound the page). HEAD BLOCKERS ARE NEVER STRIPPED: a session another row in the same capture names as its blocker is on the page whatever its text (including a WAITFOR shell holding locks), flagged is_head_blocker. A victim whose blocker is NOT on the page says why in blocker_not_shown: not_captured (the blocker held no running request at that capture — an idle open transaction is the classic case; get_blocking has its input buffer from the blocked-process report), filtered (your database_name filter excluded it), or past_page (it is in the filtered population but beyond limit).")] public static async Task GetActiveQueries( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, [Description("Hours of data to retrieve. Default 1.")] int hours_back = 1, - [Description("Filter to a specific database.")] string? database_name = null, - [Description("Show only queries involved in blocking (blocking_session_id > 0 or is a head blocker).")] bool blocking_only = false, - [Description("Maximum number of rows to return. Default 50.")] int limit = 50, + [Description("Filter to a specific database. Applied in SQL; a head blocker in ANOTHER database is then not on the page, and its victims say blocker_not_shown = filtered.")] string? database_name = null, + [Description("Show only queries involved in blocking: rows with blocking_session_id > 0, plus the head blockers those rows name in the same capture. Applied in SQL, so total_snapshots counts the blocking population and truncated is measured against it.")] bool blocking_only = false, + [Description("Maximum number of rows to return. Default 50. The page is bounded by limit, not by hours_back — truncated says whether the filtered window held more.")] int limit = 50, [Description(McpHelpers.AsOfDescription)] string? as_of = null) { var (resolved, error) = await DarlingServerResolver.ResolveOrErrorAsync(postgres, server_name); @@ -108,22 +108,40 @@ public static async Task GetActiveQueries( try { var now = windowEnd; - var rows = await DarlingSessionReader.GetActiveQueriesAsync( - postgres, resolved.ServerId, now.AddHours(-hours_back), now); + var filter = string.IsNullOrWhiteSpace(database_name) ? null : database_name.Trim(); + + /* #3541 A13: the filters ride INTO the read (see ActiveQueriesSql), and the read is asked for one + row past the cap so truncation is OBSERVED on the filtered population rather than inferred from + the page. total_snapshots is the SQL's COUNT(*) OVER () of that same population — the number + used to be rows.Count of an unfiltered window read, a different population from the rows. */ + var page = await DarlingSessionReader.GetActiveQueriesAsync( + postgres, resolved.ServerId, now.AddHours(-hours_back), now, limit + 1, filter, blocking_only); + var rows = page.Rows; + if (rows.Count == 0) + { + /* A filtered miss is not a collection miss: with the filters in the query, an empty page under + database_name or blocking_only means the window held no snapshot matching them. */ + if (filter != null || blocking_only) + { + return McpHelpers.Status( + "empty", + $"No active query snapshots on {resolved.ServerName} in the last {hours_back} hour(s) matched " + + DescribeActiveQueryFilters(filter, blocking_only) + + ". The filters were applied in SQL over the whole window, so unfiltered snapshots may well exist — drop them to see what the window holds."); + } + return await DarlingEngineCapability.NotCollectedStatusAsync(postgres, resolved.ServerId, resolved.ServerName, "query_snapshots") ?? McpHelpers.Status("empty", "No active query snapshots found in the requested time range."); + } - IEnumerable filtered = rows; - - if (!string.IsNullOrEmpty(database_name)) - filtered = filtered.Where(r => (r.DatabaseName ?? "").Equals(database_name, StringComparison.OrdinalIgnoreCase)); + var truncated = rows.Count > limit; + var shown = truncated ? rows.GetRange(0, limit) : rows; - if (blocking_only) - filtered = filtered.Where(r => r.BlockingSessionId > 0 - || rows.Any(other => other.BlockingSessionId == r.SessionId)); + /* The page's own (capture, session) pairs, so a victim can say whether its blocker made the page. */ + var onPage = new HashSet<(DateTime, int)>(shown.Select(r => (r.CollectionTime, r.SessionId))); - var result = filtered.Take(limit).Select(r => new + var result = shown.Select(r => new { collection_time = r.CollectionTime.ToString("o"), session_id = r.SessionId, @@ -138,6 +156,10 @@ public static async Task GetActiveQueries( wait_type = string.IsNullOrEmpty(r.WaitType) ? null : r.WaitType, wait_time_ms = r.WaitTimeMs > 0 ? r.WaitTimeMs : (long?)null, blocking_session_id = r.BlockingSessionId > 0 ? r.BlockingSessionId : (int?)null, + /* #3541 A13: the two blocking disclosures. is_head_blocker is why a WAITFOR row can be here; + blocker_not_shown names the ONE reason a victim's blocker is not, or is null when it is. */ + is_head_blocker = r.IsHeadBlocker ? true : (bool?)null, + blocker_not_shown = BlockerNotShown(r, onPage), dop = r.Dop > 0 ? r.Dop : (int?)null, parallel_worker_count = r.ParallelWorkerCount > 0 ? r.ParallelWorkerCount : (int?)null, granted_query_memory_gb = r.GrantedQueryMemoryGb > 0 ? r.GrantedQueryMemoryGb : (double?)null, @@ -153,8 +175,18 @@ public static async Task GetActiveQueries( { server = resolved.ServerName, hours_back, - total_snapshots = rows.Count, - shown = result.Count, + filters_applied = new + { + database_name = filter, + blocking_only, + }, + /* The FILTERED population's size, computed in SQL on the same statement as the rows. */ + total_snapshots = page.PopulationCount, + snapshots_returned = result.Count, + truncated, + order = "collection_time_desc", + oldest_returned_collection_time = shown[^1].CollectionTime.ToString("o"), + newest_returned_collection_time = shown[0].CollectionTime.ToString("o"), queries = result }, McpHelpers.JsonOptions); } @@ -164,6 +196,33 @@ public static async Task GetActiveQueries( } } + /// + /// The one reason a victim's head blocker is not on the page, or null when it is (or the row is not a + /// victim). Ordered from the reader's facts outward: never captured (no running request — the idle + /// open-transaction case) beats filtered beats past the page, because each later reason presupposes the + /// earlier one did not apply. + /// + private static string? BlockerNotShown(DarlingSessionReader.ActiveQueryRow row, HashSet<(DateTime, int)> onPage) + { + if (row.BlockingSessionId <= 0) + return null; + if (!row.BlockerInCapture) + return "not_captured"; + if (!row.BlockerInPopulation) + return "filtered"; + return onPage.Contains((row.CollectionTime, row.BlockingSessionId)) ? null : "past_page"; + } + + /// Names the active filters for the filtered-miss sentence, in the caller's own vocabulary. + private static string DescribeActiveQueryFilters(string? databaseName, bool blockingOnly) + { + if (databaseName != null && blockingOnly) + return $"database_name '{databaseName}' with blocking_only"; + if (databaseName != null) + return $"database_name '{databaseName}'"; + return "blocking_only"; + } + [McpServerTool(Name = "get_waiting_tasks"), Description("Gets recently captured waiting tasks — queries that were actively waiting on a resource at collection time — NEWEST CAPTURE FIRST, longest wait first within a capture. Shows session ID, wait type, duration, blocking session, and database. Complements get_wait_stats by showing individual waiting queries rather than aggregated stats. THE PAGE IS BOUNDED BY limit, NOT BY hours_back: tasks_returned is how many rows you got, truncated says the window held more than limit, and oldest_returned_collection_time / newest_returned_collection_time bound the page — under newest-first ordering the oldest stamp IS how far back this read reached, and one busy capture can fill the whole page by itself. Raise limit or narrow hours_back when truncated is true.")] public static async Task GetWaitingTasks( NpgsqlDataSource postgres, diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTools.cs index 84f61efc5..7b7bbd117 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTools.cs @@ -36,6 +36,15 @@ namespace PerformanceMonitor.Darling.Service.Mcp; [McpServerToolType] public sealed class DarlingMcpTools { + /// + /// The source parameter's description on both SKUs' get_analysis_facts, built from + /// so the documented set IS the accepted set (#3541 A13). The old + /// text named four sources of fifteen; a caller who typed any of the other eleven got an empty list. + /// + internal const string FactSourceFilterDescription = + "Filter to one source category. Accepted values (the engine's complete source registry, refused otherwise): " + + "anomaly, bad_actor, blocking, config, coverage, cpu, database_config, disk, io, jobs, memory, queries, sessions, tempdb, waits. Omit for all."; + [McpServerTool(Name = "analyze_server"), Description("Runs the diagnostic inference engine against a server's collected data. Scores wait stats, blocking, memory, config, and other facts, then traverses a relationship graph to build evidence-backed stories about what's wrong and why. Anomaly detection compares the analysis window against 30-day time-bucketed baselines (hour-of-day x day-of-week) to identify deviations that are unusual for this specific time slot, not just unusual overall. Returns structured findings with severity scores, evidence chains, baseline context for anomalies, and recommended next tools to call. Each finding's confidence is an EVIDENCE score, not a probability: 0.20 for the fired symptom alone, plus up to 0.48 for the share of the root fact's amplifier checks (its expected companions) that matched and up to 0.32 for the depth of the evidence chain, so a lone uncorroborated symptom reads 0.20 and a fully corroborated deep chain approaches 1.0; confidence_basis says in words what each value rests on. Rank by severity for impact and by confidence for how much of the engine's own corroboration showed up; do not multiply them. A remediable finding also carries remediation_command: the full copy-paste T-SQL remediation (identical to the viewer card), including a two-sided risk-disclosure comment header on destructive changes; it is advisory only and never executed. A force-plan remediation additionally carries structured_remediation: the same decision as machine-readable fields — eligible, named blockers (parameter_sensitivity_cofired, secondary_replica_evidence), evidence numbers, and split force_sql/unforce_sql/verify_sql artifacts — so agents consume the verdict as data instead of parsing comment prose. Set as_of to analyze a PAST window instead of the present — hours_back stays the window's LENGTH, and the anomaly baseline moves with it, so the findings are the ones that window deserves rather than today's findings over older rows. An anchored run is EXPLORATORY: its findings are returned in full but deliberately NOT written to the store, because a finding row is stamped with the time the analysis RAN and would then be read as this server's current state by get_analysis_findings and by the viewer. The result says so in persisted / persistence_note.")] public static async Task AnalyzeServer( DarlingAnalysisService analysisService, @@ -225,7 +234,7 @@ public static async Task GetAnalysisFacts( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, [Description("Hours of data to analyze. Default 4.")] int hours_back = 4, - [Description("Filter to a specific source category: waits, blocking, config, memory. Omit for all.")] string? source = null, + [Description(FactSourceFilterDescription)] string? source = null, [Description("Minimum severity to include. Default 0 (all facts). Use 0.5 to see only significant facts.")] double min_severity = 0, [Description(McpHelpers.AsOfDescription)] string? as_of = null) { @@ -235,6 +244,11 @@ public static async Task GetAnalysisFacts( var validation = McpHelpers.ValidateWindow(hours_back, as_of, out var windowEnd); if (validation != null) return validation; + /* #3541 A13: an unknown source is refused with the whole accepted set, never applied as a filter + that matches nothing. The set is the scorer's registry, not a copy of it. */ + validation = McpHelpers.ValidateChoice(source, FactScorer.KnownSources, "source"); + if (validation != null) return validation; + /* Null for an absent anchor — see analyze_server's note. Nothing here persists, so the distinction costs nothing; it is kept so AnalysisContext.AsOfUtc means one thing everywhere. */ var anchor = string.IsNullOrWhiteSpace(as_of) ? (DateTime?)null : windowEnd; diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingSessionReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingSessionReader.cs index 033b42349..0c78698f1 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingSessionReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingSessionReader.cs @@ -47,7 +47,20 @@ public sealed record ActiveQueryRow( long TotalElapsedTimeMs, string? ElapsedTimeFormatted, long LogicalReads, long Reads, long Writes, string? WaitType, long WaitTimeMs, int BlockingSessionId, int Dop, int ParallelWorkerCount, double GrantedQueryMemoryGb, string? TransactionIsolationLevel, int OpenTransactionCount, - string? LoginName, string? HostName, string? ProgramName, string? QueryText); + string? LoginName, string? HostName, string? ProgramName, string? QueryText) + { + /// Some row in the SAME capture names this session as its blocker (#3541 A13) — the reason a + /// WAITFOR row can be on the page. + public bool IsHeadBlocker { get; init; } + + /// For a victim: its blocker had a row in the same capture at all. False is the idle + /// open-transaction head blocker sys.dm_exec_requests never lists. + public bool BlockerInCapture { get; init; } + + /// For a victim: its blocker also passes the caller's filters, so it is in the population the + /// page is drawn from (it may still be past the page — the tool checks that). + public bool BlockerInPopulation { get; init; } + } /// One waiting-task snapshot row. public sealed record WaitingTaskRow( @@ -110,48 +123,146 @@ public static async Task> GetLatestSessionStatsAsync( /// /// The captured query snapshots over the window — the viewer's LatestQuerySnapshotsSql projected /// to the columns Lite's get_active_queries surfaces, from the base query_snapshots table (the - /// viewer reads base here too). granted_query_memory_gb is numeric(18,2) → double precision. WAITFOR - /// shells are trimmed. $1 server_id, $2/$3 window (naive UTC). + /// viewer reads base here too). granted_query_memory_gb is numeric(18,2) → double precision. + /// $1 server_id, $2/$3 window (naive UTC), $4 row cap, $5 database filter (NULL = all), $6 blocking_only. + /// + /// Every filter is part of the query (#3541 A13). This read used to return the whole window + /// unfiltered and unbounded; the tool then applied database_name and blocking_only in C#, + /// took limit, and published the pre-filter row count as total_snapshots beside the page — + /// so the "total" was of a different population from the rows, and a page could be empty while the + /// window held matches. Now the filters are predicates on the same statement, the population count is + /// COUNT(*) OVER () on the FILTERED rows above the parameterised LIMIT (the #3613 idiom), and + /// the cap is the caller's, fetched at limit + 1 so truncation is observed rather than inferred. + /// + /// Head blockers are never stripped (#3541 A13). The WAITFOR trim exists to drop the idle + /// shells a monitoring session leaves in dm_exec_requests, but the classic head blocker IS a session + /// sitting in WAITFOR with an open transaction — and this read dropped it while its victims' + /// blocking_session_id pointed at the session that was no longer on the page. A row is kept + /// whatever its text when some row in the SAME capture names it as its blocker. Same capture, not same + /// window: session ids are reused, so "any row in the window points at this session id" would resurrect + /// unrelated sessions from other snapshots, which is what the old C# arm did. + /// + /// The two blocker-presence flags. blocker_in_capture: the victim's blocker had a row + /// in the same capture at all — FALSE is the other classic head blocker, a session idle in an open + /// transaction, which sys.dm_exec_requests never lists and so was never captured; the tool says so on the + /// row rather than leaving a dangling id. blocker_in_population: the blocker also passes the + /// caller's own filters (a head blocker in another database under a database_name filter does + /// not) — the caller asked for that database, so the row is honoured and the victim says its blocker was + /// filtered. A blocker that passes both but falls past the page is detected by the tool, which has the + /// page. /// public const string ActiveQueriesSql = """ + WITH window_rows AS ( + SELECT + collection_time, + session_id, + database_name, + status, + cpu_time_ms, + total_elapsed_time_ms, + elapsed_time_formatted, + logical_reads, + reads, + writes, + wait_type, + wait_time_ms, + blocking_session_id, + dop, + parallel_worker_count, + CAST(granted_query_memory_gb AS double precision) AS granted_query_memory_gb, + transaction_isolation_level, + open_transaction_count, + login_name, + host_name, + program_name, + query_text + FROM query_snapshots + WHERE server_id = $1 + AND collection_time >= $2 + AND collection_time <= $3 + ), + heads AS ( + /* (capture, session) pairs some victim in the SAME capture points at. */ + SELECT DISTINCT collection_time, blocking_session_id AS session_id + FROM window_rows + WHERE blocking_session_id > 0 + ), + population AS ( + SELECT + w.*, + (h.session_id IS NOT NULL) AS is_head_blocker, + EXISTS ( + SELECT 1 + FROM window_rows b + WHERE b.collection_time = w.collection_time + AND b.session_id = w.blocking_session_id + ) AS blocker_in_capture + FROM window_rows AS w + LEFT JOIN heads AS h + ON h.collection_time = w.collection_time + AND h.session_id = w.session_id + WHERE (w.query_text NOT LIKE 'WAITFOR%' OR h.session_id IS NOT NULL) + AND ($5::text IS NULL OR w.database_name = $5) + AND (NOT $6::boolean OR w.blocking_session_id > 0 OR h.session_id IS NOT NULL) + ) SELECT - collection_time, - session_id, - database_name, - status, - cpu_time_ms, - total_elapsed_time_ms, - elapsed_time_formatted, - logical_reads, - reads, - writes, - wait_type, - wait_time_ms, - blocking_session_id, - dop, - parallel_worker_count, - CAST(granted_query_memory_gb AS double precision), - transaction_isolation_level, - open_transaction_count, - login_name, - host_name, - program_name, - query_text - FROM query_snapshots - WHERE server_id = $1 - AND collection_time >= $2 - AND collection_time <= $3 - AND query_text NOT LIKE 'WAITFOR%' - ORDER BY collection_time DESC, cpu_time_ms DESC + p.collection_time, + p.session_id, + p.database_name, + p.status, + p.cpu_time_ms, + p.total_elapsed_time_ms, + p.elapsed_time_formatted, + p.logical_reads, + p.reads, + p.writes, + p.wait_type, + p.wait_time_ms, + p.blocking_session_id, + p.dop, + p.parallel_worker_count, + p.granted_query_memory_gb, + p.transaction_isolation_level, + p.open_transaction_count, + p.login_name, + p.host_name, + p.program_name, + p.query_text, + p.is_head_blocker, + p.blocker_in_capture, + EXISTS ( + SELECT 1 + FROM population q + WHERE q.collection_time = p.collection_time + AND q.session_id = p.blocking_session_id + ) AS blocker_in_population, + COUNT(*) OVER () AS population_count + FROM population AS p + ORDER BY p.collection_time DESC, p.cpu_time_ms DESC + LIMIT $4 """; - public static async Task> GetActiveQueriesAsync( - NpgsqlDataSource postgres, int serverId, DateTime startUtc, DateTime endUtc, CancellationToken cancellationToken = default) + /// The filtered page plus the filtered population's size, from one statement. + public sealed record ActiveQueriesPage(List Rows, long PopulationCount); + + /// + /// The newest snapshots over the window that pass the filters, with the filtered + /// population's count (#3541 A13). Callers detecting truncation pass limit + 1 and read the extra + /// row as the signal. null = every database; + /// keeps victims and the head blockers of victims in the same capture. + /// + public static async Task GetActiveQueriesAsync( + NpgsqlDataSource postgres, int serverId, DateTime startUtc, DateTime endUtc, int cap, + string? databaseName = null, bool blockingOnly = false, CancellationToken cancellationToken = default) { var rows = new List(); + long populationCount = 0; await using var command = postgres.CreateCommand(ActiveQueriesSql); command.CommandTimeout = McpCommandDeadlines.ReadSeconds; DarlingMcpReadParameters.AddWindow(command, serverId, startUtc, endUtc); + DarlingMcpReadParameters.AddInt(command, cap); + DarlingMcpReadParameters.AddNullableText(command, databaseName); + DarlingMcpReadParameters.AddBoolean(command, blockingOnly); await using var reader = await command.ExecuteReaderAsync(cancellationToken); while (await reader.ReadAsync(cancellationToken)) { @@ -177,10 +288,16 @@ public static async Task> GetActiveQueriesAsync( reader.IsDBNull(18) ? null : reader.GetString(18), reader.IsDBNull(19) ? null : reader.GetString(19), reader.IsDBNull(20) ? null : reader.GetString(20), - reader.IsDBNull(21) ? null : reader.GetString(21))); + reader.IsDBNull(21) ? null : reader.GetString(21)) + { + IsHeadBlocker = !reader.IsDBNull(22) && reader.GetBoolean(22), + BlockerInCapture = !reader.IsDBNull(23) && reader.GetBoolean(23), + BlockerInPopulation = !reader.IsDBNull(24) && reader.GetBoolean(24), + }); + populationCount = reader.GetInt64(25); } - return rows; + return new ActiveQueriesPage(rows, populationCount); } /* ─────────────────────────── waiting tasks (base table over the window) ─────────────────────────── */ diff --git a/Darling/PerformanceMonitor.Darling.Storage/DailySummarySql.cs b/Darling/PerformanceMonitor.Darling.Storage/DailySummarySql.cs index f762bfa73..a45618e23 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/DailySummarySql.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/DailySummarySql.cs @@ -154,8 +154,24 @@ UNION SELECT d FROM alerts the count. 0 when the blocking came from a source without a wait time. */ COALESCE(CASE WHEN COALESCE(b.c, 0) > 0 THEN b.max_wait_ms ELSE dm.max_wait_ms END, 0) AS peak_block_wait_ms, /* Every collector run in the window (#3539 A2): the denominator that turns collection_errors - into a share. Appended LAST so every existing ordinal read stays where it was. */ - COALESCE(cl.runs, 0) AS collection_runs + into a share. Appended after peak_block_wait_ms so every existing ordinal read stays where it was. */ + COALESCE(cl.runs, 0) AS collection_runs, + /* #3541 A9: how many of the seven per-signal sources hold at least one row for the day — the + retention arm's PRESENCE fact. The spine is a UNION over nine sources aging out at different + horizons, each LEFT JOINed and COALESCEd to zero, so a day the collection log (60 days) or the + alert log (90) still names can have every signal (30) purged and read as measured-zero-Healthy. + The reader judges such a day by this count against the store's retention horizon: zero sources + past the horizon is a purged shell, some sources past it is a day the purge has not reached. + A source counts as present when its grouped CTE produced a row for the day, which for cpu is + "any sample" (the FILTER is inside the aggregate) and for waits is "any positive delta". + Appended LAST, after collection_runs, for the same ordinal reason. */ + (CASE WHEN w.d IS NULL THEN 0 ELSE 1 END) + + (CASE WHEN q.d IS NULL THEN 0 ELSE 1 END) + + (CASE WHEN dl.d IS NULL THEN 0 ELSE 1 END) + + (CASE WHEN b.d IS NULL THEN 0 ELSE 1 END) + + (CASE WHEN dm.d IS NULL THEN 0 ELSE 1 END) + + (CASE WHEN cp.d IS NULL THEN 0 ELSE 1 END) + + (CASE WHEN m.d IS NULL THEN 0 ELSE 1 END) AS signal_sources_present FROM day_spine s LEFT JOIN waits w ON w.d = s.d LEFT JOIN queries q ON q.d = s.d diff --git a/Lite.Tests/DailyHealthBandTests.cs b/Lite.Tests/DailyHealthBandTests.cs index ffd30eb31..2c1265a91 100644 --- a/Lite.Tests/DailyHealthBandTests.cs +++ b/Lite.Tests/DailyHealthBandTests.cs @@ -590,3 +590,103 @@ public void TodayCell_MinutesOld_FallsToTheUnrateableArm() DailyHealthBandCalculator.Classify(Signals(deadlocks: 1, window: window))); } } + + +/// +/// #3541 A9: the one decision both SKUs' daily-summary readers make about a returned day — is it a +/// measurement, or the shape retention left behind — pinned identically here and in the twin project. +/// +/// The defect: the daily aggregate's spine is a UNION over nine sources aging out at different +/// horizons, each COALESCEd to zero, so a day between the shortest horizon (the signals' 30 days) and the +/// longest (the alert log's 90) kept its spine row while every signal the band reads was gone — and zeros +/// band Healthy. The state is decided from two inputs, the day and the horizon; the run count only decides +/// between the two INSIDE-retention states. The horizon test comes first, because runs > 0 alone was +/// half the fix and called the whole second month collected. +/// +public class DailySummaryRetentionTests +{ + private static readonly DateTime Horizon = new(2026, 8, 19); + + [Theory] + [InlineData("2026-08-18", 1_440, 0, DailySummaryDataState.Purged)] /* the day before: a surviving run record does not rescue it */ + [InlineData("2026-08-18", 0, 0, DailySummaryDataState.Purged)] /* nor does the absence of one change the verdict */ + [InlineData("2026-07-01", 5, 0, DailySummaryDataState.Purged)] + [InlineData("2026-08-18", 1_440, 7, DailySummaryDataState.PastHorizon)] /* signal rows still there: the purge has not reached it */ + [InlineData("2026-08-18", 0, 1, DailySummaryDataState.PastHorizon)] /* even one source present withholds "purged" */ + [InlineData("2026-08-19", 1, 0, DailySummaryDataState.Collected)] /* the horizon day itself is held */ + [InlineData("2026-09-01", 1_440, 7, DailySummaryDataState.Collected)] + [InlineData("2026-09-01", 1_440, 0, DailySummaryDataState.Collected)] /* inside retention a quiet day needs no signal rows to be collected */ + [InlineData("2026-09-01", 0, 3, DailySummaryDataState.NoRunRecord)] /* inside retention, signals but no run recorded — a disclosure, the band stands */ + public void TheState_IsDecidedByTheHorizonAndPresenceFirst_ThenByTheRunCount(string day, long runs, int present, DailySummaryDataState expected) + { + var date = DateTime.ParseExact(day, "yyyy-MM-dd", System.Globalization.CultureInfo.InvariantCulture); + Assert.Equal(expected, DailySummaryRetention.StateFor(date, runs, present, Horizon)); + /* A time-of-day on either side changes nothing: the decision is on DATES. */ + Assert.Equal(expected, DailySummaryRetention.StateFor(date.AddHours(23), runs, present, Horizon.AddHours(5))); + } + + /// The horizon is the cutoff instant's DATE — see + /// for why that is exact on the TimescaleDB path and at most a partial day generous on the DELETE path. + [Fact] + public void TheHorizon_IsTheCutoffsDate() + { + var now = new DateTime(2026, 9, 18, 14, 30, 0, DateTimeKind.Utc); + Assert.Equal(new DateTime(2026, 8, 19), DailySummaryRetention.HorizonFor(now, 30)); + Assert.Equal(new DateTime(2026, 9, 17), DailySummaryRetention.HorizonFor(now, 1)); + Assert.Equal(DateTimeKind.Utc, DailySummaryRetention.HorizonFor(now, 30).Kind); + Assert.Throws(() => DailySummaryRetention.HorizonFor(now, 0)); + } + + [Fact] + public void TheVocabulary_IsOneWordPerState_AndTheNoteSaysWhatTheZerosAre() + { + Assert.Equal("collected", DailySummaryRetention.Label(DailySummaryDataState.Collected)); + Assert.Equal("purged", DailySummaryRetention.Label(DailySummaryDataState.Purged)); + Assert.Equal("past_horizon", DailySummaryRetention.Label(DailySummaryDataState.PastHorizon)); + Assert.Equal("no_run_record", DailySummaryRetention.Label(DailySummaryDataState.NoRunRecord)); + + Assert.Null(DailySummaryRetention.Note(DailySummaryDataState.Collected, Horizon)); + var purged = DailySummaryRetention.Note(DailySummaryDataState.Purged, Horizon)!; + Assert.StartsWith("PURGED", purged, StringComparison.Ordinal); + Assert.Contains("2026-08-19", purged, StringComparison.Ordinal); + Assert.Contains("absences, not measurements", purged, StringComparison.Ordinal); + var past = DailySummaryRetention.Note(DailySummaryDataState.PastHorizon, Horizon, 3)!; + Assert.StartsWith("PAST HORIZON", past, StringComparison.Ordinal); + Assert.Contains("3 of 7 signal sources", past, StringComparison.Ordinal); + Assert.Equal(7, DailySummaryRetention.SignalSourceCount); + var noRun = DailySummaryRetention.Note(DailySummaryDataState.NoRunRecord, Horizon)!; + Assert.StartsWith("NO RUN RECORD", noRun, StringComparison.Ordinal); + Assert.Contains("the band stands", noRun, StringComparison.Ordinal); + } + + /// The band's own contract, end to end: a signals projection with HasData folded from a + /// past-horizon state is No Data even under a Critical count — the calendar's grey, never green or red. + /// The two inside-retention states keep the band, because inside retention a zero is a measurement. + [Fact] + public void APastHorizonDay_BandsNoData_WhateverItsCounts_AndAnInsideRetentionDayKeepsItsBand() + { + foreach (var state in new[] { DailySummaryDataState.Collected, DailySummaryDataState.NoRunRecord }) + { + var signals = new DailyHealthSignals + { + HasData = state is not (DailySummaryDataState.Purged or DailySummaryDataState.PastHorizon), + Deadlocks = 480, + CollectionRuns = 1_440, + Window = TimeSpan.FromDays(1), + }; + Assert.Equal(DailyHealthBand.Critical, DailyHealthBandCalculator.Classify(signals)); + } + + foreach (var state in new[] { DailySummaryDataState.Purged, DailySummaryDataState.PastHorizon }) + { + var signals = new DailyHealthSignals + { + HasData = state is not (DailySummaryDataState.Purged or DailySummaryDataState.PastHorizon), + Deadlocks = 480, + CollectionRuns = 1_440, + Window = TimeSpan.FromDays(1), + }; + Assert.Equal(DailyHealthBand.NoData, DailyHealthBandCalculator.Classify(signals)); + } + } +} diff --git a/Lite.Tests/FindingsRetentionHorizonPinTests.cs b/Lite.Tests/FindingsRetentionHorizonPinTests.cs index 06383f769..fb75895b9 100644 --- a/Lite.Tests/FindingsRetentionHorizonPinTests.cs +++ b/Lite.Tests/FindingsRetentionHorizonPinTests.cs @@ -116,7 +116,10 @@ public void TheParquetArchiveHorizon_IsASeparateKnob_InADifferentUnit() { var archive = ReadRepoSource("Lite", "Services", "RetentionService.cs"); - Assert.Contains("int retentionMonths = 3", archive, StringComparison.Ordinal); + /* #3541 A9 promoted the literal to a named constant so the daily-summary reader publishes the SAME + horizon the cleanup enforces; the knob is still months, still 3, still RetentionService's own. */ + Assert.Contains("public const int ArchiveRetentionMonths = 3;", archive, StringComparison.Ordinal); + Assert.Contains("int retentionMonths = ArchiveRetentionMonths", archive, StringComparison.Ordinal); Assert.False( archive.Contains("AnalysisRetentionDefaults", StringComparison.Ordinal), "RetentionService took on the findings horizon — the parquet archive window is a separate retention"); diff --git a/Lite.Tests/McpMissMessageParityPinTests.cs b/Lite.Tests/McpMissMessageParityPinTests.cs index b7ee844ef..d9ce6b23c 100644 --- a/Lite.Tests/McpMissMessageParityPinTests.cs +++ b/Lite.Tests/McpMissMessageParityPinTests.cs @@ -178,6 +178,16 @@ what lives twice is the tool-description sentence teaching a caller what the num "For ANOMALY_* facts the metadata carries baseline_confidence — the baseline's own trustworthiness (tier x sample density), which the scorer multiplies into that fact's severity; it is a different quantity from a finding's confidence in analyze_server.", "Each finding's `confidence` is an EVIDENCE score (0.20 for the symptom alone, more as the root fact's amplifier checks match and the chain deepens — a lone uncorroborated symptom is 0.20, never 1.0) and `confidence_basis` says in words what it rests on", "(an ANOMALY_* fact's `baseline_confidence` is the baseline's trustworthiness, not a finding's `confidence`)", + + /* #3541 A9 / A13. The purged-day refusal (get_daily_summary), the filtered-miss sentences + (get_top_queries_by_cpu under parallel_only / min_dop; get_active_queries under database_name / + blocking_only) and the negative-span refusal all live twice; each is a sentence whose whole job is + to say WHICH kind of nothing came back, and a SKU that reworded one alone would send its callers to + a different next step. The negative-span refusal is built by McpHelpers.ValidateUncappedWindow and + is shared by construction; what lives twice is the call, pinned by McpPageContractTests. */ + "): the per-signal tables the health band reads (deadlocks, blocking, CPU, memory, waits) have been purged for that day, so no health verdict is possible and the counts would be zeros by construction, not by measurement. Longer-lived sources may still record the day — collection_runs and alert_count below are real where non-zero.", + ". The filter was applied in SQL over the whole window, so this is the window's answer rather than a page artefact — drop parallel_only / min_dop to see the unfiltered ranking, or confirm current parallelism with analyze_query_plan.", + ". The filters were applied in SQL over the whole window, so unfiltered snapshots may well exist — drop them to see what the window holds.", }; [Theory] diff --git a/Lite.Tests/McpPageContractTests.cs b/Lite.Tests/McpPageContractTests.cs index 05b05c26b..369a8d796 100644 --- a/Lite.Tests/McpPageContractTests.cs +++ b/Lite.Tests/McpPageContractTests.cs @@ -316,6 +316,180 @@ public async Task GetWaitStats_CapBindsToLimit_AndTruncationIsObserved() AssertPage(whole, "waits", "wait_types_returned", returned: 3, truncated: false); } + /* ───────────────────────── #3541 A13: a filter is part of the query ───────────────────────── */ + + /// + /// parallel_only / min_dop used to be a .Where over the top-N page the SQL had already + /// cut, so a box whose hottest N plans were serial answered an EMPTY page while the window held a parallel + /// plan just past the cut. The fixture is that shape — three serial groups hotter than one parallel group, + /// read at top = 2. The old code returned nothing; the fixed read returns the parallel group, and the + /// unfiltered read at the same cap still returns the two hottest serial ones. Darling's twin proves the + /// same pair against live Postgres. + /// + [Fact] + public async Task GetTopQueriesByCpu_ParallelFilter_RanksTheFilteredPopulation_NotTheFilteredPage() + { + var now = WholeSecondsNow().AddMinutes(-2); + await SeedQueryStatAsync(now, "0xSERIAL1", cpuUs: 900_000L, maxDop: 1); + await SeedQueryStatAsync(now, "0xSERIAL2", cpuUs: 800_000L, maxDop: 1); + await SeedQueryStatAsync(now, "0xSERIAL3", cpuUs: 700_000L, maxDop: 1); + await SeedQueryStatAsync(now, "0xPARALLEL", cpuUs: 100_000L, maxDop: 8); + + var unfiltered = Parse(await McpQueryTools.GetTopQueriesByCpu(_dataService, _serverManager, ServerName, 1, 2)); + Assert.Equal(new[] { "0xSERIAL1", "0xSERIAL2" }, Hashes(unfiltered)); + Assert.Equal(JsonValueKind.Null, unfiltered.GetProperty("filter_applied").ValueKind); + + var parallel = Parse(await McpQueryTools.GetTopQueriesByCpu(_dataService, _serverManager, ServerName, 1, 2, parallel_only: true)); + Assert.Equal(new[] { "0xPARALLEL" }, Hashes(parallel)); + Assert.Contains("max_dop >= 2", parallel.GetProperty("filter_applied").GetString(), StringComparison.Ordinal); + + /* min_dop above the seeded DOP: an empty FILTERED page is the window's answer, not a collection miss. */ + var tooHigh = Parse(await McpQueryTools.GetTopQueriesByCpu(_dataService, _serverManager, ServerName, 1, 2, min_dop: 16)); + Assert.Equal("empty", tooHigh.GetProperty("status").GetString()); + Assert.Contains("max_dop >= 16", tooHigh.GetProperty("message").GetString(), StringComparison.Ordinal); + } + + /// + /// One capture holding the three blocker situations the tool now names: a victim whose blocker is a + /// WAITFOR shell in the same capture (the classic head blocker — kept and flagged, where the old trim + /// dropped it while the victim pointed at it), a victim whose blocker was never captured (an idle open + /// transaction), and a victim in one database whose blocker is in another. A stale WAITFOR row under the + /// same session id in an EARLIER capture must not be resurrected: the match is same capture, not same + /// window. The count beside the page is the FILTERED population's, and truncation is observed on it. + /// + [Fact] + public async Task GetActiveQueries_FiltersInTheQuery_KeepsHeadBlockers_AndNamesAbsentOnes() + { + var t = WholeSecondsNow().AddMinutes(-2); + await SeedSnapshotAsync(t, 55, "Db", "UPDATE Posts SET Score = 1", blockingSessionId: 60, cpuMs: 500); + await SeedSnapshotAsync(t, 60, "Db", "WAITFOR DELAY '00:05'", blockingSessionId: 0, cpuMs: 1); + await SeedSnapshotAsync(t, 56, "Db", "DELETE FROM Votes", blockingSessionId: 61, cpuMs: 400); + await SeedSnapshotAsync(t, 57, "OtherDb", "SELECT * FROM Sales", blockingSessionId: 62, cpuMs: 300); + await SeedSnapshotAsync(t, 62, "Db", "UPDATE Users SET Reputation = 0", blockingSessionId: 0, cpuMs: 900); + await SeedSnapshotAsync(t, 70, "Db", "SELECT COUNT(*) FROM Comments", blockingSessionId: 0, cpuMs: 200); + await SeedSnapshotAsync(t.AddMinutes(-2), 60, "Db", "WAITFOR DELAY '00:05'", blockingSessionId: 0, cpuMs: 1); + + var all = Parse(await McpSessionTools.GetActiveQueries(_dataService, _serverManager, ServerName, 1, limit: 50)); + var rows = all.GetProperty("queries").EnumerateArray().ToArray(); + Assert.Equal(6, rows.Length); + Assert.Equal(6, all.GetProperty("total_snapshots").GetInt64()); + Assert.Equal(6, all.GetProperty("snapshots_returned").GetInt32()); + Assert.False(all.GetProperty("truncated").GetBoolean()); + Assert.Equal("collection_time_desc", all.GetProperty("order").GetString()); + Assert.Equal(Stamp(t), all.GetProperty("newest_returned_collection_time").GetString()); + + var head = Row(rows, 60); + Assert.True(head.GetProperty("is_head_blocker").GetBoolean()); + Assert.StartsWith("WAITFOR", head.GetProperty("query_text").GetString(), StringComparison.Ordinal); + Assert.Equal(JsonValueKind.Null, Row(rows, 55).GetProperty("blocker_not_shown").ValueKind); + Assert.Equal("not_captured", Row(rows, 56).GetProperty("blocker_not_shown").GetString()); + Assert.Equal(JsonValueKind.Null, Row(rows, 57).GetProperty("blocker_not_shown").ValueKind); + Assert.Equal(JsonValueKind.Null, Row(rows, 70).GetProperty("is_head_blocker").ValueKind); + + /* database_name, in the query: the population is OtherDb's one victim; its blocker is in Db, so filtered. */ + var other = Parse(await McpSessionTools.GetActiveQueries(_dataService, _serverManager, ServerName, 1, "OtherDb")); + Assert.Equal(1, other.GetProperty("total_snapshots").GetInt64()); + Assert.Equal("filtered", Assert.Single(other.GetProperty("queries").EnumerateArray()).GetProperty("blocker_not_shown").GetString()); + + /* blocking_only, in the query: victims 55/56/57 + heads 60/62 = 5; 70 is out. */ + var blocking = Parse(await McpSessionTools.GetActiveQueries(_dataService, _serverManager, ServerName, 1, blocking_only: true)); + Assert.Equal(5, blocking.GetProperty("total_snapshots").GetInt64()); + Assert.DoesNotContain(blocking.GetProperty("queries").EnumerateArray(), r => r.GetProperty("session_id").GetInt32() == 70); + + /* Truncation on the FILTERED population, as a pair; the WAITFOR head (1 ms CPU) falls past a 4-row + page and its victim says so. */ + var cut = Parse(await McpSessionTools.GetActiveQueries(_dataService, _serverManager, ServerName, 1, blocking_only: true, limit: 4)); + Assert.True(cut.GetProperty("truncated").GetBoolean()); + Assert.Equal(4, cut.GetProperty("snapshots_returned").GetInt32()); + Assert.Equal(5, cut.GetProperty("total_snapshots").GetInt64()); + Assert.Equal("past_page", Row(cut.GetProperty("queries").EnumerateArray().ToArray(), 55).GetProperty("blocker_not_shown").GetString()); + var whole = Parse(await McpSessionTools.GetActiveQueries(_dataService, _serverManager, ServerName, 1, blocking_only: true, limit: 5)); + Assert.False(whole.GetProperty("truncated").GetBoolean()); + + /* A filtered miss names the filter rather than calling the window empty. */ + var miss = Parse(await McpSessionTools.GetActiveQueries(_dataService, _serverManager, ServerName, 1, "NoSuchDb")); + Assert.Equal("empty", miss.GetProperty("status").GetString()); + Assert.Contains("database_name 'NoSuchDb'", miss.GetProperty("message").GetString(), StringComparison.Ordinal); + + /* A13's third item: the uncapped reads refuse a negative span rather than flipping its sign. */ + Assert.StartsWith("Invalid hours_back value '-24'", await McpHealthTools.GetCollectionLog(_dataService, _serverManager, ServerName, -24), StringComparison.Ordinal); + Assert.StartsWith("Invalid hours_back value '0'", await McpHealthTools.GetCurrentWaitsTrend(_dataService, _serverManager, ServerName, 0), StringComparison.Ordinal); + Assert.StartsWith("Invalid hours_back value '-1'", await McpHealthTools.GetBlockingStats(_dataService, _serverManager, ServerName, -1), StringComparison.Ordinal); + } + + /* ───────────────────────── #3541 A9: retention ghosts ───────────────────────── */ + + /// + /// Lite's one horizon is the archive retention (three months, every table together), so a day older than + /// that which the spine still holds is a shell whatever its run count: purged, NoData, never + /// Healthy. A day inside the horizon with signal rows but no run record is no_run_record and keeps its band. Today's run + /// is collected and Healthy beside them. The single-day read of a purged day refuses a verdict, and + /// summary_date is exact ISO-8601. + /// + [Fact] + public async Task DailySummary_StopsPaintingPurgedDaysGreen_AndPublishesTheHorizon() + { + var today = DateTime.UtcNow.Date; + var horizon = LocalDataService.DailySummaryRetentionHorizon(DateTime.UtcNow); + var ghostDay = horizon.AddDays(-10); + var uncollectedDay = today.AddDays(-3); + + await SeedRunAsync(DateTime.UtcNow.AddMinutes(-2)); + await SeedRunAsync(ghostDay.AddHours(12)); + /* Signal rows and no run record: the spine holds the day from wait_stats alone. */ + await SeedWaitStatAsync(uncollectedDay.AddHours(12), "CXPACKET", 4000); + + var daysBack = (int)(today - ghostDay).TotalDays + 1; + var range = Parse(await McpHealthTools.GetDailySummaryRange(_dataService, _serverManager, ServerName, daysBack)); + Assert.Equal(horizon.ToString("yyyy-MM-dd"), range.GetProperty("retention_horizon").GetString()); + Assert.Equal(3, range.GetProperty("day_count").GetInt32()); + Assert.Equal(1, range.GetProperty("days_before_horizon").GetInt32()); + Assert.Equal(1, range.GetProperty("purged_day_count").GetInt32()); + Assert.Equal(1, range.GetProperty("collected_day_count").GetInt32()); + + var days = range.GetProperty("days").EnumerateArray().ToArray(); + var ghost = Assert.Single(days, d => d.GetProperty("summary_date").GetString() == ghostDay.ToString("yyyy-MM-dd")); + Assert.Equal(1, ghost.GetProperty("collection_runs").GetInt64()); + Assert.Equal("purged", ghost.GetProperty("data_state").GetString()); + Assert.Equal("NoData", ghost.GetProperty("health_band").GetString()); + Assert.Equal("No Data", ghost.GetProperty("overall_health").GetString()); + Assert.StartsWith("PURGED", ghost.GetProperty("data_note").GetString(), StringComparison.Ordinal); + + /* Inside retention with signal rows and no run record: a disclosure, not a withheld verdict — the day + keeps its band (Healthy here; PerformanceCalendarDataTests pins an alert-only day as Warning). */ + var uncollected = Assert.Single(days, d => d.GetProperty("summary_date").GetString() == uncollectedDay.ToString("yyyy-MM-dd")); + Assert.Equal("no_run_record", uncollected.GetProperty("data_state").GetString()); + Assert.Equal("Healthy", uncollected.GetProperty("health_band").GetString()); + Assert.StartsWith("NO RUN RECORD", uncollected.GetProperty("data_note").GetString(), StringComparison.Ordinal); + Assert.Equal(0, uncollected.GetProperty("collection_runs").GetInt64()); + Assert.Equal("CXPACKET", uncollected.GetProperty("top_wait_type").GetString()); + + var live = Assert.Single(days, d => d.GetProperty("summary_date").GetString() == today.ToString("yyyy-MM-dd")); + Assert.Equal("collected", live.GetProperty("data_state").GetString()); + Assert.Equal("Healthy", live.GetProperty("overall_health").GetString()); + Assert.Equal(JsonValueKind.Null, live.GetProperty("data_note").ValueKind); + + var single = Parse(await McpHealthTools.GetDailySummary(_dataService, _serverManager, ServerName, ghostDay.ToString("yyyy-MM-dd"))); + Assert.Equal("unavailable", single.GetProperty("status").GetString()); + Assert.Equal("purged", single.GetProperty("hints").GetProperty("data_state").GetString()); + Assert.Equal(horizon.ToString("yyyy-MM-dd"), single.GetProperty("hints").GetProperty("retention_horizon").GetString()); + + var todayRow = Parse(await McpHealthTools.GetDailySummary(_dataService, _serverManager, ServerName)); + Assert.Equal("collected", todayRow.GetProperty("data_state").GetString()); + Assert.Equal(horizon.ToString("yyyy-MM-dd"), todayRow.GetProperty("retention_horizon").GetString()); + + Assert.StartsWith("Invalid summary_date value '01/02/2026'", await McpHealthTools.GetDailySummary(_dataService, _serverManager, ServerName, "01/02/2026"), StringComparison.Ordinal); + } + + /// The Lite horizon is the archive retention constant, in months, from the reader's clock. + [Fact] + public void TheLiteHorizon_IsTheArchiveRetention_InMonths() + { + var now = new DateTime(2026, 9, 18, 14, 30, 0, DateTimeKind.Utc); + Assert.Equal(now.AddMonths(-RetentionService.ArchiveRetentionMonths).Date, LocalDataService.DailySummaryRetentionHorizon(now)); + Assert.Equal(3, RetentionService.ArchiveRetentionMonths); + } + /* ───────────────────────── the contract, as a census ───────────────────────── */ /// @@ -334,6 +508,8 @@ public static readonly (Type Tools, string ToolName)[] PagedTools = (typeof(McpPlanCorrectionTools), "get_plan_corrections"), (typeof(McpWaitTools), "get_waiting_tasks"), (typeof(McpWaitTools), "get_wait_stats"), + /* #3541 A13: joined the dialect when its filters moved into the SQL — see the A13 tests above. */ + (typeof(McpSessionTools), "get_active_queries"), ]; [Fact] @@ -480,4 +656,35 @@ INSERT INTO wait_stats delta_waiting_tasks, delta_wait_time_ms, delta_signal_wait_time_ms) VALUES ($1, $2, $3, $4, $5, 0, 0, 0, 10, $6, 100)", _nextId--, Naive(at), _serverId, ServerName, waitType, deltaMs); + + /* #3541 A13 / A9 fixtures. */ + + private Task SeedQueryStatAsync(DateTime at, string queryHash, long cpuUs, int maxDop) => ExecAsync(@" +INSERT INTO query_stats + (collection_id, collection_time, server_id, server_name, database_name, query_hash, query_plan_hash, sql_handle, plan_handle, + query_text, last_execution_time, creation_time, delta_execution_count, delta_worker_time, delta_elapsed_time, delta_logical_reads, + min_dop, max_dop) +VALUES ($1, $2, $3, $4, 'Db', $5, '0xPLANHASH', $6, '0xPLANH', $7, $2, $2, 10, $8, $8, 100, 1, $9)", + _nextId--, Naive(at), _serverId, ServerName, queryHash, "0xSQLH" + queryHash, "SELECT " + queryHash, cpuUs, maxDop); + + private Task SeedSnapshotAsync(DateTime at, int sessionId, string database, string text, int blockingSessionId, long cpuMs) => ExecAsync(@" +INSERT INTO query_snapshots + (collection_id, collection_time, server_id, server_name, session_id, database_name, query_text, status, blocking_session_id, + wait_type, cpu_time_ms, total_elapsed_time_ms) +VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11, $12)", + _nextId--, Naive(at), _serverId, ServerName, sessionId, database, text, + blockingSessionId > 0 ? "suspended" : "running", blockingSessionId, blockingSessionId > 0 ? "LCK_M_X" : null, cpuMs, cpuMs * 2); + + private Task SeedRunAsync(DateTime at) => ExecAsync(@" +INSERT INTO collection_log + (log_id, server_id, server_name, collector_name, collection_time, + duration_ms, status, error_message, rows_collected, sql_duration_ms, duckdb_duration_ms) +VALUES ($1, $2, $3, 'wait_stats', $4, 100, 'SUCCESS', NULL, 10, 80, 20)", + _nextId--, _serverId, ServerName, Naive(at)); + + private static string[] Hashes(JsonElement root) => + root.GetProperty("queries").EnumerateArray().Select(q => q.GetProperty("query_hash").GetString()!).ToArray(); + + private static JsonElement Row(JsonElement[] rows, int sessionId) => + Assert.Single(rows, r => r.GetProperty("session_id").GetInt32() == sessionId); } diff --git a/Lite.Tests/PerformanceCalendarDataTests.cs b/Lite.Tests/PerformanceCalendarDataTests.cs index 7b044df0b..5602140cb 100644 --- a/Lite.Tests/PerformanceCalendarDataTests.cs +++ b/Lite.Tests/PerformanceCalendarDataTests.cs @@ -32,8 +32,15 @@ public class PerformanceCalendarDataTests : IClassFixture, private const string ServerName = "CalTestServer"; private long _nextId = -1; - private static readonly DateTime MonthStart = new(2026, 7, 1, 0, 0, 0, DateTimeKind.Utc); - private static readonly DateTime MonthEnd = new(2026, 8, 1, 0, 0, 0, DateTimeKind.Utc); + /* The fixture month is the calendar month TWO months before the current one, not a fixed month + (#3541 A9): the daily summary now judges each day against the store's retention horizon (Lite: + RetentionService.ArchiveRetentionMonths back from the wall clock), and a fixed July 2026 would have + drifted past that horizon within weeks of landing, turning every band below into No Data on a date + nobody changed. Two months back is always inside a three-month horizon and always a finished month, + so the still-forming-day clamp never applies; the day-number comments below ("07-10") read as + "day 10 of the fixture month". */ + private static readonly DateTime MonthStart = new DateTime(DateTime.UtcNow.Year, DateTime.UtcNow.Month, 1, 0, 0, 0, DateTimeKind.Utc).AddMonths(-2); + private static readonly DateTime MonthEnd = MonthStart.AddMonths(1); public PerformanceCalendarDataTests(SharedDuckDbFixture fixture) { @@ -109,7 +116,7 @@ private Task SeedAlertAsync(DateTime day, string metric, bool dismissed) => "INSERT INTO config_alert_log (alert_time, server_id, server_name, metric_name, current_value, threshold_value, dismissed) VALUES ($1,$2,$3,$4,1.0,1.0,$5)", day.AddHours(1), ServerId, ServerName, metric, dismissed); - private static DateTime Day(int d) => new(2026, 7, d, 0, 0, 0, DateTimeKind.Utc); + private static DateTime Day(int d) => MonthStart.AddDays(d - 1); [Fact] public async Task GetDailySummaryRange_BucketsEachDay_AndBandsViaSharedCalculator() diff --git a/Lite/Mcp/McpAnalysisTools.cs b/Lite/Mcp/McpAnalysisTools.cs index 29fe2ae0a..a149da1d0 100644 --- a/Lite/Mcp/McpAnalysisTools.cs +++ b/Lite/Mcp/McpAnalysisTools.cs @@ -11,6 +11,14 @@ namespace PerformanceMonitorLite.Mcp; [McpServerToolType] public sealed class McpAnalysisTools { + /// + /// The source parameter's description, verbatim Darling's DarlingMcpTools.FactSourceFilterDescription, + /// built from so the documented set IS the accepted set (#3541 A13). + /// + internal const string FactSourceFilterDescription = + "Filter to one source category. Accepted values (the engine's complete source registry, refused otherwise): " + + "anomaly, bad_actor, blocking, config, coverage, cpu, database_config, disk, io, jobs, memory, queries, sessions, tempdb, waits. Omit for all."; + [McpServerTool(Name = "analyze_server"), Description("Runs the diagnostic inference engine against a server's collected data. Scores wait stats, blocking, memory, config, and other facts, then traverses a relationship graph to build evidence-backed stories about what's wrong and why. Anomaly detection compares the analysis window against 30-day time-bucketed baselines (hour-of-day x day-of-week) to identify deviations that are unusual for this specific time slot, not just unusual overall. Returns structured findings with severity scores, evidence chains, baseline context for anomalies, and recommended next tools to call. Each finding's confidence is an EVIDENCE score, not a probability: 0.20 for the fired symptom alone, plus up to 0.48 for the share of the root fact's amplifier checks (its expected companions) that matched and up to 0.32 for the depth of the evidence chain, so a lone uncorroborated symptom reads 0.20 and a fully corroborated deep chain approaches 1.0; confidence_basis says in words what each value rests on. Rank by severity for impact and by confidence for how much of the engine's own corroboration showed up; do not multiply them. A remediable finding also carries remediation_command: the full copy-paste T-SQL remediation (identical to the viewer card), including a two-sided risk-disclosure comment header on destructive changes; it is advisory only and never executed. A force-plan remediation additionally carries structured_remediation: the same decision as machine-readable fields — eligible, named blockers (parameter_sensitivity_cofired, secondary_replica_evidence), evidence numbers, and split force_sql/unforce_sql/verify_sql artifacts — so agents consume the verdict as data instead of parsing comment prose. Set as_of to analyze a PAST window instead of the present — hours_back stays the window's LENGTH, and the anomaly baseline moves with it, so the findings are the ones that window deserves rather than today's findings over older rows. An anchored run is EXPLORATORY: its findings are returned in full but deliberately NOT written to the store, because a finding row is stamped with the time the analysis RAN and would then be read as this server's current state by get_analysis_findings and by the viewer. The result says so in persisted / persistence_note.")] public static async Task AnalyzeServer( AnalysisService analysisService, @@ -200,7 +208,7 @@ public static async Task GetAnalysisFacts( ServerManager serverManager, [Description("Server name or display name.")] string? server_name = null, [Description("Hours of data to analyze. Default 4.")] int hours_back = 4, - [Description("Filter to a specific source category: waits, blocking, config, memory. Omit for all.")] string? source = null, + [Description(FactSourceFilterDescription)] string? source = null, [Description("Minimum severity to include. Default 0 (all facts). Use 0.5 to see only significant facts.")] double min_severity = 0, [Description(McpHelpers.AsOfDescription)] string? as_of = null) { @@ -210,6 +218,11 @@ public static async Task GetAnalysisFacts( var validation = McpHelpers.ValidateWindow(hours_back, as_of, out var windowEnd); if (validation != null) return validation; + /* #3541 A13: an unknown source is refused with the whole accepted set, never applied as a filter + that matches nothing. The set is the scorer's registry, not a copy of it. */ + validation = McpHelpers.ValidateChoice(source, FactScorer.KnownSources, "source"); + if (validation != null) return validation; + /* Null for an absent anchor — see analyze_server's note. Nothing here persists, so the distinction costs nothing; it is kept so AnalysisContext.AsOfUtc means one thing everywhere. */ var anchor = string.IsNullOrWhiteSpace(as_of) ? (DateTime?)null : windowEnd; diff --git a/Lite/Mcp/McpHealthTools.cs b/Lite/Mcp/McpHealthTools.cs index a94cb14d3..ad8aa679c 100644 --- a/Lite/Mcp/McpHealthTools.cs +++ b/Lite/Mcp/McpHealthTools.cs @@ -63,27 +63,42 @@ collection of ANY collector — a live collection log beside a dead CPU collecto } } - [McpServerTool(Name = "get_daily_summary"), Description("Gets a daily health summary: overall composite health band (Healthy/Warning/Critical), total wait time, top wait type, unique query count, deadlocks, blocking events, memory pressure (and severe memory pressure), high-CPU samples, collection errors, and actionable alert count for one day. Use this for a quick overview to decide which areas need investigation.")] + [McpServerTool(Name = "get_daily_summary"), Description("Gets a daily health summary: overall composite health band (Healthy/Warning/Critical), total wait time, top wait type, unique query count, deadlocks, blocking events, memory pressure (and severe memory pressure), high-CPU samples, collection errors, and actionable alert count for one day. Use this for a quick overview to decide which areas need investigation. A day before the store's retention_horizon (the oldest day the shortest-lived signal table still holds) returns status=unavailable with data_state=purged rather than a health band: its per-signal counts would be COALESCEd zeros, not measurements, and a zero is only a measurement inside retention. A returned day carries data_state=collected (a verdict), past_horizon (before the horizon but some signal table still holds rows — the purge has not reached it; No Data, non-zero counts real) or no_run_record (inside retention, no collector run recorded — banded on the counts as read, which are measurements there; the collection-error share has no denominator).")] public static async Task GetDailySummary( LocalDataService dataService, ServerManager serverManager, [Description("Server name or display name.")] string? server_name = null, - [Description("Summary date (yyyy-MM-dd), interpreted as a UTC day. Default is today.")] string? summary_date = null) + [Description("Summary date, ISO-8601 yyyy-MM-dd ONLY (e.g. 2026-07-09), interpreted as a UTC day; any other spelling is refused rather than guessed at. Default is today.")] string? summary_date = null) { var (resolved, error) = ServerResolver.ResolveOrError(serverManager, server_name); if (error != null) return error; - DateTime? date = null; - if (!string.IsNullOrEmpty(summary_date)) - { - if (!DateTime.TryParse(summary_date, System.Globalization.CultureInfo.InvariantCulture, System.Globalization.DateTimeStyles.None, out var parsed)) - return $"Invalid date format '{summary_date}'. Use yyyy-MM-dd format (e.g., 2026-07-09)."; - date = parsed; - } + /* #3541 A9: exact ISO-8601, refused otherwise — McpHelpers.ParseSummaryDate says why the general + parse this replaced was the wrong tool; Darling's twin makes the same call. */ + var dateError = McpHelpers.ParseSummaryDate(summary_date, out var date); + if (dateError != null) return dateError; try { var row = await dataService.GetDailySummaryAsync(resolved.ServerId, date); + + /* #3541 A9: a day before the retention horizon is "unavailable" in the miss vocabulary's own + sense — it existed and is not retrievable now — told apart from a never-collected day because + the two send a caller to different places. Darling's twin says the same words. */ + if (row is { DataState: DailySummaryDataState.Purged }) + return McpHelpers.Status( + "unavailable", + $"{row.SummaryDate:yyyy-MM-dd} is before {resolved.ServerName}'s retention_horizon ({row.RetentionHorizon:yyyy-MM-dd}): the per-signal tables the health band reads (deadlocks, blocking, CPU, memory, waits) have been purged for that day, so no health verdict is possible and the counts would be zeros by construction, not by measurement. Longer-lived sources may still record the day — collection_runs and alert_count below are real where non-zero.", + new + { + summary_date = row.SummaryDate.ToString("yyyy-MM-dd"), + overall_health = row.OverallHealth, + data_state = DailySummaryRetention.Label(row.DataState), + retention_horizon = row.RetentionHorizon?.ToString("yyyy-MM-dd"), + collection_runs = row.CollectionRuns, + alert_count = row.AlertCount, + }); + if (row == null || !row.HasData) { var missDate = row?.SummaryDate ?? date ?? DateTime.UtcNow.Date; @@ -99,6 +114,11 @@ public static async Task GetDailySummary( summary_date = row.SummaryDate.ToString("yyyy-MM-dd"), overall_health = row.OverallHealth, health_band = row.HealthBand.ToString(), + /* #3541 A9: collected, past_horizon or no_run_record here (purged returned above); the note + says what the zeros are on a non-collected day, null on a collected one. */ + data_state = DailySummaryRetention.Label(row.DataState), + data_note = row.RetentionHorizon is { } horizon ? DailySummaryRetention.Note(row.DataState, horizon, row.SignalSourcesPresent) : null, + retention_horizon = row.RetentionHorizon?.ToString("yyyy-MM-dd"), total_wait_time_sec = row.TotalWaitTimeSec, top_wait_type = row.TopWaitType, unique_queries = row.UniqueQueries, @@ -134,7 +154,7 @@ public static async Task GetDailySummary( /// branch on a parameter it may not have sent. They share the ONE aggregate underneath, which is what /// stops them ever disagreeing about a day. /// - [McpServerTool(Name = "get_daily_summary_range"), Description("Gets the daily health summary for a SPAN of days rather than one: one row per collected day, each with its composite health band (Healthy/Warning/Critical), total wait time, top wait type, unique query count, deadlocks, blocking events with the peak block wait, high-CPU samples, memory pressure, collection errors and actionable alert count. This is what the desktop viewer's Performance Calendar month grid draws, and it is the read to use when the question is WHICH day rather than how one day went — scan the bands, then call get_daily_summary for the day that stands out. A day on which anything at all was collected appears here even if every signal was quiet (that day is Healthy, not missing), so a gap in the returned days is a gap in COLLECTION.")] + [McpServerTool(Name = "get_daily_summary_range"), Description("Gets the daily health summary for a SPAN of days rather than one: one row per collected day, each with its composite health band (Healthy/Warning/Critical), total wait time, top wait type, unique query count, deadlocks, blocking events with the peak block wait, high-CPU samples, memory pressure, collection errors and actionable alert count. This is what the desktop viewer's Performance Calendar month grid draws, and it is the read to use when the question is WHICH day rather than how one day went — scan the bands, then call get_daily_summary for the day that stands out. A day on which anything at all was collected appears here even if every signal was quiet (that day is Healthy, not missing), so a gap in the returned days is a gap in COLLECTION — INSIDE RETENTION. The per-signal tables age out at the store's shortest retention while the collection log and alert log live longer, so retention_horizon is the oldest day every signal can still answer for; a returned day before it carries data_state=purged (no signal table holds it) or past_horizon (some still do — the purge has not reached it), health_band=NoData and a data_note, NEVER Healthy — a purged day's zeros are absences, and days_before_horizon counts both kinds. A day inside retention with no collector run recorded is data_state=no_run_record — it keeps its band (an alert-only day is Warning), with the caveat that the error share has no denominator. Purged and past_horizon rows carry no verdict.")] public static async Task GetDailySummaryRange( LocalDataService dataService, ServerManager serverManager, @@ -200,14 +220,24 @@ cannot report a healthy server as uncollected. Darling's twin uses the same word otherwise tell which days they were given from the days they got. */ from_date = fromDate.ToString("yyyy-MM-dd"), to_date = lastDay.ToString("yyyy-MM-dd"), - /* Days WITH data, not days in the span. The two differ exactly where collection has a hole, - and that difference is the most useful thing on this payload. */ + /* Days the spine holds, not days in the span. The two differ exactly where collection has a + hole, and that difference is the most useful thing on this payload — read it together with + days_before_horizon, because a held day before the horizon is a shell, not a collected day. */ day_count = rows.Count, + /* #3541 A9: the store's horizon (reader clock, Lite's one archive retention — see the reader) + and how many returned days fall before it. Every row carries the same horizon, so the + first row's is the range's. */ + retention_horizon = rows[0].RetentionHorizon?.ToString("yyyy-MM-dd"), + days_before_horizon = rows.Count(row => row.DataState is DailySummaryDataState.Purged or DailySummaryDataState.PastHorizon), + purged_day_count = rows.Count(row => row.DataState == DailySummaryDataState.Purged), + collected_day_count = rows.Count(row => row.DataState == DailySummaryDataState.Collected), days = rows.Select(row => new { summary_date = row.SummaryDate.ToString("yyyy-MM-dd"), overall_health = row.OverallHealth, health_band = row.HealthBand.ToString(), + data_state = DailySummaryRetention.Label(row.DataState), + data_note = row.RetentionHorizon is { } rowHorizon ? DailySummaryRetention.Note(row.DataState, rowHorizon, row.SignalSourcesPresent) : null, total_wait_time_sec = row.TotalWaitTimeSec, top_wait_type = row.TopWaitType, unique_queries = row.UniqueQueries, @@ -574,7 +604,7 @@ public static async Task GetCollectionLog( LocalDataService dataService, ServerManager serverManager, [Description("Server name or display name.")] string? server_name = null, - [Description("Hours of history. Default 24.")] int hours_back = 24, + [Description("Hours of history. Default 24. No upper bound (this read exists to look further back than the 168-hour reads allow); a negative or zero value is refused rather than read as its absolute value.")] int hours_back = 24, [Description("Maximum rows to return. Default 200. Applied AFTER the two filters, so it caps the matching rows rather than the window.")] int limit = 200, [Description(McpHelpers.AsOfDescription)] string? as_of = null, /* @@ -600,17 +630,18 @@ to say the filter was ignored. Darling's twin refuses the same value with the sa var invalidFloor = McpHelpers.ValidateMinMs(min_duration_ms, "min_duration_ms"); if (invalidFloor != null) return invalidFloor; - /* ResolveAsOf here, deliberately NOT ValidateWindow. These three reads have never capped - hours_back -- they Math.Abs() it and window on the result -- so routing them through the - shared validator would impose the 168-hour ceiling every other read carries, and take reach - away from exactly the read whose premise is looking FURTHER back than the default. The anchor - is validated because it is new; the span keeps the behaviour callers already have. */ - var anchorError = McpHelpers.ResolveAsOf(as_of, out var windowEnd); + /* ValidateUncappedWindow, deliberately NOT ValidateWindow. These three reads have never capped + hours_back, so routing them through the shared validator would impose the 168-hour ceiling every + other read carries and take reach away from exactly the read whose premise is looking FURTHER back + than the default. What they no longer do is Math.Abs() a negative span (#3541 A13): a window that + ends before it starts is a caller error, and flipping the sign answered a different question with + nothing to say so. Refused, with Darling's twin's words. */ + var anchorError = McpHelpers.ValidateUncappedWindow(hours_back, as_of, out var windowEnd); if (anchorError != null) return anchorError; try { - var hours = Math.Abs(hours_back); + var hours = hours_back; /* Over-fetch by one so truncation is observed, not inferred -- see Darling's twin. The filters go INTO the read for the same reason they do there: the over-fetch is then of the FILTERED set, so @@ -739,24 +770,25 @@ public static async Task GetCurrentWaitsTrend( LocalDataService dataService, ServerManager serverManager, [Description("Server name or display name.")] string? server_name = null, - [Description("Hours of history. Default 4.")] int hours_back = 4, + [Description("Hours of history. Default 4. No upper bound (this read exists to look further back than the 168-hour reads allow); a negative or zero value is refused rather than read as its absolute value.")] int hours_back = 4, [Description("Limit the blocked-session series to one database. Omit for all databases.")] string? database_name = null, [Description(McpHelpers.AsOfDescription)] string? as_of = null) { var (resolved, error) = ServerResolver.ResolveOrError(serverManager, server_name); if (error != null) return error; - /* ResolveAsOf here, deliberately NOT ValidateWindow. These three reads have never capped - hours_back -- they Math.Abs() it and window on the result -- so routing them through the - shared validator would impose the 168-hour ceiling every other read carries, and take reach - away from exactly the read whose premise is looking FURTHER back than the default. The anchor - is validated because it is new; the span keeps the behaviour callers already have. */ - var anchorError = McpHelpers.ResolveAsOf(as_of, out var windowEnd); + /* ValidateUncappedWindow, deliberately NOT ValidateWindow. These three reads have never capped + hours_back, so routing them through the shared validator would impose the 168-hour ceiling every + other read carries and take reach away from exactly the read whose premise is looking FURTHER back + than the default. What they no longer do is Math.Abs() a negative span (#3541 A13): a window that + ends before it starts is a caller error, and flipping the sign answered a different question with + nothing to say so. Refused, with Darling's twin's words. */ + var anchorError = McpHelpers.ValidateUncappedWindow(hours_back, as_of, out var windowEnd); if (anchorError != null) return anchorError; try { - var hours = Math.Abs(hours_back); + var hours = hours_back; var filter = string.IsNullOrWhiteSpace(database_name) ? null : new[] { database_name }; var waits = await dataService.GetWaitingTaskTrendAsync(resolved.ServerId, hours, asOfUtc: windowEnd); @@ -814,23 +846,24 @@ public static async Task GetBlockingStats( LocalDataService dataService, ServerManager serverManager, [Description("Server name or display name.")] string? server_name = null, - [Description("Hours of history. Default 24.")] int hours_back = 24, + [Description("Hours of history. Default 24. No upper bound (this read exists to look further back than the 168-hour reads allow); a negative or zero value is refused rather than read as its absolute value.")] int hours_back = 24, [Description(McpHelpers.AsOfDescription)] string? as_of = null) { var (resolved, error) = ServerResolver.ResolveOrError(serverManager, server_name); if (error != null) return error; - /* ResolveAsOf here, deliberately NOT ValidateWindow. These three reads have never capped - hours_back -- they Math.Abs() it and window on the result -- so routing them through the - shared validator would impose the 168-hour ceiling every other read carries, and take reach - away from exactly the read whose premise is looking FURTHER back than the default. The anchor - is validated because it is new; the span keeps the behaviour callers already have. */ - var anchorError = McpHelpers.ResolveAsOf(as_of, out var windowEnd); + /* ValidateUncappedWindow, deliberately NOT ValidateWindow. These three reads have never capped + hours_back, so routing them through the shared validator would impose the 168-hour ceiling every + other read carries and take reach away from exactly the read whose premise is looking FURTHER back + than the default. What they no longer do is Math.Abs() a negative span (#3541 A13): a window that + ends before it starts is a caller error, and flipping the sign answered a different question with + nothing to say so. Refused, with Darling's twin's words. */ + var anchorError = McpHelpers.ValidateUncappedWindow(hours_back, as_of, out var windowEnd); if (anchorError != null) return anchorError; try { - var hours = Math.Abs(hours_back); + var hours = hours_back; var blocking = await dataService.GetBlockingDurationStatsAsync(resolved.ServerId, hours, asOfUtc: windowEnd); var deadlocks = await dataService.GetDeadlockSeverityStatsAsync(resolved.ServerId, hours, asOfUtc: windowEnd); diff --git a/Lite/Mcp/McpInstructions.cs b/Lite/Mcp/McpInstructions.cs index b957c8001..336158085 100644 --- a/Lite/Mcp/McpInstructions.cs +++ b/Lite/Mcp/McpInstructions.cs @@ -60,8 +60,8 @@ You are connected to a SQL Server performance monitoring tool via Performance Mo | `get_current_waits_trend` | The two Current Waits series over time: waiting-task total wait per wait type per collection, and blocked-session counts per database per collection. `get_waiting_tasks` gives the snapshot and can never say whether now is worse than an hour ago; this is that question. Read the two series together — a wait-type spike with no blocked sessions is a resource wait, the same spike with them is contention. An empty result distinguishes a genuine all-clear (`empty`) from a server the collector has never sampled (`unavailable`), which is NOT an all-clear | `server_name`, `hours_back`, `database_name`, `as_of` | | `get_blocking_stats` | Blocking SEVERITY per minute: blocking duration (event count, total, max, avg wait) and deadlock severity (victim count plus total/max/avg wait across EVERY process in the graphs, not just victims). `get_blocking_trend` and `get_deadlock_trend` say how OFTEN; this says how BAD — ten one-second blocks and one ten-minute block are the same count and a different problem. An empty result distinguishes a genuinely clear window (`empty`) from a server where neither capture path has ever produced a row (`unavailable`), which is NOT a clean bill of health | `server_name`, `hours_back`, `as_of` | | `get_server_summary` | Quick health overview: CPU %, memory, blocking/deadlock counts; three clocks named (`cpu_captured_at`, `memory_captured_at`, `last_collection` = newest collection of ANY collector) | `server_name` | - | `get_daily_summary` | Daily composite health band + wait/query/deadlock/blocking/CPU/memory/alert rollup for one day | `server_name`, `summary_date` (yyyy-MM-dd, default today) | - | `get_daily_summary_range` | The SAME rollup across a span of days — one row per collected day, the Performance Calendar's month grid. Use it when the question is WHICH day rather than how one day went: scan the bands, then call `get_daily_summary` for the day that stands out. A day with ANY collection appears even when every signal was quiet, so a day absent from the result is a gap in COLLECTION. `as_of` anchors the LAST day of the range. An empty result distinguishes a range outside this server's history (`empty`) from a server nothing has ever been collected for (`unavailable`) | `server_name`, `days_back` (default 30, max 366), `as_of` | + | `get_daily_summary` | Daily composite health band + wait/query/deadlock/blocking/CPU/memory/alert rollup for one day. A day before the store's `retention_horizon` is `unavailable` with `data_state: purged` — never a band, because its zeros are absences | `server_name`, `summary_date` (yyyy-MM-dd ONLY, refused otherwise; default today) | + | `get_daily_summary_range` | The SAME rollup across a span of days — one row per collected day, the Performance Calendar's month grid. Use it when the question is WHICH day rather than how one day went: scan the bands, then call `get_daily_summary` for the day that stands out. A day with ANY collection appears even when every signal was quiet, so a day absent from the result is a gap in COLLECTION — inside retention. The payload publishes `retention_horizon` (the oldest day every signal table still holds) and `days_before_horizon`; a returned day before the horizon is `data_state: purged` (no signal table holds it) or `past_horizon` (some still do) and `NoData`, NEVER Healthy, because a purged day's per-signal zeros are absences left by the purge. Purged and past-horizon rows carry no verdict; `no_run_record` (inside retention, no collector run recorded) keeps its band with a caveat. `as_of` anchors the LAST day of the range. An empty result distinguishes a range outside this server's history (`empty`) from a server nothing has ever been collected for (`unavailable`) | `server_name`, `days_back` (default 30, max 366), `as_of` | ### Wait Statistics Tools | Tool | Purpose | Key Parameters | @@ -194,7 +194,7 @@ You are connected to a SQL Server performance monitoring tool via Performance Mo ### Session & Active Query Tools | Tool | Purpose | Key Parameters | |------|---------|----------------| - | `get_active_queries` | Active query snapshots from sys.dm_exec_requests — what was running at each collection point | `server_name`, `hours_back` (default 1), `database_name`, `blocking_only`, `limit`, `as_of` | + | `get_active_queries` | Active query snapshots from sys.dm_exec_requests — what was running at each collection point. `database_name` / `blocking_only` are applied IN the query; `total_snapshots` counts the filtered population, `snapshots_returned` the page, `truncated` says the population held more. Head blockers a victim names are never stripped (`is_head_blocker`); a victim whose blocker is absent says why in `blocker_not_shown` (`not_captured` / `filtered` / `past_page`) | `server_name`, `hours_back` (default 1), `database_name`, `blocking_only`, `limit`, `as_of` | | `get_session_stats` | Connection counts and resource usage grouped by application | `server_name` | ### Execution Plan Analysis Tools @@ -220,7 +220,7 @@ You are connected to a SQL Server performance monitoring tool via Performance Mo | Tool | Purpose | Key Parameters | |------|---------|----------------| | `analyze_server` | Runs the inference engine: scores facts, traverses relationship graph, returns evidence-backed findings with severity and recommended next tools. Each finding's `confidence` is an EVIDENCE score (0.20 for the symptom alone, more as the root fact's amplifier checks match and the chain deepens — a lone uncorroborated symptom is 0.20, never 1.0) and `confidence_basis` says in words what it rests on; rank by `severity` for impact and read `confidence` as how much of the engine's own corroboration showed up. A remediable finding also carries `remediation_command` — the full copy-paste T-SQL remediation (identical to the viewer card), with a two-sided risk-disclosure header on destructive changes; advisory only, never executed. Force-plan findings additionally carry `structured_remediation`: the verdict as machine-readable fields (eligible + named blockers), evidence, and split force/unforce/verify SQL. With `as_of` it analyzes a PAST window — anomaly baseline included — and is EXPLORATORY: the findings come back in full but are NOT persisted, which `persisted` / `persistence_note` state on every result | `server_name`, `hours_back` (default 4), `as_of` | - | `get_analysis_facts` | Exposes raw scored facts from the collect+score pipeline — every observation the engine sees with base severity, amplifiers, and metadata (an ANOMALY_* fact's `baseline_confidence` is the baseline's trustworthiness, not a finding's `confidence`) | `server_name`, `hours_back` (default 4), `source` (filter), `min_severity`, `as_of` | + | `get_analysis_facts` | Exposes raw scored facts from the collect+score pipeline — every observation the engine sees with base severity, amplifiers, and metadata (an ANOMALY_* fact's `baseline_confidence` is the baseline's trustworthiness, not a finding's `confidence`) | `server_name`, `hours_back` (default 4), `source` (one of the engine's 15 registered sources — anomaly, bad_actor, blocking, config, coverage, cpu, database_config, disk, io, jobs, memory, queries, sessions, tempdb, waits; an unknown value is refused with the set), `min_severity`, `as_of` | | `compare_analysis` | Compares two time periods (e.g., peak vs off-peak, yesterday vs today, the windows around a change), banding each fact worse / better / stable by how far its VALUE moved on the server's own scale — in the stored per-server baseline's robust sigma where one exists (`delta_sigma`, `band_source` `baseline`), otherwise only when the value moved at least a quarter of the larger side AND registers a quarter of the way up its own severity ladder (`band_source` `absolute`); `band_rules` states the rules on every payload. Rows are grouped into physical-cause `families` (one I/O stall is one family row), and `BAD_ACTOR_` appearances are `plan_cache_churn`, not new or resolved issues. A verdict is a DIFFERENCE, not an experiment: same-hour-yesterday at N=1 vs N=1 cannot show that a change caused anything. When NEITHER window produced facts the result is `unavailable` rather than an all-zero comparison, because "nothing to compare" is not "nothing changed"; when only ONE window is empty the payload carries a `caveat` saying so; a partly collected side flags every verdict row with `coverage_caveat`. `baseline_hours_back` is measured from the comparison window's END, so `as_of` moves BOTH windows together | `server_name`, `hours_back` (default 4), `baseline_hours_back` (default 28), `as_of` | | `audit_config` | Edition-aware configuration audit: evaluates CTFP, MAXDOP, max memory, and max worker threads against best practices | `server_name` | | `get_analysis_findings` | Retrieves persisted findings from previous analysis runs (each with `confidence_basis`: rows persisted before `confidence` measured corroboration are labelled `path-shape (pre-#3538)` — under that formula a lone symptom read 1.0, so do not read those as corroborated), deduplicated to one entry per diagnostic chain (`story_path_hash` + `incident_id`): the latest occurrence plus `occurrences`/`first_seen`/`last_seen`/`peak_severity` spanning the window; each remediable finding carries `remediation_command` — the full copy-paste T-SQL remediation (identical to the viewer card), rendered from the persisted action, advisory only and never executed; force-plan findings additionally carry `structured_remediation` (verdict + evidence + split artifacts, machine-readable). Its window is on ANALYSIS TIME, so `as_of` asks what analysis was saying then rather than re-analyzing that window now | `server_name`, `hours_back` (default 24), `as_of` | @@ -317,7 +317,7 @@ You are connected to a SQL Server performance monitoring tool via Performance Mo - `get_waiting_tasks` shows what's actively waiting, complementing the aggregated view from `get_wait_stats`. - `get_wait_types` helps you discover available wait types before drilling into `get_wait_trend`. - Trend tools (`get_wait_trend`, `get_file_io_trend`, `get_memory_trend`, `get_blocking_trend`, `get_deadlock_trend`, `get_query_duration_trend`) confirm whether a problem is new, worsening, or steady-state. - - Query tools support `database_name` filtering and `parallel_only`/`min_dop` filtering to narrow results. + - Query tools support `database_name` filtering and `parallel_only`/`min_dop` filtering to narrow results. Every filter is applied IN the query before the ranking and the cap (`filter_applied` names the parallelism floor in force), so the page is the top-N of the filtered population and an empty filtered page is the window's answer. ## Important Limitations diff --git a/Lite/Mcp/McpQueryTools.cs b/Lite/Mcp/McpQueryTools.cs index a07d3a5d8..65a56bb4d 100644 --- a/Lite/Mcp/McpQueryTools.cs +++ b/Lite/Mcp/McpQueryTools.cs @@ -9,7 +9,7 @@ namespace PerformanceMonitorLite.Mcp; [McpServerToolType] public sealed class McpQueryTools { - [McpServerTool(Name = "get_top_queries_by_cpu"), Description("Gets expensive queries from sys.dm_exec_query_stats (plan cache). Best for: currently cached queries with detailed per-execution stats, DOP, spills, and query_hash for trending. Returns query_hash, query_plan_hash, sql_handle, plan_handle, and host_object (the hosting procedure/function for proc-hosted statements, null for ad-hoc) — groups key on (database, query_hash, host_object), so INSERT...EXEC callers in different procedures report separately with their own text. distinct_texts counts statement texts merged into a group (>1 = ad-hoc literal variants or pre-upgrade history; query_text is one representative, 0 means no stored text for the group). Supports database and parallelism filtering. min/max_cpu_ms and min/max_elapsed_ms are LIFETIME extremes for the plan's time in cache (same semantics as max_dop), not windowed — totals and avgs are windowed deltas; rows where an extreme provably predates the window carry extremes_note. Also returns cpu_attribution: the returned rows' summed CPU-seconds against the SQL process's measured CPU-seconds for the window (avg cpu_utilization % x core count x window) - attributed_cpu_ratio says how much of the box the ranking explains; when the CPU series or core count is missing, or covers too little of the window, the ratio is omitted rather than invented.")] + [McpServerTool(Name = "get_top_queries_by_cpu"), Description("Gets expensive queries from sys.dm_exec_query_stats (plan cache). Best for: currently cached queries with detailed per-execution stats, DOP, spills, and query_hash for trending. Returns query_hash, query_plan_hash, sql_handle, plan_handle, and host_object (the hosting procedure/function for proc-hosted statements, null for ad-hoc) — groups key on (database, query_hash, host_object), so INSERT...EXEC callers in different procedures report separately with their own text. distinct_texts counts statement texts merged into a group (>1 = ad-hoc literal variants or pre-upgrade history; query_text is one representative, 0 means no stored text for the group). Supports database and parallelism filtering; every filter is applied IN the query before the ranking and the cap, so the page is the top-N of the FILTERED population (filter_applied names the parallelism floor in force, null when none), and an empty page under parallel_only/min_dop is the window's answer rather than a page artefact. min/max_cpu_ms and min/max_elapsed_ms are LIFETIME extremes for the plan's time in cache (same semantics as max_dop), not windowed — totals and avgs are windowed deltas; rows where an extreme provably predates the window carry extremes_note. Also returns cpu_attribution: the returned rows' summed CPU-seconds against the SQL process's measured CPU-seconds for the window (avg cpu_utilization % x core count x window) - attributed_cpu_ratio says how much of the box the ranking explains; when the CPU series or core count is missing, or covers too little of the window, the ratio is omitted rather than invented.")] public static async Task GetTopQueriesByCpu( LocalDataService dataService, ServerManager serverManager, @@ -38,17 +38,34 @@ shrinks the numerator/denominator window skew from the ranking query's full dura signature is the only way to zero it, and sub-microsecond against an hours window does not buy that churn). */ var nowUtc = windowEnd; - var rows = await dataService.GetTopQueriesByCpuAsync(resolved.ServerId, hours_back, top, databaseNames: string.IsNullOrEmpty(database_name) ? null : new[] { database_name }, asOfUtc: windowEnd); + + /* #3541 A13: the parallelism filter goes INTO the read as a lifetime max_dop floor on the grouped + population, before the CPU ranking and the cap (Darling's twin: TopQueriesSql's HAVING note). It + used to be a .Where over the returned top-N page, so parallel_only=true on a box whose hottest + plans were serial came back EMPTY while the window held parallel plans. 2 for parallel_only, + min_dop when set above that (min_dop implies parallel filtering, as its description says), 0 + (admit all) otherwise. */ + var minMaxDop = min_dop > 1 ? min_dop : parallel_only ? 2 : 0; + var filterApplied = minMaxDop > 0 + ? $"lifetime max_dop >= {minMaxDop} (applied in SQL before the top-{top} ranking; the page is the top-{top} of the parallel population)" + : null; + + var rows = await dataService.GetTopQueriesByCpuAsync(resolved.ServerId, hours_back, top, databaseNames: string.IsNullOrEmpty(database_name) ? null : new[] { database_name }, asOfUtc: windowEnd, minMaxDop: minMaxDop); if (rows.Count == 0) { + /* A filtered miss is not a collection miss — same words as Darling's twin. */ + if (minMaxDop > 0) + { + return McpHelpers.Status( + "empty", + $"No query-stats group on {resolved.ServerName} in the last {hours_back} hour(s) has a cached plan with lifetime max_dop >= {minMaxDop}. The filter was applied in SQL over the whole window, so this is the window's answer rather than a page artefact — drop parallel_only / min_dop to see the unfiltered ranking, or confirm current parallelism with analyze_query_plan.", + new { filter_applied = filterApplied }); + } + return await McpEngineCapability.NotCollectedStatusAsync(dataService, resolved.ServerId, resolved.ServerName, "query_stats") ?? McpHelpers.Status("unavailable", "No query stats available for the specified time range."); } - var filtered = rows - .Where(r => !(parallel_only || min_dop > 1) || (r.MaxDop > 1 && r.MaxDop >= (min_dop > 1 ? min_dop : 2))) - .ToList(); - /* #2320: what fraction of the box's measured CPU the RETURNED rows explain — numerator is the caller-visible ranking (post top-N, post filters), denominator is measured, and the ratio is omitted rather than invented when a denominator piece is missing. One nowUtc @@ -60,12 +77,12 @@ ratio is omitted rather than invented when a denominator piece is missing. One n var cpuAggregate = await cpuAggregateTask; var properties = await propertiesTask; var attribution = CpuAttribution.Compute( - filtered.Sum(r => r.TotalCpuMs) / 1000.0, + rows.Sum(r => r.TotalCpuMs) / 1000.0, nowUtc.AddHours(-hours_back), nowUtc, cpuAggregate.SampleCount, cpuAggregate.FirstSample, cpuAggregate.LastSample, cpuAggregate.AvgSqlCpuPercent, properties?.CpuCount ?? 0); - var result = filtered.Select(r => new + var result = rows.Select(r => new { database_name = r.DatabaseName, query_hash = r.QueryHash, @@ -110,6 +127,8 @@ ratio is omitted rather than invented when a denominator piece is missing. One n { server = resolved.ServerName, hours_back, + /* #3541 A13: the filter that shaped the population, stated on the payload; null when none. */ + filter_applied = filterApplied, cpu_attribution = new { ranked_cpu_seconds = attribution.RankedCpuSeconds, diff --git a/Lite/Mcp/McpSessionTools.cs b/Lite/Mcp/McpSessionTools.cs index 0557c2600..ca34101a8 100644 --- a/Lite/Mcp/McpSessionTools.cs +++ b/Lite/Mcp/McpSessionTools.cs @@ -9,15 +9,15 @@ namespace PerformanceMonitorLite.Mcp; [McpServerToolType] public sealed class McpSessionTools { - [McpServerTool(Name = "get_active_queries"), Description("Gets active query snapshots captured from sys.dm_exec_requests. Shows what queries were running at each collection point: session ID, query text, wait type, CPU time, elapsed time, blocking info, DOP, and memory grants. Use hours_back to look at a specific time window — critical for finding what was running during a CPU spike or blocking event.")] + [McpServerTool(Name = "get_active_queries"), Description("Gets active query snapshots captured from sys.dm_exec_requests. Shows what queries were running at each collection point: session ID, query text, wait type, CPU time, elapsed time, blocking info, DOP, and memory grants. Use hours_back to look at a specific time window — critical for finding what was running during a CPU spike or blocking event. EVERY FILTER IS PART OF THE QUERY: database_name and blocking_only are applied in SQL before the page is cut, total_snapshots is the count of snapshot rows in the window that pass your filters, snapshots_returned is how many you got, and truncated says the filtered population held more than limit — raise limit or narrow hours_back when it is true (NEWEST CAPTURE FIRST, highest CPU first within a capture; oldest_returned_collection_time / newest_returned_collection_time bound the page). HEAD BLOCKERS ARE NEVER STRIPPED: a session another row in the same capture names as its blocker is on the page whatever its text (including a WAITFOR shell holding locks), flagged is_head_blocker. A victim whose blocker is NOT on the page says why in blocker_not_shown: not_captured (the blocker held no running request at that capture — an idle open transaction is the classic case; get_blocked_process_reports has its input buffer from the blocked-process report), filtered (your database_name filter excluded it), or past_page (it is in the filtered population but beyond limit).")] public static async Task GetActiveQueries( LocalDataService dataService, ServerManager serverManager, [Description("Server name or display name.")] string? server_name = null, [Description("Hours of data to retrieve. Default 1.")] int hours_back = 1, - [Description("Filter to a specific database.")] string? database_name = null, - [Description("Show only queries involved in blocking (blocking_session_id > 0 or is a head blocker).")] bool blocking_only = false, - [Description("Maximum number of rows to return. Default 50.")] int limit = 50, + [Description("Filter to a specific database. Applied in SQL; a head blocker in ANOTHER database is then not on the page, and its victims say blocker_not_shown = filtered.")] string? database_name = null, + [Description("Show only queries involved in blocking: rows with blocking_session_id > 0, plus the head blockers those rows name in the same capture. Applied in SQL, so total_snapshots counts the blocking population and truncated is measured against it.")] bool blocking_only = false, + [Description("Maximum number of rows to return. Default 50. The page is bounded by limit, not by hours_back — truncated says whether the filtered window held more.")] int limit = 50, [Description(McpHelpers.AsOfDescription)] string? as_of = null) { var (resolved, error) = ServerResolver.ResolveOrError(serverManager, server_name); @@ -25,24 +25,42 @@ public static async Task GetActiveQueries( var validation = McpHelpers.ValidateWindow(hours_back, as_of, out var windowEnd); if (validation != null) return validation; + validation = McpHelpers.ValidateTop(limit); + if (validation != null) return validation; try { - var rows = await dataService.GetLatestQuerySnapshotsAsync(resolved.ServerId, hours_back, asOfUtc: windowEnd); + var filter = string.IsNullOrWhiteSpace(database_name) ? null : database_name.Trim(); + + /* #3541 A13: the filters ride INTO the read (GetActiveQueriesPageAsync), asked for one row past + the cap so truncation is OBSERVED on the filtered population; total_snapshots is that read's + COUNT(*) OVER () of the same population. Darling's twin is DarlingSessionReader.ActiveQueriesSql. */ + var (rows, populationCount) = await dataService.GetActiveQueriesPageAsync( + resolved.ServerId, hours_back, limit + 1, filter, blocking_only, asOfUtc: windowEnd); + if (rows.Count == 0) + { + /* A filtered miss is not a collection miss — Darling's twin's words. */ + if (filter != null || blocking_only) + { + return McpHelpers.Status( + "empty", + $"No active query snapshots on {resolved.ServerName} in the last {hours_back} hour(s) matched " + + DescribeActiveQueryFilters(filter, blocking_only) + + ". The filters were applied in SQL over the whole window, so unfiltered snapshots may well exist — drop them to see what the window holds."); + } + return await McpEngineCapability.NotCollectedStatusAsync(dataService, resolved.ServerId, resolved.ServerName, "query_snapshots") ?? McpHelpers.Status("empty", "No active query snapshots found in the requested time range."); + } - IEnumerable filtered = rows; + var truncated = rows.Count > limit; + var shown = truncated ? rows.GetRange(0, limit) : rows; - if (!string.IsNullOrEmpty(database_name)) - filtered = filtered.Where(r => r.DatabaseName.Equals(database_name, StringComparison.OrdinalIgnoreCase)); + /* The page's own (capture, session) pairs, so a victim can say whether its blocker made the page. */ + var onPage = new HashSet<(DateTime, int)>(shown.Select(r => (r.CollectionTime, r.SessionId))); - if (blocking_only) - filtered = filtered.Where(r => r.BlockingSessionId > 0 - || rows.Any(other => other.BlockingSessionId == r.SessionId)); - - var result = filtered.Take(limit).Select(r => new + var result = shown.Select(r => new { collection_time = r.CollectionTime.ToString("o"), session_id = r.SessionId, @@ -57,6 +75,9 @@ public static async Task GetActiveQueries( wait_type = string.IsNullOrEmpty(r.WaitType) ? null : r.WaitType, wait_time_ms = r.WaitTimeMs > 0 ? r.WaitTimeMs : (long?)null, blocking_session_id = r.BlockingSessionId > 0 ? r.BlockingSessionId : (int?)null, + /* #3541 A13: the two blocking disclosures — see Darling's twin. */ + is_head_blocker = r.IsHeadBlocker ? true : (bool?)null, + blocker_not_shown = BlockerNotShown(r, onPage), dop = r.Dop > 0 ? r.Dop : (int?)null, parallel_worker_count = r.ParallelWorkerCount > 0 ? r.ParallelWorkerCount : (int?)null, granted_query_memory_gb = r.GrantedQueryMemoryGb > 0 ? r.GrantedQueryMemoryGb : (double?)null, @@ -72,8 +93,18 @@ public static async Task GetActiveQueries( { server = resolved.ServerName, hours_back, - total_snapshots = rows.Count, - shown = result.Count, + filters_applied = new + { + database_name = filter, + blocking_only, + }, + /* The FILTERED population's size, computed in SQL on the same statement as the rows. */ + total_snapshots = populationCount, + snapshots_returned = result.Count, + truncated, + order = "collection_time_desc", + oldest_returned_collection_time = shown[^1].CollectionTime.ToString("o"), + newest_returned_collection_time = shown[0].CollectionTime.ToString("o"), queries = result }, McpHelpers.JsonOptions); } @@ -83,6 +114,30 @@ public static async Task GetActiveQueries( } } + /// The one reason a victim's head blocker is not on the page, or null — Darling's twin's ladder: + /// never captured beats filtered beats past the page, because each later reason presupposes the earlier + /// one did not apply. + private static string? BlockerNotShown(QuerySnapshotRow row, HashSet<(DateTime, int)> onPage) + { + if (row.BlockingSessionId <= 0) + return null; + if (!row.BlockerInCapture) + return "not_captured"; + if (!row.BlockerInPopulation) + return "filtered"; + return onPage.Contains((row.CollectionTime, row.BlockingSessionId)) ? null : "past_page"; + } + + /// Names the active filters for the filtered-miss sentence, in the caller's own vocabulary. + private static string DescribeActiveQueryFilters(string? databaseName, bool blockingOnly) + { + if (databaseName != null && blockingOnly) + return $"database_name '{databaseName}' with blocking_only"; + if (databaseName != null) + return $"database_name '{databaseName}'"; + return "blocking_only"; + } + [McpServerTool(Name = "get_session_stats"), Description("Gets connection and session statistics grouped by application. Shows connection counts, running/sleeping/dormant breakdown, and aggregate resource usage per application.")] public static async Task GetSessionStats( LocalDataService dataService, diff --git a/Lite/Services/CollectionBackgroundService.cs b/Lite/Services/CollectionBackgroundService.cs index 75555628d..bf951ee00 100644 --- a/Lite/Services/CollectionBackgroundService.cs +++ b/Lite/Services/CollectionBackgroundService.cs @@ -370,7 +370,7 @@ private void RunRetentionIfDue() try { - _retentionService.CleanupOldArchives(retentionMonths: 3); + _retentionService.CleanupOldArchives(retentionMonths: RetentionService.ArchiveRetentionMonths); _lastRetentionTime = DateTime.UtcNow; } catch (Exception ex) diff --git a/Lite/Services/LocalDataService.Blocking.cs b/Lite/Services/LocalDataService.Blocking.cs index 90b567965..4fbf5e8fd 100644 --- a/Lite/Services/LocalDataService.Blocking.cs +++ b/Lite/Services/LocalDataService.Blocking.cs @@ -285,6 +285,189 @@ AND query_text NOT LIKE 'WAITFOR%' return items; } + /// + /// The get_active_queries MCP read (#3541 A13): the newest snapshot rows over the + /// window that pass the caller's filters, plus the FILTERED population's size from the same statement. + /// A sibling of rather than a change to it: that read is the + /// grids' whole-window snapshot (unfiltered, unbounded — the Active Queries grid wants every row and + /// filters in the UI), and the two questions are different enough that one signature serving both would + /// carry a page cap the grid must remember to disable. + /// + /// Every filter is part of the query. The tool used to read the whole window, filter + /// database_name and blocking_only in C#, take limit, and publish the PRE-filter row + /// count as total_snapshots beside the page — a total of a different population from the rows, and a + /// page that could be empty while the window held matches. Here the filters are predicates, the population + /// count is COUNT(*) OVER () on the filtered rows above the cap (Darling's #3613 idiom), and the cap + /// is the caller's, fetched at limit + 1 so truncation is observed rather than inferred. + /// + /// Head blockers are never stripped. The WAITFOR trim exists to drop idle monitoring shells, + /// but the classic head blocker IS a session sitting in WAITFOR with an open transaction, and it was + /// dropped while its victims' blocking_session_id pointed at a session no longer on the page. A row + /// is kept whatever its text when a row in the SAME capture names it as its blocker — same capture, not + /// same window, because session ids are reused and the old C# arm matched across the whole window. The two + /// flags tell a victim's story when its blocker is absent: blocker_in_capture false is the idle + /// open-transaction blocker sys.dm_exec_requests never lists; blocker_in_population false is a + /// blocker the caller's own database filter excluded. + /// + /// The predicates are composed as SQL text from two booleans (a database name is still bound), the + /// way, because DuckDB cannot infer a type for a bare $N IS NULL + /// parameter the way PostgreSQL's $N::text cast lets Darling's twin do it. + /// + public async Task<(List Rows, long PopulationCount)> GetActiveQueriesPageAsync( + int serverId, int hoursBack, int cap, string? databaseName = null, bool blockingOnly = false, DateTime? asOfUtc = null) + { + using var _q = TimeQuery("GetActiveQueriesPageAsync", "v_query_snapshots filtered page (MCP)"); + using var connection = await OpenConnectionAsync(); + using var command = connection.CreateCommand(); + + var (startTime, endTime) = GetTimeRange(hoursBack, null, null, asOfUtc, SelectedServerTabUtcOffsetMinutes); + var dbClause = BuildDbInClause( + string.IsNullOrWhiteSpace(databaseName) ? null : new[] { databaseName.Trim() }, "w.database_name", 5, out var dbValues); + var blockingClause = blockingOnly ? " AND (w.blocking_session_id > 0 OR h.session_id IS NOT NULL)" : ""; + + command.CommandText = @" +WITH window_rows AS ( + SELECT + session_id, + database_name, + elapsed_time_formatted, + query_text, + status, + blocking_session_id, + wait_type, + wait_time_ms, + wait_resource, + cpu_time_ms, + total_elapsed_time_ms, + reads, + writes, + logical_reads, + granted_query_memory_gb, + transaction_isolation_level, + dop, + parallel_worker_count, + collection_time, + login_name, + host_name, + program_name, + open_transaction_count, + percent_complete, + query_hash + FROM v_query_snapshots + WHERE server_id = $1 + AND collection_time >= $2 + AND collection_time <= $3 +), +heads AS ( + /* (capture, session) pairs some victim in the SAME capture points at. */ + SELECT DISTINCT collection_time, blocking_session_id AS session_id + FROM window_rows + WHERE blocking_session_id > 0 +), +population AS ( + SELECT + w.*, + (h.session_id IS NOT NULL) AS is_head_blocker, + EXISTS ( + SELECT 1 + FROM window_rows b + WHERE b.collection_time = w.collection_time + AND b.session_id = w.blocking_session_id + ) AS blocker_in_capture + FROM window_rows w + LEFT JOIN heads h + ON h.collection_time = w.collection_time + AND h.session_id = w.session_id + WHERE (w.query_text NOT LIKE 'WAITFOR%' OR h.session_id IS NOT NULL)" + dbClause + blockingClause + @" +) +SELECT + p.session_id, + p.database_name, + p.elapsed_time_formatted, + p.query_text, + p.status, + p.blocking_session_id, + p.wait_type, + p.wait_time_ms, + p.wait_resource, + p.cpu_time_ms, + p.total_elapsed_time_ms, + p.reads, + p.writes, + p.logical_reads, + p.granted_query_memory_gb, + p.transaction_isolation_level, + p.dop, + p.parallel_worker_count, + p.collection_time, + p.login_name, + p.host_name, + p.program_name, + p.open_transaction_count, + p.percent_complete, + p.query_hash, + p.is_head_blocker, + p.blocker_in_capture, + EXISTS ( + SELECT 1 + FROM population q + WHERE q.collection_time = p.collection_time + AND q.session_id = p.blocking_session_id + ) AS blocker_in_population, + COUNT(*) OVER () AS population_count +FROM population p +ORDER BY p.collection_time DESC, p.cpu_time_ms DESC +LIMIT $4"; + + command.Parameters.Add(new DuckDBParameter { Value = serverId }); + command.Parameters.Add(new DuckDBParameter { Value = startTime }); + command.Parameters.Add(new DuckDBParameter { Value = endTime }); + command.Parameters.Add(new DuckDBParameter { Value = cap }); + foreach (var db in dbValues) + command.Parameters.Add(new DuckDBParameter { Value = db }); + + var items = new List(); + long populationCount = 0; + using var reader = await command.ExecuteReaderAsync(); + while (await reader.ReadAsync()) + { + items.Add(new QuerySnapshotRow + { + SessionId = reader.IsDBNull(0) ? 0 : reader.GetInt32(0), + DatabaseName = reader.IsDBNull(1) ? "" : reader.GetString(1), + ElapsedTimeFormatted = reader.IsDBNull(2) ? "" : reader.GetString(2), + QueryText = reader.IsDBNull(3) ? "" : reader.GetString(3), + Status = reader.IsDBNull(4) ? "" : reader.GetString(4), + BlockingSessionId = reader.IsDBNull(5) ? 0 : reader.GetInt32(5), + WaitType = reader.IsDBNull(6) ? "" : reader.GetString(6), + WaitTimeMs = reader.IsDBNull(7) ? 0 : reader.GetInt64(7), + WaitResource = reader.IsDBNull(8) ? "" : reader.GetString(8), + CpuTimeMs = reader.IsDBNull(9) ? 0 : reader.GetInt64(9), + TotalElapsedTimeMs = reader.IsDBNull(10) ? 0 : reader.GetInt64(10), + Reads = reader.IsDBNull(11) ? 0 : reader.GetInt64(11), + Writes = reader.IsDBNull(12) ? 0 : reader.GetInt64(12), + LogicalReads = reader.IsDBNull(13) ? 0 : reader.GetInt64(13), + GrantedQueryMemoryGb = reader.IsDBNull(14) ? 0 : ToDouble(reader.GetValue(14)), + TransactionIsolationLevel = reader.IsDBNull(15) ? "" : reader.GetString(15), + Dop = reader.IsDBNull(16) ? 0 : reader.GetInt32(16), + ParallelWorkerCount = reader.IsDBNull(17) ? 0 : reader.GetInt32(17), + CollectionTime = reader.IsDBNull(18) ? DateTime.MinValue : reader.GetDateTime(18), + LoginName = reader.IsDBNull(19) ? "" : reader.GetString(19), + HostName = reader.IsDBNull(20) ? "" : reader.GetString(20), + ProgramName = reader.IsDBNull(21) ? "" : reader.GetString(21), + OpenTransactionCount = reader.IsDBNull(22) ? 0 : reader.GetInt32(22), + PercentComplete = reader.IsDBNull(23) ? 0m : Convert.ToDecimal(reader.GetValue(23)), + QueryHash = reader.IsDBNull(24) ? "" : reader.GetString(24), + IsHeadBlocker = !reader.IsDBNull(25) && reader.GetBoolean(25), + BlockerInCapture = !reader.IsDBNull(26) && reader.GetBoolean(26), + BlockerInPopulation = !reader.IsDBNull(27) && reader.GetBoolean(27), + }); + populationCount = Convert.ToInt64(reader.GetValue(28)); + } + + return (items, populationCount); + } + /// /// Gets lightweight blocking + deadlock counts and latest event time for alert badge updates. /// Much cheaper than fetching full rows with XML — just COUNT(*) and MAX(time). @@ -1153,6 +1336,19 @@ public class QuerySnapshotRow public int OpenTransactionCount { get; set; } public decimal PercentComplete { get; set; } public string QueryHash { get; set; } = ""; + + /// Some row in the SAME capture names this session as its blocker (#3541 A13) — the reason a + /// WAITFOR row can be on the MCP page. Set by only. + public bool IsHeadBlocker { get; set; } + + /// For a victim: its blocker had a row in the same capture at all. False is the idle + /// open-transaction head blocker sys.dm_exec_requests never lists. MCP read only. + public bool BlockerInCapture { get; set; } + + /// For a victim: its blocker also passes the caller's filters, so it is in the population the + /// page is drawn from (it may still be past the page — the tool checks that). MCP read only. + public bool BlockerInPopulation { get; set; } + public bool HasQueryPlan => !string.IsNullOrEmpty(QueryPlan); public bool HasLiveQueryPlan => !string.IsNullOrEmpty(LiveQueryPlan); public string CollectionTimeLocal => CollectionTime == DateTime.MinValue ? "" : ServerTimeHelper.FormatServerTime(CollectionTime); diff --git a/Lite/Services/LocalDataService.DailySummary.cs b/Lite/Services/LocalDataService.DailySummary.cs index bd81fe33d..972030488 100644 --- a/Lite/Services/LocalDataService.DailySummary.cs +++ b/Lite/Services/LocalDataService.DailySummary.cs @@ -140,8 +140,19 @@ UNION SELECT d FROM alerts count. 0 when the blocking came from a source without a wait time. */ COALESCE(CASE WHEN COALESCE(b.c, 0) > 0 THEN b.max_wait_ms ELSE dm.max_wait_ms END, 0) AS peak_block_wait_ms, /* Every collector run in the window (#3539 A2): the denominator that turns collection_errors into a - share. Appended LAST so every existing ordinal read stays where it was. */ - COALESCE(cl.runs, 0) AS collection_runs + share. Appended after peak_block_wait_ms so every existing ordinal read stays where it was. */ + COALESCE(cl.runs, 0) AS collection_runs, + /* #3541 A9: how many of the seven per-signal sources hold at least one row for the day — the retention + arm's PRESENCE fact, Darling's DailySummarySql column of the same name (its comment carries the + reasoning). Lite's sources share one archive horizon, so the ghost is narrower here, but the reader + judges the day the same way from the same fact. Appended LAST, after collection_runs. */ + (CASE WHEN w.d IS NULL THEN 0 ELSE 1 END) + + (CASE WHEN q.d IS NULL THEN 0 ELSE 1 END) + + (CASE WHEN dl.d IS NULL THEN 0 ELSE 1 END) + + (CASE WHEN b.d IS NULL THEN 0 ELSE 1 END) + + (CASE WHEN dm.d IS NULL THEN 0 ELSE 1 END) + + (CASE WHEN cp.d IS NULL THEN 0 ELSE 1 END) + + (CASE WHEN m.d IS NULL THEN 0 ELSE 1 END) AS signal_sources_present FROM day_spine s LEFT JOIN waits w ON w.d = s.d LEFT JOIN queries q ON q.d = s.d @@ -174,16 +185,34 @@ MCP read hands its resolved window end so a backdated as_of never clamps against clock; the live calendar read leaves this null. */ var referenceUtc = asOfUtc ?? DateTime.UtcNow; + /* #3541 A9: the retention horizon, from the READER's wall clock and never the anchor — a purge is a + wall-clock event and a backdated as_of cannot un-purge an archive; DailySummaryRetention.HorizonFor + says why in full. Lite has ONE horizon for every source (RetentionService.ArchiveRetentionMonths), + so a day past it is gone from every table together and cannot ghost the way Darling's per-signal + horizons let a day do; the state is stamped all the same so the two SKUs publish one vocabulary + and the same day is judged the same way on both. */ + var retentionHorizon = DailySummaryRetentionHorizon(DateTime.UtcNow); + var results = new List(); using var reader = await command.ExecuteReaderAsync(); while (await reader.ReadAsync()) { - results.Add(ReadDailySummaryRow(reader, referenceUtc)); + results.Add(ReadDailySummaryRow(reader, referenceUtc, retentionHorizon)); } return results; } + /// + /// The oldest UTC day every daily-summary source still holds on Lite (#3541 A9): today minus the archive + /// retention. Month arithmetic rather than 's day count + /// because Lite's horizon is DECLARED in months () + /// and the cleanup that enforces it subtracts months; converting to a day count here would make the two + /// disagree by a day or two around month ends. + /// + internal static DateTime DailySummaryRetentionHorizon(DateTime utcNow) => + utcNow.AddMonths(-RetentionService.ArchiveRetentionMonths).Date; + /// /// Gets the daily summary for a specific date (or today if null). Delegates to the range query for a /// single day so the single-day contract shares the one aggregate; returns a No-Data row when the day @@ -193,16 +222,30 @@ MCP read hands its resolved window end so a backdated as_of never clamps against { var targetDate = summaryDate?.Date ?? DateTime.UtcNow.Date; var rows = await GetDailySummaryRangeAsync(serverId, targetDate, targetDate.AddDays(1)); - return rows.Count > 0 - ? rows[0] - : new DailySummaryRow { SummaryDate = targetDate, HasData = false, HealthBand = DailyHealthBand.NoData }; + if (rows.Count > 0) + { + return rows[0]; + } + + /* #3541 A9: a day the spine does not hold is not "collected" either — before the horizon it is purged + like any other, inside it simply without a run record — so the single-day tool can say which. */ + var horizon = DailySummaryRetentionHorizon(DateTime.UtcNow); + return new DailySummaryRow + { + SummaryDate = targetDate, + HasData = false, + HealthBand = DailyHealthBand.NoData, + DataState = DailySummaryRetention.StateFor(targetDate, 0, 0, horizon), + RetentionHorizon = horizon, + }; } - private static DailySummaryRow ReadDailySummaryRow(System.Data.Common.DbDataReader reader, DateTime referenceUtc) + private static DailySummaryRow ReadDailySummaryRow(System.Data.Common.DbDataReader reader, DateTime referenceUtc, DateTime retentionHorizon) { var row = new DailySummaryRow { ReferenceUtc = referenceUtc, + RetentionHorizon = retentionHorizon, SummaryDate = reader.IsDBNull(0) ? DateTime.MinValue : Convert.ToDateTime(reader.GetValue(0)), TotalWaitTimeSec = reader.IsDBNull(1) ? 0m : Convert.ToDecimal(reader.GetValue(1)), TopWaitType = reader.IsDBNull(2) ? "" : reader.GetString(2), @@ -217,8 +260,14 @@ private static DailySummaryRow ReadDailySummaryRow(System.Data.Common.DbDataRead MaxBlockDurationMs = reader.IsDBNull(11) ? 0L : Convert.ToInt64(reader.GetValue(11)), /* #3539 A2: the trailing collection_runs column — the collection-error share's denominator. */ CollectionRuns = reader.IsDBNull(12) ? 0L : Convert.ToInt64(reader.GetValue(12)), + /* #3541 A9: the signal-presence count, after collection_runs. */ + SignalSourcesPresent = reader.IsDBNull(13) ? 0 : Convert.ToInt32(reader.GetValue(13)), HasData = true, }; + /* #3541 A9: judged from the day, its run count, its signal presence and the horizon; ToSignals folds + a non-collected state into HasData = false so the shared band reads NoData rather than + measured-zero-Healthy. */ + row.DataState = DailySummaryRetention.StateFor(row.SummaryDate, row.CollectionRuns, row.SignalSourcesPresent, retentionHorizon); row.HealthBand = DailyHealthBandCalculator.Classify(row.ToSignals()); return row; } @@ -252,9 +301,24 @@ public class DailySummaryRow /// (#3539 A2). public long MaxBlockDurationMs { get; set; } - /// True when the day had any collection. False renders the calendar cell as No-Data (grey). + /// True when the spine holds the day at all. Together with this decides + /// the band's HasData: a held day that is purged or past the horizon renders the calendar cell No-Data (grey) + /// exactly as an absent day does (#3541 A9). public bool HasData { get; set; } + /// Whether this row's counts are a measurement or the shape retention left behind (#3541 A9) — + /// see . Defaults to Collected so a row built without the reader + /// (the tests' hand-built rows) bands as it always did. + public DailySummaryDataState DataState { get; set; } = DailySummaryDataState.Collected; + + /// The horizon was judged against; null on a row nobody judged. + public DateTime? RetentionHorizon { get; set; } + + /// How many of the seven per-signal sources hold at least one row for the day (#3541 A9) — the + /// aggregate's trailing signal_sources_present column, the fact that tells a purged shell from a + /// day the purge has not reached. + public int SignalSourcesPresent { get; set; } + /// The composite health band that colors this day's calendar cell. public DailyHealthBand HealthBand { get; set; } = DailyHealthBand.NoData; @@ -272,7 +336,10 @@ public class DailySummaryRow /// Projects this row's counts into the shared banding input. public DailyHealthSignals ToSignals() => new() { - HasData = HasData, + /* #3541 A9: a purged or past-horizon day is a NoData day to the band, whatever the spine still holds + for it — its COALESCEd zeros may be absences, and measured-zero-Healthy was the lie. Inside retention + (Collected, NoRunRecord) a zero IS a measurement and the band stands. */ + HasData = HasData && DataState is not (DailySummaryDataState.Purged or DailySummaryDataState.PastHorizon), Deadlocks = DeadlockCount, CollectionErrors = CollectionErrors, CollectionRuns = CollectionRuns, diff --git a/Lite/Services/LocalDataService.QueryStats.cs b/Lite/Services/LocalDataService.QueryStats.cs index b9fc19889..d120b4e2e 100644 --- a/Lite/Services/LocalDataService.QueryStats.cs +++ b/Lite/Services/LocalDataService.QueryStats.cs @@ -89,14 +89,23 @@ GROUP BY date_trunc('hour', collection_time) return items; } - public async Task> GetTopQueriesByCpuAsync(int serverId, int hoursBack = 24, int top = 50, DateTime? fromDate = null, DateTime? toDate = null, int utcOffsetMinutes = 0, IReadOnlyList? databaseNames = null, DateTime? asOfUtc = null) + /// + /// The top-N query-stats groups by CPU over the window. is the lifetime + /// max_dop floor a group must reach to be RANKED at all (#3541 A13): 0 for no parallelism filter + /// (the grids' read, byte-identical to before), 2 for the MCP tool's parallel_only, the caller's + /// min_dop otherwise. It is a HAVING predicate on the grouped population, before the CPU ordering + /// and the cap, so the page is the top-N of the filtered population — the tool used to filter the + /// returned top-N page in C#, and a box whose hottest plans were all serial answered an empty page while + /// the window held parallel plans. Darling's TopQueriesSql carries the same floor as its $6. + /// + public async Task> GetTopQueriesByCpuAsync(int serverId, int hoursBack = 24, int top = 50, DateTime? fromDate = null, DateTime? toDate = null, int utcOffsetMinutes = 0, IReadOnlyList? databaseNames = null, DateTime? asOfUtc = null, int minMaxDop = 0) { using var _q = TimeQuery("GetTopQueriesByCpuAsync", "v_query_stats top N by CPU"); using var connection = await OpenConnectionAsync(); using var command = connection.CreateCommand(); var (startTime, endTime) = GetTimeRange(hoursBack, fromDate, toDate, asOfUtc, SelectedServerTabUtcOffsetMinutes); - var dbClause = BuildDbInClause(databaseNames, "database_name", 6, out var dbValues); + var dbClause = BuildDbInClause(databaseNames, "database_name", 7, out var dbValues); command.CommandText = @" WITH ranked AS ( @@ -157,7 +166,11 @@ FROM v_query_stats AND collection_time <= $3 AND last_execution_time >= $2 + $5 * INTERVAL '1' MINUTE" + dbClause + @" GROUP BY database_name, query_hash, host_object_name - HAVING SUM(delta_execution_count) > 0 OR SUM(delta_elapsed_time) > 0 + HAVING (SUM(delta_execution_count) > 0 OR SUM(delta_elapsed_time) > 0) + /* #3541 A13: the parallelism floor is part of the QUERY, on the grouped population before the ranking + and the cap — see the method's note. $6 = 0 admits every group; NULL max_dop (never captured) reads + as 0 and stays out of a filtered page, as the C# arm it replaces did. */ + AND COALESCE(MAX(max_dop), 0) >= $6 ORDER BY SUM(delta_worker_time) DESC LIMIT $4 + 5 ), @@ -218,6 +231,7 @@ ORDER BY r.total_cpu_us DESC command.Parameters.Add(new DuckDBParameter { Value = endTime }); command.Parameters.Add(new DuckDBParameter { Value = top }); command.Parameters.Add(new DuckDBParameter { Value = utcOffsetMinutes }); + command.Parameters.Add(new DuckDBParameter { Value = minMaxDop }); foreach (var db in dbValues) command.Parameters.Add(new DuckDBParameter { Value = db }); diff --git a/Lite/Services/RetentionService.cs b/Lite/Services/RetentionService.cs index a6447027f..ef078edc3 100644 --- a/Lite/Services/RetentionService.cs +++ b/Lite/Services/RetentionService.cs @@ -18,6 +18,19 @@ namespace PerformanceMonitorLite.Services; /// public class RetentionService { + /// + /// How long an archived Parquet file is kept, in months — Lite's ONE retention horizon (#3541 A9). + /// + /// Lite does not purge per collector: hot rows leave DuckDB for Parquet after the archive service's + /// hot-data week, the v_* views union live and archive so every read sees both, and this is the + /// age at which an archive file is deleted. Because every table — the signal tables, the collection log + /// and the alert log alike — shares it, a day older than this is gone from every source at once, which is + /// why the daily summary's retention horizon on Lite is this single number rather than the shortest of + /// several. Named so the horizon the daily-summary reader publishes and the horizon the cleanup enforces + /// are the same constant, not two literals that happen to agree. + /// + public const int ArchiveRetentionMonths = 3; + private readonly string _archivePath; private readonly ILogger? _logger; @@ -35,7 +48,7 @@ public RetentionService(string archivePath, ILogger? logger = /// - Consolidated daily: "20260221_wait_stats.parquet" (yyyyMMdd prefix) /// - Legacy monthly: "2026-02_wait_stats.parquet" (yyyy-MM prefix) /// - public void CleanupOldArchives(int retentionMonths = 3) + public void CleanupOldArchives(int retentionMonths = ArchiveRetentionMonths) { if (!Directory.Exists(_archivePath)) { diff --git a/PerformanceMonitor.Analysis/FactScorer.cs b/PerformanceMonitor.Analysis/FactScorer.cs index 8b41aecae..a2cc36bf4 100644 --- a/PerformanceMonitor.Analysis/FactScorer.cs +++ b/PerformanceMonitor.Analysis/FactScorer.cs @@ -17,6 +17,31 @@ namespace PerformanceMonitor.Analysis; /// public class FactScorer { + /// + /// The source registry: every value a collector on either SKU emits, in the + /// spelling the facts carry (#3541 A13). + /// + /// Why a registry exists at all. Sources were only ever string literals — in the collectors + /// that stamp them, in the switch below that scores them, and in the four the get_analysis_facts + /// description happened to mention ("waits, blocking, config, memory") out of the fifteen that exist. A + /// caller filtering on any of the other eleven got [], which reads as "no facts of that kind" for + /// a value that could never have matched. The MCP tools now publish THIS list as the accepted set and + /// refuse anything outside it; the analysis-side tests pin it against every Source = "..." literal + /// in the three collector assemblies AND against the switch arms below, so a new source that lands in a + /// collector without landing here fails a test rather than becoming the sixteenth silent value. + /// + /// Sorted, and kept sorted, because the list is published verbatim in a refusal message and a + /// description on both SKUs. coverage and sessions are emitted but not scored (they carry + /// context, base severity 0); they are still filterable, so they are still members. perfmon is + /// named in the amplifier context set below but no collector emits it, so it is NOT a member — a filter + /// on it would always be empty, which is the outcome this registry exists to refuse. + /// + public static readonly IReadOnlyList KnownSources = new[] + { + "anomaly", "bad_actor", "blocking", "config", "coverage", "cpu", "database_config", "disk", "io", + "jobs", "memory", "queries", "sessions", "tempdb", "waits", + }; + /// /// Scores all facts: Layer 1 (base severity), then Layer 2 (amplifiers). /// diff --git a/PerformanceMonitor.Common/DailySummaryDataState.cs b/PerformanceMonitor.Common/DailySummaryDataState.cs new file mode 100644 index 000000000..0ec5fba08 --- /dev/null +++ b/PerformanceMonitor.Common/DailySummaryDataState.cs @@ -0,0 +1,161 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Globalization; + +namespace PerformanceMonitor.Common +{ + /// + /// Whether a daily-summary row's counts are a MEASUREMENT of that day or the shape retention left behind + /// (#3541 A9). + /// + /// The defect this names. The daily aggregate's day spine is a UNION over nine sources, and + /// each source is LEFT JOINed and COALESCEd to zero. The sources age out at DIFFERENT + /// horizons: the per-signal collector tables at their 30-day default, the collection log at twice that, + /// the alert log at ninety days. So for every day between the shortest horizon and the longest, the spine + /// still has the day (the collection log or an alert still holds a row for it) while every signal the + /// band reads — deadlocks, blocking, CPU, memory — has been purged, and the COALESCE renders each of them + /// as a measured zero. Measured zeros band Healthy. A 366-day range therefore painted the purged months + /// green, under a description that said a returned day meant collection happened and a quiet day was a + /// healthy one. + /// + /// Shared by both SKUs so the two products, and the MCP tools and the calendar within each, make + /// the SAME decision from the same two inputs: the day, and the store's retention horizon. + /// + public enum DailySummaryDataState + { + /// The day is inside every signal's retention and the collection log recorded runs for + /// it: the counts are measurements and the band is a verdict. + Collected = 0, + + /// The day is BEFORE the store's retention horizon — the oldest day the shortest-lived + /// signal source can still answer for — and NO signal source holds a row for it: the spine has the + /// day from a longer-lived source alone (a run record, an alert). The signals the band reads are + /// gone, so its zeros are absences and its band is . + Purged = 1, + + /// The day is inside retention but has NO collector run recorded for it: the spine has the + /// day from a signal row or an alert alone. Inside retention every zero is a measurement (the tables + /// hold whatever the day had), so the band STANDS — an alert-only day is Warning, as it always was — + /// and the state is a disclosure: the collection-error share has no denominator (the band's own arm + /// then fails away from Healthy on a non-zero error count), and nothing records that the day was + /// fully collected. Named for the fact, not for a claim: a deadlock row on such a day IS evidence of + /// collection — what is missing is the record. + NoRunRecord = 2, + + /// The day is before the retention horizon yet SOME signal source still holds rows for it — + /// the purge has not reached it (held by the rollup-coverage gate, paused, or the boundary day), or + /// the sources' retentions differ so one aged out ahead of the others. The rows that are there are + /// real; the zeros beside them may be absences; the band cannot tell which, so it is + /// and the row says how many sources still answer. + PastHorizon = 3, + } + + /// + /// The one decision both SKUs' daily-summary readers make about a returned day (#3541 A9), plus the + /// payload vocabulary for it. + /// + public static class DailySummaryRetention + { + /// + /// The number of per-signal sources the daily aggregate reads: waits, queries, deadlocks, blocked-process + /// reports, DMV blocking snapshots, CPU samples, memory-pressure events. The collection log and the alert + /// log are the spine's other two members and are NOT signals — they are the longer-lived sources whose + /// survival is what lets a day outlive its signals. + /// + public const int SignalSourceCount = 7; + + /// + /// The state of one returned day from the three facts that decide it. Only the two past-horizon states + /// withhold the band: inside retention a zero is a measurement, and a day the alert log alone names is + /// still a day an alert fired on. + /// + /// The horizon test comes FIRST, so a purged day with a surviving run record (the collection + /// log outlives the signals by design — its horizon is twice theirs so a failure's evidence outlives + /// the metric rows it explains) reads as purged rather than as collected-and-quiet. That ordering IS + /// the fix: runs > 0 alone would have called every day in the second month collected. + /// + /// Presence is what keeps the horizon honest. The horizon is ARITHMETIC — now minus the shortest + /// retention — while a purge is an EVENT that may not have happened yet: the sweep runs daily, the + /// TimescaleDB path drops whole chunks, and the rollup-coverage gate can hold a table's purge for + /// weeks. A day before the horizon that still has signal rows is therefore not called purged, because + /// "the tables no longer hold it" would be false; it is , + /// which keeps the rows and withholds the verdict. + /// + /// The day the row describes (UTC date). + /// Collector runs of every status the collection log holds for the day. + /// How many of the signal sources hold + /// at least one row for the day — the aggregate's signal_sources_present column. + /// The oldest UTC day the shortest-lived signal source still holds — + /// see . + public static DailySummaryDataState StateFor(DateTime summaryDate, long collectionRuns, int signalSourcesPresent, DateTime retentionHorizon) + { + if (summaryDate.Date < retentionHorizon.Date) + return signalSourcesPresent > 0 ? DailySummaryDataState.PastHorizon : DailySummaryDataState.Purged; + + return collectionRuns > 0 ? DailySummaryDataState.Collected : DailySummaryDataState.NoRunRecord; + } + + /// + /// The retention horizon: the oldest UTC day on which EVERY signal the band reads is still present, + /// from the SHORTEST retention among the sources. + /// + /// The purge cutoff is now − retention, an instant; the day containing it is the first + /// day the shortest-lived source still holds rows for. On the TimescaleDB path that whole day is + /// present until its chunk's END passes the cutoff (drop_chunks removes only chunks entirely older + /// than the cutoff, and raw chunks are one day wide), so the cutoff's own day is complete there; on + /// the DELETE path the cutoff's day may hold only the hours after the cutoff. The horizon is the + /// cutoff's DATE, which is exact on the production path and at most one partial day generous on the + /// other — the direction that never calls a day with real rows purged. + /// + /// The clock is the READER's wall clock, never the caller's as_of: a purge is a + /// wall-clock event, and a backdated anchor cannot un-purge a table. Anchoring the horizon to + /// as_of would let "as_of 25 days ago, days_back 30" paint the purged stretch green again — + /// the exact defect. The tools' own no-DateTime.UtcNow rule is about the WINDOW (which must + /// follow the anchor); the horizon is a property of the store, and it belongs in the reader. + /// + public static DateTime HorizonFor(DateTime utcNow, int shortestRetentionDays) + { + if (shortestRetentionDays < 1) + throw new ArgumentOutOfRangeException(nameof(shortestRetentionDays), shortestRetentionDays, "A retention horizon needs at least one day."); + + return utcNow.AddDays(-shortestRetentionDays).Date; + } + + /// The payload spelling of a state — one vocabulary on both SKUs. + public static string Label(DailySummaryDataState state) => state switch + { + DailySummaryDataState.Collected => "collected", + DailySummaryDataState.Purged => "purged", + DailySummaryDataState.PastHorizon => "past_horizon", + _ => "no_run_record", + }; + + /// + /// The one-sentence reason a non-collected day carries on its row; null for a collected day so + /// the common case pays nothing. Says what the zeros ARE, because a reader that sees + /// deadlock_count: 0 beside health_band: NoData otherwise has to guess which of the two + /// to believe. + /// + public static string? Note(DailySummaryDataState state, DateTime retentionHorizon, int signalSourcesPresent = 0) => state switch + { + DailySummaryDataState.Purged => + "PURGED: this day is before the store's retention_horizon (" + retentionHorizon.ToString("yyyy-MM-dd", CultureInfo.InvariantCulture) + + ") and none of the per-signal tables the band reads (waits, queries, deadlocks, blocking, CPU, memory) holds a row for it; their zeros are absences, not measurements. " + + "The day is listed because a longer-lived source (the collection log or the alert log) still records it — collection_runs and alert_count are real where non-zero. No health verdict is possible.", + DailySummaryDataState.PastHorizon => + "PAST HORIZON: this day is before the store's retention_horizon (" + retentionHorizon.ToString("yyyy-MM-dd", CultureInfo.InvariantCulture) + + ") but " + signalSourcesPresent.ToString(CultureInfo.InvariantCulture) + " of " + SignalSourceCount.ToString(CultureInfo.InvariantCulture) + + " signal sources still hold rows for it — the purge has not reached it, or the sources' retentions differ. Non-zero counts are real; a zero may be a measurement or an absence, and the band cannot tell which, so no health verdict is given.", + DailySummaryDataState.NoRunRecord => + "NO RUN RECORD: no collector run is recorded for this day, so the collection-error share has no denominator and nothing records that the day was fully collected; the day appears because a signal table or the alert log holds rows for it. Inside retention its zeros are measurements, so the band stands on the counts as read.", + _ => null, + }; + } +} diff --git a/PerformanceMonitor.Common/Mcp/McpHelpers.cs b/PerformanceMonitor.Common/Mcp/McpHelpers.cs index fd1b521f1..3dc4e9e35 100644 --- a/PerformanceMonitor.Common/Mcp/McpHelpers.cs +++ b/PerformanceMonitor.Common/Mcp/McpHelpers.cs @@ -7,6 +7,7 @@ */ using System; +using System.Collections.Generic; using System.Globalization; using System.Text.Json; @@ -102,6 +103,105 @@ internal static class McpHelpers return ResolveAsOf(asOf, out endUtc); } + /// + /// The window validation for the three UNCAPPED reads (get_collection_log, get_current_waits_trend, + /// get_blocking_stats): a positive span of any length, then the anchor — + /// minus its ceiling. + /// + /// What this replaces (#3541 A13). Those three tools never went through + /// because its 168-hour ceiling would take reach away from exactly the reads whose premise is looking further + /// back than the default — and having stepped around the validator they Math.Abs'd the span instead. + /// A caller who sent hours_back = -24 asked a question with no meaning (a window that ends before it + /// starts), and got the last 24 hours back with nothing to say the sign had been flipped: an answer to a + /// different question, indistinguishable from a correct one, which is the silently-different-answer class + /// every validator in this file exists to remove. Zero is refused with it — a zero-length window holds + /// nothing by construction, and an empty result under a "genuinely quiet" sentence would be a lie. + /// + /// The refusal borrows 's first sentence so a caller who has seen the + /// capped reads' message recognises it, and then says the one thing that differs: there is no ceiling. + /// + public static string? ValidateUncappedWindow(int hoursBack, string? asOf, out DateTime endUtc) + { + endUtc = DateTime.UtcNow; + + if (hoursBack <= 0) + { + return $"Invalid hours_back value '{hoursBack}'. Must be a positive integer — a negative or zero window has no meaning and is refused rather than read as its absolute value. This read has no upper bound on hours_back."; + } + + return ResolveAsOf(asOf, out endUtc); + } + + /// + /// The ONLY spelling summary_date accepts on both SKUs' daily-summary tools: the ISO-8601 calendar + /// date its own description has always promised. + /// + public const string SummaryDateFormat = "yyyy-MM-dd"; + + /// + /// Parses get_daily_summary's summary_date: is null when the + /// caller sent nothing (today, resolved by the reader), the UTC date when they sent a usable one. Returns + /// null when usable, the refusal when not. + /// + /// Exact, not general (#3541 A9). This sat in the same file as 's + /// strict allowlist and used a general , + /// which under the invariant culture also accepts 01/02/2026 as M/d/yyyy — so a caller who + /// meant 1 February was answered about 2 January, correctly formatted, with nothing to say so. The tool's + /// description promised yyyy-MM-dd; the parser now agrees with it instead of exceeding it, exactly as + /// does for as_of. The refusal names the one accepted form. + /// + public static string? ParseSummaryDate(string? summaryDate, out DateTime? date) + { + date = null; + if (string.IsNullOrWhiteSpace(summaryDate)) + { + return null; + } + + if (!DateTime.TryParseExact( + summaryDate.Trim(), + SummaryDateFormat, + CultureInfo.InvariantCulture, + DateTimeStyles.AdjustToUniversal | DateTimeStyles.AssumeUniversal, + out var parsed)) + { + return $"Invalid summary_date value '{summaryDate}'. Expected an ISO-8601 calendar date, yyyy-MM-dd (e.g. 2026-07-09), read as a UTC day. Other spellings — including 07/09/2026 — are refused rather than guessed at, because 01/02/2026 reads as two different days depending on who wrote it."; + } + + date = DateTime.SpecifyKind(parsed, DateTimeKind.Utc).Date; + return null; + } + + /// + /// Validates an optional ENUMERATED filter — a parameter whose usable values are a closed set the + /// caller cannot see. Returns null when the caller sent nothing or a member of the set, the refusal + /// naming the whole set when not. The match is case-insensitive, and the caller is expected to use the + /// canonical spelling from downstream rather than the caller's. + /// + /// Refuses rather than filters to nothing (#3541 A13). get_analysis_facts applied an + /// unknown source as an equality filter and returned an empty list, under a description that + /// documented four of the engine's source names — so a caller who typed the fifth read "no facts of that + /// kind" for a value that could never have matched. An unknown member of a closed set is a caller error, + /// and the refusal that lists the set is the only answer that lets the caller fix it. + /// + public static string? ValidateChoice(string? value, IReadOnlyCollection accepted, string paramName) + { + if (string.IsNullOrWhiteSpace(value)) + { + return null; + } + + foreach (var candidate in accepted) + { + if (string.Equals(candidate, value.Trim(), StringComparison.OrdinalIgnoreCase)) + { + return null; + } + } + + return $"Invalid {paramName} value '{value}'. Accepted values: {string.Join(", ", accepted)}. Omit it for all."; + } + /// /// The as_of parameter's description, shared VERBATIM by every windowed read on both SKUs. /// From 6b837f97eceaa76e1e097479b01efd02da8612a2 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 17:06:29 -0400 Subject: [PATCH 64/69] The interval-hourly refresh recomputes one fleet-wide bucket per dirty hour and paid twelve index inserts per row for eleven indexes nothing reads; the capture-down alert read decompressed a server's whole collection_log to learn two statuses (#3597, partial) (#3647) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * The interval-hourly refresh stops paying twelve index inserts per re-materialized row for eleven indexes nothing reads, and the capture-down alert read stops decompressing a server's whole collection_log to learn two collectors' latest status (#3597, partial) Rig-measured on PostgreSQL 18.4 / TimescaleDB 2.28.1 at one tenth of the largest production store's scale: a refresh of query_store_stats_interval_hourly recomputes exactly the hour-buckets the invalidation log marks dirty — the newly closed one every hour, plus one whole fleet-wide bucket per distinct past hour any backdated row landed in (a ONE-row backdated insert re-materialized 36,120 rows; twelve backdated rows in one transaction across twelve hours re-materialized twelve buckets, clean hours included). The window's width is not the cost; the dirty-bucket count times the per-bucket cost is. The only backdated writer is QueryStoreBackfill, by design (collection_time = slice ceiling). The per-bucket cost carried TimescaleDB's default create_group_indexes: eleven (column, bucket DESC) btrees on the L1 materialization that no reader uses — its three child aggregates read it by bucket range, the coverage probe and arming gate read min(bucket), retention drops chunks. EXPLAIN (ANALYZE, BUFFERS, WAL) of one bucket's materialization INSERT: 45.5 MB WAL / 483,689 records / 1.64 M buffer touches / 6,040 dirtied with them; 10.7 MB / 72,735 / 447 K / 56 with the bucket index alone. L1 is now created without them and EnsureIntervalDedupMaterializationIndexesAsync drops them on an existing store at startup, per index under a 10 s lock_timeout so a refresh in flight is yielded to rather than convoyed. The capture-down read was the #3496 shape #3496 named and left: ROW_NUMBER() OVER (ORDER BY log_id DESC) over every collection_log row the server had for two collectors — 9,573 buffers across 61 chunks (59 decompressed) per alert pass per server on the rig. It is now one chunk-orderable LIMIT 1 per collector: 14 buffers, the newest chunk only, 120 of 122 chunk scans never executed. Partial: the dirty-bucket multiplier is the backfill's designed behaviour and is not changed here; the production dirty-bucket count per hour is the read that decides how much of the 417 s average this removes, and the exact statements for it are on the issue. * IntervalDedupMaterializationIndexesTests records why it is not serialized against the live collection: its gated arm mints a scratch database, so it cannot race the shared store (#1776 own-store, the census's own guidance) --- .../CaptureDownChunkOrderTests.cs | 223 +++++++++++++++ .../CollectionSignalsChunkOrderTests.cs | 7 +- ...ntervalDedupMaterializationIndexesTests.cs | 266 ++++++++++++++++++ .../DarlingSelfAlertEvaluator.cs | 66 ++++- .../DarlingWorker.cs | 9 + .../TimescaleSupport.cs | 206 +++++++++++++- 6 files changed, 758 insertions(+), 19 deletions(-) create mode 100644 Darling/Darling.Tests/CaptureDownChunkOrderTests.cs create mode 100644 Darling/Darling.Tests/IntervalDedupMaterializationIndexesTests.cs diff --git a/Darling/Darling.Tests/CaptureDownChunkOrderTests.cs b/Darling/Darling.Tests/CaptureDownChunkOrderTests.cs new file mode 100644 index 000000000..5f7d49e77 --- /dev/null +++ b/Darling/Darling.Tests/CaptureDownChunkOrderTests.cs @@ -0,0 +1,223 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Linq; +using System.Text; +using System.Text.RegularExpressions; +using System.Threading; +using System.Threading.Tasks; +using Npgsql; +using PerformanceMonitor.Darling.Service; +using PerformanceMonitor.Darling.Storage; +using Xunit; + +namespace Darling.Tests; + +/// +/// #3597: the capture-down self-alert read asks "what did this collector's LATEST run log?" once per +/// collector as a chunk-orderable LIMIT 1, not as a window function over the server's whole +/// collection_log history. +/// +/// This is the read #3496 named and deliberately left — ROW_NUMBER() OVER (PARTITION BY +/// collector_name ORDER BY log_id DESC) cannot early-stop by reordering alone. collection_log is a +/// hypertable partitioned on collection_time with no index on log_id, so the window's only legal +/// plan decompressed EVERY chunk in the server's retention horizon, Merge-Appended them and numbered ~100 K +/// rows to keep two: 9,573 buffers over 61 chunks on a rig with 60 days of one server's log, every alert +/// pass, every 30 seconds, per server — the heaviest read on the pass by two to three orders of magnitude, +/// and one of the four sites the issue saw die at the 10 s deadline while the interval-hourly refresh +/// starved the store. Per collector, ORDER BY collection_time DESC LIMIT 1 lets ChunkAppend walk the +/// chunks newest-first and stop at the first row: 14 buffers, the newest chunk only, 120 of 122 chunk scans +/// never executed. The property is horizon-independent, so a longer retention cannot regress it. +/// +/// The regression is QUIET, exactly as #3496's was: a revert to the window returns the same two rows on +/// any store small enough for a test and only shows up as deadline breaches once a store's retention has +/// filled. So the shape is pinned at the source, and the gated arm asks the planner. +/// +/* Live-fixture tests share one Postgres store; the collection serializes them so cross-test row churn + cannot race another class's assertions. */ +[Collection("live-postgres")] +public sealed class CaptureDownChunkOrderTests +{ + /// Distinctive fake id — a real server_id is a storage-name hash, never this. + private const int TestServerId = -735971; + private const string TestServerName = "capture-down-chunk-order-e2e"; + + [Fact] + public void EachCollector_IsAskedOnce_OrderedByThePartitionColumn_LimitOne() + { + var sql = DarlingSelfAlertEvaluator.MissingCaptureSessionsSql; + + /* One arm per capture collector, UNION ALL between them, each stopping at its newest row. */ + Assert.Equal(2, Regex.Matches(sql, @"ORDER BY cl\.collection_time DESC\s+LIMIT 1").Count); + Assert.Single(Regex.Matches(sql, @"\bUNION ALL\b")); + Assert.Contains("cl.collector_name = 'deadlocks'", sql, StringComparison.Ordinal); + Assert.Contains("cl.collector_name = 'blocked_process_report'", sql, StringComparison.Ordinal); + + /* The verdict is still on the LATEST run's status, applied after each arm has picked its row. */ + Assert.Contains("WHERE x.status = 'SESSION_MISSING'", sql, StringComparison.Ordinal); + } + + /// + /// The negative half: no window function, and no ordering by log_id in any spelling — the two + /// shapes that force every chunk to execute. Matched as shapes rather than the literals that shipped, so a + /// re-spelling cannot slip past the pin the way the original slipped past review. + /// + [Fact] + public void NoWindowFunction_AndNoOrderingByLogId() + { + var sql = DarlingSelfAlertEvaluator.MissingCaptureSessionsSql; + + Assert.DoesNotMatch(new Regex(@"\bOVER\s*\(", RegexOptions.IgnoreCase), sql); + Assert.DoesNotMatch(new Regex(@"ROW_NUMBER", RegexOptions.IgnoreCase), sql); + Assert.DoesNotMatch(new Regex(@"log_id", RegexOptions.IgnoreCase), sql); + } + + /// + /// The evidence no string pin can give: that the planner stops at the newest chunk. Builds the store the + /// way the service does (ladder, then collection_log's hypertable conversion where TimescaleDB is + /// present), seeds one server's capture-collector rows across eight days — eight 1-day chunks — and + /// EXPLAINs the shipped statement with its real bound parameter: no WindowAgg, and at most one + /// chunk executed per arm, every other chunk scan reported never executed. Then the read answers + /// through the same path, and it is the NEWEST run that decides: a SESSION_MISSING three days ago + /// followed by a success is not a missing session; a success three days ago followed by + /// SESSION_MISSING is. + /// + [Fact] + public async Task TheShippedRead_ExecutesOnlyTheNewestChunk_AndTheNewestRunDecides_AgainstDevPostgres() + { + var connectionString = Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + Assert.SkipWhen(string.IsNullOrEmpty(connectionString), + "Set DARLING_TEST_PG to a Postgres connection string to run the live capture-down access-path test."); + + var ct = TestContext.Current.CancellationToken; + + using var connection = new NpgsqlConnection(connectionString); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + + /* #1922: probe on its own connection. The service's own runtime conversion for collection_log, which sits + outside the collector catalog and so outside ConvertToHypertablesAsync. */ + var timescaleEnabled = await LiveTimescaleProbe.TryEnableAsync(connectionString!, ct); + if (timescaleEnabled) + { + Assert.True(await TimescaleSupport.EnsureCollectionLogHypertableAsync(connection, null, ct)); + } + + var bodySucceeded = false; + try + { + await DeleteTestRowsAsync(connection, ct); + + /* Eight days of one-minute blocked_process_report and five-minute deadlocks rows, plus a filler + collector so the newest chunk holds rows the arms must skip past. All Kind-Unspecified: naive-UTC + storage, see DarlingObservability.LogCollectionAsync. */ + var utcNow = DateTime.SpecifyKind(DateTime.UtcNow, DateTimeKind.Unspecified); + await SeedHistoryAsync(connection, utcNow, ct); + + /* The two verdict rows. deadlocks: SESSION_MISSING three days ago, then a SUCCESS a minute ago — NOT + missing. blocked_process_report: SUCCESS all along, then SESSION_MISSING a minute ago — missing. */ + await InsertAsync(connection, 9_100_000_001, "deadlocks", utcNow.AddDays(-3).AddSeconds(7), "SESSION_MISSING", ct); + await InsertAsync(connection, 9_100_000_002, "deadlocks", utcNow.AddMinutes(-1), "SUCCESS", ct); + await InsertAsync(connection, 9_100_000_003, "blocked_process_report", utcNow.AddMinutes(-1), "SESSION_MISSING", ct); + + using (var analyze = new NpgsqlCommand("ANALYZE collect.collection_log", connection)) + { + await analyze.ExecuteNonQueryAsync(ct); + } + + var plan = await ExplainShippedReadAsync(connection, ct); + + Assert.DoesNotContain("WindowAgg", plan, StringComparison.Ordinal); + + if (timescaleEnabled) + { + /* Every chunk scan the plan carries, split into executed and never-executed. ChunkAppend orders + the chunks newest-first for ORDER BY collection_time DESC, so each arm's LIMIT 1 is satisfied + by the newest chunk and the rest never start. */ + var chunkScans = plan.Split('\n').Where(l => Regex.IsMatch(l, @"Scan .* on _hyper_\d+_\d+_chunk")).ToList(); + Assert.True(chunkScans.Count >= 2 * 8, + "expected the plan to carry at least eight chunk scans per arm (eight seeded days):\n" + plan); + var executed = chunkScans.Where(l => !l.Contains("never executed", StringComparison.Ordinal)).ToList(); + Assert.True(executed.Count <= 2, + "more than one chunk executed per arm — the read is walking history again:\n" + plan); + } + + await using var postgres = NpgsqlDataSource.Create(connectionString!); + var missing = await DarlingSelfAlertEvaluator.ReadMissingCaptureSessionsAsync(postgres, TestServerId, ct); + Assert.Equal(new[] { "Blocking" }, missing); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(connectionString!, bodySucceeded, async (cleanup, cleanupCt) => + await DeleteTestRowsAsync(cleanup, cleanupCt)); + } + } + + /// + /// One multi-row INSERT per collector over the eight days, so the seed is three statements rather than + /// fifteen thousand. log_id is derived from the row's instant so id order and time order agree, as + /// CollectionIdGenerator's do. + /// + private static async Task SeedHistoryAsync(NpgsqlConnection connection, DateTime utcNow, CancellationToken ct) + { + var start = utcNow.AddDays(-8); + foreach (var (collector, stepMinutes) in new[] { ("blocked_process_report", 1), ("deadlocks", 5), ("wait_stats", 1) }) + { + using var insert = new NpgsqlCommand( + "INSERT INTO collect.collection_log (log_id, server_id, server_name, collector_name, collection_time, duration_ms, status, rows_collected) " + + "SELECT 9_000_000_000 + (EXTRACT(EPOCH FROM t)::bigint * 10) + $5, $1, $2, $3, t, 20, 'SUCCESS', 0 " + + "FROM generate_series($4::timestamp, $4::timestamp + interval '8 days' - interval '2 minutes', ($6::text || ' minutes')::interval) AS t", connection); + insert.Parameters.AddWithValue(TestServerId); + insert.Parameters.AddWithValue(TestServerName); + insert.Parameters.AddWithValue(collector); + insert.Parameters.AddWithValue(start); + insert.Parameters.AddWithValue((long)stepMinutes); + insert.Parameters.AddWithValue(stepMinutes.ToString(System.Globalization.CultureInfo.InvariantCulture)); + await insert.ExecuteNonQueryAsync(ct); + } + } + + private static async Task InsertAsync(NpgsqlConnection connection, long logId, string collector, DateTime when, string status, CancellationToken ct) + { + using var insert = new NpgsqlCommand( + "INSERT INTO collect.collection_log (log_id, server_id, server_name, collector_name, collection_time, duration_ms, status, rows_collected) " + + "VALUES ($1, $2, $3, $4, $5, 20, $6, 0)", connection); + insert.Parameters.AddWithValue(logId); + insert.Parameters.AddWithValue(TestServerId); + insert.Parameters.AddWithValue(TestServerName); + insert.Parameters.AddWithValue(collector); + insert.Parameters.AddWithValue(when); + insert.Parameters.AddWithValue(status); + await insert.ExecuteNonQueryAsync(ct); + } + + private static async Task ExplainShippedReadAsync(NpgsqlConnection connection, CancellationToken ct) + { + using var explain = new NpgsqlCommand( + "EXPLAIN (ANALYZE, COSTS OFF, TIMING OFF, SUMMARY OFF) " + DarlingSelfAlertEvaluator.MissingCaptureSessionsSql, connection); + explain.Parameters.AddWithValue(TestServerId); + var plan = new StringBuilder(); + using var reader = await explain.ExecuteReaderAsync(ct); + while (await reader.ReadAsync(ct)) + { + plan.AppendLine(reader.GetString(0)); + } + + return plan.ToString(); + } + + private static async Task DeleteTestRowsAsync(NpgsqlConnection connection, CancellationToken ct) + { + using var cleanup = new NpgsqlCommand("DELETE FROM collect.collection_log WHERE server_id = $1", connection); + cleanup.Parameters.AddWithValue(TestServerId); + await cleanup.ExecuteNonQueryAsync(ct); + } +} diff --git a/Darling/Darling.Tests/CollectionSignalsChunkOrderTests.cs b/Darling/Darling.Tests/CollectionSignalsChunkOrderTests.cs index c67bfc8c7..a8c363bc9 100644 --- a/Darling/Darling.Tests/CollectionSignalsChunkOrderTests.cs +++ b/Darling/Darling.Tests/CollectionSignalsChunkOrderTests.cs @@ -32,9 +32,10 @@ namespace Darling.Tests; /// same rows on any store small enough for a test, passes every behavioral assertion, and only shows up /// months later as tail-latency deadline breaches on the store whose retention has filled — exactly how /// the defect presented the first time. So the ORDER the statement asks for is pinned at the source, -/// scoped to the one statement (the capture-down read's ROW_NUMBER ... ORDER BY cl.log_id DESC in -/// the same file is a different shape — a window function cannot early-stop by ordering swap alone — and -/// is deliberately not swept). +/// scoped to the one statement. The capture-down read in the same file was the different shape this pin +/// originally left alone — ROW_NUMBER ... ORDER BY cl.log_id DESC, which no ordering swap could +/// early-stop; #3597 reshaped it into one chunk-orderable LIMIT 1 per collector, and +/// pins that one. /// public sealed class CollectionSignalsChunkOrderTests { diff --git a/Darling/Darling.Tests/IntervalDedupMaterializationIndexesTests.cs b/Darling/Darling.Tests/IntervalDedupMaterializationIndexesTests.cs new file mode 100644 index 000000000..ad7dcbe81 --- /dev/null +++ b/Darling/Darling.Tests/IntervalDedupMaterializationIndexesTests.cs @@ -0,0 +1,266 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading; +using System.Threading.Tasks; +using Npgsql; +using PerformanceMonitor.Darling.Storage; +using Xunit; + +namespace Darling.Tests; + +/// +/// #3597: the interval-dedup L1 aggregate () +/// carries no per-GROUP-BY-column index on its materialization, and an existing store is brought to that +/// shape at startup. +/// +/// What the rig measured, restated here because the pins below only make sense against it. +/// TimescaleDB's default create_group_indexes built eleven (column, bucket DESC) btrees on +/// L1's materialization beside the bucket index. Nothing reads L1 by any of those columns — its three +/// child aggregates refresh over it by bucket range, the coverage probe and the arming gate read +/// min(bucket), retention drops chunks — but every hourly refresh re-materializes a bucket by +/// DELETE + INSERT, and each inserted row cost twelve index inserts. EXPLAIN (ANALYZE, BUFFERS, WAL) +/// of one bucket's materialization INSERT at one tenth of the largest store's scale: 45.5 MB of WAL and +/// 1.64 M buffer touches with the group indexes, 10.7 MB and 447 K with the bucket index alone. The +/// refresh, the child refreshes, compress_chunk and decompress_chunk all ran unchanged +/// without them. +/// +/// Why the pins are scoped to L1 and only L1. The option is earned by that measurement, not +/// applied for symmetry: the composer-grain rollups ARE read through their group indexes (by +/// server_id, by query_hash), and the day-grain L2 refreshes once a day and was not measured. +/// A second aggregate wanting create_group_indexes = false should arrive with its own rig figures and +/// move the scope pin deliberately. +/// +/// #1776 own-store — the gated arm mints a scratch database (it creates the continuous +/// aggregates the shared fixture deliberately leaves to the tests, plants and drops an index on one of +/// their materializations, and refreshes over seeded rows), so it cannot race the shared store and is not +/// serialized against the live-postgres collection. The same shape as +/// and QueryStoreCorrectedRollupLiveTests. +/// +public sealed class IntervalDedupMaterializationIndexesTests +{ + private const string NoGroupIndexes = "timescaledb.create_group_indexes = false"; + + [Fact] + public void L1_IsCreatedWithoutGroupIndexes() + { + var sql = TimescaleSupport.CreateQueryStoreStatsIntervalHourlySql; + + Assert.Contains("WITH (timescaledb.continuous, " + NoGroupIndexes + ") AS", sql, StringComparison.Ordinal); + /* Still materialized-only (#1759): the option rides beside `continuous`, it does not displace the + absence TimescaleContinuousAggregateTests pins for the query-acceleration tier. */ + Assert.DoesNotContain("materialized_only", sql, StringComparison.Ordinal); + } + + /// + /// The scope pin: every OTHER registered aggregate keeps TimescaleDB's default. Enumerated from the three + /// registries the ensure sweep builds from, so a new aggregate is covered the day it is registered. + /// + [Fact] + public void EveryOtherAggregate_KeepsTheDefaultGroupIndexes_UntilMeasured() + { + var others = TimescaleSupport.HourlyAggregates + .Concat(TimescaleSupport.DailyAggregates) + .Concat(TimescaleSupport.BaselineAggregates) + .Where(a => !string.Equals(a.View, TimescaleSupport.QueryStoreStatsIntervalHourlyView, StringComparison.Ordinal)) + .ToList(); + + Assert.NotEmpty(others); + foreach (var (createSql, view) in others) + { + Assert.DoesNotContain("create_group_indexes", createSql, StringComparison.Ordinal); + Assert.True(view.Length > 0); + } + + /* And L1 is in the hourly registry, so the converge below finds a materialization to act on. */ + Assert.Contains(TimescaleSupport.HourlyAggregates, a => string.Equals(a.View, TimescaleSupport.QueryStoreStatsIntervalHourlyView, StringComparison.Ordinal)); + } + + /// + /// The converge's catalog read selects by SHAPE — one column then bucket DESC — on L1's + /// materialization only, so the bucket index ((bucket DESC) alone) can never match, and a + /// materialization other than L1's is never touched. Pinned as text because the regex runs in + /// PostgreSQL; the gated test below asks the server. + /// + [Fact] + public void TheGroupIndexRead_SelectsByShape_OnL1Only() + { + var sql = TimescaleSupport.IntervalDedupMaterializationGroupIndexesSql; + + Assert.Contains($"ca.view_name = '{TimescaleSupport.QueryStoreStatsIntervalHourlyView}'", sql, StringComparison.Ordinal); + Assert.Contains(@"i.indexdef ~ 'USING btree \([a-z_]+, bucket DESC\)$'", sql, StringComparison.Ordinal); + Assert.Contains("i.tablename = ca.materialization_hypertable_name", sql, StringComparison.Ordinal); + /* Resolved from the view name, never a hard-coded _materialized_hypertable_N. */ + Assert.DoesNotContain("_materialized_hypertable_", sql, StringComparison.Ordinal); + } + + /// + /// The drop yields to a refresh in flight rather than queueing behind it (the queued-exclusive convoy + /// HourlyRefreshStartOffset documents): a bounded lock timeout, short against the grid's hour and + /// long against an idle lock. + /// + [Fact] + public void TheIndexDrop_WaitsABoundedTimeForItsLock() + { + var timeout = TimescaleSupport.IntervalDedupIndexDropLockTimeout; + + Assert.EndsWith("s", timeout, StringComparison.Ordinal); + var seconds = int.Parse(timeout.TrimEnd('s'), System.Globalization.CultureInfo.InvariantCulture); + Assert.InRange(seconds, 1, 60); + } + + /// + /// The evidence no string pin can give, on a scratch database with TimescaleDB: (1) TimescaleDB honours the + /// option — a freshly created L1 materialization carries exactly one index, on (bucket DESC); + /// (2) the converge finds a pre-#3597 store's group index (planted by hand in TimescaleDB's own shape and + /// name), drops it, reports one, and reports zero on the next call; (3) the bucket index is never + /// selected; (4) the refresh still materializes rows afterwards, through the same policy window the + /// product uses, and the child corrected hourly still reads them. + /// + [Fact] + public async Task OnAFreshStore_L1HasOnlyTheBucketIndex_AndTheConvergeDropsAPlantedGroupIndex_AgainstDevPostgres() + { + var baseConnectionString = Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + Assert.SkipWhen(string.IsNullOrEmpty(baseConnectionString), + "Set DARLING_TEST_PG to a Postgres connection string (with TimescaleDB installed) to run the live #3597 materialization-index test (it mints its own scratch database)."); + + var ct = TestContext.Current.CancellationToken; + + await using var scratch = await ScratchPostgres.CreateAsync(baseConnectionString!, ct); + await using var connection = new NpgsqlConnection(scratch.ConnectionString); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + + Assert.True(await TimescaleSupport.TryEnableAsync(connection, null, ct), + "the dev fixture is expected to have TimescaleDB installed"); + await TimescaleSupport.ConvertToHypertablesAsync(connection, null, ct); + await TimescaleSupport.EnsureContinuousAggregatesAsync(connection, null, ct); + + /* Manual refreshes below assert on exact ranges; strip the policies so a background first run + cannot race them (the QueryStoreTrendRoutingLiveTests discipline). */ + foreach (var (view, _, _, _, _) in TimescaleSupport.RollupViews) + { + await using var remove = new NpgsqlCommand( + $"SELECT remove_continuous_aggregate_policy('collect.{view}', if_exists => true)", connection); + await remove.ExecuteNonQueryAsync(ct); + } + + var (matSchema, matName) = await MaterializationOfAsync(connection, TimescaleSupport.QueryStoreStatsIntervalHourlyView, ct); + + /* (1) A fresh L1 honours the option: the bucket index and nothing else. */ + var fresh = await IndexDefinitionsAsync(connection, matSchema, matName, ct); + var only = Assert.Single(fresh); + Assert.EndsWith("USING btree (bucket DESC)", only, StringComparison.Ordinal); + + /* A settled store: nothing to find, nothing dropped. */ + Assert.Equal(0, await TimescaleSupport.EnsureIntervalDedupMaterializationIndexesAsync(connection, null, ct)); + + /* (2) Plant what an earlier build's CREATE left behind, in TimescaleDB's own name and shape. */ + var planted = matName + "_server_id_bucket_idx"; + await using (var plant = new NpgsqlCommand( + $"CREATE INDEX \"{planted}\" ON \"{matSchema}\".\"{matName}\" (server_id, bucket DESC)", connection)) + { + await plant.ExecuteNonQueryAsync(ct); + } + + Assert.Equal(2, (await IndexDefinitionsAsync(connection, matSchema, matName, ct)).Count); + + Assert.Equal(1, await TimescaleSupport.EnsureIntervalDedupMaterializationIndexesAsync(connection, null, ct)); + var afterDrop = await IndexDefinitionsAsync(connection, matSchema, matName, ct); + var survivor = Assert.Single(afterDrop); + /* (3) The bucket index survived, by shape. */ + Assert.EndsWith("USING btree (bucket DESC)", survivor, StringComparison.Ordinal); + + Assert.Equal(0, await TimescaleSupport.EnsureIntervalDedupMaterializationIndexesAsync(connection, null, ct)); + + /* (4) The aggregate still materializes without the group indexes. Fixed instants, not now-relative. */ + var hour = new DateTime(2026, 3, 4, 10, 0, 0, DateTimeKind.Unspecified); + await using (var seed = new NpgsqlCommand(@" +INSERT INTO collect.query_store_stats (collection_id, collection_time, server_id, server_name, database_name, query_id, plan_id, + execution_type_desc, first_execution_time, module_name, query_hash, execution_count, avg_duration_us, avg_cpu_time_us, + max_duration_us, max_cpu_time_us, replica_role, runtime_stats_interval_id, interval_start_time_utc) +SELECT 1, $1 + (k * interval '10 minutes'), -735970, 'interval-dedup-index-e2e', 'db', 100 + i, 1000 + i, + 'Regular', $1, 'mod', md5('q' || i), 10 * (k + 1), 500, 300, 900, 700, 'PRIMARY', 77, $1 +FROM generate_series(0, 9) AS i CROSS JOIN generate_series(0, 2) AS k", connection)) + { + seed.Parameters.AddWithValue(hour); + await seed.ExecuteNonQueryAsync(ct); + } + + await RefreshAsync(connection, TimescaleSupport.QueryStoreStatsIntervalHourlyView, hour, hour.AddHours(1), ct); + await RefreshAsync(connection, TimescaleSupport.QueryStoreStatsCorrectedHourlyView, hour, hour.AddHours(1), ct); + + await using (var l1 = new NpgsqlCommand( + $"SELECT count(*), sum(execution_count) FROM collect.{TimescaleSupport.QueryStoreStatsIntervalHourlyView} WHERE server_id = -735970", connection)) + await using (var reader = await l1.ExecuteReaderAsync(ct)) + { + Assert.True(await reader.ReadAsync(ct)); + /* Ten interval identities, each deduped to its LAST snapshot (30 executions). */ + Assert.Equal(10L, reader.GetInt64(0)); + Assert.Equal(300L, Convert.ToInt64(reader.GetValue(1), System.Globalization.CultureInfo.InvariantCulture)); + } + + await using (var corrected = new NpgsqlCommand( + $"SELECT sum(execution_count_sum) FROM collect.{TimescaleSupport.QueryStoreStatsCorrectedHourlyView} WHERE server_id = -735970", connection)) + { + /* Ten query_hash groups, one per identity; the child sums L1's deduped counts, never the raw snapshots. */ + Assert.Equal(300L, Convert.ToInt64(await corrected.ExecuteScalarAsync(ct), System.Globalization.CultureInfo.InvariantCulture)); + } + } + + private static async Task<(string Schema, string Name)> MaterializationOfAsync(NpgsqlConnection connection, string view, CancellationToken ct) + { + await using var command = new NpgsqlCommand( + "SELECT materialization_hypertable_schema, materialization_hypertable_name FROM timescaledb_information.continuous_aggregates WHERE view_schema = 'collect' AND view_name = $1", connection); + command.Parameters.AddWithValue(view); + await using var reader = await command.ExecuteReaderAsync(ct); + Assert.True(await reader.ReadAsync(ct), $"{view} was not created"); + return (reader.GetString(0), reader.GetString(1)); + } + + private static async Task> IndexDefinitionsAsync(NpgsqlConnection connection, string schema, string table, CancellationToken ct) + { + var definitions = new List(); + await using var command = new NpgsqlCommand( + "SELECT indexdef FROM pg_indexes WHERE schemaname = $1 AND tablename = $2 ORDER BY indexname", connection); + command.Parameters.AddWithValue(schema); + command.Parameters.AddWithValue(table); + await using var reader = await command.ExecuteReaderAsync(ct); + while (await reader.ReadAsync(ct)) + { + definitions.Add(reader.GetString(0)); + } + + return definitions; + } + + /// Bounded retry on 55P03 — the same reason QueryStoreTrendRoutingLiveTests retries: a policy's + /// creation-time first run can still be finishing when the manual refresh lands. + private static async Task RefreshAsync(NpgsqlConnection connection, string view, DateTime from, DateTime to, CancellationToken ct) + { + for (var attempt = 1; ; attempt++) + { + try + { + await using var refresh = new NpgsqlCommand( + $"CALL refresh_continuous_aggregate('collect.{view}', $1::timestamp, $2::timestamp)", connection); + refresh.Parameters.AddWithValue(from); + refresh.Parameters.AddWithValue(to); + await refresh.ExecuteNonQueryAsync(ct); + return; + } + catch (PostgresException ex) when (ex.SqlState == "55P03" && attempt < 12) + { + await Task.Delay(TimeSpan.FromSeconds(1), ct); + } + } + } +} diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingSelfAlertEvaluator.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingSelfAlertEvaluator.cs index 06ad5082c..7b2ecef93 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingSelfAlertEvaluator.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingSelfAlertEvaluator.cs @@ -4475,33 +4475,69 @@ AND status IN ('SUCCESS', 'SKIPPED')) A return (lastSuccess, recentRuns, recentSuccess); } + /// The shipped capture-down statement, a constant so the #3597 pins and the gated plan test read + /// the text the method executes rather than a copy of it. + internal const string MissingCaptureSessionsSql = @" +SELECT x.collector_name +FROM +( + (SELECT cl.collector_name, cl.status + FROM collection_log AS cl + WHERE cl.server_id = $1 + AND cl.collector_name = 'deadlocks' + ORDER BY cl.collection_time DESC + LIMIT 1) + UNION ALL + (SELECT cl.collector_name, cl.status + FROM collection_log AS cl + WHERE cl.server_id = $1 + AND cl.collector_name = 'blocked_process_report' + ORDER BY cl.collection_time DESC + LIMIT 1) +) AS x +WHERE x.status = 'SESSION_MISSING' +ORDER BY x.collector_name"; + /// /// The blocking/deadlock XE collectors whose LATEST run logged SESSION_MISSING — the session is /// absent and couldn't be created, so capture is non-functional even though the tolerant reader "succeeds" /// with zero rows. The Darling twin of the Dashboard's GetMissingCaptureSessionsAsync, on Darling's /// collector names. Returns the friendly capture names ("Blocking" / "Deadlock"). + /// + /// #3597: one chunk-orderable LIMIT 1 per collector, not a window function over the + /// server's whole history. This read shipped as ROW_NUMBER() OVER (PARTITION BY collector_name + /// ORDER BY log_id DESC) over every collection_log row the server had for the two collectors — + /// the #3496 shape, which that fix named and deliberately left: a window function cannot early-stop by + /// reordering alone. collection_log is a hypertable partitioned on collection_time with no + /// index on log_id, so the window had exactly one legal plan: decompress EVERY chunk in the + /// retention horizon for the server, Merge Append them, and number 100 K rows to keep two. Measured on a + /// rig with 60 days of one server's log across 61 chunks (59 compressed): 9,573 buffers and every chunk + /// executed, per alert pass, every 30 seconds, per server — the largest read on the pass by two to three + /// orders of magnitude, and one of the issue's four sites that died at the 10 s deadline while the + /// interval-hourly refresh starved the store for I/O. The same question asked per collector as + /// ORDER BY collection_time DESC LIMIT 1 lets ChunkAppend order the chunks newest-first and stop + /// at the first row: 14 buffers, the newest chunk only, 120 of 122 chunk scans never executed — and the + /// property is horizon-independent, so a longer retention cannot regress it. Two literal arms rather than + /// a LATERAL over a VALUES list because the collector names are a closed set the alert owns, and a plan + /// with a literal predicate is the one the gated test can pin. + /// + /// Why no log_id tiebreak here, when #3496 kept one. The recent-N read orders across ALL + /// of a server's collectors, where many rows share a collection instant and the id decides among them. + /// Each arm here is ONE collector on ONE server, and a collector logs one row per run, stamped + /// DateTime.UtcNow at log time (DarlingObservability.LogCollectionAsync) — two runs of the + /// same collector on the same server cannot share a microsecond, so there is nothing for a tiebreak to + /// decide. What it would COST is measured: with , log_id DESC appended, the index on + /// (server_id, collection_time) cannot serve the second key, and each arm top-N-heapsorts the + /// server's whole newest chunk (1,158 buffers) instead of stopping at its first hit (5). /// + internal static async Task> ReadMissingCaptureSessionsAsync( NpgsqlDataSource postgres, int serverId, CancellationToken cancellationToken) { var missing = new List(); await using var connection = await postgres.OpenConnectionAsync(cancellationToken); - using var command = new NpgsqlCommand(@" -SELECT x.collector_name -FROM -( - SELECT - cl.collector_name, - cl.status, - ROW_NUMBER() OVER (PARTITION BY cl.collector_name ORDER BY cl.log_id DESC) AS n - FROM collection_log AS cl - WHERE cl.server_id = $1 - AND cl.collector_name IN ('deadlocks', 'blocked_process_report') -) AS x -WHERE x.n = 1 -AND x.status = 'SESSION_MISSING' -ORDER BY x.collector_name", connection) { CommandTimeout = DarlingAlertReadAdapter.AlertPassCommandTimeoutSeconds }; + using var command = new NpgsqlCommand(MissingCaptureSessionsSql, connection) { CommandTimeout = DarlingAlertReadAdapter.AlertPassCommandTimeoutSeconds }; command.Parameters.AddWithValue(serverId); await using var reader = await command.ExecuteReaderAsync(cancellationToken); diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs index 5ae9d5148..0dde9b00f 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs @@ -1411,6 +1411,15 @@ overlaps with an existing continuous aggregate policy". So on any store that eve await TimescaleSupport.EnsureContinuousAggregatesAsync(timescaleConnection, _logger, stoppingToken); + /* #3597: AFTER the aggregates exist and BEFORE compression, take the eleven per-column group + indexes off the interval-dedup materialization on any store whose aggregate predates + create_group_indexes = false on its CREATE. Nothing reads them, and every hourly refresh paid + twelve index inserts per re-materialized row for them — measured at 4.3x the WAL per bucket. + Before compression so the nightly pass compresses a relation already without them. Its own + catalog read, its own per-index isolation under a lock timeout that yields to a refresh in + flight, its own summary line; a no-op on every start after the first. */ + await TimescaleSupport.EnsureIntervalDedupMaterializationIndexesAsync(timescaleConnection, _logger, stoppingToken); + /* #3581: AFTER the aggregates exist, put their materializations on the compression ladder the raw tier has been on since the archival tier existed — none of the twenty ever was, and on the largest store they were 235 GiB of a 415 GiB database, larger than the 9x-compressed raw they diff --git a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs index 2e012f132..e4b4f6670 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs @@ -1402,9 +1402,32 @@ hourly grain the residual is irreducible — an interval genuinely collected in /// (): nothing reads it, so it only has to outlive raw for the /// arming gate and outlive its consumers' refresh windows ( for /// the corrected hourly, for the corrected daily). + /// + /// create_group_indexes = false (#3597) — no per-column index on the materialization, + /// because nothing reads one and every refresh paid for eleven. TimescaleDB's default builds one btree + /// per GROUP BY column, (column, bucket DESC), on a continuous aggregate's materialization hypertable; + /// here that was eleven of them beside the bucket index — twelve indexes on the one materialization in this + /// file whose row count is near-raw. Nothing reads this relation by any of those columns: its three + /// consumers (, + /// , ) + /// refresh over it by bucket range, the coverage probe and the arming gate read min(bucket), + /// and retention drops whole chunks. What the eleven indexes DID do was tax the refresh: the hourly policy + /// re-materializes a bucket by DELETE + INSERT, and every inserted row cost twelve index inserts. Measured + /// on a rig at one tenth of the largest production store's scale (PostgreSQL 18.4 / TimescaleDB 2.28.1, the + /// production pair), EXPLAIN (ANALYZE, BUFFERS, WAL) of one bucket's materialization INSERT + /// (36,121 rows from 216,721 raw): with the group indexes 45.5 MB of WAL over 483,689 records, + /// 1.64 M buffer touches, 6,040 buffers dirtied; with only the bucket index 10.7 MB over 72,735 records, + /// 447 K buffer touches, 56 dirtied — the indexes were 4.3x the WAL and 1.19 M of the buffer touches per + /// bucket. On the rig those touches are memory hits and the wall clock barely moves; on a store whose + /// materialization is 71.5 GiB they are the leaf pages of eleven cold indexes, which is the I/O the issue's + /// alert-read victims were starved by. brings + /// an existing store to the same shape — this option only speaks at CREATE. Scoped to THIS aggregate on + /// purpose: the composer-grain rollups are read by server_id and query_hash through exactly + /// these indexes, and refreshes once a day and was not + /// measured — the option is earned by a measurement, not applied for symmetry. /// public const string CreateQueryStoreStatsIntervalHourlySql = @"CREATE MATERIALIZED VIEW IF NOT EXISTS collect.query_store_stats_interval_hourly -WITH (timescaledb.continuous) AS +WITH (timescaledb.continuous, timescaledb.create_group_indexes = false) AS SELECT server_id, server_name, @@ -5734,6 +5757,187 @@ public static async Task EnsureMaterializationChunkIntervalAsync(NpgsqlConn return changed; } + /* ─────────────── interval-dedup materialization indexes (#3597) ─────────────── */ + + /// + /// The per-GROUP-BY-column indexes TimescaleDB's default create_group_indexes built on + /// 's materialization hypertable, resolved from the catalog + /// by SHAPE rather than by name: a btree on the materialization whose key is exactly one column followed by + /// bucket DESC. The bucket index ((bucket DESC) alone) does not match and stays — the refresh's + /// DELETE, the coverage probe's min(bucket) and the children's bucket-range reads all use it. The + /// materialization's schema and name are TimescaleDB's (_timescaledb_internal._materialized_hypertable_N), + /// never stable across stores, so the statement resolves them from the view name, the same rule as + /// . Hypertable-level indexes only (pg_indexes on the + /// parent): dropping the parent drops every chunk's copy, so the per-chunk names never need to be known — + /// they are read here only to SIZE what a drop releases, and for that the naming convention is the only + /// map 2.28.1 offers (no catalog row and no pg_inherits/pg_depend edge ties a chunk's copy to + /// its parent index): a copy is named <chunk>_<parent index> truncated to 63 characters. + /// A copy TimescaleDB had to suffix to keep unique after truncation is missed by that join, which + /// understates the logged figure and changes nothing else. + /// + public static string IntervalDedupMaterializationGroupIndexesSql => + $@" +SELECT + i.schemaname, + i.indexname, + i.indexdef, + pg_relation_size(format('%I.%I', i.schemaname, i.indexname)::regclass) + + COALESCE((SELECT sum(pg_relation_size(format('%I.%I', ci.schemaname, ci.indexname)::regclass)) + FROM timescaledb_information.chunks AS ch + JOIN pg_indexes AS ci + ON ci.schemaname = ch.chunk_schema + AND ci.tablename = ch.chunk_name + AND ci.indexname = left(ch.chunk_name || '_' || i.indexname, 63) + WHERE ch.hypertable_schema = ca.materialization_hypertable_schema + AND ch.hypertable_name = ca.materialization_hypertable_name), 0) AS bytes +FROM timescaledb_information.continuous_aggregates AS ca +JOIN pg_indexes AS i + ON i.schemaname = ca.materialization_hypertable_schema + AND i.tablename = ca.materialization_hypertable_name +WHERE ca.view_schema = 'collect' +AND ca.view_name = '{QueryStoreStatsIntervalHourlyView}' +AND i.indexdef ~ 'USING btree \([a-z_]+, bucket DESC\)$' +ORDER BY i.indexname"; + + /// + /// The lock wait the index drops below will tolerate before giving the start back, as a PostgreSQL + /// lock_timeout literal. DROP INDEX takes AccessExclusiveLock on the materialization and + /// its chunks, and the hourly refresh that this exists to lighten holds RowExclusiveLock on the same + /// relation for its whole run — up to fifteen minutes on the largest store. A drop that queued behind it + /// would not merely wait: a QUEUED exclusive request blocks every later shared request too (the convoy + /// documents), so the three child aggregates' refreshes and the + /// coverage probe would pile up behind a lock that was only requested. Ten seconds is long enough for the + /// lock to be free whenever no refresh is running and short enough that a running refresh costs this start + /// nothing but a warning; the next start retries. Set with SET LOCAL inside each drop's own + /// transaction, so it never outlives the statement it guards. + /// + public const string IntervalDedupIndexDropLockTimeout = "10s"; + + /// + /// Drops the per-column group indexes an earlier build's CREATE MATERIALIZED VIEW left on + /// 's materialization (#3597), so an existing store reaches + /// the shape 's create_group_indexes = false gives + /// a fresh one. Idempotent under the catalog — a settled store reads pg_indexes once and issues + /// nothing — and failure-isolated per index. Returns the number of indexes dropped this start. + /// + /// What was measured and what was not, stated apart because the lever is licensed by the first + /// and not the second. The refresh's cost per re-materialized bucket was measured with and without + /// these indexes on a rig at one tenth of the largest store's scale (the figures are on + /// ): 4.3x the WAL, 1.19 M extra buffer touches and + /// 108x the dirtied buffers per bucket with them, from twelve index inserts per row where one suffices. + /// That is the write amplification, and it is a property of the statement, not of the rig. What the rig + /// could NOT reproduce is the production I/O regime — its indexes fit in shared buffers, so its wall clock + /// barely moved — and so this file does not claim a refresh-duration figure for the largest store. The + /// issue's own job_history series after this lands is that measurement; the trough value + /// (276–286 s at the quietest hours, when only the newly-closed bucket is dirty) is the per-bucket floor + /// this should lower. + /// + /// Why a drop at startup rather than a recreate. The option that keeps a fresh store from + /// building these speaks only at CREATE, and recreating the aggregate would discard seven days of + /// materialization the corrected tiers are gated on. DROP INDEX on the parent hypertable is + /// transactional, propagates to every chunk, and is measured harmless to what remains: the hourly refresh, + /// the three child aggregates' refreshes, compress_chunk on a materialization chunk and + /// decompress_chunk all ran unchanged on the rig with the bucket index alone. + /// + /// Ordering. After (the aggregate must exist) + /// and before (so the nightly compression pass compresses a + /// relation that is already smaller). Under a so a refresh + /// in flight at startup is yielded to rather than convoyed — that arm logs at Warning and the next start + /// retries, which on an hourly grid is at most one refresh away from succeeding. + /// + public static async Task EnsureIntervalDedupMaterializationIndexesAsync(NpgsqlConnection connection, ILogger? logger, CancellationToken cancellationToken = default) + { + if (connection is null) + { + throw new ArgumentNullException(nameof(connection)); + } + + var indexes = new List<(string Schema, string Name, string Definition, long Bytes)>(); + try + { + using var probe = new NpgsqlCommand(IntervalDedupMaterializationGroupIndexesSql, connection) { CommandTimeout = SetupTimeoutSeconds }; + await using var reader = await probe.ExecuteReaderAsync(cancellationToken); + while (await reader.ReadAsync(cancellationToken)) + { + indexes.Add(( + reader.GetString(0), + reader.GetString(1), + reader.GetString(2), + reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3), CultureInfo.InvariantCulture))); + } + } + catch (Exception ex) when (ex is not OperationCanceledException) + { + logger?.LogWarning( + "TimescaleDB: could not read {View}'s materialization indexes, so none was dropped this start — its hourly refresh keeps paying twelve index inserts per re-materialized row until the next start reads it (#3597): {Message}", + QueryStoreStatsIntervalHourlyView, ex.Message); + return 0; + } + + if (indexes.Count == 0) + { + logger?.LogInformation( + "TimescaleDB: {View}'s materialization carries no per-column group index — the bucket index alone, the shape its refresh is cheapest in (#3597)", + QueryStoreStatsIntervalHourlyView); + return 0; + } + + var dropped = 0; + long freed = 0; + foreach (var (schema, name, definition, bytes) in indexes) + { + /* One transaction per index, each with its own lock timeout: a drop that cannot get its lock leaves + the others untried THIS start rather than half-done, because the convoy argument on the timeout + constant applies to every one of them equally — if the first is blocked by a refresh, so are the + rest, and eleven ten-second waits is a startup stalled for two minutes behind a lock it decided + not to wait for. */ + try + { + await using var transaction = await connection.BeginTransactionAsync(cancellationToken); + using (var timeout = new NpgsqlCommand($"SET LOCAL lock_timeout = '{IntervalDedupIndexDropLockTimeout}'", connection, transaction) { CommandTimeout = SetupTimeoutSeconds }) + { + await timeout.ExecuteNonQueryAsync(cancellationToken); + } + + using (var drop = new NpgsqlCommand($"DROP INDEX IF EXISTS {QuoteIdentifier(schema)}.{QuoteIdentifier(name)}", connection, transaction) { CommandTimeout = SetupTimeoutSeconds }) + { + await drop.ExecuteNonQueryAsync(cancellationToken); + } + + await transaction.CommitAsync(cancellationToken); + dropped++; + freed += bytes; + logger?.LogInformation( + "TimescaleDB: dropped {Index} ({SizeMiB:0.#} MiB across the materialization and its chunks) from {View}'s materialization — a per-column group index nothing read, whose maintenance every hourly refresh paid on every re-materialized row (#3597). Definition was: {Definition}", + name, bytes / 1048576d, QueryStoreStatsIntervalHourlyView, definition); + } + catch (PostgresException ex) when (string.Equals(ex.SqlState, PostgresErrorCodes.LockNotAvailable, StringComparison.Ordinal)) + { + logger?.LogWarning( + "TimescaleDB: {Index} on {View}'s materialization could not be dropped within {Timeout} — its hourly refresh is holding the relation, and waiting would queue every reader behind this drop; the remaining {Remaining} group index(es) are left for the next start rather than each waiting its own turn (#3597).", + name, QueryStoreStatsIntervalHourlyView, IntervalDedupIndexDropLockTimeout, indexes.Count - dropped); + break; + } + catch (Exception ex) when (ex is not OperationCanceledException) + { + logger?.LogWarning( + "TimescaleDB: could not drop {Index} from {View}'s materialization — its refresh keeps maintaining it until the next restart retries (often a permission issue: the store login must own the materialization) (#3597): {Message}", + name, QueryStoreStatsIntervalHourlyView, ex.Message); + } + } + + logger?.LogInformation( + "TimescaleDB: {Dropped}/{Found} per-column group index(es) dropped from {View}'s materialization this start, {FreedMiB:0.#} MiB released; {Remaining} remain (#3597)", + dropped, indexes.Count, QueryStoreStatsIntervalHourlyView, freed / 1048576d, indexes.Count - dropped); + + return dropped; + } + + /// Double-quotes one SQL identifier, doubling any embedded quote — the catalog names the + /// drops above interpolate are TimescaleDB's own, but a name is a name and gets quoted. + private static string QuoteIdentifier(string identifier) + => "\"" + identifier.Replace("\"", "\"\"", StringComparison.Ordinal) + "\""; + /// /// Puts every continuous aggregate this product owns on the compression ladder (#3581): enables columnar /// compression on each materialization, attaches a once-a-day compression policy on the daily band, stages From dd1061afa84c823b62de71dd957c41b5195b64d1 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 17:07:42 -0400 Subject: [PATCH 65/69] Top CPU Queries renders as one record per query on Slack, not seven interleaved fields: a drill-down's rows travel as records beside their flat fields, and a five-fact page's fixed items cost 20 blocks instead of 33 (#3644) (#3649) --- Lite.Tests/RecordDetailRenderingTests.cs | 504 ++++++++++++++++++ Lite.Tests/SlackDetailsSizeTests.cs | 95 +++- .../AlertContext.cs | 49 ++ .../AnalysisNotificationService.cs | 107 +++- .../EmailTemplateBuilder.cs | 42 ++ .../WebhookAlertService.cs | 207 ++++++- 6 files changed, 964 insertions(+), 40 deletions(-) create mode 100644 Lite.Tests/RecordDetailRenderingTests.cs diff --git a/Lite.Tests/RecordDetailRenderingTests.cs b/Lite.Tests/RecordDetailRenderingTests.cs new file mode 100644 index 000000000..8869b9d6c --- /dev/null +++ b/Lite.Tests/RecordDetailRenderingTests.cs @@ -0,0 +1,504 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Globalization; +using System.Linq; +using System.Text; +using System.Text.Json; +using PerformanceMonitor.Analysis; +using PerformanceMonitor.Notifications; +using PerformanceMonitorLite.Services; +using Xunit; + +namespace PerformanceMonitorLite.Tests; + +/// +/// #3644: the High CPU card's "Top Cpu Queries" was unreadable on Slack. The producer flattened every +/// drill-down row into one field per attribute (#1 Database, #1 Query Hash, … #1 Query +/// Text, #2 Database, …) and Slack lays a section's fields out two across in submission +/// order, so seven attributes per query meant no query's attributes ever stayed together: #1's text beside +/// #2's hash, #3's database beside #2's SQL, a multi-line text inflating its grid row so the label/value +/// adjacency below it broke. Read live on a production page during a sustained CPU arc. +/// +/// The fix is in the producer's shape and in three renderers. FlattenInto now ALSO packs +/// each row of an array of objects as an — its ordinal, one summary line of +/// its non-empty scalars in the row's own order, and its SQL text(s) separately — beside the flat fields it +/// has always produced. Slack renders records as one bold summary line over a code-blocked text per row, +/// packed into as few sections as fit the text ceiling; email and Teams render the same compact list. Every +/// surface that lays fields out in one column (the persisted row, the in-app grid, PagerDuty, the generic +/// webhook, the redundancy oracle) keeps the fields and is untouched. +/// +/// The two things a change like this can break. It can repaint a payload that was fine — every +/// engine alert, every hand-built detail — so the byte-identity arms build a context with every non-record +/// shape (fields, body, code block, plain object, scalar array) and assert each surface's output equals the +/// pre-#3644 rendering, restated here. And a renderer with a ceiling can cut silently, so the cut arm puts +/// a text past the record cap and asserts the note states its count in characters and lands on a whole +/// character (#3622), inside every Slack ceiling. +/// +public class RecordDetailRenderingTests +{ + private static AlertBranding Branding => EmailAlertService.Branding; + + private static readonly DateTime s_now = new(2026, 9, 18, 15, 42, 0, DateTimeKind.Utc); + + private const int TextObjectLimit = 3000; + private const string Pointer = "see email or in-app Alert Details for the full text"; + + /* ---------------- fixtures ---------------- */ + + private static string QueryText(int i) => string.Create(CultureInfo.InvariantCulture, + $"SELECT o.OrderId, o.OrderDate, ol.Quantity FROM Sales.Orders AS o JOIN Sales.OrderLines AS ol ON ol.OrderId = o.OrderId WHERE o.OrderDate >= @Start AND c.Region = @Region{i} OPTION (RECOMPILE)"); + + /// A CPU_SPIKE finding carrying exactly the Top Cpu Queries drill-down the Lite collector + /// emits — the shape and property order of CollectTopCpuQueries, five rows as it writes them. + private static AnalysisFinding HighCpuFinding(int rows = 5, Func? row = null) + { + var topCpu = new List(); + for (var i = 0; i < rows; i++) + { + topCpu.Add(row?.Invoke(i) ?? new + { + database = "ReportingDB", + query_hash = string.Create(CultureInfo.InvariantCulture, $"0x9A3F5C21D4E8B7A{i}"), + total_cpu_ms = 3_088_689.0 - i * 100_000, + execution_count = 159_665L - i * 1_000, + max_dop = 16, + spills = 0L + i, + query_text = QueryText(i) + }); + } + + return new AnalysisFinding + { + ServerId = 3, + ServerName = "SQLPROD-A", + Category = "cpu", + StoryPath = "CPU_SPIKE", + StoryPathHash = "cpu-0001", + Severity = 0.9, + Confidence = 0.8, + FactCount = 1, + RootFactKey = "CPU_SPIKE", + RootFactValue = 96.4, + TimeRangeStart = s_now.AddHours(-4), + TimeRangeEnd = s_now, + DrillDown = new Dictionary { ["top_cpu_queries"] = topCpu } + }; + } + + private static AlertDetailItem TopCpuItem(AnalysisFinding finding) => + Assert.Single(FindingMessageFormatter.BuildContext(finding, 1.5).Details, d => d.Heading == "Top Cpu Queries"); + + /// Every detail shape a producer emits that is NOT a record array: fields, advice body, a code + /// block, a plain object, a scalar array, and an incident-style item. None carries records. + private static AlertContext NonRecordContext() + { + var context = new AlertContext(); + var diagnosis = new AlertDetailItem { Heading = "Diagnosis" }; + diagnosis.Fields.Add(("Story", "BLOCKING → LONG_RUNNING")); + diagnosis.Fields.Add(("Severity", "1.20")); + diagnosis.Fields.Add(("Database", "SalesDB")); + context.Details.Add(diagnosis); + context.Details.Add(new AlertDetailItem { Heading = "Find the head blocker", Body = "Investigation: look at the head.\n\nRemediation: fix the index." }); + context.Details.Add(new AlertDetailItem { Heading = "Remediation T-SQL", Body = "SELECT 1;", IsCodeBlock = true }); + var chain = new AlertDetailItem { Heading = "Blocking Chain 1" }; + chain.Fields.Add(("Database", "SalesDB")); + chain.Fields.Add(("Blocked SQL", "UPDATE dbo.Orders SET x = 1 WHERE id = 5")); + chain.Fields.Add(("Blocking SQL", "SELECT * FROM dbo.Orders WITH (HOLDLOCK)")); + chain.Fields.Add(("Wait Time Ms", "12400")); + context.Details.Add(chain); + var plain = new AlertDetailItem { Heading = "Spike Peak" }; + plain.Fields.Add(("Time", "2026-09-18T11:40:00Z")); + plain.Fields.Add(("Cpu Percent", "96.4")); + context.Details.Add(plain); + var scalars = new AlertDetailItem { Heading = "Objects" }; + scalars.Fields.Add(("#1", "dbo.Orders")); + scalars.Fields.Add(("#2", "dbo.OrderLines")); + context.Details.Add(scalars); + Assert.All(context.Details, d => Assert.Empty(d.Records)); + return context; + } + + private static List SlackBlocks(string payload) => + JsonDocument.Parse(payload).RootElement.GetProperty("attachments")[0].GetProperty("blocks").EnumerateArray().ToList(); + + private static string? SectionText(JsonElement block) => + block.GetProperty("type").GetString() == "section" && block.TryGetProperty("text", out var t) + ? t.GetProperty("text").GetString() + : null; + + private static string Slack(AlertContext context) => + WebhookAlertService.BuildSlackPayload("Analysis finding", "SQLPROD-A", "0.90", "1.5", Branding, + context: context, triageUrl: "https://example.invalid/triage/1", nowUtc: s_now); + + private static string Teams(AlertContext context) => + WebhookAlertService.BuildTeamsPayload("Analysis finding", "SQLPROD-A", "0.90", "1.5", Branding, + context: context, triageUrl: "https://example.invalid/triage/1", nowUtc: s_now); + + private static (string Html, string Text) Email(AlertContext context) => + EmailTemplateBuilder.BuildAlertEmail("Analysis finding", "SQLPROD-A", "0.90", "1.5", 30, Branding, context); + + /* ---------------- the producer ---------------- */ + + /// + /// An array of objects flattens to BOTH shapes: the flat #N Label fields exactly as before — + /// seven per row, three rows, the 300-character cut on the text — and one record per kept row. Rows + /// past the third are not carried in either shape. + /// + [Fact] + public void AnArrayOfObjects_CarriesTheFlatFieldsUnchanged_AndOneRecordPerRow() + { + var item = TopCpuItem(HighCpuFinding()); + + Assert.Equal(21, item.Fields.Count); + Assert.Equal(new[] { "#1 Database", "#1 Query Hash", "#1 Total Cpu Ms", "#1 Execution Count", "#1 Max Dop", "#1 Spills", "#1 Query Text" }, + item.Fields.Take(7).Select(f => f.Label)); + Assert.Equal("ReportingDB", item.Fields[0].Value); + Assert.Equal("3088689", item.Fields[2].Value); + Assert.Equal(QueryText(0), item.Fields[6].Value); + Assert.StartsWith("#3 ", item.Fields[^1].Label, StringComparison.Ordinal); + + Assert.Equal(3, item.Records.Count); + Assert.Equal(new[] { 1, 2, 3 }, item.Records.Select(r => r.Ordinal)); + } + + /// + /// The summary line is the row's scalars in the ROW's property order — identity first, measures after, + /// as the collector wrote them — as Label: value joined by ·, numbers group-separated, + /// without the ordinal (each renderer sets it) and without the text. The text is the property whose + /// name says it is SQL, labelled as its flat field is, and carried WHOLE: the flat field's 300-character + /// cut is not applied, because the renderer with a ceiling states its own. + /// + [Fact] + public void TheSummaryIsTheScalarsInRowOrder_AndTheTextIsTheSqlProperty() + { + var item = TopCpuItem(HighCpuFinding()); + var first = item.Records[0]; + + Assert.Equal("Database: ReportingDB · Query Hash: 0x9A3F5C21D4E8B7A0 · Total Cpu Ms: 3,088,689 · Execution Count: 159,665 · Max Dop: 16 · Spills: 0", first.Summary); + var (label, text) = Assert.Single(first.Texts); + Assert.Equal("Query Text", label); + Assert.Equal(QueryText(0), text); + Assert.DoesNotContain("Query Text", first.Summary, StringComparison.Ordinal); + } + + /// A long text is carried whole in the record and cut at 300 in the flat field — the two + /// projections of one row differ only there. + [Fact] + public void ALongText_IsWholeInTheRecord_AndCutInTheFlatField() + { + var text = new string('x', 500); + var item = TopCpuItem(HighCpuFinding(1, _ => new { database = "D", query_hash = "0x1", query_text = text })); + + Assert.Equal(text, Assert.Single(item.Records[0].Texts).Text); + var field = Assert.Single(item.Fields, f => f.Label == "#1 Query Text"); + Assert.Equal(301, field.Value.Length); + Assert.Equal(text[..300] + "…", field.Value); + } + + /// Nulls and empty strings are left out of the summary (the non-AG replica_role is empty + /// on nearly every server) but kept in the flat fields; booleans and fractions render as written; + /// a nested value renders as its compact JSON; several SQL properties become several labelled texts in + /// property order; an empty text is not carried. + [Fact] + public void TheSummarySkipsEmptyScalars_AndARowCanCarrySeveralTexts() + { + var item = TopCpuItem(HighCpuFinding(1, _ => new + { + database = "SalesDB", + replica_role = "", + regression_factor = 14.25, + worker_ratio = 221.376, + cofired = true, + plan = new { id = 7 }, + nothing = (string?)null, + blocked_sql = "UPDATE dbo.Orders SET x = 1", + blocking_sql = "SELECT * FROM dbo.Orders WITH (HOLDLOCK)", + victim_sql = "" + })); + + var record = item.Records[0]; + Assert.Equal("Database: SalesDB · Regression Factor: 14.25 · Worker Ratio: 221.38 · Cofired: true · Plan: {\"id\":7}", record.Summary); + Assert.Equal(new[] { ("Blocked Sql", "UPDATE dbo.Orders SET x = 1"), ("Blocking Sql", "SELECT * FROM dbo.Orders WITH (HOLDLOCK)") }, record.Texts); + + Assert.Contains(item.Fields, f => f.Label == "#1 Replica Role" && f.Value == ""); + Assert.Contains(item.Fields, f => f.Label == "#1 Victim Sql" && f.Value == ""); + } + + /// An array of SCALARS and a plain OBJECT keep the flat shape alone — they are genuinely paired + /// scalars, which is what fields are for — and an array that mixes objects and scalars does too. + [Fact] + public void ScalarArrays_PlainObjects_AndMixedArrays_CarryNoRecords() + { + var finding = HighCpuFinding(); + finding.DrillDown!["objects"] = new List { "dbo.Orders", "dbo.OrderLines" }; + finding.DrillDown!["spike_peak"] = new { time = "2026-09-18T11:40:00Z", cpu_percent = 96.4 }; + finding.DrillDown!["mixed"] = new List { new { a = 1 }, "loose" }; + var context = FindingMessageFormatter.BuildContext(finding, 1.5); + + var objects = Assert.Single(context.Details, d => d.Heading == "Objects"); + Assert.Empty(objects.Records); + Assert.Equal(new[] { ("#1", "dbo.Orders"), ("#2", "dbo.OrderLines") }, objects.Fields); + + var peak = Assert.Single(context.Details, d => d.Heading == "Spike Peak"); + Assert.Empty(peak.Records); + Assert.Equal(2, peak.Fields.Count); + + var mixed = Assert.Single(context.Details, d => d.Heading == "Mixed"); + Assert.Empty(mixed.Records); + Assert.Equal(new[] { ("#1 A", "1"), ("#2", "loose") }, mixed.Fields); + } + + /* ---------------- Slack ---------------- */ + + /// + /// The shape the reader sees: the divider, then ONE section — *Top Cpu Queries*, and per row a + /// bold *#N · summary* line over the SQL in a triple-backtick block — in row order, top to bottom. + /// No fields array anywhere in the detail: the grid that interleaved is gone. Two blocks where + /// the field grid cost four (heading + 21 fields = 22, in sections of ten). + /// + [Fact] + public void Slack_RendersARecordDetailAsOneSectionOfStackedRecords_AndNoFieldGrid() + { + var finding = HighCpuFinding(); + var context = FindingMessageFormatter.BuildContext(finding, 1.5); + var item = TopCpuItem(finding); + var blocks = SlackBlocks(Slack(context)); + + var at = blocks.FindIndex(b => SectionText(b)?.StartsWith("*Top Cpu Queries*", StringComparison.Ordinal) == true); + Assert.True(at > 0); + Assert.Equal("divider", blocks[at - 1].GetProperty("type").GetString()); + Assert.False(blocks[at].TryGetProperty("fields", out _)); + /* The detail's run ends where the next detail (or the footer) starts: exactly divider + one section. */ + var next = blocks[at + 1].GetProperty("type").GetString(); + Assert.True(next is "divider" or "actions" or "context", $"a second block of type {next} followed the record section"); + + var text = SectionText(blocks[at])!; + Assert.True(text.Length <= TextObjectLimit); + var expected = new StringBuilder("*Top Cpu Queries*"); + foreach (var record in item.Records) + { + expected.Append('\n').Append(string.Create(CultureInfo.InvariantCulture, $"*#{record.Ordinal} · {record.Summary}*")) + .Append("\n```\n").Append(record.Texts[0].Text).Append("\n```"); + } + + Assert.Equal(expected.ToString(), text); + + /* And the reading order is vertical: #1's text comes before #2's summary. */ + var text1 = text.IndexOf(QueryText(0), StringComparison.Ordinal); + var summary2 = text.IndexOf("*#2 · ", StringComparison.Ordinal); + Assert.True(text1 > 0 && summary2 > text1); + Assert.DoesNotContain("*#1 Database:*", text, StringComparison.Ordinal); + } + + /// + /// Records pack: one section while they fit the text ceiling, and a record that would cross it opens the + /// next section with no repeated heading. Three records with 900-character texts fit one section (under + /// 3,000 with the heading); three with 1,400-character texts do not — the third opens a second section + /// — and no record is ever split across two. + /// + [Theory] + [InlineData(900, 1)] + [InlineData(1400, 2)] + public void Slack_PacksRecordsIntoAsFewSectionsAsFit_AndNeverSplitsOne(int textLength, int expectedSections) + { + var finding = HighCpuFinding(3, i => new { database = "D", query_hash = string.Create(CultureInfo.InvariantCulture, $"0x{i}"), query_text = new string((char)('a' + i), textLength) }); + var context = FindingMessageFormatter.BuildContext(finding, 1.5); + var blocks = SlackBlocks(Slack(context)); + + var at = blocks.FindIndex(b => SectionText(b)?.StartsWith("*Top Cpu Queries*", StringComparison.Ordinal) == true); + var sections = new List(); + for (var i = at; i < blocks.Count && SectionText(blocks[i]) is { } t; i++) + { + sections.Add(t); + } + + Assert.Equal(expectedSections, sections.Count); + Assert.All(sections, s => Assert.True(s.Length <= TextObjectLimit, $"{s.Length} chars")); + Assert.Single(sections, s => s.StartsWith("*Top Cpu Queries*", StringComparison.Ordinal)); + /* Each record's whole text is inside exactly one section. */ + for (var i = 0; i < 3; i++) + { + var text = new string((char)('a' + i), textLength); + Assert.Single(sections, s => s.Contains(text, StringComparison.Ordinal)); + } + } + + /// + /// A text past the record's room is cut with the count stated, in the words a cut field uses, and the + /// cut lands on a whole character: a 🔥 astride the boundary is left out whole, no U+FFFD reaches the + /// reader, and kept plus omitted is the text's own character count. The section stays inside every + /// Slack ceiling. No producer emits such a text (the collectors bound them at 500); this is the ceiling's + /// hygiene, pinned so it cannot cut silently. + /// + [Fact] + public void Slack_CutsAnOversizedRecordText_StatingTheCountInCharacters_OnAWholeCharacter() + { + /* Control: an ASCII text of the same length reads the builder's own kept length K from the payload. */ + const int length = 6000; + static string Rendered(string text) + { + var finding = HighCpuFinding(1, _ => new { database = "D", query_hash = "0x1", query_text = text }); + var blocks = SlackBlocks(Slack(FindingMessageFormatter.BuildContext(finding, 1.5))); + var section = Assert.Single(blocks, b => SectionText(b)?.StartsWith("*Top Cpu Queries*", StringComparison.Ordinal) == true); + var s = SectionText(section)!; + Assert.True(s.Length <= TextObjectLimit, $"{s.Length} chars"); + return s; + } + + static (string Kept, int Omitted) Cut(string section) + { + var fence = section.IndexOf("```\n", StringComparison.Ordinal) + 4; + var noteAt = section.IndexOf("... (", fence, StringComparison.Ordinal); + Assert.True(noteAt > fence, "no stated cut in the record text"); + var omitted = int.Parse(section[(noteAt + 5)..section.IndexOf(" more characters", noteAt, StringComparison.Ordinal)], NumberStyles.AllowThousands, CultureInfo.InvariantCulture); + return (section[fence..noteAt], omitted); + } + + var control = Rendered(new string('x', length)); + var (controlKept, controlOmitted) = Cut(control); + Assert.Equal(length, controlKept.Length + controlOmitted); + Assert.Contains(Pointer, control, StringComparison.Ordinal); + var k = controlKept.Length; + + /* The character under test straddles the old cut: units K-1..K. */ + const string fire = "\U0001F525"; + var value = new string('x', k - 1) + fire + new string('x', length - k - 1); + Assert.Equal(length, value.Length); + var section = Rendered(value); + var (kept, omitted) = Cut(section); + + Assert.DoesNotContain('\uFFFD', section); + Assert.Equal(new string('x', k - 1), kept); + Assert.Equal(new StringInfo(value).LengthInTextElements - new StringInfo(kept).LengthInTextElements, omitted); + } + + /// A record with two texts leads each with its label in italics so the reader can tell blocked + /// from blocking; a record with one text carries no label — the heading and the summary already say + /// what it is. + [Fact] + public void Slack_LabelsTextsOnlyWhenARecordHasSeveral() + { + var finding = HighCpuFinding(1, _ => new { database = "D", blocked_sql = "UPDATE t SET x = 1", blocking_sql = "SELECT * FROM t" }); + var section = Assert.Single(SlackBlocks(Slack(FindingMessageFormatter.BuildContext(finding, 1.5))), b => SectionText(b)?.StartsWith("*Top Cpu Queries*", StringComparison.Ordinal) == true); + Assert.Equal("*Top Cpu Queries*\n*#1 · Database: D*\n_Blocked Sql_\n```\nUPDATE t SET x = 1\n```\n_Blocking Sql_\n```\nSELECT * FROM t\n```", SectionText(section)); + + var one = Assert.Single(SlackBlocks(Slack(FindingMessageFormatter.BuildContext(HighCpuFinding(1), 1.5))), b => SectionText(b)?.StartsWith("*Top Cpu Queries*", StringComparison.Ordinal) == true); + Assert.DoesNotContain("_Query Text_", SectionText(one), StringComparison.Ordinal); + } + + /* ---------------- email and Teams ---------------- */ + + /// Email HTML: one data row per record labelled #N with the summary, then a query row + /// per text in the monospace <pre> idiom — six table rows for three queries instead of + /// twenty-one, no #1 Database label anywhere. Plain text: the same list, the text indented under + /// its label. + [Fact] + public void Email_RendersRecordsAsACompactList_InBothBodies() + { + var context = FindingMessageFormatter.BuildContext(HighCpuFinding(), 1.5); + var item = Assert.Single(context.Details, d => d.Heading == "Top Cpu Queries"); + var (html, text) = Email(context); + + Assert.Contains(">#1", html, StringComparison.Ordinal); + Assert.Contains(System.Net.WebUtility.HtmlEncode(item.Records[0].Summary), html, StringComparison.Ordinal); + Assert.Contains(">Query Text", html, StringComparison.Ordinal); + Assert.Contains("
" + System.Net.WebUtility.HtmlEncode(QueryText(0)) + "
", html, StringComparison.Ordinal); + Assert.DoesNotContain("#1 Database", html, StringComparison.Ordinal); + Assert.Equal(3, CountOf(html, ">Query Text")); + + Assert.Contains($" #1: {item.Records[0].Summary}\r\n Query Text:\r\n {QueryText(0)}\r\n", text, StringComparison.Ordinal); + Assert.DoesNotContain("#1 Database", text, StringComparison.Ordinal); + } + + /// Teams: one fact per record — name #N, value the summary over the text in inline code + /// — instead of one per attribute; three facts for three queries. + [Fact] + public void Teams_RendersRecordsAsOneFactPerRow() + { + var context = FindingMessageFormatter.BuildContext(HighCpuFinding(), 1.5); + var item = Assert.Single(context.Details, d => d.Heading == "Top Cpu Queries"); + using var doc = JsonDocument.Parse(Teams(context)); + + var section = Assert.Single(doc.RootElement.GetProperty("sections").EnumerateArray(), + s => s.TryGetProperty("activityTitle", out var t) && t.GetString() == "Top Cpu Queries"); + var facts = section.GetProperty("facts").EnumerateArray().ToList(); + Assert.Equal(3, facts.Count); + Assert.Equal(new[] { "#1", "#2", "#3" }, facts.Select(f => f.GetProperty("name").GetString())); + Assert.Equal(item.Records[0].Summary + " \n`" + QueryText(0) + "`", facts[0].GetProperty("value").GetString()); + } + + /* ---------------- byte identity ---------------- */ + + /// + /// Nothing that is not a record array changes by a byte on any surface. The context carries every other + /// detail shape; each renderer's output equals the pre-#3644 rendering of the same context, restated + /// here from the shapes the renderers have always emitted. The email stamps are normalized because + /// BuildAlertEmail reads the clock. + /// + [Fact] + public void NonRecordDetails_RenderByteIdentically_OnEverySurface() + { + var context = NonRecordContext(); + + /* Slack: the pre-#3644 details loop, verbatim (the #3612 oracle's shapes). */ + var expectedSlack = new List(); + foreach (var detail in context.Details) + { + expectedSlack.Add(new { type = "divider" }); + if (detail.IsCodeBlock) + { + expectedSlack.Add(new { type = "section", text = new { type = "mrkdwn", text = $"*{detail.Heading}*\nSee email or in-app Alert Details for the copy-paste T-SQL." } }); + continue; + } + + if (!string.IsNullOrEmpty(detail.Body)) + { + expectedSlack.Add(new { type = "section", text = new { type = "mrkdwn", text = $"*{detail.Heading}*\n{detail.Body}" } }); + continue; + } + + var fields = new List { new { type = "mrkdwn", text = $"*{detail.Heading}*" } }; + foreach (var (label, value) in detail.Fields) + { + fields.Add(new { type = "mrkdwn", text = $"*{label}:*\n{value}" }); + } + + expectedSlack.Add(new { type = "section", fields }); + } + + var slackRun = JsonSerializer.Serialize(expectedSlack); + Assert.Contains(slackRun[1..^1], Slack(context), StringComparison.Ordinal); + + /* Teams: one fact per field, per detail section. */ + using var teams = JsonDocument.Parse(Teams(context)); + var chain = Assert.Single(teams.RootElement.GetProperty("sections").EnumerateArray(), s => s.TryGetProperty("activityTitle", out var t) && t.GetString() == "Blocking Chain 1"); + Assert.Equal(context.Details[3].Fields.Select(f => (f.Label, f.Value)), + chain.GetProperty("facts").EnumerateArray().Select(f => (f.GetProperty("name").GetString()!, f.GetProperty("value").GetString()!))); + + /* Email: the fields table, label column then value, the SQL-labelled ones in
. */
+        var (html, text) = Email(context);
+        Assert.Contains(">Blocked SQL", html, StringComparison.Ordinal);
+        Assert.Contains("
UPDATE dbo.Orders SET x = 1 WHERE id = 5
", html, StringComparison.Ordinal); + Assert.Contains(" Blocking Chain 1\r\n Database: SalesDB\r\n Blocked SQL: UPDATE dbo.Orders SET x = 1 WHERE id = 5\r\n Blocking SQL: SELECT * FROM dbo.Orders WITH (HOLDLOCK)\r\n Wait Time Ms: 12400\r\n", text, StringComparison.Ordinal); + Assert.Contains(" Objects\r\n #1: dbo.Orders\r\n #2: dbo.OrderLines\r\n", text, StringComparison.Ordinal); + } + + private static int CountOf(string haystack, string needle) + { + var count = 0; + for (var at = haystack.IndexOf(needle, StringComparison.Ordinal); at >= 0; at = haystack.IndexOf(needle, at + needle.Length, StringComparison.Ordinal)) + { + count++; + } + + return count; + } +} diff --git a/Lite.Tests/SlackDetailsSizeTests.cs b/Lite.Tests/SlackDetailsSizeTests.cs index 9454b25e3..be7e5cc43 100644 --- a/Lite.Tests/SlackDetailsSizeTests.cs +++ b/Lite.Tests/SlackDetailsSizeTests.cs @@ -36,9 +36,19 @@ namespace PerformanceMonitorLite.Tests; /// and it goes through — the same call the notification /// service makes — so the items, their order and their field counts are the producer's, not a lookalike. /// The only knob is how many distinct query hashes the drill-downs surface, because the formatter derives -/// one incident item per distinct hash and that is the count production varied on. Measured from source: -/// ten fixed items cost 33 blocks, the head 2, the footer 2, each incident 2 — 37 + 2×incidents — so six -/// incidents (49 blocks) is the last shape that fits and the seventh is the fifty-first block. +/// one incident item per distinct hash and that is the count production varied on. Measured from source +/// at #3612: ten fixed items cost 33 blocks, the head 2, the footer 2, each incident 2 — 37 + 2×incidents — +/// so six incidents (49 blocks) was the last shape that fit and the seventh the fifty-first block. +/// +/// #3644 re-measured the fixed items. Six of the seven drill-downs are arrays of rows, and each +/// used to render as one field per attribute per row — three to five blocks each, 25 of the 33. They now +/// render as record sections (one bold summary line over the row's SQL in a code block, packed into as few +/// sections as fit the text ceiling: one apiece for every shape the producer emits), two blocks each, so +/// the ten fixed items cost 20 and the message is 24 + 2×incidents: thirteen incidents (50 blocks) is the +/// last shape that fits, the fourteenth is the fifty-second block, and the three production pages that +/// #3612 could only deliver with five of their incidents now deliver whole. The fitting arms below pin the +/// new numbers and the past-the-limit arms moved to fourteen and fifteen; the oracle gained a record arm. +/// The engine alerts and every hand-built detail never carry records, and those pins did not move. /// /// The two hazards this suite is built around. A budget is trivially satisfied by dropping /// things silently, so every over-budget arm asserts the omission item names the count AND every dropped @@ -319,6 +329,12 @@ private static (string Kept, int Omitted) CutField(List blocks, str /// The pre-#3612 details loop, verbatim: divider, then a pointer section, a body section, or heading + /// fields in sections of ten. Serialized the way the builder serializes, this is the oracle every /// "fits" arm compares against — the bounding pass must not change a byte of a page that fits. + /// #3644 added one arm, restated here independently of the builder: a detail carrying + /// renders *Heading* and then one unit per record — + /// *#N · summary* over each text in a triple-backtick block, an italic label above the block only + /// when a record has several texts — appended to the current section while it stays inside the + /// text-object ceiling and opening a new section otherwise. No text in the fixture is anywhere near the + /// per-record cap, so the oracle states no cut; pins the cut. /// private static string LegacyDetailBlocksJson(AlertContext context) { @@ -338,6 +354,37 @@ private static string LegacyDetailBlocksJson(AlertContext context) continue; } + if (detail.Records.Count > 0) + { + var section = new StringBuilder($"*{detail.Heading}*"); + foreach (var record in detail.Records) + { + var unit = new StringBuilder(string.Create(CultureInfo.InvariantCulture, $"*#{record.Ordinal} · {record.Summary}*")); + foreach (var (label, text) in record.Texts) + { + if (record.Texts.Count > 1) + { + unit.Append($"\n_{label}_"); + } + + unit.Append($"\n```\n{text}\n```"); + } + + if (section.Length + 1 + unit.Length > TextObjectLimit) + { + blocks.Add(new { type = "section", text = new { type = "mrkdwn", text = section.ToString() } }); + section.Clear().Append(unit); + } + else + { + section.Append('\n').Append(unit); + } + } + + blocks.Add(new { type = "section", text = new { type = "mrkdwn", text = section.ToString() } }); + continue; + } + var fields = new List { new { type = "mrkdwn", text = $"*{detail.Heading}*" } }; foreach (var (label, value) in detail.Fields) { @@ -361,14 +408,21 @@ private static int LegacyBlockCount(AlertContext context) => /* ---------------- the shape that must not change ---------------- */ /// - /// The regression pin, on the real producer: the lost story with three and with six distinct hashes - /// — six is the LAST count that fits (49 blocks) — renders its details exactly as the pre-#3612 loop - /// rendered them, byte for byte, with no omission item. Every analysis page that ever delivered lives - /// on this path, and so does every engine alert. + /// The regression pin, on the real producer: the lost story renders its details exactly as the oracle + /// renders them, byte for byte, with no omission item. Three and six distinct hashes are #3612's fitting + /// shapes (43 and 49 blocks then; 30 and 36 now that the six record-shaped drill-downs cost two blocks + /// each, #3644); nine, ten and eleven are the THREE PRODUCTION PAGES #3612 could deliver only by dropping + /// incidents from the sixth on, and they now deliver whole at 42, 44 and 46; thirteen is the LAST count + /// that fits, on exactly fifty. Every analysis page that ever delivered lives on this path, and so does + /// every engine alert. /// [Theory] - [InlineData(3, 13, 43)] - [InlineData(6, 16, 49)] + [InlineData(3, 13, 30)] + [InlineData(6, 16, 36)] + [InlineData(9, 19, 42)] + [InlineData(10, 20, 44)] + [InlineData(11, 21, 46)] + [InlineData(13, 23, 50)] public void AFiveFactStoryThatFits_RendersItsDetailsByteForByte_WithNoOmission(int distinctHashes, int expectedDetails, int expectedBlocks) { var finding = FiveFactPlanRegression(distinctHashes); @@ -388,19 +442,16 @@ public void AFiveFactStoryThatFits_RendersItsDetailsByteForByte_WithNoOmission(i /* ---------------- the block budget and the stated omission ---------------- */ /// - /// The live failure and the fix. Seven distinct hashes is the first shape past the line (51 blocks - /// under the old loop — the guard asserts it, so the arm can never pass vacuously); nine, ten and - /// eleven are the three production pages (19, 20 and 21 details); fifteen is the most the three - /// hash-bearing drill-downs can surface. Every one delivers inside every ceiling; the finding — the - /// Diagnosis, the advice, the T-SQL pointer and all seven drill-downs — is whole; the incidents kept - /// are the LEADING ones in order; and the omission item names how many were dropped and every one - /// of them by heading. + /// The live failure and the fix. At #3612 seven distinct hashes was the first shape past the line and + /// nine, ten and eleven were the three production pages; since #3644 halved the drill-downs' cost those + /// all fit (the arm above), and the line moved to fourteen — 52 blocks unbudgeted, the guard asserts it + /// so the arm can never pass vacuously — with fifteen the most the three hash-bearing drill-downs can + /// surface. Every one delivers inside every ceiling; the finding — the Diagnosis, the advice, the T-SQL + /// pointer and all seven drill-downs — is whole; the incidents kept are the LEADING ones in order; and + /// the omission item names how many were dropped and every one of them by heading. /// [Theory] - [InlineData(7, 17)] - [InlineData(9, 19)] - [InlineData(10, 20)] - [InlineData(11, 21)] + [InlineData(14, 24)] [InlineData(15, 25)] public void AFiveFactStoryPastTheBlockLimit_DeliversInsideFiftyBlocks_WithTheOmissionStated(int distinctHashes, int expectedDetails) { @@ -408,9 +459,9 @@ public void AFiveFactStoryPastTheBlockLimit_DeliversInsideFiftyBlocks_WithTheOmi var context = FindingMessageFormatter.BuildContext(finding, 1.5); Assert.Equal(expectedDetails, context.Details.Count); - /* Head (2) + footer (2) + the old loop's details must cross the limit, or this arm exercises nothing. */ + /* Head (2) + footer (2) + the unbudgeted details must cross the limit, or this arm exercises nothing. */ Assert.True(4 + LegacyBlockCount(context) > BlockLimit, - $"the pre-#3612 rendering is only {4 + LegacyBlockCount(context)} blocks — this arm no longer reproduces the failure"); + $"the unbudgeted rendering is only {4 + LegacyBlockCount(context)} blocks — this arm no longer reproduces the failure"); var payload = AnalysisPayload(finding, context); using var doc = JsonDocument.Parse(payload); diff --git a/PerformanceMonitor.Notifications/AlertContext.cs b/PerformanceMonitor.Notifications/AlertContext.cs index 345aed411..42b8010c6 100644 --- a/PerformanceMonitor.Notifications/AlertContext.cs +++ b/PerformanceMonitor.Notifications/AlertContext.cs @@ -187,6 +187,36 @@ public class AlertDetailItem public string Heading { get; set; } = ""; public List<(string Label, string Value)> Fields { get; set; } = new(); + /// + /// The same content as , regrouped by the record it came from (#3644) — populated + /// ONLY when the item was flattened from an array of objects (a drill-down's top-N rows: the High CPU + /// card's Top Cpu Queries, Queries At Spike, Top Spilling Queries, Parameter Sensitive Queries, + /// Regressed Queries, Tempdb Breakdown). Empty for every other item. + /// Why a second projection of the same pairs, rather than a replacement. The flat pairs are + /// correct on every surface that lays them out in ONE column: the persisted context_json and the + /// in-app Alert Details grid that reads it, (the persisted + /// detail_text and the redundancy oracle behind ProseForDelivery), the email table, the + /// Teams fact list, PagerDuty's custom_details dictionary, the generic webhook's parts, and the + /// incident-roster match in IncidentDeliveryFilter. Every one of those reads #1 Database, + /// #1 Query Hash, … #2 Database top to bottom and the record stays together. The one + /// surface that does NOT is a Slack section's fields array, which Slack lays out in a two-across + /// grid filled left to right in submission order: seven attributes per query means query #1's text + /// lands beside query #2's hash, #3's database floats beside #2's SQL, and a multi-line text inflates + /// its grid row so the label/value adjacency below it breaks. Read live on a production High CPU page + /// (#3644), the card a person reads while production is on fire. Fields are Slack's tool for short, + /// genuinely PAIRED scalars — the card's own Current Value / Threshold pair renders fine — and + /// the wrong tool for a repeating record. + /// So a record-shaped item carries both: for the flat surfaces, unchanged to + /// the byte, and this list for the renderers that can keep a record together as one visual unit (Slack + /// renders each as a bold summary line over the record's text in a code block; email and Teams render + /// the same compact list). A renderer that knows this list prefers it and skips the fields; one that + /// does not sees exactly what it saw before. Not persisted: it is derivable from the same drill-down + /// rows the fields already persist, nothing re-renders a rehydrated context to Slack, and doubling the + /// drill-down payload of every analysis row for a reader that renders the fields would buy nothing. + /// Producers that build items by hand (the engine alerts, the incident renderer) never populate it. + /// + public List Records { get; set; } = new(); + /// /// Multi-paragraph prose for this item (advice Investigation / Remediation). /// When non-null, renderers emit this as flowing paragraph text rather than @@ -212,6 +242,25 @@ public class AlertDetailItem public RemediationAction? Remediation { get; set; } } +/// +/// One record of a record-shaped detail item (#3644): the row's scalar attributes packed into ONE +/// summary line, and its long text(s) — the query text, a blocked/blocking SQL pair, a CREATE or ALTER +/// statement — carried separately so a renderer can set them in a code block under the summary rather +/// than beside it. +/// is the row's 1-based position (#1, #2, …), the same number +/// the flat fields prefix their labels with, so a reader moving between Slack and the email finds the same +/// row under the same number. is the non-empty scalar properties in the record's OWN +/// order — the collectors write identity first (database, hash, id) and measures after — as +/// Label: value pairs joined by ·, numbers with group separators; it never carries the +/// ordinal, which each renderer sets in its own idiom. is every string property whose +/// name says it is SQL (query_text, blocked_sql, blocking_sql, victim_sql, +/// create_statement, alter_statement), labelled the way the flat field is, in property +/// order; empty texts are not carried. The text is the row's full value as the collector bounded it (500 +/// characters on the query-store drill-downs), NOT the 300-character cut the flat field applies — the +/// renderer that has a ceiling states its own cut. +/// +public sealed record AlertDetailRecord(int Ordinal, string Summary, List<(string Label, string Text)> Texts); + /// /// Serialization DTO for persisting as JSON. /// is a List<(string,string)> diff --git a/PerformanceMonitor.Notifications/AnalysisNotificationService.cs b/PerformanceMonitor.Notifications/AnalysisNotificationService.cs index 1fafeb2ff..180366f0b 100644 --- a/PerformanceMonitor.Notifications/AnalysisNotificationService.cs +++ b/PerformanceMonitor.Notifications/AnalysisNotificationService.cs @@ -9,6 +9,7 @@ using System; using System.Collections.Concurrent; using System.Collections.Generic; +using System.Globalization; using System.Linq; using System.Text; using System.Text.Json; @@ -587,7 +588,7 @@ which the MCP findings output also reads. */ var item = new AlertDetailItem { Heading = Humanize(key) }; try { - FlattenInto(item.Fields, JsonSerializer.SerializeToElement(value)); + FlattenInto(item, JsonSerializer.SerializeToElement(value)); } catch { @@ -729,19 +730,38 @@ private static string[] SplitObjects(string joined) => return sb.ToString(); } + /// How many rows of an array-shaped drill-down reach the alert. The collectors write five; + /// three is what a chat card can carry and still be read as a card. + private const int DrillDownRowLimit = 3; + /// - /// Flattens one drill-down value into label/value field pairs. Arrays are capped at - /// the first 3 elements; nested objects/arrays are rendered as compact JSON. + /// Flattens one drill-down value into the item's label/value field pairs. Arrays are capped at + /// the first elements; nested objects/arrays are rendered as compact JSON. + /// #3644: an array of OBJECTS is also carried as . The + /// flat pairs (#1 Database, #1 Query Hash, … #1 Query Text, #2 Database, …) + /// are right for every surface that lays them out in one column, and wrong for exactly one: Slack's + /// two-across fields grid, where a seven-attribute record never stays together — #1's text lands + /// beside #2's hash, #3's database beside #2's SQL — and the reader reconstructs each query by hunting + /// #N labels across two columns and several screen-heights. Read live on a production High CPU + /// page during a sustained CPU arc. Fields are for PAIRED scalars; a repeating record is a list of rows, + /// so each row is ALSO packed as one record: the ordinal, one summary line of its non-empty scalars in + /// the record's own order ( decides which properties are text), and its + /// text(s) separately, so a renderer can set the summary over the text instead of beside it. The fields + /// are not changed by a byte — the persisted row, the email table, the in-app grid and every other flat + /// reader see what they saw — and the records are only populated when every kept element is an object, + /// so a mixed or scalar array keeps the flat shape alone (a scalar has no record to be). /// - private static void FlattenInto(List<(string Label, string Value)> fields, JsonElement element) + private static void FlattenInto(AlertDetailItem item, JsonElement element) { + var fields = item.Fields; switch (element.ValueKind) { case JsonValueKind.Array: var index = 0; + var allObjects = true; foreach (var child in element.EnumerateArray()) { - if (index >= 3) + if (index >= DrillDownRowLimit) break; index++; @@ -749,12 +769,17 @@ private static void FlattenInto(List<(string Label, string Value)> fields, JsonE { foreach (var prop in child.EnumerateObject()) fields.Add(($"#{index} {Humanize(prop.Name)}", ScalarText(prop.Value))); + item.Records.Add(ToRecord(index, child)); } else { + allObjects = false; fields.Add(($"#{index}", ScalarText(child))); } } + + if (!allObjects) + item.Records.Clear(); break; case JsonValueKind.Object: @@ -768,6 +793,78 @@ private static void FlattenInto(List<(string Label, string Value)> fields, JsonE } } + /// + /// One drill-down row as an (#3644). The summary is the row's non-empty + /// scalar properties, in the row's own property order, as Label: value joined by · — the + /// same humanized label the flat field carries, so a reader moving between surfaces matches them by eye + /// — with numbers group-separated (), because a summary line is read, where + /// the flat field's raw digits are copied. Nulls and empty strings are left out of the summary (the + /// non-AG replica_role is empty on nearly every server, and "Replica Role:" followed by nothing + /// is noise on a line meant to be scanned); the flat field keeps them. Nested values render as the + /// compact JSON the flat field renders. Text properties go to + /// whole — not through : the collectors bound them (LEFT(…, 500)), the email has no + /// ceiling, and the Slack renderer states its own cut. + /// + private static AlertDetailRecord ToRecord(int ordinal, JsonElement row) + { + var summary = new StringBuilder(); + var texts = new List<(string Label, string Text)>(); + foreach (var prop in row.EnumerateObject()) + { + var label = Humanize(prop.Name); + if (IsTextProperty(prop.Name) && prop.Value.ValueKind == JsonValueKind.String) + { + var text = prop.Value.GetString(); + if (!string.IsNullOrEmpty(text)) + texts.Add((label, text)); + continue; + } + + var value = prop.Value.ValueKind switch + { + JsonValueKind.Null => string.Empty, + JsonValueKind.Number => SummaryNumber(prop.Value), + _ => ScalarText(prop.Value) + }; + if (value.Length == 0) + continue; + + if (summary.Length > 0) + summary.Append(" · "); + summary.Append(label).Append(": ").Append(value); + } + + return new AlertDetailRecord(ordinal, summary.ToString(), texts); + } + + /// + /// Whether a drill-down property carries SQL text rather than a scalar (#3644): every text-bearing + /// property the collectors emit is named for it — query_text, blocked_sql / blocking_sql / + /// victim_sql, create_statement / alter_statement — so the suffix is the rule, and + /// a new collector that names its text the same way is grouped correctly without a change here. A + /// hash (query_hash, query_plan_hash) is a scalar and stays in the summary. + /// + private static bool IsTextProperty(string name) => + name.EndsWith("_text", StringComparison.OrdinalIgnoreCase) + || name.EndsWith("_sql", StringComparison.OrdinalIgnoreCase) + || name.EndsWith("_statement", StringComparison.OrdinalIgnoreCase) + || name.Equals("sql", StringComparison.OrdinalIgnoreCase) + || name.Equals("text", StringComparison.OrdinalIgnoreCase); + + /// A JSON number for a summary line: integral values with group separators + /// (3,088,689), fractional ones to at most two decimals (14.2, 221.37), invariant + /// culture. The flat field keeps the raw digits. + private static string SummaryNumber(JsonElement number) + { + if (number.TryGetInt64(out var whole)) + return whole.ToString("N0", CultureInfo.InvariantCulture); + if (number.TryGetDouble(out var real)) + return Math.Floor(real) == real && Math.Abs(real) < 1e15 + ? real.ToString("N0", CultureInfo.InvariantCulture) + : real.ToString("#,##0.##", CultureInfo.InvariantCulture); + return number.GetRawText(); + } + /// Renders a single JSON value as truncated display text. private static string ScalarText(JsonElement element) { diff --git a/PerformanceMonitor.Notifications/EmailTemplateBuilder.cs b/PerformanceMonitor.Notifications/EmailTemplateBuilder.cs index 5f11ab3e0..1c891c516 100644 --- a/PerformanceMonitor.Notifications/EmailTemplateBuilder.cs +++ b/PerformanceMonitor.Notifications/EmailTemplateBuilder.cs @@ -266,6 +266,31 @@ private static void AppendDetailSection(StringBuilder sb, AlertContext? context, } sb.Append(""); } + else if (item.Records.Count > 0) + { + /* #3644: a record-shaped item (a drill-down's top-N rows) as a compact list in the same + table idiom as the fields below — one data row per record, labelled "#N", carrying the + summary line; then one query row per text, labelled as the flat field is. Three rows + with seven attributes are six table rows instead of twenty-one, and the row number a + reader saw on Slack is the row number here. The plain-text body renders the same list. */ + sb.Append(""); + sb.Append($""); + + for (int i = 0; i < item.Records.Count; i++) + { + var record = item.Records[i]; + bool lastRecord = i == item.Records.Count - 1; + AppendDataRow(sb, $"#{record.Ordinal}", record.Summary, lastRecord && record.Texts.Count == 0); + for (int t = 0; t < record.Texts.Count; t++) + { + var (label, text) = record.Texts[t]; + AppendQueryRow(sb, label, text, lastRecord && t == record.Texts.Count - 1); + } + } + + sb.Append("
"); + sb.Append(""); + } else { /* Detail item fields */ @@ -363,6 +388,23 @@ bodies are maintained together — a change to one must touch the other. */ sb.Append($" {para}\r\n"); } } + else if (item.Records.Count > 0) + { + /* #3644: the HTML body's compact record list, in text — the summary on the "#N" line, + each text indented under it, one line per line of the statement. */ + foreach (var record in item.Records) + { + sb.Append($" #{record.Ordinal}: {record.Summary}\r\n"); + foreach (var (label, text) in record.Texts) + { + sb.Append($" {label}:\r\n"); + foreach (var line in text.Replace("\r\n", "\n").Split('\n')) + { + sb.Append($" {line}\r\n"); + } + } + } + } else { foreach (var (label, value) in item.Fields) diff --git a/PerformanceMonitor.Notifications/WebhookAlertService.cs b/PerformanceMonitor.Notifications/WebhookAlertService.cs index ef699011c..c515d885d 100644 --- a/PerformanceMonitor.Notifications/WebhookAlertService.cs +++ b/PerformanceMonitor.Notifications/WebhookAlertService.cs @@ -644,9 +644,43 @@ which skip Fields when Body is present. */ } var itemFacts = new List(); - foreach (var (label, value) in detail.Fields) + if (detail.Records.Count > 0) { - itemFacts.Add(new { name = label, value }); + /* #3644: a record-shaped detail is one fact per ROW — name "#N", value the summary line + over the text(s) in backticks — rather than one fact per attribute (seven per row). The + MessageCard fact list is a name/value table, so the flat fields did not interleave here + the way Slack's grid did; the compact list is the same reading order in a third of the + rows, and the same row numbers a reader sees on Slack and in the email. Fact values + render markdown; a MessageCard supports inline code but no fenced block, and the + text's own backticks would end the span early, so they are replaced with a straight + quote (a T-SQL text never carries one; a PostgreSQL text rarely). */ + foreach (var record in detail.Records) + { + var value = new StringBuilder(record.Summary); + foreach (var (label, text) in record.Texts) + { + if (value.Length > 0) + { + value.Append(" \n"); + } + + if (record.Texts.Count > 1) + { + value.Append('_').Append(label).Append("_ \n"); + } + + value.Append('`').Append(text.Replace('`', '\'')).Append('`'); + } + + itemFacts.Add(new { name = string.Create(CultureInfo.InvariantCulture, $"#{record.Ordinal}"), value = value.ToString() }); + } + } + else + { + foreach (var (label, value) in detail.Fields) + { + itemFacts.Add(new { name = label, value }); + } } itemSections.Add(new @@ -851,6 +885,24 @@ private static void AddSlackFieldSections(List blocks, List fiel /// private const int SlackDetailHeadingLimit = 500; + /// + /// The most one record of a record-shaped detail (, #3644) may + /// occupy in a section's text: its bold summary line plus its code-blocked text(s), markup included. + /// Sized so that a record ALWAYS fits a section beside the detail's heading — the heading is bounded at + /// and wrapped in *…*\n (four more units) — which is + /// what lets pack records greedily and open a fresh section on the + /// first one that does not fit, with no record ever cut between sections. No producer comes near it: + /// the widest drill-down row (Regressed Queries, twelve scalars and a 500-character text) is under + /// 900. A text that would push a record past it is cut with the count stated, through + /// — the same note, in the same characters, as a cut field. + /// + private const int SlackRecordLimit = SlackTextObjectLimit - SlackDetailHeadingLimit - 4; + + /// The most of a record's summary line a Slack section carries before the texts take the rest + /// of the record's room. The producer packs a row's scalars into it — twelve labelled numbers is under + /// 400 — so this never fires on a real page; it exists so the texts' share below is always positive. + private const int SlackRecordSummaryLimit = 1000; + /// /// What the stated-omission item for dropped details costs: its own divider, so it reads as the /// final item rather than as a continuation of whichever detail happened to be last, and one section. @@ -1245,11 +1297,11 @@ is for. */ /// /// One detail's block run: its leading divider, then the shape its kind has always rendered — a /// fixed pointer for remediation T-SQL (never inlined on a chat surface), a *Heading*-led - /// mrkdwn section for a body, or field sections (heading first, then *label:* fields) for - /// everything else. #3612 bounds each text object in the run: the body goes through the prose - /// splitter () with the heading as its header and - /// as its block budget, and every field through - /// . A detail inside every cap renders the pre-#3612 bytes. + /// mrkdwn section for a body, record sections for a record-shaped detail (#3644, below), or field + /// sections (heading first, then *label:* fields) for everything else. #3612 bounds each text + /// object in the run: the body goes through the prose splitter () + /// with the heading as its header and as its block budget, and every + /// field through . A detail inside every cap renders the pre-#3612 bytes. /// private static List RenderSlackDetail(AlertDetailItem detail, int bodyBlockBudget) { @@ -1272,6 +1324,14 @@ private static List RenderSlackDetail(AlertDetailItem detail, int bodyBl return run; } + if (detail.Records.Count > 0) + { + /* #3644: a repeating record is a list of rows, not a grid of pairs. The fields this detail also + carries are the same content, and they are what every other surface renders. */ + AddSlackRecordSections(run, heading, detail.Records); + return run; + } + var detailFields = new List(); detailFields.Add(new { type = "mrkdwn", text = $"*{heading}*" }); @@ -1284,6 +1344,114 @@ private static List RenderSlackDetail(AlertDetailItem detail, int bodyBl return run; } + /// + /// A record-shaped detail (#3644) as mrkdwn sections: *Heading* leads the first, and each record + /// is one visual unit under it — a bold line *#N · Database: … · Total Cpu Ms: 3,088,689 · …* + /// over its text in a triple-backtick code block — stacked top to bottom. Read live on a production High + /// CPU page: the same rows as seven fields per query put query #1's text beside query #2's hash and + /// #3's database beside #2's SQL, because Slack fills a section's fields two across in submission + /// order; a section's text has one column, so a record cannot be pulled apart by its neighbour, + /// and a long text dislocates nothing but itself. The code block also takes the SQL OUT of mrkdwn + /// interpretation, which a field never did (> at a line start is a quote, * pairs + /// bold). + /// + /// Block cost: the divider plus as many sections as the records pack into under the + /// text-object ceiling — ONE for every drill-down the producer emits. #3644 sketched one section per + /// record (three records, three sections); that costs MORE than the fields did for a narrow record — + /// three three-attribute rows are ten fields, one section, two blocks, and would become four — and the + /// #3612 budget is measured in blocks, so a layout fix that widened the card would push incidents off + /// the page to buy readability. Packing does not: records are appended to the current section while + /// the section stays inside , and a record that would cross it opens + /// the next section (no heading on a continuation — a repeated heading would read as a second detail), + /// so the cost is 1 + ceil(records ÷ what fits), at most one section per record and never more + /// than the fields cost. Each record is bounded by , which is what + /// guarantees any record fits a section with the heading; the fixture's six record-shaped drill-downs + /// (3 rows each, texts at the collector's 500) measure 474 to 2,614 characters and one section apiece, + /// so the five-fact page's fixed items fall from 33 blocks to 20 and the message from 37 + 2×incidents + /// to 24 + 2×incidents — thirteen incidents fit where six did, and the three production pages #3612 + /// could deliver only by dropping incidents now deliver whole (SlackDetailsSizeTests). Visually + /// one section holding three records and three sections holding one each are the same stack; Slack + /// puts no rule between consecutive sections. + /// + /// Inside a record. The summary is bounded at with a + /// trailing ellipsis (it never fires; see the constant), then the texts share what is left of the + /// record's room equally, each through so a cut states its count in + /// characters the reader can count, on a whole character (#3622). One text carries no label — the + /// heading says what the drill-down is and the summary says whose row this is; "Query Text" would add + /// nothing — and two or more (a blocking chain's blocked and blocking SQL) are each led by their label + /// in italics so the reader can tell them apart. + /// + private static void AddSlackRecordSections(List run, string heading, List records) + { + var section = new StringBuilder("*").Append(heading).Append('*'); + + void Flush() + { + run.Add(new { type = "section", text = new { type = "mrkdwn", text = section.ToString() } }); + section.Clear(); + } + + foreach (var record in records) + { + var unit = SlackRecordText(record); + if (section.Length + 1 + unit.Length > SlackTextObjectLimit) + { + Flush(); + section.Append(unit); + } + else + { + section.Append('\n').Append(unit); + } + } + + Flush(); + } + + /// One record's unit of text — see — inside + /// . + private static string SlackRecordText(AlertDetailRecord record) + { + var summary = record.Summary.Length <= SlackRecordSummaryLimit + ? record.Summary + : record.Summary[..SlackCutLength(record.Summary, SlackRecordSummaryLimit)] + "..."; + var unit = new StringBuilder(); + unit.Append("*#").Append(record.Ordinal.ToString(CultureInfo.InvariantCulture)); + if (summary.Length > 0) + { + unit.Append(" · ").Append(summary); + } + + unit.Append('*'); + + if (record.Texts.Count == 0) + { + return unit.ToString(); + } + + /* Each text's share of the record's remaining room, after its own markup: the code fence + ("\n```\n" + "\n```", nine units) and, when there are several, the italic label line. */ + var labelled = record.Texts.Count > 1; + var markup = 0; + foreach (var (label, _) in record.Texts) + { + markup += 9 + (labelled ? label.Length + 3 : 0); + } + + var share = Math.Max(0, (SlackRecordLimit - unit.Length - markup) / record.Texts.Count); + foreach (var (label, text) in record.Texts) + { + if (labelled) + { + unit.Append("\n_").Append(label).Append('_'); + } + + unit.Append("\n```\n").Append(SlackBoundedText(string.Empty, text, share)).Append("\n```"); + } + + return unit.ToString(); + } + /// A detail heading bounded to ; unchanged for every /// heading a producer has ever emitted. The cut lands on a whole character (#3622): a heading IS the /// detail's identity, and one ending in half an emoji or a letter shorn of its accent names a @@ -1307,10 +1475,19 @@ private static string SlackHeading(string heading) => /// only be at or below the UTF-16 count the first sizing pass allowed for, so the note can only be /// narrower than the room held for it, and the field stays inside its limit. /// - private static string SlackFieldText(string label, string value) + private static string SlackFieldText(string label, string value) => + SlackBoundedText($"*{label}:*\n", value, SlackFieldTextLimit); + + /// + /// over inside , the cut + /// stated inline when there is one — the mechanics of , which is this at + /// with a *label:* prefix, lifted out (#3644) so a record's + /// code-blocked text (, empty prefix, the record's share) states its cut in + /// exactly the words and the characters a cut field does. One note, one arithmetic, two sites. + /// + private static string SlackBoundedText(string prefix, string value, int limit) { - var prefix = $"*{label}:*\n"; - if (prefix.Length + value.Length <= SlackFieldTextLimit) + if (prefix.Length + value.Length <= limit) { return prefix + value; } @@ -1320,12 +1497,12 @@ static string Note(int omitted) => string.Create(CultureInfo.InvariantCulture, /* The label is producer-controlled and short; a label that alone crowds the field is bounded so the arithmetic below stays positive. */ - if (prefix.Length > SlackFieldTextLimit / 2) + if (prefix.Length > limit / 2) { - prefix = prefix[..SlackCutLength(prefix, SlackFieldTextLimit / 2)]; + prefix = prefix[..SlackCutLength(prefix, limit / 2)]; } - var keep = SlackCutLength(value, Math.Max(0, SlackFieldTextLimit - prefix.Length - Note(value.Length).Length)); + var keep = SlackCutLength(value, Math.Max(0, limit - prefix.Length - Note(value.Length).Length)); /* Counting the omitted characters IS a walk of the omitted tail — the pre-#3622 subtraction counted units, which is the thing that was wrong — so this costs the tail's length, once, on the truncation path only. Every producer bounds its values upstream (the analysis formatter cuts @@ -1859,6 +2036,10 @@ private static AlertContext RedactForWebhook(AlertContext context) { Heading = detail.Heading, Fields = detail.Fields, + /* #3644: the copy is the whole item less its payload. Inert today — this copy is only + serialized, and the serializer carries fields, not records — but a copy that silently + dropped a member is the kind of drift the next renderer of this copy would inherit. */ + Records = detail.Records, Body = detail.Body, IsCodeBlock = false }); From fb686edc6e6fcde5ada19af9d430b3eec4d4a37a Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 17:37:46 -0400 Subject: [PATCH 66/69] A stock PostgreSQL target without pg_wait_sampling has no wait history at all and its empty chart reads as a broken product: the collector grows a service-side pg_stat_activity sampler arm as the floor, the three wait tiers become one connect-time decision, and every wait read discloses its instrument (#3604) (#3645) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * A stock PostgreSQL target without pg_wait_sampling gets no wait accumulation at all, so its empty wait chart reads as a broken product: pg_wait_sampling grows a service-side pg_stat_activity sampler arm as the floor under stock targets, the three wait tiers are one connect-time decision, and every wait read discloses which instrument fed it (#3604) The pg_wait_sampling collector forks on a new connect-time fact, CollectorTargetInfo.HasPgWaitSamplingExtension: with the extension it reads the 10 ms profile as before; without it, it runs one multi-statement command of thirty one-second pg_stat_activity snapshots (each behind pg_stat_clear_snapshot, because the view is cached per transaction) and accumulates the tallies across cycles in its own collector_state, so the rows it writes are cumulative and the existing newest-minus-oldest read differences them unchanged. Which arm ran is recorded in state; get_pg_wait_sampling, get_pg_wait_stats and the Viewer panel disclose it as instrument: engine_cumulative | extension_sampled | service_sampled, with the floor caveat on the service tier. Aurora is gated off pg_wait_sampling (it cannot preload the module; the hourly EXTENSION_MISSING there was a permanent gap dressed as a precondition), and CoveredInsteadBy now points an Aurora caller at pg_wait_stats. The collector moves from hourly to five minutes and is detached from the sequential sweep body beside query_store and plan_correction, because a deliberate 30 s window inline would delay every other collector on the target; the window is 30 s and not the cycle because SweepPressureClassifier sums single-run costs against a 60 s body budget. No migration: same table, same columns, same collector name. The sampler needs nothing beyond pg_monitor. * Three pins encoded 'Aurora is a strict superset of what the PostgreSQL collectors read'; pg_wait_sampling is now the one documented exception (the module cannot be preloaded there), so the kind-axis split, the pre-flight probe line and the compose measure census each name it rather than assume it away; the empty-arm helper moves above the JSON builder's doc block it had split (#3604) * The sampler's pg_sleep argument is formatted under InvariantCulture, as every other number this file puts into SQL or state is: harmless at a whole-second period, a comma-decimal host's pg_sleep(0,5) the day it is not (#3645 review) * The extension arm clears the sampler's tally every cycle it runs: the host persists only PendingState keys, so a tally left alone would sit in collector_state for as long as the extension stayed installed and be resumed as a months-old baseline the day the target fell back to the sampler arm — splicing two eras into one series when the extension's own profile had just been reset (#3645 review) * The service-tier note claimed a series resets when the service restarts; the tally is persisted collector state reloaded every cycle, so it survives one — the note now names what actually starts a series over (cap eviction, arm reversion) and a pin keeps the false claim from returning (#3645 review) * Two Lite pins catch up: the state-declaring set is two collectors now, and get_pg_wait_stats' instrument_note naming aurora_stat_system_waits() is accounted for as PROSE in the Aurora-surface allow-list (#3604) --------- Co-authored-by: erikdarlingdata --- ...ollectorEngineCapabilityDerivationTests.cs | 21 +- .../CollectorStateContractTests.cs | 15 +- .../Darling.Tests/DarlingCliCommandsTests.cs | 34 +- Darling/Darling.Tests/DarlingComposeTests.cs | 20 + .../PgWaitInstrumentDisclosureTests.cs | 200 ++++++++++ .../Darling.Tests/PgWaitSamplerLiveTests.cs | 211 ++++++++++ .../SweepBodyDetachPolicyTests.cs | 12 +- .../DarlingServerConnector.cs | 53 ++- .../DarlingWorker.cs | 20 +- .../Mcp/DarlingMcpInstructions.cs | 4 + .../Mcp/DarlingMcpPgWaitSamplingTools.cs | 77 +++- .../Mcp/DarlingMcpPgWaitTools.cs | 13 +- .../DarlingPgWaitSamplingReader.cs | 43 ++ .../ViewerDataService.Postgres.cs | 7 + .../ViewerServerTab.Postgres.cs | 27 +- Darling/README.md | 2 +- Lite.Tests/AuroraOnlySqlIsGatedTests.cs | 5 + ...aultTraceEventsCollectorDefinitionTests.cs | 10 +- Lite.Tests/PgWaitExclusionParityTests.cs | 3 + Lite.Tests/PgWaitSamplerArmTests.cs | 368 ++++++++++++++++++ .../PgWaitSamplingCollectorDefinitionTests.cs | 3 + .../CollectorEngineCapability.cs | 31 +- .../CollectorScheduleDefaults.cs | 22 +- .../CollectorTargetInfo.cs | 26 ++ .../PgWaitInstrument.cs | 78 ++++ .../PgWaitSamplingCollector.cs | 323 ++++++++++++++- docs/postgres-first-target-runbook.md | 27 +- 27 files changed, 1597 insertions(+), 58 deletions(-) create mode 100644 Darling/Darling.Tests/PgWaitInstrumentDisclosureTests.cs create mode 100644 Darling/Darling.Tests/PgWaitSamplerLiveTests.cs create mode 100644 Lite.Tests/PgWaitSamplerArmTests.cs create mode 100644 PerformanceMonitor.Collectors/PgWaitInstrument.cs diff --git a/Darling/Darling.Tests/CollectorEngineCapabilityDerivationTests.cs b/Darling/Darling.Tests/CollectorEngineCapabilityDerivationTests.cs index ebfb1ab1c..8bac92cd2 100644 --- a/Darling/Darling.Tests/CollectorEngineCapabilityDerivationTests.cs +++ b/Darling/Darling.Tests/CollectorEngineCapabilityDerivationTests.cs @@ -477,12 +477,21 @@ public void TheShippedCatalogSplitsCleanlyOnTheKindAxis() Assert.All(sqlServer, c => Assert.True(CollectorEngineCapability.IsCollectedOnEngineKind(c, MonitoredEngineKind.SqlServer))); Assert.All(postgres, c => Assert.False(CollectorEngineCapability.IsCollectedOnEngineKind(c, MonitoredEngineKind.SqlServer))); - /* Aurora is a strict superset of the surfaces the PostgreSQL collectors read, so every one of them - applies there. This is the half that would break if a new collector were written against something - Aurora removes rather than adds. */ - Assert.All(postgres, c => Assert.True( - CollectorEngineCapability.IsCollectedOnEngineKind(c, MonitoredEngineKind.AuroraPostgres), - $"{c.Name} is reported as a permanent gap on Aurora PostgreSQL")); + /* Aurora is a strict superset of the surfaces the PostgreSQL collectors read - with ONE exception + since #3604, and it is the case this half was written to catch: pg_wait_sampling is written against + something Aurora REMOVES (the ability to preload the module; Aurora permits a fixed list of + libraries and pg_wait_sampling is not on it), so it is a permanent gap there by design, and + CoveredInsteadBy sends an Aurora caller to pg_wait_stats, which reads the engine's own counters. + Named here rather than filtered generically, so a SECOND collector gated off Aurora has to argue + its case in this comment rather than slip through. */ + var auroraGaps = postgres + .Where(c => !CollectorEngineCapability.IsCollectedOnEngineKind(c, MonitoredEngineKind.AuroraPostgres)) + .Select(c => c.Name) + .ToArray(); + Assert.Equal(new[] { "pg_wait_sampling" }, auroraGaps); + Assert.Contains("pg_wait_stats", + CollectorEngineCapability.NotCollectedMessage("aurora-01", 0, MonitoredEngineKind.AuroraPostgres, "pg_wait_sampling"), + StringComparison.Ordinal); /* Stock PostgreSQL is where the Aurora-only surfaces become a real gap (#2532), so the two tokens genuinely differ — counted from the catalog rather than listed, and asserted as a PROPER subset so diff --git a/Darling/Darling.Tests/CollectorStateContractTests.cs b/Darling/Darling.Tests/CollectorStateContractTests.cs index 52ddbf57f..a69b13c3a 100644 --- a/Darling/Darling.Tests/CollectorStateContractTests.cs +++ b/Darling/Darling.Tests/CollectorStateContractTests.cs @@ -110,13 +110,16 @@ public void TheCollectorsDeclaringStateArePinned() collector is a two-host concern rather than a definition-local one. Pinned on the catalog surface both hosts iterate. - ONE of them now: default_trace_events' last-seen trace FILE (#1962). pg_index_bloat's - per-database rotation cursor (#3153) was the other and #3234 retired it — the statistics - estimate covers every index in one statement, so there is no position to resume from. It did - not need host CODE, and neither does the survivor: the wiring below is generic. Enumerated in - the same file that pins that wiring, so a collector newly declaring state lands here first. */ + TWO of them now. default_trace_events' last-seen trace FILE (#1962); pg_index_bloat's + per-database rotation cursor (#3153) was the other until #3234 retired it — the statistics + estimate covers every index in one statement, so there is no position to resume from. Then + pg_wait_sampling (#3604): the arm that ran (its instrument token, which the reads disclose) and, + on the service-sampler arm, the cumulative tally the next cycle adds to — state a MAX() over the + table cannot recover, since the table holds only the tally's last written value per key. Neither + needed host CODE: the wiring below is generic. Enumerated in the same file that pins that wiring, + so a collector newly declaring state lands here first. */ Assert.Equal( - new[] { "default_trace_events" }, + new[] { "default_trace_events", "pg_wait_sampling" }, CollectorCatalog.All .Where(c => c.StateKeys.Count > 0) .Select(c => c.Name) diff --git a/Darling/Darling.Tests/DarlingCliCommandsTests.cs b/Darling/Darling.Tests/DarlingCliCommandsTests.cs index e6ec32957..1cc5bbf4e 100644 --- a/Darling/Darling.Tests/DarlingCliCommandsTests.cs +++ b/Darling/Darling.Tests/DarlingCliCommandsTests.cs @@ -145,16 +145,40 @@ public void FormatProbeLine_PostgresTarget_ReportsPostgresFactsAndNoSqlServerOne Assert.DoesNotContain("Unknown (0)", line, StringComparison.Ordinal); } - /// An Aurora writer clears every gate, so the count says so rather than listing nothing. + /// + /// An Aurora writer clears every gate but ONE, and the line names it. Until #3604 Aurora was a strict + /// superset of what the PostgreSQL collectors read and the line said "all N apply"; pg_wait_sampling + /// is now gated off Aurora — the engine cannot preload the module and has pg_wait_stats instead — + /// so the pre-flight is where an operator first sees that one collector, by name, does not run there. + /// Counted from the catalog and the collectors' own gates rather than hard-coded, so a second Aurora + /// gap shows up here as a changed count rather than a silently passing pin. + /// [Fact] - public void FormatProbeLine_AuroraWriter_ReportsEveryPostgresCollectorApplies() + public void FormatProbeLine_AuroraWriter_ReportsEveryPostgresCollectorButTheOneAuroraCannotHave() { - var expected = CollectorCatalog.All.Count(d => d.TargetEngine == CollectorTargetEngine.PostgreSql); + var target = PostgresProbe().ToTargetInfo(); + var postgres = CollectorCatalog.All.Where(d => d.TargetEngine == CollectorTargetEngine.PostgreSql).ToList(); + var skipped = postgres.Where(d => !CollectorCatalog.AppliesTo(d, target)).Select(d => d.Name).ToList(); + Assert.Equal(new[] { PgWaitSamplingCollector.Instance.Name }, skipped); var line = DarlingCliCommands.FormatProbeLine("aurora-writer", PostgresProbe()); - Assert.Contains($"all {expected} PostgreSQL collectors apply", line, StringComparison.Ordinal); - Assert.DoesNotContain("skipped", line, StringComparison.Ordinal); + Assert.Contains($"{postgres.Count - 1} of {postgres.Count} PostgreSQL collectors apply", line, StringComparison.Ordinal); + Assert.Contains("skipped: pg_wait_sampling", line, StringComparison.Ordinal); + } + + /// The line an Aurora writer used to get, every PostgreSQL collector applying, is now the STOCK + /// writer's with the extension present — the shape #3604 made the finest stock tier. + [Fact] + public void FormatProbeLine_StockWriterWithTheExtension_ReportsEveryPostgresCollectorApplies() + { + var expected = CollectorCatalog.All.Count(d => d.TargetEngine == CollectorTargetEngine.PostgreSql); + var probe = PostgresProbe() with { IsAurora = false, HasPgWaitSamplingExtension = true }; + + /* pg_wait_stats and pg_cpu_utilization are Aurora-only, so a stock target skips those two instead. */ + var line = DarlingCliCommands.FormatProbeLine("stock-writer", probe); + Assert.Contains($"{expected - 2} of {expected} PostgreSQL collectors apply", line, StringComparison.Ordinal); + Assert.DoesNotContain("pg_wait_sampling", line, StringComparison.Ordinal); } /// diff --git a/Darling/Darling.Tests/DarlingComposeTests.cs b/Darling/Darling.Tests/DarlingComposeTests.cs index 6a0c9cb99..21ea8af89 100644 --- a/Darling/Darling.Tests/DarlingComposeTests.cs +++ b/Darling/Darling.Tests/DarlingComposeTests.cs @@ -297,6 +297,16 @@ SQL Server target when the engine half is skipped. */ PostgresMajorVersion = 17, PostgresVersionNum = 170_005, IsInRecovery = false, }; + /* #3604: pg_wait_sampling is gated OFF Aurora (the module cannot be preloaded there; pg_wait_stats is + the instrument on that engine), so its measures surface for a stock writer instead. Every other + PostgreSQL measure still clears the Aurora writer, and the two shapes together still cover the + whole PostgreSQL measure set - which is the property this test exists for. */ + var stockWriterWithExtension = new CollectorTargetInfo + { + Engine = CollectorTargetEngine.PostgreSql, IsAurora = false, HasPgWaitSamplingExtension = true, + PostgresMajorVersion = 17, PostgresVersionNum = 170_005, IsInRecovery = false, + }; + var pg = PostgresMeasures(); Assert.NotEmpty(pg); foreach (var measure in pg) @@ -308,6 +318,16 @@ SQL Server target when the engine half is skipped. */ $"pg measure '{measure.Key}' source '{measure.SourceTable}' engine gate must exclude SQL Server."); Assert.False(CollectorCatalog.AppliesTo(collector!, sqlServer), $"pg measure '{measure.Key}' source '{measure.SourceTable}' must not apply to a SQL Server target."); + + if (string.Equals(measure.SourceTable, PgWaitSamplingCollector.Instance.TargetTable, StringComparison.Ordinal)) + { + Assert.False(CollectorCatalog.AppliesTo(collector!, auroraWriter), + $"pg measure '{measure.Key}' reads pg_wait_sampling, which #3604 gated off Aurora - it must not surface there."); + Assert.True(CollectorCatalog.AppliesTo(collector!, stockWriterWithExtension), + $"pg measure '{measure.Key}' source '{measure.SourceTable}' must apply to a stock writer with the extension."); + continue; + } + Assert.True(CollectorCatalog.AppliesTo(collector!, auroraWriter), $"pg measure '{measure.Key}' source '{measure.SourceTable}' must apply to a modern Aurora writer."); } diff --git a/Darling/Darling.Tests/PgWaitInstrumentDisclosureTests.cs b/Darling/Darling.Tests/PgWaitInstrumentDisclosureTests.cs new file mode 100644 index 000000000..4d29c0cbb --- /dev/null +++ b/Darling/Darling.Tests/PgWaitInstrumentDisclosureTests.cs @@ -0,0 +1,200 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.ComponentModel; +using System.Linq; +using System.Reflection; +using System.Text.Json; +using ModelContextProtocol.Server; +using PerformanceMonitor.Collectors; +using PerformanceMonitor.Darling.Service; +using PerformanceMonitor.Darling.Service.Mcp; +using PerformanceMonitor.Darling.Storage; +using Xunit; + +namespace Darling.Tests; + +/// +/// The read side of #3604: every surface that serves PostgreSQL waits discloses which of the three +/// instruments fed it, the service tier carries its floor caveat, and the service-side pieces the sampler +/// arm depends on (the connector's probe, the worker's detach, the connection names it excludes) agree with +/// the collector's declarations. +/// +public sealed class PgWaitInstrumentDisclosureTests +{ + private static readonly DateTime T0 = new(2026, 9, 18, 12, 0, 0, DateTimeKind.Utc); + + private static DarlingPgWaitSamplingReader.PgWaitSamplingRow Row(string type, string evt, long samples, int periodMs) => new( + EventType: type, Event: evt, QueryId: 7, SampleCount: samples, EstimatedWaitMs: samples * periodMs, + BackendCount: 1, CounterReset: false, CaptureTime: T0); + + private static JsonElement Build(DarlingPgWaitSamplingReader.WaitInstrumentState? instrument) => + JsonDocument.Parse(DarlingMcpPgWaitSamplingTools.BuildWaitSamplingJson( + "srv", 24, + new DarlingPgWaitSamplingReader.PgWaitSamplingPage([Row("Lock", "relation", 30, 1000)], 30), + limit: 20, instrument)).RootElement; + + /* ───────────────────────── the sampled read ───────────────────────── */ + + [Fact] + public void ServiceTier_IsNamedAndCarriesTheFloorCaveat() + { + var root = Build(new DarlingPgWaitSamplingReader.WaitInstrumentState(PgWaitInstrument.ServiceSampled, T0)); + + Assert.Equal("service_sampled", root.GetProperty("instrument").GetString()); + Assert.Equal(PgWaitInstrument.ServiceSampledCaveat, root.GetProperty("instrument_note").GetString()); + Assert.Contains("FLOOR", root.GetProperty("instrument_note").GetString(), StringComparison.Ordinal); + Assert.Contains("distinct backends seen in the LAST window", root.GetProperty("note").GetString(), StringComparison.Ordinal); + Assert.Equal(T0, root.GetProperty("instrument_recorded_at").GetDateTime().ToUniversalTime()); + + /* #3645 review: the tally is persisted collector state, reloaded every cycle, so a SERVICE restart does + not reset the series — the note must say what does (cap eviction, arm reversion) and not claim what + does not. Head 3 shipped the false claim; this is what stops it coming back. */ + var note = root.GetProperty("note").GetString()!; + Assert.Contains("SURVIVES a service restart", note, StringComparison.Ordinal); + Assert.DoesNotContain("resets when the service restarts", note, StringComparison.Ordinal); + Assert.Contains("500-key cap", note, StringComparison.Ordinal); + } + + [Fact] + public void ExtensionTier_IsNamedWithoutTheFloorCaveat() + { + var root = Build(new DarlingPgWaitSamplingReader.WaitInstrumentState(PgWaitInstrument.ExtensionSampled, T0)); + + Assert.Equal("extension_sampled", root.GetProperty("instrument").GetString()); + Assert.Contains("10 ms", root.GetProperty("instrument_note").GetString(), StringComparison.Ordinal); + Assert.DoesNotContain("FLOOR", root.GetProperty("instrument_note").GetString(), StringComparison.Ordinal); + Assert.DoesNotContain("LAST window", root.GetProperty("note").GetString(), StringComparison.Ordinal); + } + + /// A store written before the arm existed, or a token a future arm introduces, reads as unknown + /// — never as a grain the caller might act on. + [Fact] + public void NoRecordedInstrument_OrAnUnknownToken_ReadsAsUnknown() + { + Assert.Equal("unknown", Build(null).GetProperty("instrument").GetString()); + Assert.Equal(JsonValueKind.Null, Build(null).GetProperty("instrument_recorded_at").ValueKind); + + var foreign = Build(new DarlingPgWaitSamplingReader.WaitInstrumentState("kernel_traced", T0)); + Assert.Equal("unknown", foreign.GetProperty("instrument").GetString()); + Assert.Contains("profile_period_ms", foreign.GetProperty("instrument_note").GetString(), StringComparison.Ordinal); + } + + [Fact] + public void TheEmptyArm_StatesTheFloorOnlyOnTheServiceTier() + { + Assert.Contains(PgWaitInstrument.ServiceSampledCaveat, + DarlingMcpPgWaitSamplingTools.DescribeInstrumentForEmpty( + new DarlingPgWaitSamplingReader.WaitInstrumentState(PgWaitInstrument.ServiceSampled, T0)), StringComparison.Ordinal); + Assert.Equal(string.Empty, + DarlingMcpPgWaitSamplingTools.DescribeInstrumentForEmpty( + new DarlingPgWaitSamplingReader.WaitInstrumentState(PgWaitInstrument.ExtensionSampled, T0))); + Assert.Contains("No collection cycle has recorded", DarlingMcpPgWaitSamplingTools.DescribeInstrumentForEmpty(null), StringComparison.Ordinal); + } + + /// The instrument is read off the collector's own state row, under the collector's own name and + /// key — so a rename of either fails here rather than silently reading nothing forever. + [Fact] + public void TheInstrumentSqlReadsTheCollectorsOwnStateRow() + { + Assert.Contains("FROM collector_state", DarlingPgWaitSamplingReader.InstrumentSql, StringComparison.Ordinal); + Assert.Contains($"collector_name = '{PgWaitSamplingCollector.Instance.Name}'", DarlingPgWaitSamplingReader.InstrumentSql, StringComparison.Ordinal); + Assert.Contains($"state_key = '{PgWaitSamplingCollector.InstrumentStateKey}'", DarlingPgWaitSamplingReader.InstrumentSql, StringComparison.Ordinal); + } + + /* ───────────────────────── the Aurora read ───────────────────────── */ + + [Fact] + public void TheAuroraRead_DisclosesEngineCumulative() + { + var json = DarlingMcpPgWaitTools.BuildWaitStatsJson( + "srv", 24, + new DarlingPgWaitReader.PgWaitStatsPage( + [new DarlingPgWaitReader.PgWaitRow("IO", "DataFileRead", TotalWaits: 10, TotalWaitTimeMs: 5_000, AvgWaitTimeMs: 500)], + 5_000), + limit: 20); + var root = JsonDocument.Parse(json).RootElement; + + Assert.Equal("engine_cumulative", root.GetProperty("instrument").GetString()); + Assert.Contains("get_pg_wait_sampling", root.GetProperty("instrument_note").GetString(), StringComparison.Ordinal); + } + + /* ───────────────────────── descriptions and instructions ───────────────────────── */ + + private static string ToolDescription(Type toolType, string toolName) => + toolType.GetMethods(BindingFlags.Public | BindingFlags.Static) + .Single(m => m.GetCustomAttribute()?.Name == toolName) + .GetCustomAttribute()!.Description; + + [Fact] + public void BothToolDescriptions_NameTheThreeTiers() + { + var sampling = ToolDescription(typeof(DarlingMcpPgWaitSamplingTools), "get_pg_wait_sampling"); + Assert.Contains("extension_sampled", sampling, StringComparison.Ordinal); + Assert.Contains("service_sampled", sampling, StringComparison.Ordinal); + Assert.Contains("FLOOR", sampling, StringComparison.Ordinal); + Assert.Contains("Aurora native > pg_wait_sampling extension > service sampler", sampling, StringComparison.Ordinal); + /* The old opening claimed the extension as the ONLY source; it is one of two now. */ + Assert.DoesNotContain("from the pg_wait_sampling extension. This is", sampling, StringComparison.Ordinal); + + var stats = ToolDescription(typeof(DarlingMcpPgWaitTools), "get_pg_wait_stats"); + Assert.Contains("engine_cumulative", stats, StringComparison.Ordinal); + Assert.Contains("get_pg_wait_sampling", stats, StringComparison.Ordinal); + } + + [Fact] + public void TheInstructions_NameTheThreeTiersAndTheOrder() + { + var text = DarlingMcpInstructions.Build(DarlingPeerDirectory.Snapshot.Empty); + Assert.Contains("`engine_cumulative`", text, StringComparison.Ordinal); + Assert.Contains("`extension_sampled`", text, StringComparison.Ordinal); + Assert.Contains("`service_sampled`", text, StringComparison.Ordinal); + Assert.Contains("Aurora native > the `pg_wait_sampling` extension > the service-side sampler", text, StringComparison.Ordinal); + Assert.Contains("FLOOR, not parity", text, StringComparison.Ordinal); + } + + /* ───────────────────────── the service-side seams ───────────────────────── */ + + /// The sampler excludes this service's own backends by application_name, and the names it + /// excludes must be the names the connector presents — the Collectors assembly cannot reference the + /// service, so the agreement is pinned rather than referenced. + [Fact] + public void TheExcludedApplicationNames_AreTheOnesTheConnectorPresents() + { + Assert.Equal(MonitoredServerConnection.RemediationApplicationName, PgWaitSamplingCollector.ServiceRemediationApplicationName); + + var connector = RepoFile.ReadRepoFile("Darling", "PerformanceMonitor.Darling.Service", "MonitoredServerConnection.cs"); + Assert.Contains($"ApplicationName = \"{PgWaitSamplingCollector.ServiceApplicationName}\"", connector, StringComparison.Ordinal); + } + + [Fact] + public void TheConnectProbe_AsksPgExtensionForTheModule_AndFlowsToTheTargetInfo() + { + Assert.Contains("pg_extension", DarlingServerConnector.PostgresWaitSamplingProbeQueryText, StringComparison.Ordinal); + Assert.Contains("'pg_wait_sampling'", DarlingServerConnector.PostgresWaitSamplingProbeQueryText, StringComparison.Ordinal); + + var probe = new ConnectionProbeResult( + Success: true, MajorVersion: 0, EngineEdition: 0, EngineEditionDescription: null, + IsAzureSqlDb: false, IsAzureManagedInstance: false, IsAwsRds: false, HasMsdbAccess: true, Error: null, + Engine: CollectorTargetEngine.PostgreSql, PostgresMajorVersion: 18, PostgresVersionNum: 180001, + HasPgWaitSamplingExtension: true); + Assert.True(probe.ToTargetInfo().HasPgWaitSamplingExtension); + Assert.False((probe with { HasPgWaitSamplingExtension = false }).ToTargetInfo().HasPgWaitSamplingExtension); + } + + /// Detached from the sequential body, behind the generic gate, on a tier the detach policy + /// allows — see SweepBodyDetachPolicyTests for the invariant; this pins the predicate itself. + [Fact] + public void TheCollectorIsDetached_ByItsOwnDeclaredName() + { + Assert.True(DarlingWorker.IsPgWaitSamplingCollector(PgWaitSamplingCollector.Instance.Name)); + Assert.True(DarlingWorker.IsPgWaitSamplingCollector("PG_WAIT_SAMPLING")); + Assert.False(DarlingWorker.IsPgWaitSamplingCollector("pg_wait_stats")); + } +} diff --git a/Darling/Darling.Tests/PgWaitSamplerLiveTests.cs b/Darling/Darling.Tests/PgWaitSamplerLiveTests.cs new file mode 100644 index 000000000..055805be6 --- /dev/null +++ b/Darling/Darling.Tests/PgWaitSamplerLiveTests.cs @@ -0,0 +1,211 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Diagnostics; +using System.Linq; +using System.Text.Json; +using System.Threading; +using System.Threading.Tasks; +using Microsoft.Extensions.Logging.Abstractions; +using Npgsql; +using PerformanceMonitor.Collectors; +using PerformanceMonitor.Common; +using PerformanceMonitor.Darling.Service; +using PerformanceMonitor.Darling.Service.Mcp; +using PerformanceMonitor.Darling.Storage; +using Xunit; + +namespace Darling.Tests; + +/// +/// The service-sampler arm of pg_wait_sampling (#3604) run for real: the live store doubles as a +/// stock PostgreSQL TARGET without the extension, a lock wait and a CPU burner are held open on it, and the +/// real runs one cycle. What must come out the other end: rows in +/// pg_wait_sampling at the sampler's period, a Lock series and a CPU series among them, +/// the collector's state row saying service_sampled, and the read disclosing it. +/// +/// Gated on DARLING_TEST_PG like every live class. Takes ~30 s by construction — the window IS +/// the test — so it is one cycle, not two; the cumulative-across-cycles property is proven with a fixture +/// reader in Lite.Tests.PgWaitSamplerArmTests, where it costs nothing. +/// +[Collection("live-postgres")] +public sealed class PgWaitSamplerLiveTests +{ + private const int ServerId = 360_4; + + [Fact] + public async Task OneCycleAgainstAStockTarget_LandsServiceSampledRowsTheReadDiscloses() + { + var connectionString = Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + Assert.SkipWhen(string.IsNullOrWhiteSpace(connectionString), + "Set DARLING_TEST_PG to a Postgres connection string to run the live sampler test."); + + var ct = TestContext.Current.CancellationToken; + await using var postgres = NpgsqlDataSource.Create(connectionString!); + + /* The TARGET is the store's own instance, reached with a distinct application_name so the sampler's + self-exclusion (by the service's names) does not hide the workload, and so the workload's own + backends are attributable. */ + var targetBuilder = new NpgsqlConnectionStringBuilder(connectionString) { ApplicationName = "pm3604-target", Pooling = false }; + var workloadBuilder = new NpgsqlConnectionStringBuilder(connectionString) { ApplicationName = "pm3604-workload", Pooling = false }; + + var bodySucceeded = false; + try + { + await using (var setup = new NpgsqlConnection(workloadBuilder.ConnectionString)) + { + await setup.OpenAsync(ct); + await using var cmd = new NpgsqlCommand("CREATE TABLE IF NOT EXISTS pm3604_lock_target (id int)", setup); + await cmd.ExecuteNonQueryAsync(ct); + } + + /* Workload 1: a lock wait. Holder takes ACCESS EXCLUSIVE and sits; the waiter blocks on it for the + whole window — wait_event_type Lock, wait_event relation. */ + await using var holder = new NpgsqlConnection(workloadBuilder.ConnectionString); + await holder.OpenAsync(ct); + await using var holderTx = await holder.BeginTransactionAsync(ct); + await using (var lockCmd = new NpgsqlCommand("LOCK TABLE pm3604_lock_target IN ACCESS EXCLUSIVE MODE", holder, holderTx)) + { + await lockCmd.ExecuteNonQueryAsync(ct); + } + + using var workloadStop = new CancellationTokenSource(); + var waiter = Task.Run(async () => + { + await using var c = new NpgsqlConnection(workloadBuilder.ConnectionString); + await c.OpenAsync(workloadStop.Token); + await using var q = new NpgsqlCommand("SELECT count(*) FROM pm3604_lock_target", c) { CommandTimeout = 120 }; + try { await q.ExecuteScalarAsync(workloadStop.Token); } catch (OperationCanceledException) { } + }, CancellationToken.None); + + /* Workload 2: CPU. A backend on processor with no wait event — the arm's CPU/Running row. */ + var burner = Task.Run(async () => + { + await using var c = new NpgsqlConnection(workloadBuilder.ConnectionString); + await c.OpenAsync(workloadStop.Token); + while (!workloadStop.IsCancellationRequested) + { + await using var q = new NpgsqlCommand("SELECT count(*) FROM generate_series(1, 20000000)", c) { CommandTimeout = 120 }; + try { await q.ExecuteScalarAsync(workloadStop.Token); } catch (OperationCanceledException) { } + } + }, CancellationToken.None); + + await Task.Delay(TimeSpan.FromSeconds(2), ct); + + var runtime = new ServerRuntime + { + Config = new MonitoredServer { Name = "pm3604-stock", Host = targetBuilder.Host ?? "localhost", Engine = "postgres" }, + ConnectionString = targetBuilder.ConnectionString, + Target = new CollectorTargetInfo + { + Engine = CollectorTargetEngine.PostgreSql, + PostgresMajorVersion = 18, + PostgresVersionNum = 180000, + /* The rig has no pg_wait_sampling: the connect probe would say false, and so does this. */ + HasPgWaitSamplingExtension = false, + }, + StorageName = "pm3604-stock", + ServerId = ServerId, + }; + + var runner = new DarlingCollectorRunner(postgres, new CollectorDeltaCalculator(), NullLogger.Instance); + + var wall = Stopwatch.StartNew(); + var result = await runner.RunAsync(PgWaitSamplingCollector.Instance, runtime, ct); + wall.Stop(); + + workloadStop.Cancel(); + await holderTx.RollbackAsync(CancellationToken.None); + await Task.WhenAll(waiter, burner); + + /* 1. The run itself: rows landed, the window was the window. */ + Assert.True(result.Rows > 0, $"the sampler wrote no rows; note={result.HostNote}"); + var expectedWindowMs = (PgWaitSamplingCollector.SamplerSnapshotsPerCycle - 1) * PgWaitSamplingCollector.SamplerPeriodMs; + Assert.InRange(wall.ElapsedMilliseconds, expectedWindowMs, expectedWindowMs + 30_000); + + /* 2. The rows: at the sampler's period, with the two workloads visible. */ + var rows = new List<(string Type, string Event, long Samples, int PeriodMs, int Backends)>(); + await using (var read = postgres.CreateCommand( + "SELECT event_type, event, sample_count, profile_period_ms, backend_count FROM pg_wait_sampling WHERE server_id = $1")) + { + read.Parameters.AddWithValue(ServerId); + await using var reader = await read.ExecuteReaderAsync(ct); + while (await reader.ReadAsync(ct)) + { + rows.Add((reader.GetString(0), reader.GetString(1), reader.GetInt64(2), reader.GetInt32(3), reader.GetInt32(4))); + } + } + + Assert.NotEmpty(rows); + Assert.All(rows, r => Assert.Equal(PgWaitSamplingCollector.SamplerPeriodMs, r.PeriodMs)); + var lockRow = Assert.Single(rows, r => r.Type == "Lock" && r.Event == "relation"); + /* Held for the whole window, so seen in nearly every snapshot; allow for the first snapshot racing the waiter. */ + Assert.InRange(lockRow.Samples, PgWaitSamplingCollector.SamplerSnapshotsPerCycle / 2, PgWaitSamplingCollector.SamplerSnapshotsPerCycle); + Assert.Equal(1, lockRow.Backends); + Assert.Contains(rows, r => r.Type == "CPU" && r.Event == "Running"); + /* The shared exclusions hold on this arm too: the holder is idle in transaction (Client), and no + background sleeper (Activity/Timeout) is counted. */ + Assert.DoesNotContain(rows, r => PgWaitStatsCollector.IgnoredWaitTypes.Contains(r.Type)); + + /* 3. The instrument, recorded by the collector and read back by the reader the tools use. */ + var instrument = await DarlingPgWaitSamplingReader.GetWaitInstrumentAsync(postgres, ServerId, ct); + Assert.NotNull(instrument); + Assert.Equal(PgWaitInstrument.ServiceSampled, instrument.Instrument); + + var tally = await ReadStateAsync(postgres, PgWaitSamplingCollector.TallyStateKey, ct); + Assert.Equal(lockRow.Samples, PgWaitSamplingCollector.ParseTally(tally)[("Lock", "relation", 0)]); + + /* 4. The read discloses it. */ + var page = await DarlingPgWaitSamplingReader.GetPgWaitSamplingPageAsync( + postgres, ServerId, DateTime.UtcNow.AddHours(-1), DateTime.UtcNow.AddMinutes(1), 21, ct); + var json = JsonDocument.Parse(DarlingMcpPgWaitSamplingTools.BuildWaitSamplingJson("pm3604-stock", 1, page, 20, instrument)).RootElement; + Assert.Equal("service_sampled", json.GetProperty("instrument").GetString()); + Assert.Contains("FLOOR", json.GetProperty("instrument_note").GetString(), StringComparison.Ordinal); + + /* Reported, not asserted: the measured cost of one cycle on this rig. */ + Console.WriteLine( + $"pg_wait_sampling service arm: wall {wall.ElapsedMilliseconds} ms, sql {result.SqlMs} ms, store {result.StorageMs} ms, " + + $"{result.Rows} rows, {rows.Count} series; Lock/relation {lockRow.Samples}/{PgWaitSamplingCollector.SamplerSnapshotsPerCycle} snapshots"); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(connectionString!, bodySucceeded, async (cleanup, cleanupCt) => + { + /* One statement per command: Npgsql does not bind positional parameters across a + multi-statement command. */ + foreach (var sql in new[] + { + "DELETE FROM pg_wait_sampling WHERE server_id = $1", + "DELETE FROM collector_state WHERE server_id = $1", + "DELETE FROM collection_log WHERE server_id = $1", + }) + { + await using var command = new NpgsqlCommand(sql, cleanup); + command.Parameters.AddWithValue(ServerId); + await command.ExecuteNonQueryAsync(cleanupCt); + } + + await using var drop = new NpgsqlCommand("DROP TABLE IF EXISTS pm3604_lock_target", cleanup); + await drop.ExecuteNonQueryAsync(cleanupCt); + }); + } + } + + private static async Task ReadStateAsync(NpgsqlDataSource postgres, string key, CancellationToken ct) + { + await using var command = postgres.CreateCommand( + "SELECT state_value FROM collector_state WHERE server_id = $1 AND collector_name = 'pg_wait_sampling' AND state_key = $2"); + command.Parameters.AddWithValue(ServerId); + command.Parameters.AddWithValue(key); + return await command.ExecuteScalarAsync(ct) as string; + } +} diff --git a/Darling/Darling.Tests/SweepBodyDetachPolicyTests.cs b/Darling/Darling.Tests/SweepBodyDetachPolicyTests.cs index 37c25297d..f96b59696 100644 --- a/Darling/Darling.Tests/SweepBodyDetachPolicyTests.cs +++ b/Darling/Darling.Tests/SweepBodyDetachPolicyTests.cs @@ -39,6 +39,13 @@ namespace Darling.Tests; /// has zero fanout at p90 11,964ms. A fanout-derived split would detach the cheap collector and leave /// the expensive one starving the tier. See #2840. /// +/// The third member is detached for a different reason (#3604). pg_wait_sampling's +/// service-sampler arm holds its connection for thirty one-second pg_stat_activity snapshots per cycle +/// BY DESIGN — a deterministic 30 s run, not a bimodal tail — and awaited inline that would delay every +/// other collector on a stock PostgreSQL target by half a minute every five. The criterion generalises: a +/// single-run cost that would starve the fast tier if it sat in the body, whether measured or designed. It +/// sits on the five-minute tier, so the one-minute-tier invariant below holds for it as for the others. +/// /// The 4.5x evidence. use2 runs the same Balanced preset with Query Store dead since /// 2026-08-17 17:36. Its query_stats delivered cadence stepped from 4.69-9.02 min (Query Store /// live) to 1.44-1.57 min (dead) the following day, and held there for two weeks. @@ -46,10 +53,11 @@ namespace Darling.Tests; public sealed class SweepBodyDetachPolicyTests { /// The collectors #2700/#2717 detach, by the criterion documented on this class. - private static readonly string[] ExpectedDetached = { "query_store", "plan_correction" }; + private static readonly string[] ExpectedDetached = { "query_store", "plan_correction", "pg_wait_sampling" }; private static bool IsDetached(string name) => - DarlingWorker.IsQueryStoreCollector(name) || DarlingWorker.IsPlanCorrectionCollector(name); + DarlingWorker.IsQueryStoreCollector(name) || DarlingWorker.IsPlanCorrectionCollector(name) + || DarlingWorker.IsPgWaitSamplingCollector(name); /// /// The invariant that makes detaching safe, and the one that GENERALISES: a detached collector runs diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingServerConnector.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingServerConnector.cs index fdb6a6a85..c2fc154de 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingServerConnector.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingServerConnector.cs @@ -308,6 +308,40 @@ private static async Task ProbeAuroraAsync( } } + /// + /// Whether the pg_wait_sampling extension is created in the database this connection landed in + /// (#3604) — the connect-time fact that picks PgWaitSamplingCollector's arm on a stock target: + /// the extension's own 10 ms profile when present, the service-side pg_stat_activity sampler when + /// not. pg_extension is per database, and the collector reads pg_wait_sampling_profile from + /// exactly this database, so this is the fact that decides whether that read can succeed rather than a + /// proxy for it (pg_available_extensions would say "installable", which is a different question). + /// + public const string PostgresWaitSamplingProbeQueryText = + @"SELECT EXISTS (SELECT 1 FROM pg_extension WHERE extname = 'pg_wait_sampling')"; + + /// + /// Runs . Fails CLOSED to false on any error other than + /// cancellation, and for the same reason does: this probe decides which + /// arm an optional collector takes, and a target that answered every other connect question must still be + /// monitored. False routes the target to the sampler arm, which reads only pg_stat_activity — the + /// floor — so the failure direction produces a coarser wait history rather than none. + /// + private static async Task ProbeWaitSamplingExtensionAsync( + NpgsqlConnection connection, CancellationToken cancellationToken, ILogger? logger = null) + { + try + { + using var command = new NpgsqlCommand(PostgresWaitSamplingProbeQueryText, connection) { CommandTimeout = 15 }; + var present = await command.ExecuteScalarAsync(cancellationToken); + return present is bool b && b; + } + catch (Exception ex) when (ex is not OperationCanceledException) + { + logger?.LogDebug(ex, "pg_wait_sampling extension probe did not succeed; treating the extension as absent (#3604)"); + return false; + } + } + /// Connects, probes, and returns the runtime state for one configured server. public static async Task ConnectAsync(MonitoredServer config, ILogger? logger, CancellationToken cancellationToken) { @@ -406,9 +440,15 @@ private static async Task ConnectPostgresAsync( var isAurora = await ProbeAuroraAsync(connection, cancellationToken, logger); + /* #3604: the wait-tier decision's second fact. Probed AFTER Aurora-ness and independently of it — + an Aurora cluster cannot load the module, so the answer there is false by construction, but the + gate reads the two facts separately and the sweep varies them separately. */ + var hasWaitSampling = await ProbeWaitSamplingExtensionAsync(connection, cancellationToken, logger); + logger?.LogInformation( - "Connected to PostgreSQL target '{Server}': major {Major} (server_version_num {Num}), {Role}, Aurora: {Aurora} — {VersionText}", + "Connected to PostgreSQL target '{Server}': major {Major} (server_version_num {Num}), {Role}, Aurora: {Aurora}, pg_wait_sampling: {WaitSampling} — {VersionText}", config.DisplayName, majorVersion, versionNum, isInRecovery ? "reader (in recovery)" : "writer", isAurora, + hasWaitSampling ? "present (extension_sampled tier)" : isAurora ? "n/a (engine_cumulative tier)" : "absent (service_sampled tier)", versionText); /* A Postgres target reached through the SQL Server path would have failed on the detection @@ -434,6 +474,8 @@ read and the failure names a grant that would never have helped. Aurora was unaf IsAwsRds = RdsEndpoint.TryParse( new NpgsqlConnectionStringBuilder(connectionString).Host) is not null, IsInRecovery = isInRecovery, + /* #3604: which stock-PostgreSQL wait instrument this target gets, decided here once. */ + HasPgWaitSamplingExtension = hasWaitSampling, }, StorageName = storageName, ServerId = config.ServerId, @@ -478,7 +520,8 @@ succeeded against a different engine. */ PostgresMajorVersion: runtime.Target.PostgresMajorVersion, PostgresVersionNum: runtime.Target.PostgresVersionNum, IsAurora: runtime.Target.IsAurora, - IsInRecovery: runtime.Target.IsInRecovery); + IsInRecovery: runtime.Target.IsInRecovery, + HasPgWaitSamplingExtension: runtime.Target.HasPgWaitSamplingExtension); } catch (OperationCanceledException) { @@ -588,7 +631,10 @@ public sealed record ConnectionProbeResult( /* #2280: the database the connection ACTUALLY reached, so a registration-time collision check can compare what the SERVER says against what other registrations claim, rather than comparing two claims. Trailing and defaulted, so every existing construction of this record still compiles unchanged. */ - string? ConnectedDatabase = null) + string? ConnectedDatabase = null, + /* #3604: the wait-tier fact, so --test-connection's "which collectors would run" count is computed + against the same shape the gate reads. Trailing and defaulted for the #2280 reason. */ + bool HasPgWaitSamplingExtension = false) { /// /// Rebuilds the gate's-eye view of this target, so a caller can ask which collectors would actually @@ -607,5 +653,6 @@ public sealed record ConnectionProbeResult( PostgresVersionNum = PostgresVersionNum, IsAurora = IsAurora, IsInRecovery = IsInRecovery, + HasPgWaitSamplingExtension = HasPgWaitSamplingExtension, }; } diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs index 0dde9b00f..2c167f5c7 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs @@ -2922,6 +2922,20 @@ internal static bool IsQueryStoreCollector(string collectorName) => internal static bool IsPlanCorrectionCollector(string collectorName) => string.Equals(collectorName, PlanCorrectionCollector.Instance.Name, StringComparison.OrdinalIgnoreCase); + /// + /// #3604: whether a dispatched collector name is pg_wait_sampling — the third collector detached + /// from the sequential body, and the first detached for a DELIBERATE run length rather than a bimodal + /// one. Its sampler arm holds its connection for thirty one-second snapshots per cycle by design; awaited + /// inline that would delay every other collector on a stock PostgreSQL target by half a minute every five, + /// which is the #2700 starvation with a known cause. Detached behind the generic per-(server, collector) + /// gate, a still-running window simply skips the tick — and it cannot still be running, because the run is + /// 30 s and the cadence is 300 s; the gate is there for the day someone lengthens the window. On the + /// extension arm the run is a 500-row read and the detach costs nothing. Compared against the collector's + /// OWN declared name, for the renaming-safety reason the two siblings state. + /// + internal static bool IsPgWaitSamplingCollector(string collectorName) => + string.Equals(collectorName, PgWaitSamplingCollector.Instance.Name, StringComparison.OrdinalIgnoreCase); + /// /// #2219: refreshes this PostgreSQL server's statement text if it is due, and swallows everything if not. /// @@ -6605,7 +6619,9 @@ sys.dm_db_tuning_recommendations is re-read whole on every successful pass regar at completion is correct only for the sequential arm; a detached run finishes 100-230s later, by which time the 15s sweep has reset and rebuilt the mark from unrelated ticks. */ var peerMaxAtDispatchMs = PeerMaxOrNull(server); - if (IsQueryStoreCollector(name) || IsPlanCorrectionCollector(name)) + /* #3604: pg_wait_sampling is the third, and the reason is different in kind — see + IsPgWaitSamplingCollector: a deliberate 30 s sampling window, not a bimodal tail. */ + if (IsQueryStoreCollector(name) || IsPlanCorrectionCollector(name) || IsPgWaitSamplingCollector(name)) { _ = RunDetachedAsync(server, runner, name, peerMaxAtDispatchMs, cancellationToken); } @@ -7614,7 +7630,7 @@ collector on THIS server has not finished — skip is safe because every collect NotGated (mirroring QueryStoreServerGate's) collapses this to a single null check below — a future third collector needs only its own IsXCollector check added to this one condition, never a second one to keep in sync. */ - using var detachedGate = IsPlanCorrectionCollector(collectorName) + using var detachedGate = IsPlanCorrectionCollector(collectorName) || IsPgWaitSamplingCollector(collectorName) ? _detachedCollectorGates.GetOrAdd((runtime.ServerId, collectorName), static _ => new DetachedCollectorGate()).TryAcquire() : DetachedCollectorGate.NotGated; diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs index 483b1b2ea..436e32aae 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs @@ -68,6 +68,10 @@ public static string Build(DarlingPeerDirectory.Snapshot peers) When a read comes back with no data, the `status` word says WHICH kind of nothing it is, and the four are not interchangeable. `empty` is a true negative: we looked and there was nothing to find. `unavailable` means this server could have that data and does not have it right now, so collection health is worth a look. `not_collected` means this server does not collect that at all — and when the reason is the ENGINE, the gap is PERMANENT: the collector serving that read does not run on this server's engine (an Azure SQL Database has no system_health session, no default trace and no SQL Agent; a PostgreSQL target collects none of the SQL Server signals at all, and the `get_pg_*` reads are the ones that answer there), so there is no session to start, no collector to enable, and nothing to check. The message names the engine and the collector. Do not send anyone to go and fix it. `precondition` is the one that IS worth acting on: this server could have that data, the collector is running, and a setup step on the monitored server is in the way — a Query Store that is off or has gone READ_ONLY, an Extended Events capture session that is not running, an extension that was never created, a grant the monitoring login was refused. The message names the precondition, quotes what the monitored server itself said, and gives the statement or grant that satisfies it. It is re-derived on EVERY read rather than decided when the connection was made, so once somebody does the thing it asked for the next call answers with data — usually with nothing to restart on the monitoring side. A few preconditions are the exception and SAY SO IN THEIR OWN MESSAGE: the fact that gates them is read once when the service connects to that server and cached for the connection's life, so satisfying them also needs the service to reconnect before collection resumes. Read the message rather than assuming the general case — it tells you which kind you have, and telling somebody to retry a connect-scoped one without reconnecting sends them round a loop that never terminates. + ### PostgreSQL waits come from one of three instruments + + A PostgreSQL target's wait history is fed by exactly ONE of three instruments, chosen once when the service connects to it, in this order of preference: Aurora native > the `pg_wait_sampling` extension > the service-side sampler. Aurora targets get `engine_cumulative` — the engine's own wait counters and measured wait TIME, read by `get_pg_wait_stats`. Stock and RDS targets are read by `get_pg_wait_sampling`, whose `instrument` field says which of the other two fed the rows: `extension_sampled` is the extension's in-engine 10 ms profiler; `service_sampled` means the extension is not installed and this service polled `pg_stat_activity` once a second for a 30-second window every five minutes instead. That last tier is a FLOOR, not parity — it under-counts waits shorter than a second and sees nothing between windows, and `instrument_note` says so in the answer — but it is what stops a first-install stock target from reading as "no waits at all". A count only means something beside its instrument: read `instrument` before comparing two servers' wait profiles, and treat `service_sampled` shares as trustworthy for steady fractions of the server's time and approximate for rare short events. Installing `pg_wait_sampling` (a `shared_preload_libraries` entry, then `CREATE EXTENSION` in the monitored database, then a reconnect) moves a target to the finer tier with no other change. + ### Asking about a PAST window Every tool below that takes `hours_back` also takes `as_of`: an optional ISO-8601 UTC instant that moves the END of the window off "now". `hours_back` stays the window's LENGTH. So the four hours around last Tuesday 03:00 is `as_of=2026-08-19T05:00:00Z, hours_back=4` — not `hours_back=170`. diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgWaitSamplingTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgWaitSamplingTools.cs index 04f65bdcf..4d522c2a1 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgWaitSamplingTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgWaitSamplingTools.cs @@ -14,6 +14,7 @@ using System.Threading.Tasks; using ModelContextProtocol.Server; using Npgsql; +using PerformanceMonitor.Collectors; using PerformanceMonitor.Common; using PerformanceMonitor.Darling.Storage; @@ -45,11 +46,22 @@ namespace PerformanceMonitor.Darling.Service.Mcp; /// mostly CPU and a profile that is mostly IO call for opposite next steps, and a wait-only view cannot /// tell them apart. /// +/// +/// +/// It discloses its INSTRUMENT (#3604). Since the collector grew a service-side sampler arm for +/// targets without the extension, the same table holds two grains: the extension's 10 ms in-engine samples +/// and this service's one-second pg_stat_activity polls over a 30 s window each cycle. A count of 300 +/// is three seconds of waiting under one and five minutes under the other, and estimated_wait_ms is +/// right on both only because each arm stores its own period. So every non-empty answer carries +/// instrument (extension_sampled or service_sampled, read from the collector's own +/// per-server state) and, on the service tier, the floor caveat in +/// — the same disclose-the-instrument pattern get_pg_plan_capture_readiness established for plans. +/// /// [McpServerToolType] public sealed class DarlingMcpPgWaitSamplingTools { - [McpServerTool(Name = "get_pg_wait_sampling"), Description("Gets sampled PostgreSQL wait events attributed to query shapes, from the pg_wait_sampling extension. This is the stock-PostgreSQL counterpart of get_pg_wait_stats, which reads an Aurora-only source: use this tool on any self-hosted or non-Aurora PostgreSQL target. A sampling profiler periodically records what each backend is doing, so results are sample COUNTS, and estimated_wait_ms is samples multiplied by the sampling period rather than a measured duration - treat a rare event's estimate as approximate. Rows with event_type CPU mean the backend was running, not waiting, so the profile answers 'waiting or working' as well as 'waiting on what'. queryid joins get_pg_top_queries; queryid 0 is work belonging to no statement, such as a background worker. The profile is cluster-wide and carries no database attribution by design. THE PAGE IS BOUNDED BY limit: waits_returned is how many rows you got, truncated says the window held more, and the rows are the most-sampled so the ones past the cap are rarer. SHARES ARE OF THE WINDOW, NOT OF THE PAGE: pct_of_samples' denominator is total_samples, the WHOLE window's differenced sample count across every (event type, event, query) series, computed in the same statement as the rows - so a three-row page does not sum to 100%, and the gap between returned_samples (what the page adds up to) and total_samples is the activity the cap left out; returned_pct_of_total is that ratio stated once.")] + [McpServerTool(Name = "get_pg_wait_sampling"), Description("Gets sampled PostgreSQL wait events attributed to query shapes, for any non-Aurora PostgreSQL target. This is the stock-PostgreSQL counterpart of get_pg_wait_stats, which reads an Aurora-only source. PostgreSQL wait history comes from one of THREE instruments, chosen once per target when the service connects (Aurora native > pg_wait_sampling extension > service sampler), and the answer's instrument field says which one fed these rows: extension_sampled means the pg_wait_sampling extension's in-engine 10 ms profiler; service_sampled means this service polled pg_stat_activity once a second for a 30-second window every five minutes because the extension is not installed - a FLOOR, not parity, that under-counts waits shorter than a second and cannot see between windows (instrument_note spells out the limit and the one-step upgrade). Either way a sampler periodically records what each backend is doing, so results are sample COUNTS, and estimated_wait_ms is samples multiplied by that instrument's own period rather than a measured duration - treat a rare event's estimate as approximate. Rows with event_type CPU mean the backend was running, not waiting, so the profile answers 'waiting or working' as well as 'waiting on what'. queryid joins get_pg_top_queries; queryid 0 is work belonging to no statement, such as a background worker. The profile is cluster-wide and carries no database attribution by design. THE PAGE IS BOUNDED BY limit: waits_returned is how many rows you got, truncated says the window held more, and the rows are the most-sampled so the ones past the cap are rarer. SHARES ARE OF THE WINDOW, NOT OF THE PAGE: pct_of_samples' denominator is total_samples, the WHOLE window's differenced sample count across every (event type, event, query) series, computed in the same statement as the rows - so a three-row page does not sum to 100%, and the gap between returned_samples (what the page adds up to) and total_samples is the activity the cap left out; returned_pct_of_total is that ratio stated once.")] public static async Task GetPgWaitSampling( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -73,6 +85,11 @@ have a denominator the cap cannot shrink. */ var page = await DarlingPgWaitSamplingReader.GetPgWaitSamplingPageAsync( postgres, resolved.ServerId, windowEnd.AddHours(-hours_back), windowEnd, limit + 1); + /* #3604: which arm fed these rows, off the collector's own state. Read on the empty path too, so + a service-tier target that is genuinely idle is told it is being sampled at the floor grain + rather than left to wonder whether the extension tier was ever in play. */ + var instrument = await DarlingPgWaitSamplingReader.GetWaitInstrumentAsync(postgres, resolved.ServerId); + if (page.Rows.Count == 0) { /* Three distinct empty states, and only the last of them is "nothing happened". The @@ -90,10 +107,11 @@ have a denominator the cap cannot shrink. */ + "figures here are per-interval deltas, so a single collection has nothing to " + "difference against and the window fills on the second one. On a genuinely idle " + "server this is the healthy state: the profiler samples backends, and an idle " - + "server has none to sample."); + + "server has none to sample." + + DescribeInstrumentForEmpty(instrument)); } - return BuildWaitSamplingJson(resolved.ServerName, hours_back, page, limit); + return BuildWaitSamplingJson(resolved.ServerName, hours_back, page, limit, instrument); } catch (Exception ex) { @@ -101,6 +119,26 @@ have a denominator the cap cannot shrink. */ } } + /// + /// The sentence the empty arm appends about the instrument (#3604): the service tier's floor caveat when + /// that is what is sampling, nothing when the extension is (its idle answer needs no qualification), and a + /// plain "not yet recorded" when no cycle has stated an arm. + /// + internal static string DescribeInstrumentForEmpty(DarlingPgWaitSamplingReader.WaitInstrumentState? instrument) + { + if (instrument is null) + { + return " No collection cycle has recorded which wait instrument this server is on yet, so the " + + "first answer here will also say whether it is the extension's profiler or the service sampler."; + } + + return string.Equals(instrument.Instrument, PgWaitInstrument.ServiceSampled, StringComparison.Ordinal) + ? " This server is on the service_sampled tier (the pg_wait_sampling extension is not installed), " + + "so \"none to sample\" was measured over one-second polls in a 30-second window each cycle. " + + PgWaitInstrument.ServiceSampledCaveat + : string.Empty; + } + /// /// The response body, split out so the WIRE SHAPE can be asserted without a live store. /// @@ -119,8 +157,15 @@ internal static string BuildWaitSamplingJson( string serverName, int hoursBack, DarlingPgWaitSamplingReader.PgWaitSamplingPage page, - int limit) + int limit, + DarlingPgWaitSamplingReader.WaitInstrumentState? instrument = null) { + /* #3604: an unrecognised token is not echoed as an instrument - a future arm this build does not know + reads as unknown, which is the honest word, rather than as a grain the caller might act on. */ + var instrumentToken = instrument is not null && PgWaitInstrument.IsKnown(instrument.Instrument) + ? instrument.Instrument + : "unknown"; + var serviceTier = string.Equals(instrumentToken, PgWaitInstrument.ServiceSampled, StringComparison.Ordinal); var truncated = page.Rows.Count > limit; var rows = truncated ? page.Rows.Take(limit).ToList() : page.Rows; @@ -158,6 +203,22 @@ that joins to nothing. */ waits_returned = waits.Count, truncated, order = "samples_desc", + /* #3604: the grain. Beside the page facts rather than buried in the note, because it changes what + every number below means and a reader comparing two servers must see it first. */ + instrument = instrumentToken, + instrument_recorded_at = instrument?.RecordedAtUtc, + instrument_note = instrumentToken switch + { + PgWaitInstrument.ExtensionSampled => + "The pg_wait_sampling extension's in-engine profiler: every backend sampled every " + + "profile_period (10 ms by default) and attributed to its queryid. The finest grain " + + "available on stock PostgreSQL.", + PgWaitInstrument.ServiceSampled => PgWaitInstrument.ServiceSampledCaveat, + _ => "No collection cycle has recorded which arm fed these rows (a store written before the " + + "service sampler existed, or a collector that has not completed a cycle since). Read " + + "profile_period_ms per row with that in mind: 10 is the extension's default, 1000 is " + + "the service sampler's.", + }, /* The WINDOW's samples, across every series — the denominator of every pct_of_samples above. NOT the sum of the rows; that is returned_samples. */ total_samples = windowTotalSamples, @@ -172,6 +233,14 @@ that joins to nothing. */ + "same statement as the rows; each row's pct_of_samples divides by it, so the shares on a " + "page do not sum to 100 unless the page is the whole window (truncated = false). " + "returned_samples is what the rows returned add up to." + + (serviceTier + ? " backend_count on this tier is the distinct backends seen in the LAST window, not " + + "since the profile started. The tally behind these counts is kept in the monitoring " + + "store's own collector_state and reloaded every cycle, so it SURVIVES a service restart; " + + "a series starts over only when the tally's 500-key cap evicted it as least-sampled and " + + "it was seen again, or when the target moved to the extension tier and back — and " + + "counter_reset reports either the same way it reports an extension reset." + : string.Empty) + (resetInWindow ? " At least one series was RESET inside this window, so its figures cover only " + "the time since the reset." diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgWaitTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgWaitTools.cs index eeaa3b7c5..9756d2660 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgWaitTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgWaitTools.cs @@ -13,6 +13,7 @@ using System.Threading.Tasks; using ModelContextProtocol.Server; using Npgsql; +using PerformanceMonitor.Collectors; using PerformanceMonitor.Common; using PerformanceMonitor.Darling.Storage; @@ -24,7 +25,7 @@ namespace PerformanceMonitor.Darling.Service.Mcp; [McpServerToolType] public sealed class DarlingMcpPgWaitTools { - [McpServerTool(Name = "get_pg_wait_stats"), Description("Gets the top PostgreSQL wait events aggregated over a time period, for Amazon Aurora PostgreSQL targets. Waits reveal what the database spends time waiting on: IO events point at storage or cache misses, Lock events at blocking between sessions, LWLock at internal contention, and LSN is Aurora's storage-durability wait. Background-worker and client-idle waits are already excluded by the collector, so every row here is real work. Note this is a separate tool from get_wait_stats, which covers SQL Server: PostgreSQL has a two-level type/event taxonomy, no signal-wait concept, and reports in microseconds, so the two cannot share one result shape. THE PAGE IS BOUNDED BY limit: wait_events_returned is how many events you got, truncated says the window held more, and the rows are the heaviest so the ones past the cap are lighter. SHARES ARE OF THE WINDOW, NOT OF THE PAGE: pct_of_total_wait's denominator is total_wait_time_ms, the WHOLE window's wait time across every event, computed in the same statement as the rows - so a three-row page does not sum to 100%, and the gap between returned_wait_time_ms (what the page adds up to) and total_wait_time_ms is the waiting the cap left out; returned_pct_of_total is that ratio stated once.")] + [McpServerTool(Name = "get_pg_wait_stats"), Description("Gets the top PostgreSQL wait events aggregated over a time period, for Amazon Aurora PostgreSQL targets - the engine_cumulative tier, the finest of the three PostgreSQL wait instruments (Aurora native > pg_wait_sampling extension > service sampler; stock targets take one of the other two and get_pg_wait_sampling serves them, disclosing which). Waits reveal what the database spends time waiting on: IO events point at storage or cache misses, Lock events at blocking between sessions, LWLock at internal contention, and LSN is Aurora's storage-durability wait. Background-worker and client-idle waits are already excluded by the collector, so every row here is real work. Note this is a separate tool from get_wait_stats, which covers SQL Server: PostgreSQL has a two-level type/event taxonomy, no signal-wait concept, and reports in microseconds, so the two cannot share one result shape. THE PAGE IS BOUNDED BY limit: wait_events_returned is how many events you got, truncated says the window held more, and the rows are the heaviest so the ones past the cap are lighter. SHARES ARE OF THE WINDOW, NOT OF THE PAGE: pct_of_total_wait's denominator is total_wait_time_ms, the WHOLE window's wait time across every event, computed in the same statement as the rows - so a three-row page does not sum to 100%, and the gap between returned_wait_time_ms (what the page adds up to) and total_wait_time_ms is the waiting the cap left out; returned_pct_of_total is that ratio stated once.")] public static async Task GetPgWaitStats( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -126,6 +127,16 @@ internal static string BuildWaitStatsJson( wait_events_returned = result.Count, truncated, order = "total_wait_time_ms_desc", + /* #3604: the instrument, stated on this surface too so the three PostgreSQL wait tiers read + alike. Constant here rather than looked up: pg_wait_stats is Aurora-gated and reads the + engine's own counters, so nothing else can have fed this table. */ + instrument = PgWaitInstrument.EngineCumulative, + instrument_note = "Aurora's aurora_stat_system_waits(): every wait's count and measured time, " + + "accumulated by the engine since instance start - the finest of the three " + + "PostgreSQL wait instruments and the only one that measures time rather than " + + "estimating it from samples. Stock PostgreSQL targets are served by " + + "get_pg_wait_sampling instead, whose instrument field says whether the " + + "pg_wait_sampling extension or the service-side sampler fed them.", /* The WINDOW's wait time, across every event that accrued any — the denominator of every pct_of_total_wait above. NOT the sum of the rows; that is returned_wait_time_ms. */ total_wait_time_ms = Math.Round(windowTotalMs, 1), diff --git a/Darling/PerformanceMonitor.Darling.Storage/DarlingPgWaitSamplingReader.cs b/Darling/PerformanceMonitor.Darling.Storage/DarlingPgWaitSamplingReader.cs index a9096c17a..81c6577ac 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/DarlingPgWaitSamplingReader.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/DarlingPgWaitSamplingReader.cs @@ -120,6 +120,49 @@ AND o.event IS NOT DISTINCT FROM n.event LIMIT $4 """; + /// + /// Which instrument is feeding this server's pg_wait_sampling rows (#3604), read from the + /// collector's own state: PgWaitSamplingCollector records a PgWaitInstrument token under + /// collector_state (server_id, 'pg_wait_sampling', 'instrument') on every cycle, whichever arm ran. + /// Off the store rather than re-derived here because the decision was made once at connect and the + /// collector is the only thing that knows which arm its last cycle took; a read guessing from + /// profile_period_ms would be right until an operator set the extension's period to a second. + /// Null when no cycle has recorded one — a store written before #3604, or a server whose collector + /// has not completed a cycle since. The tool says so rather than picking a default. + /// + public const string InstrumentSql = """ + SELECT state_value, updated_at + FROM collector_state + WHERE server_id = $1 + AND collector_name = 'pg_wait_sampling' + AND state_key = 'instrument' + """; + + /// The recorded instrument and when the collector last recorded it (UTC), or null. + public sealed record WaitInstrumentState(string Instrument, DateTime RecordedAtUtc); + + /// Runs . An unrecognised token is returned as-is; the tool decides + /// whether to echo it. + public static async Task GetWaitInstrumentAsync( + NpgsqlDataSource postgres, int serverId, CancellationToken cancellationToken = default) + { + ArgumentNullException.ThrowIfNull(postgres); + + await using var command = postgres.CreateCommand(InstrumentSql); + command.CommandTimeout = StorageCommandDeadlines.McpReadSeconds; + command.Parameters.AddWithValue(serverId); + + await using var reader = await command.ExecuteReaderAsync(cancellationToken); + if (!await reader.ReadAsync(cancellationToken) || reader.IsDBNull(0)) + { + return null; + } + + return new WaitInstrumentState( + reader.GetString(0), + reader.IsDBNull(1) ? default : DateTime.SpecifyKind(reader.GetDateTime(1), DateTimeKind.Utc)); + } + /// The rows alone — the WPF Viewer's grid, which has no column for the window total. public static async Task> GetPgWaitSamplingAsync( NpgsqlDataSource postgres, int serverId, DateTime startUtc, DateTime endUtc, int limit, diff --git a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Postgres.cs b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Postgres.cs index f43ca713a..55770c39d 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Postgres.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.Postgres.cs @@ -392,6 +392,13 @@ public async Task> GetPgCpuUtilizationHistoryAsync( int serverId, DateTime startUtc, DateTime endUtc, int limit, CancellationToken cancellationToken = default) => DarlingPgWaitSamplingReader.GetPgWaitSamplingAsync(_dataSource, serverId, startUtc, endUtc, limit, cancellationToken); + /// Which instrument is feeding this server's sampled waits (#3604) — the collector's own recorded + /// arm, so the panel note can say whether the grid is the extension's 10 ms profile or the service's + /// one-second floor. Null when no cycle has recorded one. + public Task GetPgWaitInstrumentAsync( + int serverId, CancellationToken cancellationToken = default) => + DarlingPgWaitSamplingReader.GetWaitInstrumentAsync(_dataSource, serverId, cancellationToken); + /// /// The kernel's own CPU and disk per query (#2603). Deltas and reset detection live in the reader. /// diff --git a/Darling/PerformanceMonitor.Darling.Viewer/ViewerServerTab.Postgres.cs b/Darling/PerformanceMonitor.Darling.Viewer/ViewerServerTab.Postgres.cs index 7ed8a48eb..a60c9b952 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/ViewerServerTab.Postgres.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/ViewerServerTab.Postgres.cs @@ -398,6 +398,11 @@ private async Task LoadPgWaitSamplingAsync(DateTime startUtc, DateTime endUtc) } var rows = await _dataService.GetPgWaitSamplingAsync(_server.ServerId, startUtc, endUtc, PgGridRowLimit); + /* #3604: the arm that fed the grid, off the collector's own state — the same disclosure the MCP read + makes, because the same table now holds two grains and the note is where this panel says which. */ + var instrument = await _dataService.GetPgWaitInstrumentAsync(_server.ServerId); + var serviceTier = instrument is not null + && string.Equals(instrument.Instrument, PgWaitInstrument.ServiceSampled, StringComparison.Ordinal); PgWaitSamplingGrid.ItemsSource = rows; @@ -406,11 +411,25 @@ private async Task LoadPgWaitSamplingAsync(DateTime startUtc, DateTime endUtc) PgWaitSamplingNote.Text = rows.Count == 0 ? PanelNote("pg_wait_sampling", 0, - "No wait samples for this server in this window. The usual cause is that " - + "pg_wait_sampling is not in shared_preload_libraries — check the Extensions panel, " - + "which says whether it is installed, available or absent. If it IS loaded, an empty " - + "grid means the server waited on nothing worth sampling, which is the healthy answer.") + serviceTier + ? "No wait samples for this server in this window. This server is on the service-sampled " + + "tier (pg_wait_sampling is not installed), so the service polled pg_stat_activity once a " + + "second for a 30-second window each cycle and found nothing worth counting — the healthy " + + "answer at that grain, which under-counts waits shorter than a second." + : "No wait samples for this server in this window. The usual cause is that " + + "pg_wait_sampling is not in shared_preload_libraries — check the Extensions panel, " + + "which says whether it is installed, available or absent. If it IS loaded, an empty " + + "grid means the server waited on nothing worth sampling, which is the healthy answer.") : $"{rows.Count:N0} wait event(s), {attributed:N0} attributed to a query. " + + (serviceTier + ? "Instrument: SERVICE SAMPLER — pg_wait_sampling is not installed, so this is the " + + "service polling pg_stat_activity once a second for a 30-second window every five " + + "minutes. A floor, not parity: waits shorter than a second are under-counted and " + + "nothing between windows is seen. Installing the extension moves this server to " + + "the 10 ms tier. " + : instrument is not null + ? "Instrument: pg_wait_sampling extension (10 ms in-engine profiler). " + : string.Empty) + "Est. Wait is samples multiplied by the profile period — an estimate from a sampling " + "profiler, not a measured duration, so treat it as a ranking rather than a stopwatch. " + "A Query ID of 0 is a background process rather than an unknown query, and CPU/Running " diff --git a/Darling/README.md b/Darling/README.md index b2074fb88..23147de3a 100644 --- a/Darling/README.md +++ b/Darling/README.md @@ -735,7 +735,7 @@ GRANT pg_read_all_data TO darling_monitor; -- PostgreSQL 14+, see below | `pg_replication_stats` | `pg_stat_replication` | 1 min / 30 d | The CONNECTED standbys — four byte distances from `pg_current_wal_lsn()` and three time lags. The counterpart of the slot collector and a different question: a slot describes what the primary RETAINS and exists even when nothing is attached (the disk-filling case); this describes replicas actually attached, and one that disappears here while its slot persists is the dangerous combination. The lag columns keep growing during a stall rather than freezing, which is what makes them usable as an alerting signal | | `pg_buffer_usage` | `pg_buffercache` | 60 min / 30 d | What is resident in the shared buffer pool, per relation — which objects the cache is SPENT on, a different question from which are read most. Joins `pg_buffercache.relfilenode` to `pg_relation_filenode(c.oid)`, **not** to `c.oid`, which is the join every published example gets wrong, and scopes to the connected database because the pool is cluster-wide while `pg_class` is not. Hourly: the view is a scan of every buffer (6.1 ms at 512 MB, linear to ~780 ms at 64 GB), and a server without the extension records a non-fatal skip naming the one-line `CREATE EXTENSION` fix | | `pg_extension_availability` | `pg_available_extensions`, `pg_extension` | 24 h / 365 d | Four states, not a boolean — `installed`, `outdated`, `available`, `absent` — because `available` is a `CREATE EXTENSION` away and `absent` is not. Absence is derived against an enumerated roster; `auto_explain` and `pg_wait_sampling` are deliberately off it, being preload-only modules that never appear in `pg_available_extensions` even where they are loaded and working. Per database as of V95, because `installed`/`outdated` are per-database facts | -| `pg_wait_sampling` | the `pg_wait_sampling` extension's profile | 60 min / 30 d | Wait events attributed to the query that waited — wait analysis existed ONLY on Aurora before this, which was backwards, because self-hosted is where the extension story is richest. `sample_count` is a tally of observations, not a duration, and `profile_period_ms` travels beside it because the count is uninterpretable alone. Cluster-wide with no database attribution, deliberately: the profile carries none, and inventing one would be the scope error V95 removed elsewhere | +| `pg_wait_sampling` | the `pg_wait_sampling` extension's profile, or — without the extension — a service-side `pg_stat_activity` sampler (#3604) | 5 min / 30 d | Wait events attributed to the query that waited — wait analysis existed ONLY on Aurora before this, which was backwards, because self-hosted is where the extension story is richest. Two arms, one table, one instrument per target: with the extension created in the monitored database the collector reads its 10 ms in-engine profile; without it the collector polls `pg_stat_activity` once a second for a 30-second window each cycle and accumulates the tallies itself, so a stock first install gets a wait profile (a floor, not parity) rather than an empty chart. Not on Aurora, which has `pg_wait_stats` and cannot load the module. Every read of this table discloses `instrument` (`extension_sampled` / `service_sampled`). `sample_count` is a tally of observations, not a duration, and `profile_period_ms` travels beside it because the count is uninterpretable alone. Cluster-wide with no database attribution, deliberately: the profile carries none, and inventing one would be the scope error V95 removed elsewhere | | `pg_kernel_stats` | `pg_stat_kcache` | 60 min / 30 d | The kernel's own numbers per `queryid` — user/system CPU, bytes that reached the DEVICE, page faults. `pg_stat_statements` reports elapsed time, which cannot separate a query that was WAITING from one burning processor; these compose with it to make that split. `exec_read_bytes = 0` does not mean nothing was read: a read served by the OS page cache is genuinely zero. Top-level statements only, because `track = 'all'` would double-count every function body on the server | | `pg_predicate_stats` | `pg_qualstats` | 60 min / 30 d | Which columns are filtered on, how selectively, and where the planner's estimate was wrong. `pg_index_usage_stats` records indexes that were USED; this records predicates that were EVALUATED, including the ones with no index behind them — the index-candidate signal, invisible to everything else here. `sample_rate` is stored per row and is usually not 1 (the extension defaults to `1/max_connections`), so counts are never read as complete. Per database, because the OIDs its rows carry only resolve in their own database | | `pg_column_stats` | `pg_stats` | 24 h / 365 d | The planner inputs behind a misestimate — `n_distinct`, `null_frac`, `avg_width`, `correlation`, and the frequency of the single most common value. The SHAPE of the skew, never the values: `most_common_vals` and `histogram_bounds` hold raw customer data and are deliberately not stored. Per database. Daily and kept a year because these move when ANALYZE runs and are asked about retrospectively. Needs `pg_read_all_data` — without it the collector succeeds and stores zero rows, the quiet failure the paragraph above exists for | diff --git a/Lite.Tests/AuroraOnlySqlIsGatedTests.cs b/Lite.Tests/AuroraOnlySqlIsGatedTests.cs index 0ed2f4387..372534b06 100644 --- a/Lite.Tests/AuroraOnlySqlIsGatedTests.cs +++ b/Lite.Tests/AuroraOnlySqlIsGatedTests.cs @@ -84,6 +84,11 @@ public class AuroraOnlySqlIsGatedTests "aurora_stat_statements(), every other PostgreSQL the vanilla view with the Aurora-only columns " + "null. The tool reads the STORE, never the target; the dependency itself is " + "PgStatementStatsCollector's PAIRED entry above.", + + ["DarlingMcpPgWaitTools.cs"] = + "PROSE (#3604). get_pg_wait_stats' instrument_note tells the caller which source fed the rows — " + + "Aurora's aurora_stat_system_waits() — so a count can be read beside its grain. The tool reads " + + "the STORE, never the target; the collector it describes is PgWaitStatsCollector's GATED entry.", }; [Fact] diff --git a/Lite.Tests/DefaultTraceEventsCollectorDefinitionTests.cs b/Lite.Tests/DefaultTraceEventsCollectorDefinitionTests.cs index b67dd21ab..28cc6f3bb 100644 --- a/Lite.Tests/DefaultTraceEventsCollectorDefinitionTests.cs +++ b/Lite.Tests/DefaultTraceEventsCollectorDefinitionTests.cs @@ -340,18 +340,20 @@ public void StateKeys_DeclaresTheLastSeenTracePath_AndTheDeclaringSetIsPinned() declares no state and its host runs no state query. A collector appearing here is a real design decision — both hosts must load and persist its keys — not a silent addition. - ONE declares state now: default_trace_events stores the trace FILE it last read (#1962). #3153 + TWO declare state now. default_trace_events stores the trace FILE it last read (#1962). #3153 had added pg_index_bloat's per-database rotation cursor -- whose absence is what made the collector re-measure its largest index every cycle forever while labelling the rest as merely deferred -- and #3234 retired it, because the statistics estimate covers every index in one - statement and there is no position to resume from. Ordered by name so this reads as a set - rather than as an accident of catalog order. */ + statement and there is no position to resume from. pg_wait_sampling joined in #3604: the arm it + ran (its instrument token) and, on the service-sampler arm, the cumulative tally the next cycle + adds to -- which the table holds only as its last written value per key, so a MAX() cannot + recover it. Ordered by name so this reads as a set rather than as an accident of catalog order. */ var declaring = CollectorCatalog.All .Where(c => c.StateKeys.Count > 0) .Select(c => c.Name) .OrderBy(n => n, StringComparer.Ordinal) .ToArray(); - Assert.Equal(new[] { "default_trace_events" }, declaring); + Assert.Equal(new[] { "default_trace_events", "pg_wait_sampling" }, declaring); } [Fact] diff --git a/Lite.Tests/PgWaitExclusionParityTests.cs b/Lite.Tests/PgWaitExclusionParityTests.cs index a242bac1e..1e79a74b2 100644 --- a/Lite.Tests/PgWaitExclusionParityTests.cs +++ b/Lite.Tests/PgWaitExclusionParityTests.cs @@ -112,6 +112,9 @@ private static CollectorContext Context() Engine = CollectorTargetEngine.PostgreSql, PostgresMajorVersion = 17, PostgresVersionNum = 170000, + /* #3604: the EXTENSION arm. The definition forks on this fact, and every pin in this file is + about the profile query; the sampler arm has its own tests. */ + HasPgWaitSamplingExtension = true, }, }; } diff --git a/Lite.Tests/PgWaitSamplerArmTests.cs b/Lite.Tests/PgWaitSamplerArmTests.cs new file mode 100644 index 000000000..e06fdf937 --- /dev/null +++ b/Lite.Tests/PgWaitSamplerArmTests.cs @@ -0,0 +1,368 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor Lite. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Linq; +using System.Text.RegularExpressions; +using System.Threading; +using System.Threading.Tasks; +using Lite.Tests.Helpers; +using PerformanceMonitor.Collectors; +using Xunit; + +namespace Lite.Tests; + +/// +/// The service-sampler arm of (#3604) — the floor under stock +/// PostgreSQL targets without the pg_wait_sampling extension — and the three-tier selection that +/// decides which instrument a target gets. +/// +/// What is pinned here and why. The arm's whole correctness rests on four things a live run +/// cannot cheaply prove: that the batch really re-reads pg_stat_activity (each snapshot is its own +/// statement behind a pg_stat_clear_snapshot()), that the tally it writes is CUMULATIVE across cycles +/// (the read differences newest against oldest, so a per-window count would be silently wrong), that it +/// excludes its own connections and the shared idle types, and that exactly ONE wait collector applies to +/// any PostgreSQL shape. Each is a test below driven with a fixture reader, no server. +/// +public sealed class PgWaitSamplerArmTests +{ + private static CollectorContext MakeContext(bool hasExtension, int major = 17, IReadOnlyDictionary? state = null) => + new() + { + ServerId = 42, + ServerName = "test-server", + CollectionTime = new DateTime(2026, 9, 18, 12, 0, 0, DateTimeKind.Utc), + Deltas = new RecordingCollectorDeltaCalculator(), + Target = new CollectorTargetInfo + { + Engine = CollectorTargetEngine.PostgreSql, + PostgresMajorVersion = major, + PostgresVersionNum = major * 10000, + HasPgWaitSamplingExtension = hasExtension, + }, + State = state ?? new Dictionary(), + }; + + /* ───────────────────────── tier selection ───────────────────────── */ + + /// + /// Aurora native > extension > service sampler, and exactly one path per target. Asserted over the + /// engine-capability sweep's own shapes rather than three hand-picked targets, so a fourth fact added to + /// the gate later is swept too. + /// + [Fact] + public void ExactlyOneWaitCollectorAppliesToEveryPostgresShape() + { + foreach (var kind in new[] { "postgres", "aurora-postgres" }) + { + foreach (var target in CollectorEngineCapability.TargetsWithEngineKind(kind)) + { + var aurora = CollectorCatalog.AppliesTo("pg_wait_stats", target); + var sampling = CollectorCatalog.AppliesTo("pg_wait_sampling", target); + Assert.True(aurora ^ sampling, + $"IsAurora={target.IsAurora} HasExt={target.HasPgWaitSamplingExtension}: pg_wait_stats={aurora}, pg_wait_sampling={sampling} — exactly one must run"); + Assert.Equal(target.IsAurora, aurora); + } + } + } + + [Fact] + public void TheArmFollowsTheExtensionFact_AndTheAuroraTierNeverReachesEither() + { + var extension = PgWaitSamplingCollector.Instance.BuildQuery(MakeContext(hasExtension: true)).Text; + var sampler = PgWaitSamplingCollector.Instance.BuildQuery(MakeContext(hasExtension: false)).Text; + + Assert.Contains("pg_wait_sampling_profile", extension, StringComparison.Ordinal); + /* The profile query mentions pg_stat_activity in a comment; what it must not do is READ it. */ + Assert.DoesNotContain("FROM pg_stat_activity", extension, StringComparison.Ordinal); + + Assert.Contains("pg_stat_activity", sampler, StringComparison.Ordinal); + Assert.DoesNotContain("pg_wait_sampling_profile", sampler, StringComparison.Ordinal); + + Assert.False(PgWaitSamplingCollector.Instance.AppliesTo(new CollectorTargetInfo + { + Engine = CollectorTargetEngine.PostgreSql, IsAurora = true, HasPgWaitSamplingExtension = false, + }), "Aurora must take pg_wait_stats alone; the sampler arm beside it would be two answers to one question"); + } + + /// The Aurora gap now points at the instrument that DOES answer there — the mirror of the + /// pointer that already ran from pg_wait_stats to pg_wait_sampling. + [Fact] + public void OnAurora_TheSamplingReadIsAPermanentGapThatNamesPgWaitStats() + { + Assert.False(CollectorEngineCapability.IsCollectedOnEngineKind("pg_wait_sampling", "aurora-postgres")); + Assert.True(CollectorEngineCapability.IsCollectedOnEngineKind("pg_wait_sampling", "postgres")); + + var message = CollectorEngineCapability.NotCollectedMessage("db-01", 0, "aurora-postgres", "pg_wait_sampling"); + Assert.NotNull(message); + Assert.Contains("pg_wait_stats", message, StringComparison.Ordinal); + Assert.Contains("never will", message, StringComparison.Ordinal); + } + + /* ───────────────────────── the batch ───────────────────────── */ + + [Fact] + public void TheBatchIsOneSnapshotThenSleepClearSnapshot_RepeatedToTheConstant() + { + var sql = PgWaitSamplingCollector.SamplerBatchSql(hasQueryId: true); + + var snapshots = Regex.Matches(sql, @"FROM pg_stat_activity AS a").Count; + var sleeps = Regex.Matches(sql, @"SELECT pg_sleep\(").Count; + var clears = Regex.Matches(sql, @"SELECT pg_stat_clear_snapshot\(\)").Count; + + Assert.Equal(PgWaitSamplingCollector.SamplerSnapshotsPerCycle, snapshots); + Assert.Equal(PgWaitSamplingCollector.SamplerSnapshotsPerCycle - 1, sleeps); + Assert.Equal(PgWaitSamplingCollector.SamplerSnapshotsPerCycle - 1, clears); + + /* Every re-read is preceded by a clear, in that order: pg_stat_activity is cached per transaction on + first access, and the batch is one implicit transaction, so a snapshot without a clear before it + would return the previous instant again. */ + var statements = sql.Split(';', StringSplitOptions.RemoveEmptyEntries).Select(s => s.Trim()).ToList(); + for (var i = 1; i < statements.Count; i++) + { + if (statements[i].Contains("FROM pg_stat_activity", StringComparison.Ordinal)) + { + Assert.Contains("pg_stat_clear_snapshot", statements[i - 1], StringComparison.Ordinal); + } + } + + /* The period the rows carry is the period the batch sleeps. */ + Assert.Contains($"pg_sleep({PgWaitSamplingCollector.SamplerPeriodMs / 1000.0})", sql, StringComparison.Ordinal); + } + + /// The window must stay well inside the 60,000 ms body budget SweepPressureClassifier sums + /// single-run costs against, and inside the collector's own command timeout with headroom. + [Fact] + public void TheWindowFitsTheBodyBudgetAndTheCommandTimeout() + { + var windowSeconds = (PgWaitSamplingCollector.SamplerSnapshotsPerCycle - 1) * PgWaitSamplingCollector.SamplerPeriodMs / 1000.0; + Assert.InRange(windowSeconds, 1, 30); + Assert.True(PgWaitSamplingCollector.Instance.CommandTimeoutSecondsOverride is { } t && t >= windowSeconds * 2, + "the batch's results are buffered until it ends, so the command timeout must cover the whole window with headroom"); + Assert.Equal(5, CollectorScheduleDefaults.All["pg_wait_sampling"].FrequencyMinutes); + } + + [Fact] + public void TheSnapshotExcludesItselfItsSiblingsAndTheSharedIdleTypes_AndLabelsCpu() + { + var sql = PgWaitSamplingCollector.SamplerSnapshotSql(hasQueryId: true); + + Assert.Contains("a.pid <> pg_backend_pid()", sql, StringComparison.Ordinal); + Assert.Contains($"'{PgWaitSamplingCollector.ServiceApplicationName}'", sql, StringComparison.Ordinal); + Assert.Contains($"'{PgWaitSamplingCollector.ServiceRemediationApplicationName}'", sql, StringComparison.Ordinal); + + foreach (var type in PgWaitStatsCollector.IgnoredWaitTypes) + { + Assert.Contains($"'{type}'", sql, StringComparison.Ordinal); + } + + /* Same coalesce-FIRST trap the extension arm's query documents: a NULL type is CPU and must survive + the NOT IN. */ + Assert.Contains("coalesce(a.wait_event_type, 'CPU') NOT IN", sql, StringComparison.Ordinal); + Assert.Contains("coalesce(a.wait_event, 'Running')", sql, StringComparison.Ordinal); + /* Only an ACTIVE backend with no wait is on CPU; an idle one with no wait is nothing. */ + Assert.Contains("a.state = 'active'", sql, StringComparison.Ordinal); + } + + [Fact] + public void QueryIdIsSelectedOnlyWhereItExists() + { + Assert.Contains("a.query_id", PgWaitSamplingCollector.SamplerSnapshotSql(hasQueryId: true), StringComparison.Ordinal); + Assert.DoesNotContain("a.query_id", PgWaitSamplingCollector.SamplerSnapshotSql(hasQueryId: false), StringComparison.Ordinal); + + Assert.DoesNotContain("a.query_id", PgWaitSamplingCollector.Instance.BuildQuery(MakeContext(false, major: 13)).Text, StringComparison.Ordinal); + Assert.Contains("a.query_id", PgWaitSamplingCollector.Instance.BuildQuery(MakeContext(false, major: 14)).Text, StringComparison.Ordinal); + /* 0 is "unknown, assume newest" everywhere in the gates. */ + Assert.Contains("a.query_id", PgWaitSamplingCollector.Instance.BuildQuery(MakeContext(false, major: 0)).Text, StringComparison.Ordinal); + } + + /* ───────────────────────── accumulation ───────────────────────── */ + + private static object[] Snap(string type, string evt, long queryId, int pid) => new object[] { type, evt, queryId, pid }; + private static readonly object[][] Sleep = { new object[] { DBNull.Value } }; + private static readonly object[][] Clear = { new object[] { DBNull.Value } }; + + /// + /// Three snapshots, tallied by (type, event, query_id): a Lock held across all three by one backend, an + /// IO seen once, CPU twice by two different backends. Sleep and clear result sets are one column and are + /// drained, not tallied. backend_count is the WINDOW's distinct pids. + /// + [Fact] + public async Task ASnapshotSequenceIsTalliedByKey_WithWindowDistinctBackends() + { + var reader = FakeCollectorDataReader.WithResultSets( + new[] { Snap("Lock", "relation", 111, 10), Snap("CPU", "Running", 222, 20) }, + Sleep, Clear, + new[] { Snap("Lock", "relation", 111, 10), Snap("IO", "DataFileRead", 333, 30) }, + Sleep, Clear, + new[] { Snap("Lock", "relation", 111, 10), Snap("CPU", "Running", 222, 21) }); + + var context = MakeContext(hasExtension: false); + var rows = await PgWaitSamplingCollector.Instance.ReadAsync(reader, context, CancellationToken.None); + + var lock_ = Assert.Single(rows, r => r.EventType == "Lock"); + Assert.Equal(3, lock_.SampleCount); + Assert.Equal(1, lock_.BackendCount); + Assert.Equal(111, lock_.QueryId); + Assert.Equal(PgWaitSamplingCollector.SamplerPeriodMs, lock_.ProfilePeriodMs); + + var cpu = Assert.Single(rows, r => r.EventType == "CPU"); + Assert.Equal(2, cpu.SampleCount); + Assert.Equal(2, cpu.BackendCount); + + var io = Assert.Single(rows, r => r.EventType == "IO"); + Assert.Equal(1, io.SampleCount); + + /* Most-sampled first, like the extension arm's ORDER BY. */ + Assert.Equal("Lock", rows[0].EventType); + + Assert.Equal(PgWaitInstrument.ServiceSampled, context.PendingState[PgWaitSamplingCollector.InstrumentStateKey]); + } + + /// + /// The rows are CUMULATIVE: the tally carried in from the previous cycle is added to, every carried key + /// is re-emitted (so the read's newest-per-key stays current), and the new tally goes back out. This is + /// the property the read depends on — it differences newest against oldest per key over the window. + /// + [Fact] + public async Task TheTallyCarriesAcrossCycles_AndEveryCarriedKeyIsReEmitted() + { + var first = MakeContext(hasExtension: false); + await PgWaitSamplingCollector.Instance.ReadAsync( + FakeCollectorDataReader.WithResultSets(new[] { Snap("Lock", "relation", 111, 10), Snap("IO", "DataFileRead", 0, 30) }), + first, CancellationToken.None); + + var carried = new Dictionary { [PgWaitSamplingCollector.TallyStateKey] = first.PendingState[PgWaitSamplingCollector.TallyStateKey] }; + var second = MakeContext(hasExtension: false, state: carried); + var rows = await PgWaitSamplingCollector.Instance.ReadAsync( + FakeCollectorDataReader.WithResultSets(new[] { Snap("Lock", "relation", 111, 12) }), + second, CancellationToken.None); + + var lock_ = Assert.Single(rows, r => r.EventType == "Lock"); + Assert.Equal(2, lock_.SampleCount); + Assert.Equal(1, lock_.BackendCount); + + /* Not seen this window, still written with its cumulative count and a window backend count of 0. */ + var io = Assert.Single(rows, r => r.EventType == "IO"); + Assert.Equal(1, io.SampleCount); + Assert.Equal(0, io.BackendCount); + + var roundTrip = PgWaitSamplingCollector.ParseTally(second.PendingState[PgWaitSamplingCollector.TallyStateKey]); + Assert.Equal(2, roundTrip[("Lock", "relation", 111)]); + Assert.Equal(1, roundTrip[("IO", "DataFileRead", 0)]); + } + + [Fact] + public async Task AnEmptyWindowOnAnEmptyTallyWritesNothing_AndStillRecordsTheInstrument() + { + var context = MakeContext(hasExtension: false); + var rows = await PgWaitSamplingCollector.Instance.ReadAsync( + FakeCollectorDataReader.WithResultSets(Array.Empty(), Sleep, Clear, Array.Empty()), + context, CancellationToken.None); + + Assert.Empty(rows); + Assert.Equal(PgWaitInstrument.ServiceSampled, context.PendingState[PgWaitSamplingCollector.InstrumentStateKey]); + Assert.Equal(string.Empty, context.PendingState[PgWaitSamplingCollector.TallyStateKey]); + } + + [Fact] + public async Task TheTallyIsCappedByDroppingTheLeastSampled() + { + var tally = new Dictionary<(string, string, long), long>(); + for (var i = 0; i < PgWaitSamplingCollector.SamplerTallyCap + 10; i++) + { + tally[("Lock", "relation", i)] = i + 1; + } + + var text = PgWaitSamplingCollector.SerializeTally(tally); + var parsed = PgWaitSamplingCollector.ParseTally(text); + Assert.Equal(tally.Count, parsed.Count); + + /* The cap is applied in ReadSamplerAsync; drive it through the read with the oversized tally carried. */ + var context = MakeContext(hasExtension: false, state: new Dictionary { [PgWaitSamplingCollector.TallyStateKey] = text }); + var rows = await PgWaitSamplingCollector.Instance.ReadAsync( + FakeCollectorDataReader.WithResultSets(Array.Empty()), context, CancellationToken.None); + + Assert.Equal(PgWaitSamplingCollector.SamplerTallyCap, rows.Count); + Assert.DoesNotContain(rows, r => r.QueryId < 10); /* the ten least-sampled went */ + Assert.Contains(rows, r => r.QueryId == PgWaitSamplingCollector.SamplerTallyCap + 9); + } + + [Fact] + public void ABadTallyLineIsOneLostKey_NotALostTally() + { + var parsed = PgWaitSamplingCollector.ParseTally("Lock\trelation\t1\t5\ngarbage\nIO\tDataFileRead\tnotanumber\t2\nCPU\tRunning\t0\t9\n"); + Assert.Equal(2, parsed.Count); + Assert.Equal(5, parsed[("Lock", "relation", 1)]); + Assert.Equal(9, parsed[("CPU", "Running", 0)]); + } + + /// The extension arm records ITS instrument too, so a target that gains the extension flips the + /// disclosure on its next cycle rather than keeping a stale "service_sampled". + [Fact] + public async Task TheExtensionArmRecordsItsInstrument() + { + var context = MakeContext(hasExtension: true); + var rows = await PgWaitSamplingCollector.Instance.ReadAsync( + new FakeCollectorDataReader(new object[] { "Lock", "relation", 111L, 40L, 10, 2 }), + context, CancellationToken.None); + + Assert.Single(rows); + Assert.Equal(PgWaitInstrument.ExtensionSampled, context.PendingState[PgWaitSamplingCollector.InstrumentStateKey]); + /* And it CLEARS the sampler's tally rather than leaving the last one to be resumed months later if + the target ever falls back to the sampler arm (#3645 review) - the host persists only PendingState + keys, so "leave it alone" would mean "keep it forever". */ + Assert.Equal(string.Empty, context.PendingState[PgWaitSamplingCollector.TallyStateKey]); + } + + /// The whole reversion path: a tally carried from the sampler era, an extension-arm cycle, then + /// the sampler arm again — which must start from zero, not from the carried era. + [Fact] + public async Task AFallbackToTheSamplerAfterTheExtensionArm_StartsFromZero() + { + var samplerEra = MakeContext(hasExtension: false); + await PgWaitSamplingCollector.Instance.ReadAsync( + FakeCollectorDataReader.WithResultSets(new[] { Snap("Lock", "relation", 111, 10), Snap("Lock", "relation", 111, 11) }), + samplerEra, CancellationToken.None); + Assert.Equal(2, PgWaitSamplingCollector.ParseTally(samplerEra.PendingState[PgWaitSamplingCollector.TallyStateKey])[("Lock", "relation", 111)]); + + var extensionEra = MakeContext(hasExtension: true, state: new Dictionary(samplerEra.PendingState)); + await PgWaitSamplingCollector.Instance.ReadAsync( + new FakeCollectorDataReader(new object[] { "Lock", "relation", 111L, 9_999L, 10, 2 }), + extensionEra, CancellationToken.None); + + var fallback = MakeContext(hasExtension: false, state: new Dictionary(extensionEra.PendingState)); + var rows = await PgWaitSamplingCollector.Instance.ReadAsync( + FakeCollectorDataReader.WithResultSets(new[] { Snap("Lock", "relation", 111, 12) }), + fallback, CancellationToken.None); + + var lock_ = Assert.Single(rows); + Assert.Equal(1, lock_.SampleCount); + Assert.Equal(PgWaitInstrument.ServiceSampled, fallback.PendingState[PgWaitSamplingCollector.InstrumentStateKey]); + } + + [Fact] + public void StateKeysDeclareBothPieces_SoTheHostLoadsAndPersistsThem() + { + Assert.Equal( + new[] { PgWaitSamplingCollector.InstrumentStateKey, PgWaitSamplingCollector.TallyStateKey }, + PgWaitSamplingCollector.Instance.StateKeys); + } + + [Fact] + public void TheInstrumentVocabularyIsExactlyThreeTokens() + { + Assert.True(PgWaitInstrument.IsKnown(PgWaitInstrument.EngineCumulative)); + Assert.True(PgWaitInstrument.IsKnown(PgWaitInstrument.ExtensionSampled)); + Assert.True(PgWaitInstrument.IsKnown(PgWaitInstrument.ServiceSampled)); + Assert.False(PgWaitInstrument.IsKnown("Service_Sampled")); + Assert.False(PgWaitInstrument.IsKnown(null)); + Assert.Contains("FLOOR", PgWaitInstrument.ServiceSampledCaveat, StringComparison.Ordinal); + } +} diff --git a/Lite.Tests/PgWaitSamplingCollectorDefinitionTests.cs b/Lite.Tests/PgWaitSamplingCollectorDefinitionTests.cs index eae72e4ea..0fe74b135 100644 --- a/Lite.Tests/PgWaitSamplingCollectorDefinitionTests.cs +++ b/Lite.Tests/PgWaitSamplingCollectorDefinitionTests.cs @@ -35,6 +35,9 @@ private static CollectorContext MakeContext() Engine = CollectorTargetEngine.PostgreSql, PostgresMajorVersion = 17, PostgresVersionNum = 170000, + /* #3604: the EXTENSION arm. The definition forks on this fact, and every pin in this file is + about the profile query; the sampler arm has its own tests. */ + HasPgWaitSamplingExtension = true, }, }; diff --git a/PerformanceMonitor.Collectors/CollectorEngineCapability.cs b/PerformanceMonitor.Collectors/CollectorEngineCapability.cs index cb475cc95..6219f3e9f 100644 --- a/PerformanceMonitor.Collectors/CollectorEngineCapability.cs +++ b/PerformanceMonitor.Collectors/CollectorEngineCapability.cs @@ -332,14 +332,25 @@ public static IEnumerable TargetsWithEngineKind(string? eng { foreach (var isInRecovery in new[] { false, true }) { - yield return new CollectorTargetInfo + /* #3604: the pg_wait_sampling extension is a FIXABLE fact (CREATE EXTENSION plus a + preload restart), so it varies here rather than being fixed by the kind - the + sampler arm's gate reads it, and a sweep that left it false would answer that + gate from one shape. Aurora cannot load the module at all, but that is what + IsAurora above already says; the two facts are deliberately not coupled here + because the sweep's job is to be a SUPERSET of the real shapes, never a model + of which ones can co-occur. */ + foreach (var hasWaitSampling in new[] { false, true }) { - Engine = CollectorTargetEngine.PostgreSql, - IsAurora = isAurora, - PostgresMajorVersion = major, - PostgresVersionNum = versionNum, - IsInRecovery = isInRecovery, - }; + yield return new CollectorTargetInfo + { + Engine = CollectorTargetEngine.PostgreSql, + IsAurora = isAurora, + PostgresMajorVersion = major, + PostgresVersionNum = versionNum, + IsInRecovery = isInRecovery, + HasPgWaitSamplingExtension = hasWaitSampling, + }; + } } } } @@ -428,6 +439,12 @@ public static IEnumerable TargetsWithEngineKind(string? eng /* Aurora's aurora_stat_system_waits() vs. the pg_wait_sampling extension: different sources, same question - which wait events this server is spending time in. */ ["pg_wait_stats"] = "pg_wait_sampling", + /* The mirror (#3604). pg_wait_sampling is gated OFF on Aurora now - the module is not among the + libraries Aurora permits preloading, so the hourly EXTENSION_MISSING skip it used to record + there was a permanent gap wearing a fixable precondition's clothes - and an Aurora operator who + reaches get_pg_wait_sampling must be sent to the instrument that DOES answer on that engine + rather than told to install something they cannot. */ + ["pg_wait_sampling"] = "pg_wait_stats", }; /// diff --git a/PerformanceMonitor.Collectors/CollectorScheduleDefaults.cs b/PerformanceMonitor.Collectors/CollectorScheduleDefaults.cs index 5b24d2f92..484620aec 100644 --- a/PerformanceMonitor.Collectors/CollectorScheduleDefaults.cs +++ b/PerformanceMonitor.Collectors/CollectorScheduleDefaults.cs @@ -276,13 +276,14 @@ interval that produced it. An hourly grain would smear a five-minute burst of re pg_stat_statements get installed" and "when did this extension get upgraded" are asked months later, usually right after a plan changed shape and nobody can explain why. */ /* HOURLY, not every five minutes, and the reason is the fleet rather than the collector. All - four of these need an extension or a readable server log, and Aurora offers neither - so on a + three of these need an extension or a readable server log, and Aurora offers neither - so on a managed target they can only ever record a non-fatal skip, and at a five-minute cadence that is roughly 900 skip rows per target per day saying the same thing. - Hourly costs nothing for THREE of them: pg_wait_sampling_profile, pg_stat_kcache and - pg_qualstats are cumulative COUNTERS, so a longer interval loses no events - it only widens the - window each delta covers. That is the opposite of a sampled collector like pg_blocking, where + Hourly costs nothing for the two counter readers still here: pg_stat_kcache and pg_qualstats are + cumulative COUNTERS, so a longer interval loses no events - it only widens the window each delta + covers. (pg_wait_sampling_profile was the third until #3604 gave that collector a sampled arm and + its own five-minute entry below.) That is the opposite of a sampled collector like pg_blocking, where the cadence IS the resolution and stretching it genuinely loses sightings. pg_plan_capture is the fourth and it is neither. It reads a LOG, and "append-only" is not the @@ -303,7 +304,18 @@ that reads downstream as a target with no slow queries. the threshold. What is missing is that the collector cannot SAY it happened, and it already selects the file's size, so a stored previous size would make the skipped span a measurement instead of an absence. Until then a self-hosted operator lowers this per server. */ - ["pg_wait_sampling"] = new(60, 30), + /* #3604: FIVE MINUTES, split off the hourly trio above. Both halves of the hourly argument stopped + applying to this collector when it grew its sampler arm. The skip-row half: it is gated off Aurora + now and takes the sampler arm everywhere else, so it records EXTENSION_MISSING on no target at all. + The loses-nothing half is true of the extension arm's cumulative counters but false of the sampler + arm, whose window is thirty one-second snapshots per cycle - there the cadence IS the duty cycle + (30 s in 300 s, 10%), and an hour would make it 0.8%. Five minutes rather than one for two reasons + the file already states elsewhere: a 30 s single run is half the 60,000 ms body budget + SweepPressureClassifier sums single-run costs against, so it must run DETACHED from the sequential + body (DarlingWorker, beside query_store and plan_correction), and SweepBodyDetachPolicyTests pins + that nothing detached sits on the one-minute tier. The extension arm pays 12x the rows it did + hourly (at most 500 per cycle, cumulative) for deltas twelve times finer. */ + ["pg_wait_sampling"] = new(5, 30), ["pg_kernel_stats"] = new(60, 30), ["pg_predicate_stats"] = new(60, 30), ["pg_plan_capture"] = new(60, 14), diff --git a/PerformanceMonitor.Collectors/CollectorTargetInfo.cs b/PerformanceMonitor.Collectors/CollectorTargetInfo.cs index 92a0cc454..cc6451ae5 100644 --- a/PerformanceMonitor.Collectors/CollectorTargetInfo.cs +++ b/PerformanceMonitor.Collectors/CollectorTargetInfo.cs @@ -127,4 +127,30 @@ than more parallel properties. ---- */ /// surfaces are writer-only and gate off this. /// public bool IsInRecovery { get; init; } + + /// + /// True when the pg_wait_sampling extension is CREATED in the database the monitoring connection + /// lands in (pg_extension, probed once at connect), which is the fact that decides which of the two + /// stock-PostgreSQL wait instruments reads on this target (#3604). + /// + /// Why a connect-time fact rather than a per-cycle probe. The three wait tiers — Aurora's + /// engine-cumulative counters, the extension's 10 ms profiler, and the service-side sampler that is the + /// floor under both — are ONE readiness decision made at target attach, the same way + /// decides pg_wait_stats: exactly one instrument feeds each target's wait history, and a reader + /// asking "which one, and why" gets an answer that holds for the connection's life rather than one that + /// could flip between two cycles. Installing the extension is a deliberate act followed by a + /// shared_preload_libraries restart, so "re-derived at the next connect" is the natural moment for + /// the tier to move, and the connect-scoped precondition vocabulary already tells operators so. + /// + /// A FIXABLE fact, so the engine-capability sweep varies it + /// (CollectorEngineCapability.TargetsWithEngineKind): a target without the extension today may + /// have it tomorrow, so no permanence claim is ever made on it — the same rule the version floors and + /// follow. EveryFactAPostgresGateReads_IsVariedBySweepOrFixedByKind + /// fails the build if a gate reads this and the sweep leaves it at its default. + /// + /// Default false. A SQL Server target never has it; a PostgreSQL target that was not probed (a + /// test double, a hand-built shape) is treated as NOT having the extension, which routes it to the + /// sampler arm — the direction that produces a floor rather than an empty read. + /// + public bool HasPgWaitSamplingExtension { get; init; } } diff --git a/PerformanceMonitor.Collectors/PgWaitInstrument.cs b/PerformanceMonitor.Collectors/PgWaitInstrument.cs new file mode 100644 index 000000000..5f48e3214 --- /dev/null +++ b/PerformanceMonitor.Collectors/PgWaitInstrument.cs @@ -0,0 +1,78 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; + +namespace PerformanceMonitor.Collectors; + +/// +/// The three instruments a PostgreSQL target's wait history can come from, and the one token each read +/// discloses so a caller knows what GRAIN they are looking at (#3604). +/// +/// Why a vocabulary rather than a boolean. Wait accumulation on PostgreSQL had two paths and a +/// hole. Aurora targets get engine-cumulative wait TIME from aurora_stat_system_waits() +/// (); stock targets get a 10 ms sampled profile IF the +/// pg_wait_sampling extension is loaded ('s extension arm); and a +/// stock target WITHOUT the extension — most first installs, and every environment where adding a +/// preload library needs a change ticket — got nothing at all. The same collector now has a third arm that +/// samples pg_stat_activity from the service side, so the hole is a floor. But three sources with +/// three different grains feeding two tables means a number without its instrument is uninterpretable: a +/// count of 300 is 3 seconds of waiting under the extension and 5 minutes under the sampler. Every read +/// that serves any of them names which one it is serving. +/// +/// The selection is ONE decision, made at target attach, in this order: Aurora native > +/// pg_wait_sampling present > service sampler. routes the +/// first (and gates the other two off — Aurora cannot preload the module, and a second sampled series beside +/// the engine's own counters would be a worse answer, not a redundant one); +/// picks between the last two. Exactly one +/// instrument feeds each target's wait history at any time, and the collector records which one it took in +/// its own per-server state so the reads disclose it from the store rather than re-deriving it. +/// +/// Tokens are lower-snake because they travel on the MCP wire as JSON values beside status +/// words of the same shape. +/// +public static class PgWaitInstrument +{ + /// Aurora's aurora_stat_system_waits(): every wait, its count and its measured time, + /// accumulated by the engine since instance start. Parity with sys.dm_os_wait_stats. + public const string EngineCumulative = "engine_cumulative"; + + /// The pg_wait_sampling extension's in-engine profiler: every backend's wait event + /// sampled every profile_period (10 ms by default), attributed to queryid, accumulated + /// since server start or the last pg_wait_sampling_reset_profile(). + public const string ExtensionSampled = "extension_sampled"; + + /// This service polling pg_stat_activity from the outside on a one-second period for a + /// short window each cycle, accumulated across cycles in the collector's own state. A FLOOR, not parity: + /// a wait shorter than the period is observed with probability roughly its length over the period, and + /// nothing between samples or between windows is observed at all. + public const string ServiceSampled = "service_sampled"; + + /// + /// The honest limit of the service tier, in the words every read that serves it appends. One copy, so + /// the MCP read, the empty-state message and the Viewer cannot drift on what the caveat IS — the same + /// reason CollectorEngineCapability keeps its epilogue in one place. + /// + public const string ServiceSampledCaveat = + "service_sampled is a FLOOR, not parity with the pg_wait_sampling extension: this service polls " + + "pg_stat_activity once a second for a short window each cycle, so a wait shorter than a second is " + + "seen with probability roughly its length over one second, nothing between samples or between " + + "windows is seen at all, and a burst that fits inside the gap between two windows is missed " + + "entirely. Shares of the profile are trustworthy for anything that is a steady fraction of the " + + "server's time; rare short events are under-counted. estimated_wait_ms is samples multiplied by " + + "the one-second period. Installing pg_wait_sampling (shared_preload_libraries, then CREATE " + + "EXTENSION in the monitored database, then let the service reconnect) moves this target to the " + + "extension_sampled tier with no other change."; + + /// True for exactly the three tokens above; the reads use it to keep an unrecognised state + /// value from being echoed as an instrument. + public static bool IsKnown(string? instrument) => + string.Equals(instrument, EngineCumulative, StringComparison.Ordinal) + || string.Equals(instrument, ExtensionSampled, StringComparison.Ordinal) + || string.Equals(instrument, ServiceSampled, StringComparison.Ordinal); +} diff --git a/PerformanceMonitor.Collectors/PgWaitSamplingCollector.cs b/PerformanceMonitor.Collectors/PgWaitSamplingCollector.cs index 5af37e7c7..da25a2d5c 100644 --- a/PerformanceMonitor.Collectors/PgWaitSamplingCollector.cs +++ b/PerformanceMonitor.Collectors/PgWaitSamplingCollector.cs @@ -59,6 +59,61 @@ namespace PerformanceMonitor.Collectors; /// mean one thing. Grouping by (event_type, event, queryid) is the shape a read actually wants, and /// backend_count preserves the one thing the pid dimension was carrying: how many backends were /// waiting this way. +/// +/// 6. Two arms, one table, one instrument per target (#3604). A stock target WITHOUT the +/// extension used to get nothing from this collector but an hourly EXTENSION_MISSING skip — and +/// nothing from either, which is Aurora-only. For the SQL Server DBA whose +/// whole performance worldview is dm_os_wait_stats, the flagship diagnostic dimension was simply +/// absent on exactly the targets most likely to be their first, and an empty wait chart reads as "the +/// product is broken". So this definition now has a SECOND arm: when +/// is false it polls pg_stat_activity +/// itself — one-second snapshots per cycle, run as ONE multi-statement +/// command so the whole window is one collection and one collection_log row — and accumulates the +/// tallies across cycles in its own per-server state () so the rows it writes are +/// CUMULATIVE like the extension's and the existing read differences them unchanged. Which arm ran is +/// recorded beside the tally (, a token) so +/// every read can disclose the grain it is serving. It is the same shape as cpu_utilization's +/// Azure-vs-ring-buffer fork and pg_statement_stats' Aurora-vs-vanilla one: a target-aware definition, +/// not a second collector, because the catalog's one-collector-one-table rule is what the schema generator, +/// retention and the table census all rest on. +/// +/// Why the sampler is a FLOOR and says so. The extension samples in-engine every 10 ms and misses +/// nothing a backend did for longer than that. This arm samples from outside every second for +/// seconds of every five-minute cycle — a 10% duty cycle. A wait shorter +/// than a second is seen with probability roughly its length over a second; nothing between samples, or +/// between windows, is seen at all. Shares of the profile are trustworthy for anything that is a steady +/// fraction of the server's time, which is the question a wait chart answers first; rare short events are +/// under-counted and a burst that fits between two windows is missed. The window is 30 s and not the whole +/// cycle for a reason outside this file: SweepPressureClassifier sums every collector's single-run +/// cost against a 60,000 ms body budget and calls the body BODY_OVERRUN past it, so a four-minute +/// sampling run would make every stock target read as saturated. Thirty seconds at one Hz is the densest +/// honest floor that leaves that surface truthful, and it is still thirty times the resolution +/// pg_blocking and pg_lock_stats — the two existing PostgreSQL samplers — get from one snapshot a +/// minute. The period is one second rather than five because each doubling of the period halves the odds of +/// seeing a wait shorter than it, while the cost — a shared-memory read of pg_stat_activity, well under +/// a millisecond at hundreds of backends, from a connection the pool holds open regardless — does not move. +/// +/// Why the sampler reads are separate statements. pg_stat_activity is snapshotted once per +/// transaction on first access and the same rows are returned for the rest of it; a single statement that +/// slept and re-read in a loop would return thirty copies of one instant. Each snapshot is therefore its own +/// statement, preceded by pg_stat_clear_snapshot() — its documented purpose — and a pg_sleep, +/// and the three are repeated in the command text so walks result sets rather than +/// rows. Under pg_monitor (the grant every PostgreSQL collector here already needs) all three are +/// callable and every backend's wait columns are visible; the arm adds no privilege. +/// +/// What the sampler excludes. Its own backend (pg_backend_pid()), every other connection +/// this service holds open (by application_name — the two names the connector presents), and the same +/// three wait TYPES the extension arm excludes, spliced from the same set so the two arms cannot disagree +/// about what counts as a wait. A running backend (state = 'active', no wait) is CPU/ +/// Running, as on the extension arm; an idle one is Client/ClientRead and drops out by +/// type. query_id is PostgreSQL 14+, and on 13 the column is not selected rather than errored on. +/// +/// Aurora is gated OFF — is !IsAurora now, where it used to be +/// true. Aurora cannot preload pg_wait_sampling, so the hourly EXTENSION_MISSING it recorded +/// there was a permanent gap wearing a fixable precondition's clothes; and the sampler arm running beside +/// pg_wait_stats would be two answers to one question on the one engine that has the real one. +/// CollectorEngineCapability.CoveredInsteadBy now points an Aurora caller of get_pg_wait_sampling +/// at get_pg_wait_stats, the mirror of the pointer that already ran the other way. /// public sealed class PgWaitSamplingCollector : PostgresCollectorDefinitionBase { @@ -113,6 +168,83 @@ one INSTEAD of that one on stock PostgreSQL. private static readonly string IgnoredTypeList = string.Join(", ", PgWaitStatsCollector.IgnoredWaitTypes.OrderBy(t => t, StringComparer.Ordinal).Select(t => $"'{t}'")); + /// The state key under which the arm that ran records its token + /// (#3604), per server, every cycle — what the reads disclose as instrument. + public const string InstrumentStateKey = "instrument"; + + /// The state key holding the sampler arm's cumulative tally between cycles (#3604): one line per + /// (event_type, event, query_id), tab-separated, count last. Text because collector_state is text; + /// lines rather than JSON so the parser is a split and a bad line is one lost key rather than a lost + /// tally. Capped at keys. + public const string TallyStateKey = "sampler_tally"; + + /// One-second snapshots per cycle on the sampler arm — the window is this many seconds. See the + /// type header for why thirty and not the whole cycle. + public const int SamplerSnapshotsPerCycle = 30; + + /// The sampler's period, stored in every row it writes as profile_period_ms so the read's + /// samples-times-period estimate is right for this arm too. + public const int SamplerPeriodMs = 1000; + + /// Most keys the tally keeps — the same 500 the extension arm's LIMIT ships. Past it the + /// least-sampled keys are dropped; a dropped key seen again starts over, which the read shows as that + /// one series resetting rather than as anything wrong with the others. + public const int SamplerTallyCap = 500; + + /// The application_name this service's monitoring connections present, excluded from the + /// sampler so the tool does not count its own collectors as the server's workload. Pinned equal to the + /// connector's literal by test; this assembly cannot reference the service project. + public const string ServiceApplicationName = "PerformanceMonitorDarling"; + + /// The remediation connections' application_name, excluded for the same reason. + public const string ServiceRemediationApplicationName = "PerformanceMonitorDarling-Remediation"; + + /// + /// One pg_stat_activity snapshot, as the sampler arm reads it: four columns, one row per backend + /// that is waiting on something the profile counts or is on CPU. Public so a test can run it alone. + /// query_id exists from PostgreSQL 14; on 13 (or an unknown version reading as 0 — which the + /// gates treat as "newest", so it selects the column) the caller passes + /// false and the column is a constant 0, the same "belongs to no statement" value the extension arm + /// stores for a background process. + /// + public static string SamplerSnapshotSql(bool hasQueryId) => @" +SELECT + coalesce(a.wait_event_type, 'CPU')::text AS event_type, + coalesce(a.wait_event, 'Running')::text AS event, + " + (hasQueryId ? "coalesce(a.query_id, 0)::bigint" : "0::bigint") + @" AS query_id, + a.pid::int AS pid +FROM pg_stat_activity AS a +WHERE a.pid <> pg_backend_pid() + AND coalesce(a.application_name, '') NOT IN ('" + ServiceApplicationName + "', '" + ServiceRemediationApplicationName + @"') + AND (a.wait_event_type IS NOT NULL OR a.state = 'active') + AND coalesce(a.wait_event_type, 'CPU') NOT IN (" + IgnoredTypeList + ")"; + + /// + /// The sampler arm's whole cycle as ONE command: snapshot, then (sleep one period, clear the backend + /// snapshot cache, snapshot) repeated − 1 times. Statements in a + /// batch execute in order, so the sleep-clear-read sequence is guaranteed without relying on evaluation + /// order inside a statement. Public so a test can count its statements against the constant. + /// + public static string SamplerBatchSql(bool hasQueryId) + { + var snapshot = SamplerSnapshotSql(hasQueryId); + var sb = new System.Text.StringBuilder(); + sb.Append(snapshot).Append(';'); + for (var i = 1; i < SamplerSnapshotsPerCycle; i++) + { + /* InvariantCulture, like every number this file puts in SQL or state: StringBuilder.Append(double) + formats under the host's culture, and a half-second period on a comma-decimal host would emit + pg_sleep(0,5) - a review catch on #3645 before any period but 1.0 ever shipped. */ + sb.Append("\nSELECT pg_sleep(") + .Append((SamplerPeriodMs / 1000.0).ToString(System.Globalization.CultureInfo.InvariantCulture)) + .Append(");") + .Append("\nSELECT pg_stat_clear_snapshot();") + .Append(snapshot).Append(';'); + } + + return sb.ToString(); + } + private static string QueryText => @" SELECT /* A NULL wait event means the backend was NOT waiting - it was on CPU. That is PostgreSQL's own @@ -150,8 +282,14 @@ ORDER BY sum(p.count) DESC, coalesce(p.event_type, 'CPU'), coalesce(p.event, 'Ru /// /// No version gate. The extension supports PostgreSQL 13+ and the query uses nothing /// version-conditional, so a gate here would only be a second thing to keep in step with reality. + /// + /// Since #3604: every PostgreSQL target that is NOT Aurora. Aurora cannot preload the module, + /// so the skip above was permanent there rather than a resting state, and the sampler arm this definition + /// grew must not run beside pg_wait_stats on the one engine that has real counters. The + /// engine-capability sweep fixes IsAurora per kind, so this reads as a permanent gap on the + /// aurora-postgres kind with a pointer to the instrument that does answer there. /// - public override bool AppliesTo(CollectorTargetInfo target) => true; + public override bool AppliesTo(CollectorTargetInfo target) => !target.IsAurora; /// /// Cluster-wide. The profile covers every backend on the instance and carries no database column, so @@ -164,13 +302,51 @@ ORDER BY sum(p.count) DESC, coalesce(p.event_type, 'CPU'), coalesce(p.event, 'Ru /// Preloaded, and the restart is the reason an operator sees nothing here for so long. This is /// the one dependency PgExtensionAvailabilityCollector's roster deliberately omits, so a /// consumer reading that roster as the dependency set misses exactly this collector. + /// + /// Still declared after #3604, with a narrower meaning. The EXTENSION arm cannot run without + /// it; the sampler arm is what runs instead when the connect probe finds it absent, so a missing module is + /// no longer a skip — it is the coarser tier. The declaration stays because it is what the README's + /// permissions paragraph is pinned to (the operator still needs to know what to install to reach the + /// finer tier), and because a module DROPPED mid-connection on the extension arm still fails with + /// 42P01, which this declaration is what classifies as EXTENSION_MISSING rather than a + /// permissions fault until the next connect re-decides the arm. /// public override IReadOnlyList RequiredPgExtensions { get; } = new[] { new PgExtensionDependency("pg_wait_sampling", PgExtensionInstallKind.SharedPreloadLibraries), }; - public override CollectorQuery BuildQuery(CollectorContext context) => new(QueryText); + /// + /// The arm (#3604): the extension's profile when the connect probe found pg_wait_sampling created in + /// this database, the service sampler's batch when it did not. Decided off + /// rather than probed here so the choice is the connect-time one every other tier fact is. + /// + public override CollectorQuery BuildQuery(CollectorContext context) => + context.Target.HasPgWaitSamplingExtension + ? new CollectorQuery(QueryText) + : new CollectorQuery(SamplerBatchSql(HasQueryIdColumn(context.Target))); + + /// + /// The sampler batch holds − 1 seconds of deliberate sleep and its + /// results are buffered by the server until the batch ends, so the client sees nothing for the whole + /// window. The default 60 s would fit today's 30 s window but leave a cluster under load — where a snapshot + /// itself can take longer — with no headroom; 120 s bounds the arm at four times its window. Applies to + /// the extension arm too, where a longer ceiling on a 500-row read costs nothing. + /// + public override int? CommandTimeoutSecondsOverride => 120; + + /// + /// Both arms record which instrument ran (); the sampler arm also carries + /// its cumulative tally between cycles (). Declared so the host loads them + /// before the cycle and persists them after — the #1962 mechanism for state a MAX() over the table cannot + /// recover, which a tally the table only holds as its LAST written value is. + /// + public override IReadOnlyList StateKeys { get; } = new[] { InstrumentStateKey, TallyStateKey }; + + /// pg_stat_activity.query_id arrived in PostgreSQL 14. 0 is "unknown", which every gate + /// here reads as newest. + private static bool HasQueryIdColumn(CollectorTargetInfo target) => + target.PostgresMajorVersion == 0 || target.PostgresMajorVersion >= 14; public override IReadOnlyList PayloadColumns { get; } = new[] { @@ -186,6 +362,26 @@ ORDER BY sum(p.count) DESC, coalesce(p.event_type, 'CPU'), coalesce(p.event, 'Ru public override async ValueTask> ReadAsync(DbDataReader reader, CollectorContext context, CancellationToken cancellationToken) { + if (!context.Target.HasPgWaitSamplingExtension) + { + return await ReadSamplerAsync(reader, context, cancellationToken); + } + + /* The arm that ran, recorded every cycle rather than once: a target that gains or loses the extension + changes arm at its next connect, and the reads must follow on the next cycle, not the next restart. */ + context.PendingState[InstrumentStateKey] = PgWaitInstrument.ExtensionSampled; + + /* And the sampler's tally CLEARED, every cycle this arm runs (#3645 review). The host persists only + the keys in PendingState, so an extension-arm cycle that left this key alone would leave the last + sampler tally sitting in collector_state for as long as the extension stayed installed — and a + target that later fell back to the sampler arm (extension dropped, a reconnect) would resume + accumulating from a months-old baseline. Usually that reads as an honest counter_reset (the stale + tally is smaller than the extension's cumulative counts); but if the extension's own profile had + just been reset before the fallback, a stale-but-larger tally would look like continued monotonic + growth with no reset flag, splicing two unrelated eras into one series. An empty tally is what + ParseTally reads as "start over", so the fallback starts at zero and the read reports the reset. */ + context.PendingState[TallyStateKey] = string.Empty; + var rows = new List(); while (await reader.ReadAsync(cancellationToken)) @@ -204,6 +400,129 @@ public override async ValueTask> ReadAsync(DbDataReader reader, Collec return rows; } + /// + /// The sampler arm's read (#3604): walks the batch's result sets, tallies every four-column snapshot row + /// by (event_type, event, query_id), folds the window into the tally carried in from the previous cycle, + /// and emits the CUMULATIVE tally as rows in the extension arm's shape. Internal and reader-shaped so a + /// test can drive it with a fixture batch and no server. + /// + /// Result sets, not rows, are the unit. A pg_sleep or pg_stat_clear_snapshot() + /// result set is one column wide and is drained; a snapshot is four. Counting snapshots by shape rather + /// than by position means a batch that a future edit reorders still tallies only what it should. + /// + /// backend_count is the WINDOW's distinct pids, not a cumulative one: pids recycle, so a + /// distinct count across cycles would grow without meaning. The extension arm's figure is cumulative + /// because the module's is; the column means "how many backends were doing this" on both, over the + /// span each instrument can honestly speak for. + /// + internal static async ValueTask> ReadSamplerAsync(DbDataReader reader, CollectorContext context, CancellationToken cancellationToken) + { + var window = new Dictionary<(string Type, string Event, long QueryId), (long Samples, HashSet Pids)>(); + + do + { + if (reader.FieldCount == 4) + { + while (await reader.ReadAsync(cancellationToken)) + { + var key = ( + reader.IsDBNull(0) ? "CPU" : reader.GetString(0), + reader.IsDBNull(1) ? "Running" : reader.GetString(1), + reader.IsDBNull(2) ? 0L : reader.GetInt64(2)); + var pid = reader.IsDBNull(3) ? 0 : reader.GetInt32(3); + + if (!window.TryGetValue(key, out var seen)) + { + seen = (0, new HashSet()); + } + + seen.Pids.Add(pid); + window[key] = (seen.Samples + 1, seen.Pids); + } + } + else + { + while (await reader.ReadAsync(cancellationToken)) + { + } + } + } + while (await reader.NextResultAsync(cancellationToken)); + + var tally = ParseTally(context.State.TryGetValue(TallyStateKey, out var carried) ? carried : null); + foreach (var (key, seen) in window) + { + tally[key] = tally.TryGetValue(key, out var prior) ? prior + seen.Samples : seen.Samples; + } + + /* Cap by dropping the least-sampled: the extension arm ships its top 500 by count, and a tally that + grew with every distinct query_id ever seen waiting would make the state row unbounded. */ + if (tally.Count > SamplerTallyCap) + { + foreach (var victim in tally.OrderBy(kv => kv.Value).ThenBy(kv => kv.Key.Type, StringComparer.Ordinal) + .ThenBy(kv => kv.Key.Event, StringComparer.Ordinal).ThenBy(kv => kv.Key.QueryId) + .Take(tally.Count - SamplerTallyCap).Select(kv => kv.Key).ToList()) + { + tally.Remove(victim); + } + } + + context.PendingState[TallyStateKey] = SerializeTally(tally); + context.PendingState[InstrumentStateKey] = PgWaitInstrument.ServiceSampled; + + return tally + .OrderByDescending(kv => kv.Value) + .ThenBy(kv => kv.Key.Type, StringComparer.Ordinal) + .ThenBy(kv => kv.Key.Event, StringComparer.Ordinal) + .Select(kv => new Row( + EventType: kv.Key.Type, + Event: kv.Key.Event, + QueryId: kv.Key.QueryId, + SampleCount: kv.Value, + ProfilePeriodMs: SamplerPeriodMs, + BackendCount: window.TryGetValue(kv.Key, out var seen) ? seen.Pids.Count : 0)) + .ToList(); + } + + /// The tally's text form: one type\tevent\tquery_id\tcount line per key. Tabs because + /// no wait event name carries one; a line that does not parse is skipped rather than failing the cycle. + public static Dictionary<(string Type, string Event, long QueryId), long> ParseTally(string? text) + { + var tally = new Dictionary<(string, string, long), long>(); + if (string.IsNullOrEmpty(text)) + { + return tally; + } + + foreach (var line in text.Split('\n', StringSplitOptions.RemoveEmptyEntries)) + { + var parts = line.Split('\t'); + if (parts.Length != 4 + || !long.TryParse(parts[2], System.Globalization.NumberStyles.Integer, System.Globalization.CultureInfo.InvariantCulture, out var queryId) + || !long.TryParse(parts[3], System.Globalization.NumberStyles.Integer, System.Globalization.CultureInfo.InvariantCulture, out var count)) + { + continue; + } + + tally[(parts[0], parts[1], queryId)] = count; + } + + return tally; + } + + public static string SerializeTally(Dictionary<(string Type, string Event, long QueryId), long> tally) + { + var sb = new System.Text.StringBuilder(); + foreach (var (key, count) in tally.OrderByDescending(kv => kv.Value).ThenBy(kv => kv.Key.Type, StringComparer.Ordinal).ThenBy(kv => kv.Key.Event, StringComparer.Ordinal)) + { + sb.Append(key.Type).Append('\t').Append(key.Event).Append('\t') + .Append(key.QueryId.ToString(System.Globalization.CultureInfo.InvariantCulture)).Append('\t') + .Append(count.ToString(System.Globalization.CultureInfo.InvariantCulture)).Append('\n'); + } + + return sb.ToString(); + } + public override void WritePayload(Row row, ICollectorRowWriter writer, CollectorContext context) { /* No deltas here. The profile is CUMULATIVE and is stored that way on purpose: a reset or a restart diff --git a/docs/postgres-first-target-runbook.md b/docs/postgres-first-target-runbook.md index eba661232..a521ba5ac 100644 --- a/docs/postgres-first-target-runbook.md +++ b/docs/postgres-first-target-runbook.md @@ -77,6 +77,18 @@ so check it with `SHOW shared_preload_libraries;`, not `pg_extension`). Skipping the collector records a non-fatal skip naming exactly which install is missing (step 10), and the other 26 collectors are unaffected. +`pg_wait_sampling` is the one whose absence is no longer a skip (#3604). Without the extension the same +collector takes its **service-sampler arm**: it polls `pg_stat_activity` once a second for a 30-second +window every five minutes and accumulates what it saw, so a stock target has a wait profile from its first +cycle rather than an empty chart. That arm needs **nothing beyond the `pg_monitor` grant in step 1** — +`pg_stat_activity`'s wait columns for other users' backends come from `pg_read_all_stats`, which +`pg_monitor` carries, and `pg_stat_clear_snapshot()` and `pg_sleep()` are callable by any role. It is a +floor, not parity: a wait shorter than a second is seen with probability roughly its length over a second, +and nothing between windows is seen. Every read of that table (`get_pg_wait_sampling`, the Viewer's panel) +says which arm fed it as `instrument` — `extension_sampled` or `service_sampled` — and the three-tier +selection (Aurora native › extension › service sampler) is made once when the service connects, so after +installing the extension let the service reconnect to move up a tier. + ### Self-hosted: the log-reader grants (plan capture, deadlocks) `pg_deadlocks` and `pg_plan_capture` read the server log with `pg_read_file()`, and that is the one read @@ -354,7 +366,7 @@ All 27, from `CollectorScheduleDefaults` — the shared table both SKUs schedule | `pg_server_config` | 60 min | 60 min | 2 h for the *changes* read (a change needs two snapshots) | | `pg_plan_capture_readiness` | 60 min | 60 min | 60 min (facets are levels) | | `pg_plan_capture` | 60 min | **60 min** | the first plan `auto_explain` logs past its threshold | -| `pg_wait_sampling` | 60 min | 60 min | 2 h (counters) | +| `pg_wait_sampling` | 5 min | 5 min | 10 min (counters; on the service-sampler arm the second cycle also fills the first window) | | `pg_kernel_stats` | 60 min | 60 min | 2 h (counters) | | `pg_predicate_stats` | 60 min | 60 min | 2 h (counters) | | `pg_buffer_usage` | 60 min | 60 min | 60 min (residency is a level) | @@ -363,10 +375,12 @@ All 27, from `CollectorScheduleDefaults` — the shared table both SKUs schedule | `pg_column_stats` | 24 h | **24 h** | 24 h (levels; "it moved" needs two) | | `pg_extension_availability` | 24 h | **24 h** | 24 h (levels) | -The four extension-backed hourly ones (`pg_wait_sampling`, `pg_kernel_stats`, `pg_predicate_stats`, -`pg_plan_capture`) are hourly for the fleet's sake rather than the data's: on a managed target they can -only ever record a non-fatal skip, and a five-minute cadence would be ~900 skip rows per target per day -saying the same thing. Their counters lose nothing at the longer interval — with one exception the +The three extension-backed hourly ones (`pg_kernel_stats`, `pg_predicate_stats`, `pg_plan_capture`) are +hourly for the fleet's sake rather than the data's: on a managed target they can only ever record a +non-fatal skip, and a five-minute cadence would be hundreds of skip rows per target per day saying the +same thing. `pg_wait_sampling` left that group in #3604: it is gated off Aurora (which cannot load the +module) and takes the service-sampler arm everywhere else, so it records that skip nowhere, and its +five-minute cadence is the sampler arm's duty cycle (30 s of every 300). Their counters lose nothing at the longer interval — with one exception the schedule's own remarks spell out: `pg_plan_capture`'s self-hosted route is a fixed 4 MB log tail with no resume marker, so a server logging faster than that window covers silently loses plans between reads, and a self-hosted operator lowers the cadence per server rather than relying on the default. @@ -663,7 +677,8 @@ re-probes the target — which is how a promoted reader stops being gated as a s | The connect line says SQL Server, or the error mentions `SqlException` / a TDS handshake against 5432 | the target lost its `engine` on the way through the registry. Requires store schema **v70+**: `SELECT name, engine, port FROM config.config_monitored_servers;` — if `engine` is not a column, this build predates the fix and no darling.json edit will help | | Autovacuum health reports everything fine on a cluster you know is behind | you are reading a **reader**. It reports all zeros, not an error. Measured: writer 13,654,458 dead tuples, reader 0, same cluster and tables | | Added a target to darling.json and nothing happened | the store was already seeded; use `add_servers` (step 2) | -| `pg_wait_stats` empty, everything else fine | not Aurora. Core PostgreSQL has no cumulative wait counters at all, and the gap message says so — `pg_wait_sampling` answers the same question from its extension | +| `pg_wait_stats` empty, everything else fine | not Aurora. Core PostgreSQL has no cumulative wait counters at all, and the gap message says so — `pg_wait_sampling` answers the same question from its extension, or from the service sampler when the extension is absent | +| `get_pg_wait_sampling` says `instrument: service_sampled` | expected on a stock target without `pg_wait_sampling` — the floor tier (#3604). Every share is of one-second polls in a 30-second window per five-minute cycle; install the extension and let the service reconnect to reach `extension_sampled` | | `get_pg_top_queries` empty on self-hosted | `pg_stat_statements` not installed (step 1's optional half) — since #2625 the collector reads the vanilla view anywhere the extension exists | | Only some databases in `pg_autovacuum_stats` | by design: `datallowconn` and non-template only, minus `rdsadmin` on a managed instance (it rejects every customer principal) and your `excludedDatabases` | | Store stopped compressing after adding targets | background workers (step 4). Silent — check the postmaster log for "out of background workers" | From 9708928f650f498b54ef24515d4799722d28c56b Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 17:58:46 -0400 Subject: [PATCH 67/69] Top Cpu Queries' Max Dop is the newest plan's reading with the cross-plan maximum kept as a dated history, so a MAXDOP-1 instance stops reporting 16 from a plan compiled before the pin (#3648) (#3651) * Top Cpu Queries' Max Dop is the newest plan's reading with the cross-plan maximum kept as a dated history, so a MAXDOP-1 instance stops reporting 16 from a plan compiled before the pin (#3648) * Parity guard moves its cross-SKU comparison to Darling.Tests (the filter that reaches both trees), the live test cleans up through LiveStoreCleanup, and the DuckDB fixture reads UtcNow once so a round-tripped timestamp compares equal (#3648) --- .../DrillDownDopProvenanceLiveTests.cs | 231 +++++++++++++++ .../DrillDownDopProvenanceParityTests.cs | 120 ++++++++ .../Darling.Tests/QueryDopProvenanceTests.cs | 97 ++++++ .../PgDrillDownCollector.Queries.cs | 111 ++++++- .../DrillDownDopProvenanceParityTests.cs | 104 +++++++ Lite.Tests/DrillDownDopProvenanceTests.cs | 278 ++++++++++++++++++ Lite.Tests/QueryDopProvenanceTests.cs | 97 ++++++ Lite/Analysis/DrillDownCollector.Queries.cs | 106 ++++++- .../QueryDopProvenance.cs | 91 ++++++ 9 files changed, 1207 insertions(+), 28 deletions(-) create mode 100644 Darling/Darling.Tests/DrillDownDopProvenanceLiveTests.cs create mode 100644 Darling/Darling.Tests/DrillDownDopProvenanceParityTests.cs create mode 100644 Darling/Darling.Tests/QueryDopProvenanceTests.cs create mode 100644 Lite.Tests/DrillDownDopProvenanceParityTests.cs create mode 100644 Lite.Tests/DrillDownDopProvenanceTests.cs create mode 100644 Lite.Tests/QueryDopProvenanceTests.cs create mode 100644 PerformanceMonitor.Common/QueryDopProvenance.cs diff --git a/Darling/Darling.Tests/DrillDownDopProvenanceLiveTests.cs b/Darling/Darling.Tests/DrillDownDopProvenanceLiveTests.cs new file mode 100644 index 000000000..5c914a757 --- /dev/null +++ b/Darling/Darling.Tests/DrillDownDopProvenanceLiveTests.cs @@ -0,0 +1,231 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Text.Json; +using System.Threading; +using System.Threading.Tasks; +using Npgsql; +using PerformanceMonitor.Analysis; +using PerformanceMonitor.Collectors; +using PerformanceMonitor.Darling.Analysis; +using PerformanceMonitor.Darling.Storage; +using Xunit; + +namespace Darling.Tests; + +/// +/// Live-Postgres pin for #3648, Darling's twin of Lite's DrillDownDopProvenanceTests: the +/// top_cpu_queries and bad_actor_query drill-downs must headline the NEWEST plan's +/// max_dop and carry the cross-plan maximum as a history with provenance, not fold every plan the +/// hash ever had inside the window into one provenance-free number. +/// +/// The fixture is the live page. One query_hash, two plans: an old parallel plan whose +/// per-plan high-water mark reads 16 and was last seen three hours before the window's end, and a new serial +/// plan reading 1 that spent the CPU at the newest snapshot. The old read said max_dop = 16 for this +/// shape on a MAXDOP-1 instance whose stored plan was serial, and a tuning recommendation was made from it +/// and retracted. +/// +/// Live rather than a string pin because the claim is the ENGINE's answer to ROW_NUMBER() OVER +/// with explicit NULLS LAST tie-breakers combined with a partition-wide MAX() OVER — which row +/// the newest-plan CASE picks on Postgres, and that a NULL reading reaches the reader as NULL. The text half +/// (byte-identity with Lite's SQL) lives in Lite.Tests.DrillDownDopProvenanceParityTests. +/// +[Collection("live-postgres")] +public sealed class DrillDownDopProvenanceLiveTests +{ + private const int TestServerId = -364800; + private const string TestServerName = "DopProvSrv"; + private const string Db = "DopProvDb"; + private const string Hash = "0x3648HASH"; + private const string OldParallelPlan = "0x3648PLANPARALLEL"; + private const string NewSerialPlan = "0x3648PLANSERIAL"; + + private static DateTime TruncateToSeconds(DateTime t) => + DateTime.SpecifyKind(new DateTime(t.Ticks - (t.Ticks % TimeSpan.TicksPerSecond)), DateTimeKind.Unspecified); + + private static async Task OpenWithSearchPathAsync(string connectionString, CancellationToken ct) + { + var connection = new NpgsqlConnection(connectionString); + await connection.OpenAsync(ct); + await using var setPath = new NpgsqlCommand("SET search_path = " + PgSchemaGenerator.SearchPath, connection); + await setPath.ExecuteNonQueryAsync(ct); + return connection; + } + + private static async Task SeedAsync( + NpgsqlConnection c, string planHash, DateTime collectionTime, int? maxDop, long workerTimeUs, + DateTime? creationTime, CancellationToken ct) + { + await using var command = new NpgsqlCommand(@" +INSERT INTO query_stats + (collection_id, collection_time, server_id, server_name, database_name, query_hash, query_plan_hash, + sql_handle, plan_handle, creation_time, query_text, delta_execution_count, delta_worker_time, + delta_elapsed_time, delta_logical_reads, delta_spills, min_dop, max_dop) +VALUES + ($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,$11,$12,$13,$14,$15,$16,$17,$18)", c); + command.Parameters.AddWithValue(CollectionIdGenerator.Next()); + command.Parameters.AddWithValue(DateTime.SpecifyKind(collectionTime, DateTimeKind.Unspecified)); + command.Parameters.AddWithValue(TestServerId); + command.Parameters.AddWithValue(TestServerName); + command.Parameters.AddWithValue(Db); + command.Parameters.AddWithValue(Hash); + command.Parameters.AddWithValue(planHash); + command.Parameters.AddWithValue("0x3648SQLH"); + command.Parameters.AddWithValue(planHash + "H"); + command.Parameters.AddWithValue(DateTime.SpecifyKind(creationTime ?? collectionTime.AddDays(-1), DateTimeKind.Unspecified)); + command.Parameters.AddWithValue("SELECT * FROM DopProvTable"); + command.Parameters.AddWithValue(10L); + command.Parameters.AddWithValue(workerTimeUs); + command.Parameters.AddWithValue(workerTimeUs * 2); + command.Parameters.AddWithValue(1000L); + command.Parameters.AddWithValue(0L); + command.Parameters.AddWithValue(1); + command.Parameters.AddWithValue(maxDop.HasValue ? maxDop.Value : DBNull.Value); + await command.ExecuteNonQueryAsync(ct); + } + + private static async Task DeleteTestRowsAsync(NpgsqlConnection connection, CancellationToken ct) + { + await using var command = new NpgsqlCommand("DELETE FROM query_stats WHERE server_id = $1", connection); + command.Parameters.AddWithValue(TestServerId); + await command.ExecuteNonQueryAsync(ct); + } + + private static async Task CollectAsync( + NpgsqlDataSource postgres, AnalysisContext context, string factKey, string drillDownKey) + { + var finding = new AnalysisFinding + { + RootFactKey = factKey, + StoryPath = factKey, + /* Past the display gate — below it the expensive drill-downs are skipped wholesale and this + collector never runs at all. */ + Severity = 1.0, + }; + + await new PgDrillDownCollector(postgres).EnrichFindingsAsync([finding], context); + + Assert.NotNull(finding.DrillDown); + Assert.True(finding.DrillDown.TryGetValue(drillDownKey, out var raw), $"{drillDownKey} was not collected"); + return JsonSerializer.SerializeToElement(raw); + } + + [Fact] + public async Task TopCpuAndBadActor_HeadlineTheNewestPlansDop_AndCarryTheParallelHistory() + { + var connectionString = Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + Assert.SkipWhen(string.IsNullOrEmpty(connectionString), + "Set DARLING_TEST_PG to a Postgres connection string to run the live #3648 DOP-provenance test."); + + var ct = TestContext.Current.CancellationToken; + var bodySucceeded = false; + + await using (var connection = new NpgsqlConnection(connectionString)) + { + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + } + + await using (var connection = await OpenWithSearchPathAsync(connectionString!, ct)) + { + await DeleteTestRowsAsync(connection, ct); + } + + try + { + var periodEnd = TruncateToSeconds(DateTime.UtcNow); + var periodStart = periodEnd.AddHours(-4); + var parallelLastSeen = periodEnd.AddHours(-3); + var context = new AnalysisContext + { + ServerId = TestServerId, + ServerName = TestServerName, + TimeRangeStart = periodStart, + TimeRangeEnd = periodEnd, + ServerUtcOffset = TimeSpan.Zero, + }; + + var expectedNote = $"DOP 1 (a parallel plan ran at 16 until {parallelLastSeen:yyyy-MM-dd}; 2 plans in window)"; + + /* ── The live page, RED before #3648 (max_dop read 16): three parallel snapshots ending three + hours before the window's end, then the serial plan at the newest snapshot. ── */ + await using (var connection = await OpenWithSearchPathAsync(connectionString!, ct)) + { + await SeedAsync(connection, OldParallelPlan, parallelLastSeen.AddMinutes(-20), 16, 100_000, null, ct); + await SeedAsync(connection, OldParallelPlan, parallelLastSeen.AddMinutes(-10), 16, 100_000, null, ct); + await SeedAsync(connection, OldParallelPlan, parallelLastSeen, 16, 100_000, null, ct); + await SeedAsync(connection, NewSerialPlan, periodEnd.AddMinutes(-30), 1, 400_000, null, ct); + } + + await using (var postgres = NpgsqlDataSource.Create(connectionString!)) + { + var top = Assert.Single((await CollectAsync(postgres, context, "CPU_SQL_PERCENT", "top_cpu_queries")).EnumerateArray()); + Assert.Equal(1, top.GetProperty("max_dop").GetInt32()); + Assert.Equal(16, top.GetProperty("max_dop_any_plan").GetInt32()); + Assert.Equal(2, top.GetProperty("plan_count").GetInt64()); + Assert.Equal(parallelLastSeen, DateTime.Parse(top.GetProperty("max_dop_any_plan_last_seen").GetString()!, null, + System.Globalization.DateTimeStyles.RoundtripKind)); + Assert.Equal(expectedNote, top.GetProperty("dop_note").GetString()); + /* The windowed total still spans BOTH plans: 3 x 100000 + 400000 us = 700 ms. */ + Assert.Equal(700.0, top.GetProperty("total_cpu_ms").GetDouble()); + + var bad = await CollectAsync(postgres, context, "BAD_ACTOR_" + Hash, "bad_actor_query"); + Assert.Equal(1, bad.GetProperty("max_dop").GetInt32()); + Assert.Equal(16, bad.GetProperty("max_dop_any_plan").GetInt32()); + Assert.Equal(2, bad.GetProperty("plan_count").GetInt64()); + Assert.Equal(expectedNote, bad.GetProperty("dop_note").GetString()); + } + + /* ── 0 is unknown: a NULL reading arrives as JSON null on every DOP field and no history is + invented from nothing. ── */ + await using (var connection = await OpenWithSearchPathAsync(connectionString!, ct)) + { + await DeleteTestRowsAsync(connection, ct); + await SeedAsync(connection, NewSerialPlan, periodEnd.AddMinutes(-30), null, 400_000, null, ct); + } + + await using (var postgres = NpgsqlDataSource.Create(connectionString!)) + { + var top = Assert.Single((await CollectAsync(postgres, context, "CPU_SQL_PERCENT", "top_cpu_queries")).EnumerateArray()); + Assert.Equal(JsonValueKind.Null, top.GetProperty("max_dop").ValueKind); + Assert.Equal(JsonValueKind.Null, top.GetProperty("max_dop_any_plan").ValueKind); + Assert.Equal(JsonValueKind.Null, top.GetProperty("max_dop_any_plan_last_seen").ValueKind); + Assert.Equal(JsonValueKind.Null, top.GetProperty("dop_note").ValueKind); + } + + /* ── Both plans at the newest snapshot (the stale parallel plan still cached beside the serial + one that replaced it): collection_time ties, so compile time decides, and the plan + compiled LATER is the headline whatever its counter says. Postgres would default a DESC + sort to NULLS FIRST where DuckDB defaults NULLS LAST — the explicit NULLS LAST is what + keeps this row choice identical across the SKUs. ── */ + var newest = periodEnd.AddMinutes(-30); + await using (var connection = await OpenWithSearchPathAsync(connectionString!, ct)) + { + await DeleteTestRowsAsync(connection, ct); + await SeedAsync(connection, OldParallelPlan, newest, 16, 500_000, periodEnd.AddDays(-20), ct); + await SeedAsync(connection, NewSerialPlan, newest, 1, 100_000, periodEnd.AddDays(-1), ct); + } + + await using (var postgres = NpgsqlDataSource.Create(connectionString!)) + { + var top = Assert.Single((await CollectAsync(postgres, context, "CPU_SQL_PERCENT", "top_cpu_queries")).EnumerateArray()); + Assert.Equal(1, top.GetProperty("max_dop").GetInt32()); + Assert.Equal(16, top.GetProperty("max_dop_any_plan").GetInt32()); + Assert.Equal(newest, DateTime.Parse(top.GetProperty("max_dop_any_plan_last_seen").GetString()!, null, + System.Globalization.DateTimeStyles.RoundtripKind)); + } + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(connectionString!, bodySucceeded, DeleteTestRowsAsync); + } + } +} diff --git a/Darling/Darling.Tests/DrillDownDopProvenanceParityTests.cs b/Darling/Darling.Tests/DrillDownDopProvenanceParityTests.cs new file mode 100644 index 000000000..b56371f53 --- /dev/null +++ b/Darling/Darling.Tests/DrillDownDopProvenanceParityTests.cs @@ -0,0 +1,120 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Text.RegularExpressions; +using PerformanceMonitor.Darling.Analysis; +using Xunit; + +namespace Darling.Tests; + +/// +/// Darling's half of the cross-SKU text guard for #3648: the top_cpu_queries and bad_actor_query +/// drill-down SQL must be BYTE-IDENTICAL between Lite's inline cmd.CommandText and Darling's +/// / , +/// and both must carry the per-plan provenance shape rather than the old hash-folded MAX(max_dop). +/// +/// Why identical text and not two independent pins. #2705 fixed Darling's stale-max_dop +/// cross-check and closed without a record that Lite had a twin; #2999 is what that cost. The two drill-down +/// reads here were already character-for-character the same before #3648 (the port kept Lite's text), and the +/// provenance columns are engine-neutral (ROW_NUMBER() OVER, MAX() OVER, CASE aggregates, +/// explicit NULLS LAST), so the strongest guard available is equality: a future edit to one SKU's read +/// fails here until the other is brought along. +/// +/// Why this side hosts the comparison. build.yml's darling path filter covers +/// every Lite .cs file as well as the whole Darling tree, so this suite runs on an edit to EITHER +/// analysis tree; the lite filter does not reach Darling/PerformanceMonitor.Darling.Analysis, and +/// a guard comparing the two apps cannot live behind a filter that fires for only one of them (#2839, +/// CrossAppGuardCiGateTests). Lite's DrillDownDopProvenanceParityTests pins Lite's own shape and +/// meta-pins this file. Darling's text is read from the compiled constants rather than from source, so +/// nothing here depends on parsing Darling's own file. +/// +public sealed class DrillDownDopProvenanceParityTests +{ + private const string LiteFile = "Lite/Analysis/DrillDownCollector.Queries.cs"; + + public static TheoryData Reads => new() + { + { "CollectTopCpuQueries", PgDrillDownCollector.TopCpuQueriesSql }, + { "CollectBadActorDetail", PgDrillDownCollector.BadActorDetailSql }, + }; + + [Theory] + [MemberData(nameof(Reads))] + public void TheDrillDownSql_IsByteIdenticalAcrossSkus(string liteMethod, string darlingSql) + { + /* Both sides LF-normalised: the constant carries the line endings of the checkout it was compiled + from and the source read carries the checkout's on-disk endings, which agree on any one runner but + are not the claim under test. */ + Assert.Equal(Lf(darlingSql), Lf(LiteInlineSql(liteMethod))); + } + + [Theory] + [MemberData(nameof(Reads))] + public void TheDrillDownSql_CarriesThePerPlanProvenanceShape_NotTheHashFoldedMaximum(string liteMethod, string darlingSql) + { + _ = liteMethod; + var sql = Lf(darlingSql); + + /* The headline is the NEWEST plan's reading, picked by a newest-first ranking with the tie-breakers + spelled out (compile time, then CPU spent) and their null placement made explicit — DuckDB and + Postgres default DESC null placement differently, and an implicit default here would make the two + SKUs pick different rows on the same data. */ + Assert.Matches( + new Regex(@"ROW_NUMBER\(\)\s+OVER\s*\(\s*PARTITION BY database_name, query_hash\s+ORDER BY collection_time DESC, creation_time DESC NULLS LAST, delta_worker_time DESC NULLS LAST\s*\)\s+AS newest_rn", RegexOptions.Singleline), + sql); + Assert.Contains("MAX(CASE WHEN newest_rn = 1 THEN max_dop END) AS max_dop", sql, StringComparison.Ordinal); + + /* The history: how many plans the group spans, the cross-plan maximum, and when it was last seen. */ + Assert.Contains("COUNT(DISTINCT query_plan_hash) AS plan_count", sql, StringComparison.Ordinal); + Assert.Contains("MAX(max_dop) OVER (PARTITION BY database_name, query_hash) AS max_dop_any_plan", sql, StringComparison.Ordinal); + Assert.Contains("MAX(max_dop_any_plan) AS max_dop_any_plan", sql, StringComparison.Ordinal); + Assert.Contains("MAX(CASE WHEN max_dop = max_dop_any_plan THEN collection_time END) AS max_dop_any_plan_last_seen", sql, StringComparison.Ordinal); + + /* And the lie itself is gone: no bare hash-folded MAX(max_dop) projected as max_dop. */ + Assert.DoesNotContain("MAX(max_dop) AS max_dop", sql, StringComparison.Ordinal); + } + + [Fact] + public void NeitherSku_CoercesAnUnknownDopToZero() + { + /* The DMV never reports 0 — a serial plan is 1 — so `IsDBNull ? 0` on a DOP ordinal was "no reading" + rendered as a degree of parallelism. Both readers must project null and hand the four provenance + values to the shared note composer. Scoped to the two files' max_dop reads: the other drill-downs' + zero-coercions (counts, sums) are legitimately 0-when-absent. */ + foreach (var source in new[] + { + RepoFile.ReadRepoFile("Lite", "Analysis", "DrillDownCollector.Queries.cs"), + RepoFile.ReadRepoFile("Darling", "PerformanceMonitor.Darling.Analysis", "PgDrillDownCollector.Queries.cs"), + }) + { + Assert.DoesNotMatch(new Regex(@"max_dop\s*=\s*reader\.IsDBNull\(\d+\)\s*\?\s*0\b"), source); + Assert.Matches(new Regex(@"var maxDop = reader\.IsDBNull\(4\) \? \(int\?\)null"), source); + Assert.Matches(new Regex(@"var maxDop = reader\.IsDBNull\(10\) \? \(int\?\)null"), source); + Assert.Contains("dop_note = QueryDopProvenance.Note(maxDop, maxDopAnyPlan, maxDopAnyPlanLastSeen, planCount)", source, StringComparison.Ordinal); + } + } + + private static string Lf(string text) => text.Replace("\r\n", "\n", StringComparison.Ordinal); + + /// + /// Lite's inline SQL: the first cmd.CommandText = @"..." verbatim literal after the named method's + /// declaration. The two reads under test contain no doubled quotes, so the literal ends at the first + /// ";. + /// + private static string LiteInlineSql(string methodName) + { + var source = RepoFile.ReadRepoFile("Lite", "Analysis", "DrillDownCollector.Queries.cs"); + var start = source.IndexOf($"Task {methodName}(", StringComparison.Ordinal); + Assert.True(start >= 0, $"{methodName} not found in {LiteFile}"); + const string Marker = "cmd.CommandText = @\""; + var literalStart = source.IndexOf(Marker, start, StringComparison.Ordinal) + Marker.Length; + var literalEnd = source.IndexOf("\";", literalStart, StringComparison.Ordinal); + return source[literalStart..literalEnd]; + } +} diff --git a/Darling/Darling.Tests/QueryDopProvenanceTests.cs b/Darling/Darling.Tests/QueryDopProvenanceTests.cs new file mode 100644 index 000000000..53a2c48fd --- /dev/null +++ b/Darling/Darling.Tests/QueryDopProvenanceTests.cs @@ -0,0 +1,97 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using PerformanceMonitor.Common; +using Xunit; + +namespace Darling.Tests; + +/// +/// Decision-table pins for the shared (#3648) — the one sentence both +/// SKUs' top_cpu_queries / bad_actor_query drill-downs attach when a query-stats group's +/// cross-plan max_dop history disagrees with the newest plan's reading. This SAME table is pinned +/// identically in Darling.Tests (QueryDopProvenanceTests) so the two SKUs cannot drift; the shape is +/// QueryStatExtremesTests', the #2235 precedent for a lifetime-extreme annotation. +/// +public sealed class QueryDopProvenanceTests +{ + private static readonly DateTime Seen = new(2026, 9, 4, 13, 45, 0, DateTimeKind.Unspecified); + + /// + /// The live page (#3648): the newest plan is serial, an older plan ran at 16, three plans in the window. + /// The exact format renderers print verbatim. + /// + [Fact] + public void ParallelHistoryBehindASerialHeadline_IsSpelledOut() + { + Assert.Equal( + "DOP 1 (a parallel plan ran at 16 until 2026-09-04; 3 plans in window)", + QueryDopProvenance.Note(newestPlanMaxDop: 1, maxDopAnyPlan: 16, maxDopAnyPlanLastSeen: Seen, planCount: 3)); + } + + /// Row #3 on the same card: one serial plan, headline equals history, nothing to disclose. + [Fact] + public void HeadlineEqualToTheMaximum_CarriesNoNote() + { + Assert.Null(QueryDopProvenance.Note(1, 1, Seen, 1)); + Assert.Null(QueryDopProvenance.Note(8, 8, Seen, 2)); + } + + /// + /// A cumulative per-plan maximum can never be BELOW the newest plan's own reading unless the group is + /// inconsistent; the guard is <=, so that case is also silent rather than inventing a history. + /// + [Fact] + public void MaximumBelowTheHeadline_CarriesNoNote() + { + Assert.Null(QueryDopProvenance.Note(4, 2, Seen, 1)); + } + + /// No row in the group carried a reading: nothing to compare, nothing to say. + [Fact] + public void NoReadingAnywhere_CarriesNoNote() + { + Assert.Null(QueryDopProvenance.Note(null, null, null, 1)); + } + + /// + /// Newest reading unknown, only serial plans on record: "unknown" is already the whole truth and there + /// is no parallel plan to disclose. + /// + [Fact] + public void UnknownHeadline_WithOnlySerialHistory_CarriesNoNote() + { + Assert.Null(QueryDopProvenance.Note(null, 1, Seen, 2)); + } + + /// + /// Newest reading unknown but a parallel plan is on record — the case where a reader would otherwise + /// reach for the folded maximum, so it is named as unknown and the history is disclosed. + /// + [Fact] + public void UnknownHeadline_WithParallelHistory_SaysUnknownAndDiscloses() + { + Assert.Equal( + "DOP unknown for the newest plan (a parallel plan ran at 16 until 2026-09-04; 2 plans in window)", + QueryDopProvenance.Note(null, 16, Seen, 2)); + } + + /// + /// One plan SHAPE (a recompile to the same query_plan_hash after MAXDOP was lowered resets the counter) + /// still reads as a history; the count is singular and the "until" clause is dropped when the store + /// could not say when the maximum was last seen. + /// + [Fact] + public void SinglePlan_AndNoLastSeen_AreSpelledWithoutInvention() + { + Assert.Equal( + "DOP 1 (a parallel plan ran at 16; 1 plan in window)", + QueryDopProvenance.Note(1, 16, null, 1)); + } +} diff --git a/Darling/PerformanceMonitor.Darling.Analysis/PgDrillDownCollector.Queries.cs b/Darling/PerformanceMonitor.Darling.Analysis/PgDrillDownCollector.Queries.cs index 60cbcf113..53c974887 100644 --- a/Darling/PerformanceMonitor.Darling.Analysis/PgDrillDownCollector.Queries.cs +++ b/Darling/PerformanceMonitor.Darling.Analysis/PgDrillDownCollector.Queries.cs @@ -11,6 +11,7 @@ using System.Threading.Tasks; using Npgsql; using PerformanceMonitor.Analysis; +using PerformanceMonitor.Common; namespace PerformanceMonitor.Darling.Analysis; @@ -110,16 +111,56 @@ private async Task CollectQueriesAtSpike(AnalysisFinding finding, AnalysisContex } } + /// + /// #3648: max_dop here is sys.dm_exec_query_stats.max_dop — a PER-PLAN high-water mark since + /// the plan entered the cache, not a per-execution reading and not a per-statement one. This read groups + /// by (database, query_hash), so the old MAX(max_dop) folded every plan the statement text had + /// inside the window into one number and kept the largest: the highest DOP ANY plan for the hash ever ran + /// at, with nothing saying which plan, when, or whether that plan still exists. Live consequence: a High + /// CPU card read 16 for the #1 query on an instance whose MAXDOP had been 1 across its whole 14-day config + /// history and whose stored plan for that hash was serial (NonParallelPlanReason="MaxDOPSetToOne") + /// — a plan compiled before the pin, still cached with its old counter. A reader recommended a MAXDOP 1 + /// Query Store hint from the field and had to retract it after reading the plan. The same card's row #3 + /// said 1, honestly, for a hash with one serial plan — so the field was self-consistent and wrong. + /// + /// Now: max_dop is the NEWEST plan's reading (the row with the latest collection_time + /// among the rows that spent CPU in the window; ties broken by compile time, then by CPU spent), because + /// the card's question is what the query is doing to the CPU at this moment. The cross-plan maximum + /// survives as max_dop_any_plan with max_dop_any_plan_last_seen and plan_count + /// beside it — a history with provenance — and the reader coerces NULL to null, not 0: the DMV never + /// reports 0, so 0 was "no reading" rendered as a degree of parallelism. dop_note is the shared + /// sentence (PerformanceMonitor.Common.QueryDopProvenance) a renderer prints verbatim when the + /// history disagrees with the headline. New columns are appended after the old ones so the existing + /// ordinals are untouched; the SQL is byte-identical to Lite's CollectTopCpuQueries text. + /// public const string TopCpuQueriesSql = @" +WITH windowed AS +( + -- #3648: rank each hash's rows newest-first and carry the hash-wide maximum onto every row, so the + -- outer aggregate can name WHICH reading is current and WHEN the maximum was last seen. Explicit + -- NULLS LAST on the tie-breakers: DuckDB and Postgres default DESC null placement differently. + SELECT database_name, query_hash, query_plan_hash, collection_time, max_dop, + delta_worker_time, delta_execution_count, delta_spills, query_text, + ROW_NUMBER() OVER + ( + PARTITION BY database_name, query_hash + ORDER BY collection_time DESC, creation_time DESC NULLS LAST, delta_worker_time DESC NULLS LAST + ) AS newest_rn, + MAX(max_dop) OVER (PARTITION BY database_name, query_hash) AS max_dop_any_plan + FROM v_query_stats + WHERE server_id = $1 AND collection_time >= $2 AND collection_time <= $3 + AND delta_worker_time > 0 +) SELECT database_name, query_hash, SUM(delta_worker_time)::BIGINT AS total_cpu_us, SUM(delta_execution_count)::BIGINT AS exec_count, - MAX(max_dop) AS max_dop, + MAX(CASE WHEN newest_rn = 1 THEN max_dop END) AS max_dop, SUM(delta_spills)::BIGINT AS spills, - LEFT(MAX(query_text), 500) AS query_text -FROM v_query_stats -WHERE server_id = $1 AND collection_time >= $2 AND collection_time <= $3 -AND delta_worker_time > 0 + LEFT(MAX(query_text), 500) AS query_text, + COUNT(DISTINCT query_plan_hash) AS plan_count, + MAX(max_dop_any_plan) AS max_dop_any_plan, + MAX(CASE WHEN max_dop = max_dop_any_plan THEN collection_time END) AS max_dop_any_plan_last_seen +FROM windowed GROUP BY database_name, query_hash ORDER BY total_cpu_us DESC LIMIT 5"; @@ -138,15 +179,24 @@ private async Task CollectTopCpuQueries(AnalysisFinding finding, AnalysisContext using var reader = await cmd.ExecuteReaderAsync(context.CancellationToken); while (await reader.ReadAsync(context.CancellationToken)) { + var maxDop = reader.IsDBNull(4) ? (int?)null : Convert.ToInt32(reader.GetValue(4)); + var planCount = reader.IsDBNull(7) ? 0L : Convert.ToInt64(reader.GetValue(7)); + var maxDopAnyPlan = reader.IsDBNull(8) ? (int?)null : Convert.ToInt32(reader.GetValue(8)); + var maxDopAnyPlanLastSeen = reader.IsDBNull(9) ? (DateTime?)null : reader.GetDateTime(9); items.Add(new { database = reader.IsDBNull(0) ? "" : reader.GetString(0), query_hash = reader.IsDBNull(1) ? "" : reader.GetString(1), total_cpu_ms = reader.IsDBNull(2) ? 0.0 : Convert.ToDouble(reader.GetValue(2)) / 1000.0, execution_count = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)), - max_dop = reader.IsDBNull(4) ? 0 : Convert.ToInt32(reader.GetValue(4)), + /* #3648: newest plan's reading, null when unknown — never 0. History fields follow. */ + max_dop = maxDop, spills = reader.IsDBNull(5) ? 0L : Convert.ToInt64(reader.GetValue(5)), - query_text = reader.IsDBNull(6) ? "" : reader.GetString(6) + query_text = reader.IsDBNull(6) ? "" : reader.GetString(6), + plan_count = planCount, + max_dop_any_plan = maxDopAnyPlan, + max_dop_any_plan_last_seen = maxDopAnyPlanLastSeen?.ToString("o"), + dop_note = QueryDopProvenance.Note(maxDop, maxDopAnyPlan, maxDopAnyPlanLastSeen, planCount) }); } @@ -573,7 +623,32 @@ Appended after the older columns so the existing reader ordinals are untouched. finding.DrillDown!["regressed_queries"] = items; } + /// + /// #3648: the same per-plan max_dop provenance as — see the essay + /// there. This read is one hash, unfiltered by CPU spent, so the newest-plan tie-break (compile time, + /// then CPU spent) is what separates a stale parallel plan still sitting in the cache from the serial + /// one doing the work at the same collection_time. Byte-identical to Lite's + /// CollectBadActorDetail text. + /// public const string BadActorDetailSql = @" +WITH windowed AS +( + -- #3648: see CollectTopCpuQueries' windowed CTE; same ranking, same hash-wide maximum. + SELECT database_name, query_hash, query_plan_hash, collection_time, max_dop, + delta_worker_time, delta_execution_count, delta_elapsed_time, delta_logical_reads, delta_spills, + query_text, + ROW_NUMBER() OVER + ( + PARTITION BY database_name, query_hash + ORDER BY collection_time DESC, creation_time DESC NULLS LAST, delta_worker_time DESC NULLS LAST + ) AS newest_rn, + MAX(max_dop) OVER (PARTITION BY database_name, query_hash) AS max_dop_any_plan + FROM v_query_stats + WHERE server_id = $1 + AND collection_time >= $2 + AND collection_time <= $3 + AND query_hash = $4 +) SELECT database_name, query_hash, LEFT(MAX(query_text), 500) AS query_text, SUM(delta_execution_count)::BIGINT AS exec_count, @@ -589,12 +664,11 @@ THEN SUM(delta_logical_reads)::DOUBLE PRECISION / SUM(delta_execution_count) SUM(delta_worker_time)::BIGINT AS total_cpu_us, SUM(delta_logical_reads)::BIGINT AS total_reads, SUM(delta_spills)::BIGINT AS total_spills, - MAX(max_dop) AS max_dop -FROM v_query_stats -WHERE server_id = $1 -AND collection_time >= $2 -AND collection_time <= $3 -AND query_hash = $4 + MAX(CASE WHEN newest_rn = 1 THEN max_dop END) AS max_dop, + COUNT(DISTINCT query_plan_hash) AS plan_count, + MAX(max_dop_any_plan) AS max_dop_any_plan, + MAX(CASE WHEN max_dop = max_dop_any_plan THEN collection_time END) AS max_dop_any_plan_last_seen +FROM windowed GROUP BY database_name, query_hash"; private async Task CollectBadActorDetail(AnalysisFinding finding, AnalysisContext context) @@ -615,6 +689,10 @@ private async Task CollectBadActorDetail(AnalysisFinding finding, AnalysisContex using var reader = await cmd.ExecuteReaderAsync(context.CancellationToken); if (await reader.ReadAsync(context.CancellationToken)) { + var maxDop = reader.IsDBNull(10) ? (int?)null : Convert.ToInt32(reader.GetValue(10)); + var planCount = reader.IsDBNull(11) ? 0L : Convert.ToInt64(reader.GetValue(11)); + var maxDopAnyPlan = reader.IsDBNull(12) ? (int?)null : Convert.ToInt32(reader.GetValue(12)); + var maxDopAnyPlanLastSeen = reader.IsDBNull(13) ? (DateTime?)null : reader.GetDateTime(13); finding.DrillDown!["bad_actor_query"] = new { database = reader.IsDBNull(0) ? "" : reader.GetString(0), @@ -627,7 +705,12 @@ private async Task CollectBadActorDetail(AnalysisFinding finding, AnalysisContex total_cpu_ms = reader.IsDBNull(7) ? 0.0 : Convert.ToDouble(reader.GetValue(7)) / 1000.0, total_reads = reader.IsDBNull(8) ? 0L : Convert.ToInt64(reader.GetValue(8)), total_spills = reader.IsDBNull(9) ? 0L : Convert.ToInt64(reader.GetValue(9)), - max_dop = reader.IsDBNull(10) ? 0 : Convert.ToInt32(reader.GetValue(10)) + /* #3648: newest plan's reading, null when unknown — never 0. History fields follow. */ + max_dop = maxDop, + plan_count = planCount, + max_dop_any_plan = maxDopAnyPlan, + max_dop_any_plan_last_seen = maxDopAnyPlanLastSeen?.ToString("o"), + dop_note = QueryDopProvenance.Note(maxDop, maxDopAnyPlan, maxDopAnyPlanLastSeen, planCount) }; } } diff --git a/Lite.Tests/DrillDownDopProvenanceParityTests.cs b/Lite.Tests/DrillDownDopProvenanceParityTests.cs new file mode 100644 index 000000000..349ce39b8 --- /dev/null +++ b/Lite.Tests/DrillDownDopProvenanceParityTests.cs @@ -0,0 +1,104 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor Lite. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Text.RegularExpressions; +using Xunit; + +namespace Lite.Tests; + +/// +/// Lite's half of the cross-SKU text guard for #3648. The exact counterpart of +/// Darling.Tests.DrillDownDopProvenanceParityTests, built to the shape +/// QueryHighDopStaleMaxDopParityTests established: Darling's twin does the cross-STORE comparison +/// (its build.yml filter covers every Lite .cs file, so it runs on an edit to either analysis +/// tree); this file owns Lite's own census — that the top_cpu_queries and bad_actor_query reads +/// carry the per-plan provenance shape rather than the old hash-folded MAX(max_dop), and that an +/// unknown DOP is projected as null rather than 0 — and meta-pins Darling's guard, so weakening the byte- +/// identity comparison over there fails here, under the lite filter that reaches Darling's test tree. +/// +/// Why not read Darling's SQL from here. The lite filter does not reach +/// Darling/PerformanceMonitor.Darling.Analysis, and CrossAppGuardCiGateTests fails any Lite +/// guard whose source read PR CI cannot run (#2839): a guard that compares the two apps cannot live behind a +/// filter that fires for only one of them. Darling's constants are public, so its suite compares them against +/// Lite's inline text without parsing its own file. +/// +public sealed class DrillDownDopProvenanceParityTests +{ + private const string LiteFile = "Lite/Analysis/DrillDownCollector.Queries.cs"; + private const string DarlingGuard = "Darling/Darling.Tests/DrillDownDopProvenanceParityTests.cs"; + + [Theory] + [InlineData("CollectTopCpuQueries")] + [InlineData("CollectBadActorDetail")] + public void TheDrillDownSql_CarriesThePerPlanProvenanceShape_NotTheHashFoldedMaximum(string liteMethod) + { + var sql = LiteInlineSql(liteMethod); + + /* The headline is the NEWEST plan's reading, picked by a newest-first ranking with the tie-breakers + spelled out (compile time, then CPU spent) and their null placement made explicit — DuckDB and + Postgres default DESC null placement differently, and an implicit default here would make the two + SKUs pick different rows on the same data. */ + Assert.Matches( + new Regex(@"ROW_NUMBER\(\)\s+OVER\s*\(\s*PARTITION BY database_name, query_hash\s+ORDER BY collection_time DESC, creation_time DESC NULLS LAST, delta_worker_time DESC NULLS LAST\s*\)\s+AS newest_rn", RegexOptions.Singleline), + sql); + Assert.Contains("MAX(CASE WHEN newest_rn = 1 THEN max_dop END) AS max_dop", sql, StringComparison.Ordinal); + + /* The history: how many plans the group spans, the cross-plan maximum, and when it was last seen. */ + Assert.Contains("COUNT(DISTINCT query_plan_hash) AS plan_count", sql, StringComparison.Ordinal); + Assert.Contains("MAX(max_dop) OVER (PARTITION BY database_name, query_hash) AS max_dop_any_plan", sql, StringComparison.Ordinal); + Assert.Contains("MAX(max_dop_any_plan) AS max_dop_any_plan", sql, StringComparison.Ordinal); + Assert.Contains("MAX(CASE WHEN max_dop = max_dop_any_plan THEN collection_time END) AS max_dop_any_plan_last_seen", sql, StringComparison.Ordinal); + + /* And the lie itself is gone: no bare hash-folded MAX(max_dop) projected as max_dop. */ + Assert.DoesNotContain("MAX(max_dop) AS max_dop", sql, StringComparison.Ordinal); + } + + [Fact] + public void Lite_DoesNotCoerceAnUnknownDopToZero() + { + /* The DMV never reports 0 — a serial plan is 1 — so `IsDBNull ? 0` on a DOP ordinal was "no reading" + rendered as a degree of parallelism. Scoped to this file's max_dop reads: the other drill-downs' + zero-coercions (counts, sums) are legitimately 0-when-absent. */ + var source = ParitySource.ReadFile(LiteFile); + Assert.DoesNotMatch(new Regex(@"max_dop\s*=\s*reader\.IsDBNull\(\d+\)\s*\?\s*0\b"), source); + Assert.Matches(new Regex(@"var maxDop = reader\.IsDBNull\(4\) \? \(int\?\)null"), source); + Assert.Matches(new Regex(@"var maxDop = reader\.IsDBNull\(10\) \? \(int\?\)null"), source); + Assert.Contains("dop_note = QueryDopProvenance.Note(maxDop, maxDopAnyPlan, maxDopAnyPlanLastSeen, planCount)", source, StringComparison.Ordinal); + } + + [Fact] + public void DarlingsTwinGuard_StillComparesBothReadsByteForByte_SoNeitherSideCanFallBehind() + { + /* Meta-pin: Darling's guard must keep BOTH reads in its byte-identity theory and keep reading Lite's + inline text. Dropping a row, or narrowing the comparison to a Contains, is how one SKU's read would + start drifting under a green board. */ + var guard = ParitySource.ReadFile(DarlingGuard); + Assert.Contains("public void TheDrillDownSql_IsByteIdenticalAcrossSkus(", guard, StringComparison.Ordinal); + Assert.Contains("{ \"CollectTopCpuQueries\", PgDrillDownCollector.TopCpuQueriesSql }", guard, StringComparison.Ordinal); + Assert.Contains("{ \"CollectBadActorDetail\", PgDrillDownCollector.BadActorDetailSql }", guard, StringComparison.Ordinal); + Assert.Contains("Assert.Equal(Lf(darlingSql), Lf(LiteInlineSql(liteMethod)));", guard, StringComparison.Ordinal); + Assert.Contains("RepoFile.ReadRepoFile(\"Lite\", \"Analysis\", \"DrillDownCollector.Queries.cs\")", guard, StringComparison.Ordinal); + } + + /// + /// Lite's inline SQL: the first cmd.CommandText = @"..." verbatim literal after the named method's + /// declaration. The two reads under test contain no doubled quotes, so the literal ends at the first + /// ";. + /// + private static string LiteInlineSql(string methodName) + { + var source = ParitySource.ReadFile(LiteFile); + var start = source.IndexOf($"Task {methodName}(", StringComparison.Ordinal); + Assert.True(start >= 0, $"{methodName} not found in {LiteFile}"); + const string Marker = "cmd.CommandText = @\""; + var literalStart = source.IndexOf(Marker, start, StringComparison.Ordinal) + Marker.Length; + var literalEnd = source.IndexOf("\";", literalStart, StringComparison.Ordinal); + return source[literalStart..literalEnd]; + } +} diff --git a/Lite.Tests/DrillDownDopProvenanceTests.cs b/Lite.Tests/DrillDownDopProvenanceTests.cs new file mode 100644 index 000000000..247921566 --- /dev/null +++ b/Lite.Tests/DrillDownDopProvenanceTests.cs @@ -0,0 +1,278 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor Lite. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Text.Json; +using System.Threading.Tasks; +using DuckDB.NET.Data; +using PerformanceMonitor.Analysis; +using PerformanceMonitorLite.Analysis; +using PerformanceMonitorLite.Database; +using PerformanceMonitorLite.Tests; +using Xunit; + +namespace Lite.Tests; + +/// +/// Real-DuckDB pin for #3648: the top_cpu_queries and bad_actor_query drill-downs must headline +/// the NEWEST plan's max_dop and carry the cross-plan maximum as a history with provenance, not fold +/// every plan the hash ever had into one provenance-free number. +/// +/// The fixture is the live page. One query_hash, two plans inside the window: an old +/// parallel plan whose sys.dm_exec_query_stats.max_dop (a per-plan high-water mark since the plan was +/// cached) reads 16, last seen hours before the window's end, and a new serial plan reading 1 that is the one +/// spending the CPU at the newest snapshot. Before #3648 the record said max_dop = 16 for this shape — +/// on a MAXDOP-1 instance whose stored plan was serial — and a reader recommended a MAXDOP 1 Query Store hint +/// from it. After: max_dop = 1, max_dop_any_plan = 16, plan_count = 2, +/// max_dop_any_plan_last_seen = the parallel plan's last snapshot, and dop_note says so in one +/// sentence. +/// +/// Real DuckDB rather than a string pin. The claim under test is the ENGINE's answer to a +/// ROW_NUMBER() OVER (... ORDER BY collection_time DESC ...) ranking combined with a hash-wide +/// MAX() OVER — which row the newest-plan CASE picks, and that a NULL reading arrives as NULL rather +/// than 0. DrillDownDopProvenanceParityTests holds the text half across both SKUs. +/// +public sealed class DrillDownDopProvenanceTests : IClassFixture, IDisposable +{ + private const int ServerId = 36480; + private const string ServerName = "SynthDopProvSrv"; + private const string Db = "SynthDopProvDb"; + private const string Hash = "0x3648HASH"; + private const string OldParallelPlan = "0x3648PLANPARALLEL"; + private const string NewSerialPlan = "0x3648PLANSERIAL"; + + /* One UtcNow read, truncated to the second: two reads can land on different ticks, leaving a sub- + microsecond residue that DuckDB's microsecond timestamps drop on the way back out — and the + last-seen assertions below compare a round-tripped timestamp for equality. */ + private static readonly DateTime WindowEnd = TruncateToSeconds(DateTime.UtcNow); + + private static DateTime TruncateToSeconds(DateTime t) => + DateTime.SpecifyKind(new DateTime(t.Ticks - (t.Ticks % TimeSpan.TicksPerSecond)), DateTimeKind.Unspecified); + + private static readonly DateTime WindowStart = WindowEnd.AddHours(-4); + + /* The parallel plan's LAST snapshot inside the window — what max_dop_any_plan_last_seen must report. */ + private static readonly DateTime ParallelLastSeen = WindowEnd.AddHours(-3); + + private readonly DuckDbInitializer _duckDb; + private DuckDBConnection? _seedConn; + private long _nextId = 1; + + public DrillDownDopProvenanceTests(SharedDuckDbFixture fixture) + { + fixture.ResetData(); + _duckDb = fixture.DuckDb; + } + + public void Dispose() => _seedConn?.Dispose(); + + private static AnalysisContext Context() => new() + { + ServerId = ServerId, + ServerName = ServerName, + TimeRangeStart = WindowStart, + TimeRangeEnd = WindowEnd, + }; + + [Fact] + public async Task TopCpuQueries_HeadlinesTheNewestPlansDop_AndCarriesTheParallelHistory() + { + /* -- RED before #3648: max_dop read 16 — the old parallel plan's counter — for a hash whose newest + plan is serial. -- */ + await ClearAsync(); + await SeedTwoPlansOneHashAsync(); + + var row = Assert.Single(await CollectTopCpuRowsAsync()); + + Assert.Equal(1, row.GetProperty("max_dop").GetInt32()); + Assert.Equal(16, row.GetProperty("max_dop_any_plan").GetInt32()); + Assert.Equal(2, row.GetProperty("plan_count").GetInt64()); + Assert.Equal(ParallelLastSeen, DateTime.Parse(row.GetProperty("max_dop_any_plan_last_seen").GetString()!, null, + System.Globalization.DateTimeStyles.RoundtripKind)); + Assert.Equal( + $"DOP 1 (a parallel plan ran at 16 until {ParallelLastSeen:yyyy-MM-dd}; 2 plans in window)", + row.GetProperty("dop_note").GetString()); + + /* The windowed totals still span BOTH plans: the provenance columns narrow one reading, they do not + filter the read. 3 parallel snapshots x 100000 + 1 serial x 400000 = 700000 us = 700 ms. */ + Assert.Equal(700.0, row.GetProperty("total_cpu_ms").GetDouble()); + } + + [Fact] + public async Task BadActorQuery_HeadlinesTheNewestPlansDop_AndCarriesTheParallelHistory() + { + /* The same shape through the BAD_ACTOR drill-down, which is one hash unfiltered by CPU spent. */ + await ClearAsync(); + await SeedTwoPlansOneHashAsync(); + + var row = await CollectBadActorAsync(); + + Assert.Equal(1, row.GetProperty("max_dop").GetInt32()); + Assert.Equal(16, row.GetProperty("max_dop_any_plan").GetInt32()); + Assert.Equal(2, row.GetProperty("plan_count").GetInt64()); + Assert.Equal( + $"DOP 1 (a parallel plan ran at 16 until {ParallelLastSeen:yyyy-MM-dd}; 2 plans in window)", + row.GetProperty("dop_note").GetString()); + } + + [Fact] + public async Task OneSerialPlan_SaysOne_AndHasNoHistoryToDisclose() + { + /* Row #3 on the live card: one serial plan, honest before and after. No note — there is no history + that disagrees with the headline, and a note on every row would bury the one that matters. */ + await ClearAsync(); + await SeedAsync(NewSerialPlan, WindowEnd.AddMinutes(-30), maxDop: 1, workerTimeUs: 400000); + + var row = Assert.Single(await CollectTopCpuRowsAsync()); + + Assert.Equal(1, row.GetProperty("max_dop").GetInt32()); + Assert.Equal(1, row.GetProperty("max_dop_any_plan").GetInt32()); + Assert.Equal(1, row.GetProperty("plan_count").GetInt64()); + Assert.Equal(JsonValueKind.Null, row.GetProperty("dop_note").ValueKind); + } + + [Fact] + public async Task NoReading_IsNull_NotZero() + { + /* The DMV never reports a DOP of 0 — a serial plan is 1 — so 0 was always "no reading" rendered as + a number. A NULL max_dop must arrive as JSON null on every DOP field, and the note must not invent + a parallel history from nothing. */ + await ClearAsync(); + await SeedAsync(NewSerialPlan, WindowEnd.AddMinutes(-30), maxDop: null, workerTimeUs: 400000); + + var row = Assert.Single(await CollectTopCpuRowsAsync()); + + Assert.Equal(JsonValueKind.Null, row.GetProperty("max_dop").ValueKind); + Assert.Equal(JsonValueKind.Null, row.GetProperty("max_dop_any_plan").ValueKind); + Assert.Equal(JsonValueKind.Null, row.GetProperty("max_dop_any_plan_last_seen").ValueKind); + Assert.Equal(JsonValueKind.Null, row.GetProperty("dop_note").ValueKind); + } + + [Fact] + public async Task NewestReadingUnknown_WithAParallelPlanOnRecord_SaysUnknownAndDisclosesIt() + { + /* The newest plan carried no reading but an older plan ran parallel: "unknown" is the headline and + the parallel history is still disclosed, because that is exactly the case where a reader would + otherwise fall back to the folded maximum. */ + await ClearAsync(); + await SeedAsync(OldParallelPlan, ParallelLastSeen, maxDop: 16, workerTimeUs: 100000); + await SeedAsync(NewSerialPlan, WindowEnd.AddMinutes(-30), maxDop: null, workerTimeUs: 400000); + + var row = Assert.Single(await CollectTopCpuRowsAsync()); + + Assert.Equal(JsonValueKind.Null, row.GetProperty("max_dop").ValueKind); + Assert.Equal(16, row.GetProperty("max_dop_any_plan").GetInt32()); + Assert.Equal( + $"DOP unknown for the newest plan (a parallel plan ran at 16 until {ParallelLastSeen:yyyy-MM-dd}; 2 plans in window)", + row.GetProperty("dop_note").GetString()); + } + + [Fact] + public async Task TwoPlansAtTheSameNewestSnapshot_TheNewerCompiledOneIsTheHeadline() + { + /* Both plans cached at the newest snapshot — the stale parallel plan still sitting in the cache + next to the serial one that replaced it. collection_time ties, so the compile-time tie-break + decides: the plan compiled LATER is the newest plan, whatever its counter says. */ + await ClearAsync(); + var newest = WindowEnd.AddMinutes(-30); + await SeedAsync(OldParallelPlan, newest, maxDop: 16, workerTimeUs: 500000, creationTime: WindowEnd.AddDays(-20)); + await SeedAsync(NewSerialPlan, newest, maxDop: 1, workerTimeUs: 100000, creationTime: WindowEnd.AddDays(-1)); + + var row = Assert.Single(await CollectTopCpuRowsAsync()); + + Assert.Equal(1, row.GetProperty("max_dop").GetInt32()); + Assert.Equal(16, row.GetProperty("max_dop_any_plan").GetInt32()); + Assert.Equal(newest, DateTime.Parse(row.GetProperty("max_dop_any_plan_last_seen").GetString()!, null, + System.Globalization.DateTimeStyles.RoundtripKind)); + } + + private async Task CollectAsync(string factKey, string drillDownKey) + { + var finding = new AnalysisFinding + { + RootFactKey = factKey, + StoryPath = factKey, + /* Past the 0.5 display gate in EnrichFindingsAsync — below it the expensive drill-downs are + skipped wholesale and this collector never runs at all. */ + Severity = 1.0, + }; + + await new DrillDownCollector(_duckDb).EnrichFindingsAsync([finding], Context()); + + Assert.NotNull(finding.DrillDown); + Assert.True(finding.DrillDown.TryGetValue(drillDownKey, out var raw), $"{drillDownKey} was not collected"); + return JsonSerializer.SerializeToElement(raw); + } + + /// The top_cpu_queries rows (an array section). + private async Task CollectTopCpuRowsAsync() + => [.. (await CollectAsync("CPU_SQL_PERCENT", "top_cpu_queries")).EnumerateArray()]; + + /// The bad_actor_query record (a single-object section). + private Task CollectBadActorAsync() + => CollectAsync("BAD_ACTOR_" + Hash, "bad_actor_query"); + + /// + /// The live page's shape: the parallel plan spent CPU at three snapshots ending three hours before the + /// window's end, then the serial plan spent CPU at the newest snapshot. + /// + private async Task SeedTwoPlansOneHashAsync() + { + await SeedAsync(OldParallelPlan, ParallelLastSeen.AddMinutes(-20), maxDop: 16, workerTimeUs: 100000); + await SeedAsync(OldParallelPlan, ParallelLastSeen.AddMinutes(-10), maxDop: 16, workerTimeUs: 100000); + await SeedAsync(OldParallelPlan, ParallelLastSeen, maxDop: 16, workerTimeUs: 100000); + await SeedAsync(NewSerialPlan, WindowEnd.AddMinutes(-30), maxDop: 1, workerTimeUs: 400000); + } + + private async Task SeedConnectionAsync() + { + if (_seedConn is null) + { + _seedConn = _duckDb.CreateConnection(); + await _seedConn.OpenAsync(); + } + return _seedConn; + } + + private async Task ClearAsync() + { + using var readLock = _duckDb.AcquireReadLock(); + var connection = await SeedConnectionAsync(); + using var cmd = connection.CreateCommand(); + cmd.CommandText = "DELETE FROM query_stats WHERE server_id = $1"; + cmd.Parameters.Add(new DuckDBParameter { Value = ServerId }); + await cmd.ExecuteNonQueryAsync(); + } + + private async Task SeedAsync(string planHash, DateTime collectionTime, long? maxDop, long workerTimeUs, + DateTime? creationTime = null) + { + using var readLock = _duckDb.AcquireReadLock(); + var connection = await SeedConnectionAsync(); + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +INSERT INTO query_stats + (collection_id, collection_time, server_id, server_name, database_name, query_hash, query_plan_hash, + creation_time, max_dop, min_dop, delta_execution_count, delta_worker_time, delta_elapsed_time, + delta_logical_reads, delta_spills, query_text) +VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, 1, 10, $10, $11, 1000, 0, $12)"; + cmd.Parameters.Add(new DuckDBParameter { Value = _nextId++ }); + cmd.Parameters.Add(new DuckDBParameter { Value = collectionTime }); + cmd.Parameters.Add(new DuckDBParameter { Value = ServerId }); + cmd.Parameters.Add(new DuckDBParameter { Value = ServerName }); + cmd.Parameters.Add(new DuckDBParameter { Value = Db }); + cmd.Parameters.Add(new DuckDBParameter { Value = Hash }); + cmd.Parameters.Add(new DuckDBParameter { Value = planHash }); + cmd.Parameters.Add(new DuckDBParameter { Value = creationTime ?? collectionTime.AddDays(-1) }); + cmd.Parameters.Add(new DuckDBParameter { Value = maxDop.HasValue ? maxDop.Value : DBNull.Value }); + cmd.Parameters.Add(new DuckDBParameter { Value = workerTimeUs }); + cmd.Parameters.Add(new DuckDBParameter { Value = workerTimeUs * 2 }); + cmd.Parameters.Add(new DuckDBParameter { Value = "SELECT * FROM dbo.SynthDopProvTable" }); + await cmd.ExecuteNonQueryAsync(); + } +} diff --git a/Lite.Tests/QueryDopProvenanceTests.cs b/Lite.Tests/QueryDopProvenanceTests.cs new file mode 100644 index 000000000..712b392c7 --- /dev/null +++ b/Lite.Tests/QueryDopProvenanceTests.cs @@ -0,0 +1,97 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using PerformanceMonitor.Common; +using Xunit; + +namespace Lite.Tests; + +/// +/// Decision-table pins for the shared (#3648) — the one sentence both +/// SKUs' top_cpu_queries / bad_actor_query drill-downs attach when a query-stats group's +/// cross-plan max_dop history disagrees with the newest plan's reading. This SAME table is pinned +/// identically in Darling.Tests (QueryDopProvenanceTests) so the two SKUs cannot drift; the shape is +/// QueryStatExtremesTests', the #2235 precedent for a lifetime-extreme annotation. +/// +public sealed class QueryDopProvenanceTests +{ + private static readonly DateTime Seen = new(2026, 9, 4, 13, 45, 0, DateTimeKind.Unspecified); + + /// + /// The live page (#3648): the newest plan is serial, an older plan ran at 16, three plans in the window. + /// The exact format renderers print verbatim. + /// + [Fact] + public void ParallelHistoryBehindASerialHeadline_IsSpelledOut() + { + Assert.Equal( + "DOP 1 (a parallel plan ran at 16 until 2026-09-04; 3 plans in window)", + QueryDopProvenance.Note(newestPlanMaxDop: 1, maxDopAnyPlan: 16, maxDopAnyPlanLastSeen: Seen, planCount: 3)); + } + + /// Row #3 on the same card: one serial plan, headline equals history, nothing to disclose. + [Fact] + public void HeadlineEqualToTheMaximum_CarriesNoNote() + { + Assert.Null(QueryDopProvenance.Note(1, 1, Seen, 1)); + Assert.Null(QueryDopProvenance.Note(8, 8, Seen, 2)); + } + + /// + /// A cumulative per-plan maximum can never be BELOW the newest plan's own reading unless the group is + /// inconsistent; the guard is <=, so that case is also silent rather than inventing a history. + /// + [Fact] + public void MaximumBelowTheHeadline_CarriesNoNote() + { + Assert.Null(QueryDopProvenance.Note(4, 2, Seen, 1)); + } + + /// No row in the group carried a reading: nothing to compare, nothing to say. + [Fact] + public void NoReadingAnywhere_CarriesNoNote() + { + Assert.Null(QueryDopProvenance.Note(null, null, null, 1)); + } + + /// + /// Newest reading unknown, only serial plans on record: "unknown" is already the whole truth and there + /// is no parallel plan to disclose. + /// + [Fact] + public void UnknownHeadline_WithOnlySerialHistory_CarriesNoNote() + { + Assert.Null(QueryDopProvenance.Note(null, 1, Seen, 2)); + } + + /// + /// Newest reading unknown but a parallel plan is on record — the case where a reader would otherwise + /// reach for the folded maximum, so it is named as unknown and the history is disclosed. + /// + [Fact] + public void UnknownHeadline_WithParallelHistory_SaysUnknownAndDiscloses() + { + Assert.Equal( + "DOP unknown for the newest plan (a parallel plan ran at 16 until 2026-09-04; 2 plans in window)", + QueryDopProvenance.Note(null, 16, Seen, 2)); + } + + /// + /// One plan SHAPE (a recompile to the same query_plan_hash after MAXDOP was lowered resets the counter) + /// still reads as a history; the count is singular and the "until" clause is dropped when the store + /// could not say when the maximum was last seen. + /// + [Fact] + public void SinglePlan_AndNoLastSeen_AreSpelledWithoutInvention() + { + Assert.Equal( + "DOP 1 (a parallel plan ran at 16; 1 plan in window)", + QueryDopProvenance.Note(1, 16, null, 1)); + } +} diff --git a/Lite/Analysis/DrillDownCollector.Queries.cs b/Lite/Analysis/DrillDownCollector.Queries.cs index 319d6b290..b82057043 100644 --- a/Lite/Analysis/DrillDownCollector.Queries.cs +++ b/Lite/Analysis/DrillDownCollector.Queries.cs @@ -107,17 +107,56 @@ private async Task CollectTopCpuQueries(AnalysisFinding finding, AnalysisContext using var connection = _duckDb.CreateConnection(); await connection.OpenAsync(context.CancellationToken); + /* #3648: max_dop here is sys.dm_exec_query_stats.max_dop — a PER-PLAN high-water mark since the + plan entered the cache, not a per-execution reading and not a per-statement one. This read + groups by (database, query_hash), so the old MAX(max_dop) folded every plan the statement + text had inside the window into one number and kept the largest: the highest DOP ANY plan for + the hash ever ran at, with nothing saying which plan, when, or whether that plan still exists. + Live consequence: a High CPU card read 16 for the #1 query on an instance whose MAXDOP had + been 1 across its whole 14-day config history and whose stored plan for that hash was serial + (NonParallelPlanReason="MaxDOPSetToOne") — a plan compiled before the pin, still cached with + its old counter. A reader recommended a MAXDOP 1 Query Store hint from the field and had to + retract it after reading the plan. The same card's row #3 said 1, honestly, for a hash with + one serial plan — so the field was self-consistent and wrong. + + Now: max_dop is the NEWEST plan's reading (the row with the latest collection_time among the + rows that spent CPU in the window; ties broken by compile time, then by CPU spent), because + the card's question is what the query is doing to the CPU at this moment. The cross-plan + maximum survives as max_dop_any_plan with max_dop_any_plan_last_seen and plan_count beside + it — a history with provenance — and the reader coerces NULL to null, not 0: the DMV never + reports 0, so 0 was "no reading" rendered as a degree of parallelism. dop_note is the + shared sentence (PerformanceMonitor.Common.QueryDopProvenance) a renderer prints verbatim + when the history disagrees with the headline. New columns are appended after the old ones so + the existing ordinals are untouched; the SQL is byte-identical to Darling's TopCpuQueriesSql. */ using var cmd = connection.CreateCommand(); cmd.CommandText = @" +WITH windowed AS +( + -- #3648: rank each hash's rows newest-first and carry the hash-wide maximum onto every row, so the + -- outer aggregate can name WHICH reading is current and WHEN the maximum was last seen. Explicit + -- NULLS LAST on the tie-breakers: DuckDB and Postgres default DESC null placement differently. + SELECT database_name, query_hash, query_plan_hash, collection_time, max_dop, + delta_worker_time, delta_execution_count, delta_spills, query_text, + ROW_NUMBER() OVER + ( + PARTITION BY database_name, query_hash + ORDER BY collection_time DESC, creation_time DESC NULLS LAST, delta_worker_time DESC NULLS LAST + ) AS newest_rn, + MAX(max_dop) OVER (PARTITION BY database_name, query_hash) AS max_dop_any_plan + FROM v_query_stats + WHERE server_id = $1 AND collection_time >= $2 AND collection_time <= $3 + AND delta_worker_time > 0 +) SELECT database_name, query_hash, SUM(delta_worker_time)::BIGINT AS total_cpu_us, SUM(delta_execution_count)::BIGINT AS exec_count, - MAX(max_dop) AS max_dop, + MAX(CASE WHEN newest_rn = 1 THEN max_dop END) AS max_dop, SUM(delta_spills)::BIGINT AS spills, - LEFT(MAX(query_text), 500) AS query_text -FROM v_query_stats -WHERE server_id = $1 AND collection_time >= $2 AND collection_time <= $3 -AND delta_worker_time > 0 + LEFT(MAX(query_text), 500) AS query_text, + COUNT(DISTINCT query_plan_hash) AS plan_count, + MAX(max_dop_any_plan) AS max_dop_any_plan, + MAX(CASE WHEN max_dop = max_dop_any_plan THEN collection_time END) AS max_dop_any_plan_last_seen +FROM windowed GROUP BY database_name, query_hash ORDER BY total_cpu_us DESC LIMIT 5"; @@ -130,15 +169,24 @@ ORDER BY total_cpu_us DESC using var reader = await cmd.ExecuteReaderAsync(context.CancellationToken); while (await reader.ReadAsync(context.CancellationToken)) { + var maxDop = reader.IsDBNull(4) ? (int?)null : Convert.ToInt32(reader.GetValue(4)); + var planCount = reader.IsDBNull(7) ? 0L : Convert.ToInt64(reader.GetValue(7)); + var maxDopAnyPlan = reader.IsDBNull(8) ? (int?)null : Convert.ToInt32(reader.GetValue(8)); + var maxDopAnyPlanLastSeen = reader.IsDBNull(9) ? (DateTime?)null : reader.GetDateTime(9); items.Add(new { database = reader.IsDBNull(0) ? "" : reader.GetString(0), query_hash = reader.IsDBNull(1) ? "" : reader.GetString(1), total_cpu_ms = reader.IsDBNull(2) ? 0.0 : Convert.ToDouble(reader.GetValue(2)) / 1000.0, execution_count = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)), - max_dop = reader.IsDBNull(4) ? 0 : Convert.ToInt32(reader.GetValue(4)), + /* #3648: newest plan's reading, null when unknown — never 0. History fields follow. */ + max_dop = maxDop, spills = reader.IsDBNull(5) ? 0L : Convert.ToInt64(reader.GetValue(5)), - query_text = reader.IsDBNull(6) ? "" : reader.GetString(6) + query_text = reader.IsDBNull(6) ? "" : reader.GetString(6), + plan_count = planCount, + max_dop_any_plan = maxDopAnyPlan, + max_dop_any_plan_last_seen = maxDopAnyPlanLastSeen?.ToString("o"), + dop_note = QueryDopProvenance.Note(maxDop, maxDopAnyPlan, maxDopAnyPlanLastSeen, planCount) }); } @@ -567,8 +615,30 @@ private async Task CollectBadActorDetail(AnalysisFinding finding, AnalysisContex using var connection = _duckDb.CreateConnection(); await connection.OpenAsync(context.CancellationToken); + /* #3648: the same per-plan max_dop provenance as CollectTopCpuQueries — see the essay there. This + read is one hash, unfiltered by CPU spent, so the newest-plan tie-break (compile time, then CPU + spent) is what separates a stale parallel plan still sitting in the cache from the serial one + doing the work at the same collection_time. Byte-identical to Darling's BadActorDetailSql. */ using var cmd = connection.CreateCommand(); cmd.CommandText = @" +WITH windowed AS +( + -- #3648: see CollectTopCpuQueries' windowed CTE; same ranking, same hash-wide maximum. + SELECT database_name, query_hash, query_plan_hash, collection_time, max_dop, + delta_worker_time, delta_execution_count, delta_elapsed_time, delta_logical_reads, delta_spills, + query_text, + ROW_NUMBER() OVER + ( + PARTITION BY database_name, query_hash + ORDER BY collection_time DESC, creation_time DESC NULLS LAST, delta_worker_time DESC NULLS LAST + ) AS newest_rn, + MAX(max_dop) OVER (PARTITION BY database_name, query_hash) AS max_dop_any_plan + FROM v_query_stats + WHERE server_id = $1 + AND collection_time >= $2 + AND collection_time <= $3 + AND query_hash = $4 +) SELECT database_name, query_hash, LEFT(MAX(query_text), 500) AS query_text, SUM(delta_execution_count)::BIGINT AS exec_count, @@ -584,12 +654,11 @@ THEN SUM(delta_logical_reads)::DOUBLE PRECISION / SUM(delta_execution_count) SUM(delta_worker_time)::BIGINT AS total_cpu_us, SUM(delta_logical_reads)::BIGINT AS total_reads, SUM(delta_spills)::BIGINT AS total_spills, - MAX(max_dop) AS max_dop -FROM v_query_stats -WHERE server_id = $1 -AND collection_time >= $2 -AND collection_time <= $3 -AND query_hash = $4 + MAX(CASE WHEN newest_rn = 1 THEN max_dop END) AS max_dop, + COUNT(DISTINCT query_plan_hash) AS plan_count, + MAX(max_dop_any_plan) AS max_dop_any_plan, + MAX(CASE WHEN max_dop = max_dop_any_plan THEN collection_time END) AS max_dop_any_plan_last_seen +FROM windowed GROUP BY database_name, query_hash"; cmd.Parameters.Add(new DuckDBParameter { Value = context.ServerId }); @@ -600,6 +669,10 @@ FROM v_query_stats using var reader = await cmd.ExecuteReaderAsync(context.CancellationToken); if (await reader.ReadAsync(context.CancellationToken)) { + var maxDop = reader.IsDBNull(10) ? (int?)null : Convert.ToInt32(reader.GetValue(10)); + var planCount = reader.IsDBNull(11) ? 0L : Convert.ToInt64(reader.GetValue(11)); + var maxDopAnyPlan = reader.IsDBNull(12) ? (int?)null : Convert.ToInt32(reader.GetValue(12)); + var maxDopAnyPlanLastSeen = reader.IsDBNull(13) ? (DateTime?)null : reader.GetDateTime(13); finding.DrillDown!["bad_actor_query"] = new { database = reader.IsDBNull(0) ? "" : reader.GetString(0), @@ -612,7 +685,12 @@ FROM v_query_stats total_cpu_ms = reader.IsDBNull(7) ? 0.0 : Convert.ToDouble(reader.GetValue(7)) / 1000.0, total_reads = reader.IsDBNull(8) ? 0L : Convert.ToInt64(reader.GetValue(8)), total_spills = reader.IsDBNull(9) ? 0L : Convert.ToInt64(reader.GetValue(9)), - max_dop = reader.IsDBNull(10) ? 0 : Convert.ToInt32(reader.GetValue(10)) + /* #3648: newest plan's reading, null when unknown — never 0. History fields follow. */ + max_dop = maxDop, + plan_count = planCount, + max_dop_any_plan = maxDopAnyPlan, + max_dop_any_plan_last_seen = maxDopAnyPlanLastSeen?.ToString("o"), + dop_note = QueryDopProvenance.Note(maxDop, maxDopAnyPlan, maxDopAnyPlanLastSeen, planCount) }; } } diff --git a/PerformanceMonitor.Common/QueryDopProvenance.cs b/PerformanceMonitor.Common/QueryDopProvenance.cs new file mode 100644 index 000000000..68c493ad4 --- /dev/null +++ b/PerformanceMonitor.Common/QueryDopProvenance.cs @@ -0,0 +1,91 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Globalization; + +namespace PerformanceMonitor.Common; + +/// +/// #3648: the human sentence behind a query-stats drill-down row's DOP, composed once here so Lite and +/// Darling say the byte-identical thing. Sibling of , which does the same +/// job for the lifetime CPU/elapsed extremes. +/// +/// What the number was. v_query_stats.max_dop is sys.dm_exec_query_stats.max_dop: +/// a PER-PLAN high-water mark — the highest degree of parallelism that one cached plan has run at since it +/// entered the cache. The analysis drill-downs (top_cpu_queries, bad_actor_query) group query +/// stats by (database_name, query_hash), and until #3648 projected MAX(max_dop) over that +/// group — which folds every plan the statement text ever had inside the window into one number and keeps +/// the largest. On a live card that number read 16 for a statement on an instance whose MAXDOP had been 1 +/// for the whole 14-day config history, whose stored plan for that exact hash was serial +/// (NonParallelPlanReason="MaxDOPSetToOne"), and whose row #3 on the same card honestly said 1. The +/// 16 belonged to a plan compiled before the instance was pinned to 1, still sitting in the cache with its +/// old counter. A reader recommended a MAXDOP 1 Query Store hint from that field — the exact decision the +/// card exists to support — and retracted it after reading the plan the card should have summarized. +/// +/// What the record says now. max_dop is the NEWEST plan's reading — the row with the +/// latest collection_time among the rows that spent CPU in the window — because the card's question +/// is "what is this query doing to the CPU now". The cross-plan maximum survives as +/// max_dop_any_plan, with max_dop_any_plan_last_seen (the latest snapshot that carried it) +/// and plan_count (distinct query_plan_hash values in the group) beside it, so the maximum is +/// a HISTORY with provenance rather than a bare number. turns that history into the one +/// sentence a renderer can print verbatim when — and only when — the history disagrees with the headline. +/// +/// 0 is not a DOP. A serial plan reports 1; the DMV never reports 0. The readers used to coerce +/// a NULL reading to 0 and the card rendered it as a number. Every DOP parameter here is nullable and NULL +/// means "unknown"; the note says so in words rather than inventing a value. +/// +public static class QueryDopProvenance +{ + /// + /// The conditional provenance sentence for one drill-down row, or null when the headline already tells + /// the whole story. Fires only when a plan in the group ran at a HIGHER degree of parallelism than the + /// newest plan reports — the case that produced the wrong recommendation — or when the newest plan's + /// reading is unknown and a parallel plan is on record. A group whose every plan agrees with the + /// headline gets no note: there is no history to disclose. + /// + /// The newest plan's max_dop; null when the DMV row carried none. + /// The cross-plan maximum over the window; null when no row carried a reading. + /// The latest collection_time at which a row reported that maximum. + /// Distinct query_plan_hash values in the group. + public static string? Note(int? newestPlanMaxDop, int? maxDopAnyPlan, DateTime? maxDopAnyPlanLastSeen, long planCount) + { + if (maxDopAnyPlan is null) + { + /* No row in the group carried a reading at all — nothing to compare, nothing to say. */ + return null; + } + + if (newestPlanMaxDop is { } newest && maxDopAnyPlan <= newest) + { + /* The newest plan's reading IS the group's maximum: one honest number, no history behind it. */ + return null; + } + + if (newestPlanMaxDop is null && maxDopAnyPlan <= 1) + { + /* Newest reading unknown, and the only plans on record were serial: "unknown" is already the + whole truth, and there is no parallel plan to disclose. */ + return null; + } + + var headline = newestPlanMaxDop is { } known + ? $"DOP {known.ToString(CultureInfo.InvariantCulture)}" + : "DOP unknown for the newest plan"; + + var until = maxDopAnyPlanLastSeen is { } seen + ? $" until {seen.ToString("yyyy-MM-dd", CultureInfo.InvariantCulture)}" + : string.Empty; + + var plans = planCount == 1 + ? "1 plan in window" + : $"{planCount.ToString(CultureInfo.InvariantCulture)} plans in window"; + + return $"{headline} (a parallel plan ran at {maxDopAnyPlan.Value.ToString(CultureInfo.InvariantCulture)}{until}; {plans})"; + } +} From f88fe32c5a107a8ad3c45e8b842ba7576be9b540 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 18:20:20 -0400 Subject: [PATCH 68/69] Zero is a measurement: health parsers say whether their source was ever seen, a regression with no baseline stays null, and the first point of a differenced trend is no longer a fabricated 0 (#3541 A12) (#3642) * Zero is a measurement: health parsers say whether their source was ever seen, a regression with no baseline stays null, and the first point of a differenced trend is no longer a fabricated 0 (#3541 A12) Six MCP sites published a 0, an empty, or a nominal-window label where nothing had been measured. Eight of the nine get_health_parser_* tools answered a dead system_health session with the same empty a healthy hour earns; now every one carries source_observed / last_captured_at and climbs the four-rung ladder significant_waits alone had, with nothing-ever-captured answering unavailable. get_query_store_regressions kept a NULL percent NULL (a 0 baseline has no ratio) with the reason under undefined_percents, and nulls a severity banded from a ratio that does not exist. The LAG-differenced duration trends (three raw reads, the Query Store rollup builder, Lite's four) leave the window's first collection unrated - null, counted in unrated_points, explained in unrated_note - instead of 0.0. get_pg_xmin_horizon divides a holder's wins by every capture in the window (collection_log SUCCESS rows, the alert evaluator's own denominator) rather than by the source's own rows, and answers unavailable when the collector never looked. get_pvs_stats divides any MEASURED size, so 0 MB is 0.00% and only an unmeasured numerator or absent denominator yields null, with pvs_measured and the reason. get_table_index_sizes projects its baselines raw and derives each growth figure from exactly the baseline it names, publishing history_days_available and growth_over_available_history_* where the nominal windows are out of reach, plus tables_returned / truncated. * Compose with #3630 (V128 / v61): the procedure trend keeps its stored-interval derivation and the MCP reader keeps the NULL-rate row as an unrated point rather than dropping it; the SQL-mirror pin's comment lines move to the C# doc; the #3540 pins that expected the first snapshot absent now expect it unrated * CI round 1: rung 3's closing sentence no longer contains the word the sibling rungs are pinned NOT to say; the Lite trend parity test's lone in-window collection is asserted unrated rather than compared as two zeros; the eight expression-bodied growth derivations join KnownTruncatedRanges (they strand no literal) * Re-cut head: the Claude review posted nothing on 1deff83e across five attempts (the #2229 swallowed-output class); a fresh head gives the review and the guard fresh runs. No code change. --- .../DarlingMcpHealthParserToolsTests.cs | 84 +- .../DarlingMcpObjectStatsToolsTests.cs | 8 +- .../DarlingMcpTrendToolsTests.cs | 18 +- .../DarlingPerformanceTrendsReadTests.cs | 57 +- ...milyIntervalCompletionLivePostgresTests.cs | 8 +- .../EngineCapabilityMissTests.cs | 76 +- .../McpZeroIsAMeasurementTests.cs | 767 ++++++++++++++++++ .../QueryStoreTrendRoutingLiveTests.cs | 24 +- .../Darling.Tests/TsqlConventionGuardTests.cs | 14 + .../Mcp/DarlingMcpHealthParserTools.cs | 225 +++-- .../Mcp/DarlingMcpInstructions.cs | 8 +- .../Mcp/DarlingMcpObjectStatsTools.cs | 70 +- .../Mcp/DarlingMcpPgXminTools.cs | 45 +- .../Mcp/DarlingMcpPvsTools.cs | 31 +- .../DarlingMcpQueryStoreRegressionTools.cs | 35 +- .../Mcp/DarlingMcpTrendTools.cs | 14 +- .../Mcp/DarlingObjectStatsReader.cs | 107 ++- .../Mcp/DarlingQueryStoreRegressionReader.cs | 21 +- .../Mcp/DarlingSystemHealthReader.cs | 79 +- .../Mcp/DarlingTrendReader.cs | 74 +- .../DarlingPgXminReader.cs | 46 ++ .../QueryStoreTrendRouting.cs | 8 +- .../ViewerDataService.QueryTrends.cs | 8 +- .../DeltaFamilyUnknowableRowReadTests.cs | 30 +- Lite.Tests/EngineCapabilityMissTests.cs | 15 +- Lite.Tests/IndexObjectStatsTests.cs | 15 +- Lite.Tests/McpMissMessageParityPinTests.cs | 22 + Lite.Tests/McpZeroIsAMeasurementTests.cs | 160 ++++ Lite.Tests/PerformanceTrendsToolTests.cs | 41 +- Lite.Tests/QueryStoreDedupReadTests.cs | 11 +- Lite.Tests/TrendEmptyParityToolTests.cs | 11 +- Lite/Controls/ServerTab.Charts.cs | 36 +- Lite/Mcp/McpHealthParserTools.cs | 220 +++-- Lite/Mcp/McpInstructions.cs | 8 +- Lite/Mcp/McpObjectStatsTools.cs | 72 +- Lite/Mcp/McpPvsTools.cs | 31 +- Lite/Mcp/McpQueryTools.cs | 47 +- .../LocalDataService.FinOps.IndexObjects.cs | 115 ++- Lite/Services/LocalDataService.QueryStats.cs | 81 +- Lite/Services/LocalDataService.QueryStore.cs | 15 +- .../LocalDataService.QueryStoreRegressions.cs | 21 +- .../Services/LocalDataService.SystemEvents.cs | 82 +- 42 files changed, 2438 insertions(+), 422 deletions(-) create mode 100644 Darling/Darling.Tests/McpZeroIsAMeasurementTests.cs create mode 100644 Lite.Tests/McpZeroIsAMeasurementTests.cs diff --git a/Darling/Darling.Tests/DarlingMcpHealthParserToolsTests.cs b/Darling/Darling.Tests/DarlingMcpHealthParserToolsTests.cs index aa5fc6216..65a996b25 100644 --- a/Darling/Darling.Tests/DarlingMcpHealthParserToolsTests.cs +++ b/Darling/Darling.Tests/DarlingMcpHealthParserToolsTests.cs @@ -12,6 +12,7 @@ using System.IO; using System.Linq; using System.Reflection; +using System.Text.Json; using System.Threading.Tasks; using Microsoft.Extensions.DependencyInjection; using ModelContextProtocol.Server; @@ -113,27 +114,44 @@ public void DatabaseNameMapSql_LatestNamePerId_FromSizeStatsView() } /// - /// #2484: the probe that lets an empty parse-on-read answer say WHICH nothing it found must read the - /// SAME source the read itself reads. A probe on the base table would report a server as captured for - /// rows the view-backed read can never return -- picking the wrong branch in precisely the case the - /// probe exists to get right. It is also scoped to the event_type, and windowless by design. + /// #2484 → #3541 A12: the probes that let an empty parse-on-read answer say WHICH nothing it found must + /// read the SAME source the read itself reads. A probe on the base table would report a server as + /// captured for rows the view-backed read can never return -- picking the wrong branch in precisely the + /// case the probe exists to get right. Both are windowless by design (a time bound would make them + /// answer the same question the read just did) and both are a MAX over collection_time, so they + /// say WHEN as well as whether — the message needs the when. The type-scoped one is scoped to + /// event_type; the source witness deliberately is not, because it answers "has this server's ring + /// buffer ever been read into the store", which is about the session, not the category. Neither is + /// collection_log: the log records a SUCCESS for a run that read a dead session and stored + /// nothing, which is exactly the shape being mis-reported. /// [Fact] - public void HasAnyEventOfTypeSql_ProbesTheSameView_ScopedToType_AndIgnoresTheWindow() + public void TheWitnessProbes_ReadTheSameView_AreWindowless_AndSayWhen() { - var sql = DarlingSystemHealthReader.HasAnyEventOfTypeSql; - Assert.Contains("FROM v_system_health_events", sql, StringComparison.Ordinal); - Assert.Contains("WHERE server_id = $1", sql, StringComparison.Ordinal); - Assert.Contains("event_type = $2", sql, StringComparison.Ordinal); - Assert.Contains("LIMIT 1", sql, StringComparison.Ordinal); - /* Windowless: a time bound here would make the probe answer the same question the read just did. */ - Assert.DoesNotContain("event_time", sql, StringComparison.Ordinal); + var source = DarlingSystemHealthReader.LastCaptureSql; + Assert.Contains("SELECT MAX(collection_time)", source, StringComparison.Ordinal); + Assert.Contains("FROM v_system_health_events", source, StringComparison.Ordinal); + Assert.Contains("WHERE server_id = $1", source, StringComparison.Ordinal); + Assert.Contains("event_xml IS NOT NULL", source, StringComparison.Ordinal); + Assert.DoesNotContain("event_type", source, StringComparison.Ordinal); + Assert.DoesNotContain("event_time", source, StringComparison.Ordinal); + Assert.DoesNotContain("collection_log", source, StringComparison.Ordinal); + + var ofType = DarlingSystemHealthReader.LastCaptureOfTypeSql; + Assert.Contains("SELECT MAX(collection_time)", ofType, StringComparison.Ordinal); + Assert.Contains("FROM v_system_health_events", ofType, StringComparison.Ordinal); + Assert.Contains("WHERE server_id = $1", ofType, StringComparison.Ordinal); + Assert.Contains("event_type = $2", ofType, StringComparison.Ordinal); + Assert.Contains("event_xml IS NOT NULL", ofType, StringComparison.Ordinal); + Assert.DoesNotContain("event_time", ofType, StringComparison.Ordinal); + Assert.DoesNotContain("collection_log", ofType, StringComparison.Ordinal); } [Theory] [InlineData(nameof(DarlingSystemHealthReader.SystemHealthEventsByTypeSql))] [InlineData(nameof(DarlingSystemHealthReader.DatabaseNameMapSql))] - [InlineData(nameof(DarlingSystemHealthReader.HasAnyEventOfTypeSql))] + [InlineData(nameof(DarlingSystemHealthReader.LastCaptureSql))] + [InlineData(nameof(DarlingSystemHealthReader.LastCaptureOfTypeSql))] public void Reads_ArePostgresDialect_PositionalParams(string sqlName) { var sql = (string)typeof(DarlingSystemHealthReader).GetField(sqlName)!.GetValue(null)!; @@ -342,8 +360,8 @@ public void SevereError_DatabaseNameResolution_MatchesViewer() /// /// Gated (DARLING_TEST_PG) live round-trip for the health-parser tools. Plants raw system_health_events rows /// (real captured-event fixtures) across the categories + a database_size_stats mapping row, then asserts each -/// tool shreds + gates + resolves and returns its data-bearing envelope; an empty store returns the "empty" -/// miss. +/// tool shreds + gates + resolves and returns its data-bearing envelope with the source witness; a category +/// never captured on a live session is the healthy "empty"; an empty store is "unavailable" (#3541 A12). /// [Collection("live-postgres")] public sealed class DarlingMcpHealthParserToolsLivePostgresTests @@ -403,16 +421,42 @@ await DarlingMcpTestData.ExecAsync(connection, ct, DarlingMcpTestData.AssertEnvelope(await DarlingMcpHealthParserTools.GetIOIssues(postgres, ServerName), ServerName, "issues"); DarlingMcpTestData.AssertEnvelope(await DarlingMcpHealthParserTools.GetMemoryNodeOOM(postgres, ServerName), ServerName, "events"); - /* memory_conditions / memory_broker have no planted LOW rows → the "empty" miss (not a throw). */ - Assert.Equal("empty", DarlingMcpTestData.StatusOf(await DarlingMcpHealthParserTools.GetMemoryConditions(postgres, ServerName))); - Assert.Equal("empty", DarlingMcpTestData.StatusOf(await DarlingMcpHealthParserTools.GetMemoryBroker(postgres, ServerName))); + /* memory_conditions has planted sp_server_diagnostics rows and none is LOW → rung 1 of the + #3541 A12 ladder: "empty", captured and gated out, with the witness saying the source was + observed. memory_broker's event type was never planted while OTHER types were → rung 3: + still "empty" (the session IS being read; the engine recorded no broker event), never + "unavailable". Both are the healthy answer and both must say so with the witness attached. */ + var conditions = JsonDocument.Parse(await DarlingMcpHealthParserTools.GetMemoryConditions(postgres, ServerName)).RootElement; + Assert.Equal("empty", conditions.GetProperty("status").GetString()); + Assert.True(conditions.GetProperty("source_observed").GetBoolean()); + Assert.Equal(t.ToString("o"), conditions.GetProperty("last_captured_at").GetString()); + Assert.True(conditions.GetProperty("events_in_window").GetInt32() > 0); + Assert.Contains("Events ARE being captured", conditions.GetProperty("message").GetString()!, StringComparison.Ordinal); + + var broker = JsonDocument.Parse(await DarlingMcpHealthParserTools.GetMemoryBroker(postgres, ServerName)).RootElement; + Assert.Equal("empty", broker.GetProperty("status").GetString()); + Assert.True(broker.GetProperty("source_observed").GetBoolean()); + Assert.Equal(0, broker.GetProperty("events_in_window").GetInt32()); + Assert.Equal(JsonValueKind.Null, broker.GetProperty("last_captured_of_type_at").ValueKind); + Assert.Contains("the absence is a measurement", broker.GetProperty("message").GetString()!, StringComparison.Ordinal); + + /* The data envelope carries the same witness pair. */ + var scheduler = JsonDocument.Parse(await DarlingMcpHealthParserTools.GetSchedulerIssues(postgres, ServerName)).RootElement; + Assert.True(scheduler.GetProperty("source_observed").GetBoolean()); + Assert.Equal(t.ToString("o"), scheduler.GetProperty("last_captured_at").GetString()); /* an unknown server resolves to the listing error. */ Assert.StartsWith("Could not resolve server.", await DarlingMcpHealthParserTools.GetSystemHealth(postgres, "darling-no-such-server"), StringComparison.Ordinal); - /* an empty store returns the miss. */ + /* An empty store is rung 4: nothing of any type was ever captured, so this is NOT a clean bill — + "unavailable" with source_observed false (#3541 A12). It used to answer "empty", the same word + the healthy branches above earn, which is the defect. */ await DeleteRowsAsync(connection, ct, keepServer: true); - Assert.Equal("empty", DarlingMcpTestData.StatusOf(await DarlingMcpHealthParserTools.GetSchedulerIssues(postgres, ServerName))); + var dead = JsonDocument.Parse(await DarlingMcpHealthParserTools.GetSchedulerIssues(postgres, ServerName)).RootElement; + Assert.Equal("unavailable", dead.GetProperty("status").GetString()); + Assert.False(dead.GetProperty("source_observed").GetBoolean()); + Assert.Equal(JsonValueKind.Null, dead.GetProperty("last_captured_at").ValueKind); + Assert.Contains("NOT an all-clear", dead.GetProperty("message").GetString()!, StringComparison.Ordinal); bodySucceeded = true; } diff --git a/Darling/Darling.Tests/DarlingMcpObjectStatsToolsTests.cs b/Darling/Darling.Tests/DarlingMcpObjectStatsToolsTests.cs index 97b3307bc..fb8a82547 100644 --- a/Darling/Darling.Tests/DarlingMcpObjectStatsToolsTests.cs +++ b/Darling/Darling.Tests/DarlingMcpObjectStatsToolsTests.cs @@ -110,8 +110,12 @@ public void ObjectSizeGrowthSql_RollsUpPerTable_ComputesGrowth() "GROUP BY database_name, schema_name, table_name", sql, "the rollup is no longer per table"); - Assert.Contains("growth_7d_mb", sql, StringComparison.Ordinal); - Assert.Contains("growth_30d_mb", sql, StringComparison.Ordinal); + /* #3541 A12: the growth figures are no longer derived in SQL — the raw baselines come back as their own + nullable columns and ObjectSizeGrowthRow derives each figure from exactly the baseline it names + (McpZeroIsAMeasurementTests pins the fold's absence and the derivations). */ + Assert.Contains("reserved_mb_7d_ago", sql, StringComparison.Ordinal); + Assert.Contains("reserved_mb_30d_ago", sql, StringComparison.Ordinal); + Assert.Contains("reserved_mb_oldest", sql, StringComparison.Ordinal); SqlTextPin.AssertExpresses("ORDER BY l.current_reserved_mb DESC", sql, "the biggest table no longer sorts first"); } diff --git a/Darling/Darling.Tests/DarlingMcpTrendToolsTests.cs b/Darling/Darling.Tests/DarlingMcpTrendToolsTests.cs index 66a7314fa..dbfbcdfaa 100644 --- a/Darling/Darling.Tests/DarlingMcpTrendToolsTests.cs +++ b/Darling/Darling.Tests/DarlingMcpTrendToolsTests.cs @@ -206,9 +206,13 @@ public void ProcedureDurationTrendSql_SameRate_OverProcedureStats() /// /// #3540 (V128): the procedure trend reads the collection's STORED interval — MAX over the collection's - /// rows, 0 → NULL through NULLIF so a restart's marker collection drops rather than plotting 0.00 — and - /// falls back to the LAG derivation only for a pre-V128 collection. No ELSE 0. Byte-identical to the - /// viewer's copy apart from the database filter, as the pair always were. + /// rows, 0 → NULL through NULLIF so a restart's marker collection has no rate rather than plotting 0.00 — + /// and falls back to the LAG derivation only for a pre-V128 collection. No ELSE 0. Byte-identical to the + /// viewer's copy apart from the database filter, as the pair always were. The C# half: since #3541 A12 + /// the MCP reader KEEPS the NULL-rate row as an unrated point (QueryDurationTrendPoint.HasRate + /// false) rather than dropping it — a lone collection must not become an empty series the empty ladder + /// mislabels as quiet, and effective_start must be the first collection the store held. The viewer's + /// chart reader is the one that drops, because a chart has nowhere to draw "unknown". /// [Fact] public void ProcedureDurationTrendSql_PrefersTheStoredInterval_NeverFabricatesZero_AndMirrorsTheViewer() @@ -228,13 +232,15 @@ public void ProcedureDurationTrendSql_PrefersTheStoredInterval_NeverFabricatesZe var mcp = string.Join('\n', sql.Replace("\r\n", "\n", StringComparison.Ordinal).Split('\n').Select(l => l.Trim())); Assert.Equal(viewer, mcp); - /* And the shared reader DROPS a NULL-rate row rather than reading it as 0 — the C# half of the idiom. */ + /* And the shared reader KEEPS a NULL-rate row as an unrated point rather than reading it as 0 or + dropping it — the C# half of the idiom (#3541 A12). */ var source = RepoFile.ReadRepoFile("Darling", "PerformanceMonitor.Darling.Service", "Mcp", "DarlingTrendReader.cs"); var reader = source[source.IndexOf("private static async Task> ReadDurationPointsAsync(", StringComparison.Ordinal)..]; reader = reader[..reader.IndexOf("return items;", StringComparison.Ordinal)]; - Assert.Contains("if (reader.IsDBNull(1))", reader, StringComparison.Ordinal); - Assert.Contains("continue;", reader, StringComparison.Ordinal); + Assert.Contains("reader.IsDBNull(1) ? null", reader, StringComparison.Ordinal); + Assert.DoesNotContain("continue;", reader, StringComparison.Ordinal); Assert.DoesNotContain("reader.IsDBNull(1) ? 0", reader, StringComparison.Ordinal); + Assert.Equal(typeof(double?), typeof(DarlingTrendReader.QueryDurationTrendPoint).GetProperty("Value")!.PropertyType); } /// diff --git a/Darling/Darling.Tests/DarlingPerformanceTrendsReadTests.cs b/Darling/Darling.Tests/DarlingPerformanceTrendsReadTests.cs index c2c62736f..492d3c70d 100644 --- a/Darling/Darling.Tests/DarlingPerformanceTrendsReadTests.cs +++ b/Darling/Darling.Tests/DarlingPerformanceTrendsReadTests.cs @@ -125,9 +125,11 @@ an empty raw answer the requested start stands and the served span is the whole This is the whole reason executions_per_second exists. These rows carry no sample_interval_seconds (the pre-V128 shape), so the read LAG-derives - the interval — and since V128 (#3540) the FIRST snapshot, which has nothing to LAG against, - is absent rather than a fabricated 0.0 point (the correction V127 made for the wait trends). - One point comes back: the second snapshot, whose rate is the thing under test. + the interval — and the FIRST snapshot, which has nothing to LAG against, is UNRATED: two + points come back, the first with null rates (#3541 A12 — kept, not dropped, so a lone + collection is never an empty series and effective_start is the first collection the store + held; never the fabricated 0.0 it was before #3540), the envelope counting it and saying + why, and the second carrying the rate under test. */ await SeedProcedureAsync(connection, ct, MinutesAgo(20), executions: 0, elapsedUs: 0); await SeedProcedureAsync(connection, ct, MinutesAgo(15), executions: 2, elapsedUs: 600_000); @@ -135,9 +137,17 @@ These rows carry no sample_interval_seconds (the pre-V128 shape), so the read LA var procs = JsonDocument.Parse( await DarlingMcpTrendTools.GetProcedureDurationTrend(postgres, ServerName, 4)).RootElement; var procTrend = procs.GetProperty("trend"); - Assert.Equal(1, procTrend.GetArrayLength()); + Assert.Equal(2, procTrend.GetArrayLength()); - var second = procTrend[0]; + var first = procTrend[0]; + Assert.Equal(JsonValueKind.Null, first.GetProperty("value").ValueKind); + Assert.Equal(JsonValueKind.Null, first.GetProperty("elapsed_ms_per_second").ValueKind); + Assert.Equal(JsonValueKind.Null, first.GetProperty("execution_count").ValueKind); + Assert.Equal(JsonValueKind.Null, first.GetProperty("executions_per_second").ValueKind); + Assert.Equal(1, procs.GetProperty("unrated_points").GetInt32()); + Assert.Contains("no previous one inside the window", procs.GetProperty("unrated_note").GetString()!, StringComparison.Ordinal); + + var second = procTrend[1]; Assert.True(second.GetProperty("value").GetDouble() > 0, "elapsed ms/sec must be a real rate"); Assert.Equal(0, second.GetProperty("execution_count").GetInt64()); Assert.True( @@ -151,11 +161,11 @@ execution_count precedent. */ /* #3541 A2: the disclosure block. A 4-hour window anchored at now sits inside the raw horizon (the shared fixture carries no continuous aggregates — every test that builds them mints a - ScratchPostgres — so raw is also the only tier here), and the series the read SERVED begins - at its first point — the 15-minutes-ago seed, since V128 dropped the prior-less first - snapshot: effective_start says so, and the head sits three-plus hours past the requested - start, which is what `truncated` means. The point is that the label matches the data rather - than the request. + ScratchPostgres — so raw is also the only tier here), and the series the store held begins + at the 20-minutes-ago seed — the unrated first collection, kept since #3541 A12 exactly so + effective_start can say so — and the head sits three-plus hours past the requested start, + which is what `truncated` means. The point is that the label matches the data rather than + the request. */ Assert.Equal("raw", procs.GetProperty("source").GetString()); Assert.Equal("per-collection", procs.GetProperty("bucket").GetString()); @@ -187,10 +197,13 @@ separate UtcNow reads truncated to the second can land 3599 apart and quietly br await DarlingMcpTrendTools.GetQueryStoreDurationTrend(postgres, ServerName, 6)).RootElement; var storeTrend = store.GetProperty("trend"); - /* Four rows in, two points out — one per interval, not one per fetch. */ + /* Four rows in, two points out — one per interval, not one per fetch. The first interval has no + predecessor to difference against, so its rates are null, not 0 (#3541 A12). */ Assert.Equal(2, storeTrend.GetArrayLength()); Assert.StartsWith(intervalA.ToString("o")[..16], storeTrend[0].GetProperty("time").GetString()!, StringComparison.Ordinal); + Assert.Equal(JsonValueKind.Null, storeTrend[0].GetProperty("executions_per_second").ValueKind); Assert.StartsWith(intervalB.ToString("o")[..16], storeTrend[1].GetProperty("time").GetString()!, StringComparison.Ordinal); + Assert.Equal(1, store.GetProperty("unrated_points").GetInt32()); /* The surviving snapshot is the FINAL one (25 executions over the 3600 seconds between the two @@ -284,7 +297,7 @@ private static async Task DeleteRowsAsync(NpgsqlConnection connection, Cancellat /// /// Live rather than a string pin because the load-bearing claims are about which RELATION answered /// and what it computed: that the rollup-served point is the hour's summed work over the bucket width -/// (no LAG, so no fabricated first-point zero), that the payload's source / effective_start / +/// (no LAG, so no first-point question at all), that the payload's source / effective_start / /// truncated describe the served series rather than the request, and that the empty branch on this /// route names an unserved head instead of a quiet window. The raw-only read of the SAME fixture is /// asserted beside it as the revert-proof: put the raw-only read back and the rates, the point count and @@ -397,18 +410,24 @@ public async Task PastTheRawHorizon_TheTrio_ServesTheHourlyRollup_AndSaysSo() Assert.Equal(0.1, procedureTrend[1].GetProperty("value").GetDouble(), 9); /* ── the raw-only read of the SAME fixture: the estimator this replaced, and the revert-proof. - Three per-collection points, the first rated ZERO because the LAG idiom has no previous - collection to difference against — the fabricated quiet the bucket-width denominator does not - produce. Restoring the raw-only read unconditionally fails the count, the rates and the - `source` word above, not just a routing flag. ── */ + Three per-collection points, the first UNRATED (null) because the LAG idiom has no previous + collection to difference against — before #3541 A12 that point was published as a fabricated + 0.0, the quiet instant the bucket-width denominator never produced. Restoring the raw-only + read unconditionally fails the count, the rates and the `source` word above, not just a + routing flag; restoring the ELSE 0 fails the null here. ── */ var rawRoute = route with { Tier = RetentionTier.Raw }; var raw = await DarlingTrendReader.GetQueryDurationTrendAsync(postgres, ServerId, hour10.AddHours(-4), hour12, rawRoute, ct); Assert.Equal(3, raw.Points.Count); Assert.Equal(hour10.AddMinutes(5), raw.Points[0].CollectionTime); - Assert.Equal(0, raw.Points[0].Value); - Assert.Equal(8.0, raw.Points[1].Value, 9); /* 7,200 ms over the 900 s since 10:05 */ - Assert.Equal(1800d / 3300d, raw.Points[2].Value, 9); /* 1,800 ms over the 3,300 s since 10:20 */ + Assert.False(raw.Points[0].HasRate); + Assert.Null(raw.Points[0].Value); + Assert.Null(raw.Points[0].ExecutionCount); + Assert.Null(raw.Points[0].ExecutionsPerSecond); + Assert.Equal(8.0, raw.Points[1].Value!.Value, 9); /* 7,200 ms over the 900 s since 10:05 */ + Assert.Equal(1800d / 3300d, raw.Points[2].Value!.Value, 9); /* 1,800 ms over the 3,300 s since 10:20 */ + /* The unrated point is KEPT, so the series still says truthfully where the store's data begins. */ + Assert.Equal(hour10.AddMinutes(5), raw.EffectiveStartUtc); Assert.Equal("raw", rawRoute.Source); /* ── never sampled on this route: not an empty window. The raw probe finds nothing and so does the diff --git a/Darling/Darling.Tests/DeltaFamilyIntervalCompletionLivePostgresTests.cs b/Darling/Darling.Tests/DeltaFamilyIntervalCompletionLivePostgresTests.cs index d370d8b81..2f05ae870 100644 --- a/Darling/Darling.Tests/DeltaFamilyIntervalCompletionLivePostgresTests.cs +++ b/Darling/Darling.Tests/DeltaFamilyIntervalCompletionLivePostgresTests.cs @@ -46,8 +46,9 @@ public sealed class DeltaFamilyIntervalCompletionLivePostgresTests /// /// The procedure duration trend, both copies (the viewer's read and the MCP's SQL, which the pin in - /// DarlingMcpTrendToolsTests proves are one string): t1/t2 pre-V128 (NULL) — t1 no prior, no point; t2 the - /// LAG's 300 s. t3 a restart — every row 0 — absent. t4 a steady pass with a readmitted plan (its row 0) + /// DarlingMcpTrendToolsTests proves are one string): t1/t2 pre-V128 (NULL) — t1 no prior, no rate (the + /// viewer drops it, the MCP keeps it unrated); t2 the LAG's 300 s. t3 a restart — every row 0 — no rate + /// either. t4 a steady pass with a readmitted plan (its row 0) /// beside a measured 120 s row: MAX 120 wins over the LAG's 300, and the readmitted plan adds 0. /// [Fact] @@ -100,7 +101,8 @@ public async Task ProcedureDurationTrend_DropsTheUnknowableCollection_PrefersThe reader.IsDBNull(2) ? null : Convert.ToDouble(reader.GetValue(2)))); } - /* The SQL returns all four collections; t1 and t3 carry NULL rates, which the shared reader drops. */ + /* The SQL returns all four collections; t1 and t3 carry NULL rates — the viewer's chart reader drops + those (above), the MCP reader keeps them as unrated points (#3541 A12). */ Assert.Equal(new[] { t1, t2, t3, t4 }, mcp.Select(m => m.At).ToArray()); Assert.Null(mcp[0].Rate); Assert.Equal(2.0, mcp[1].Rate!.Value, precision: 6); diff --git a/Darling/Darling.Tests/EngineCapabilityMissTests.cs b/Darling/Darling.Tests/EngineCapabilityMissTests.cs index 721b1d57a..71c86fb85 100644 --- a/Darling/Darling.Tests/EngineCapabilityMissTests.cs +++ b/Darling/Darling.Tests/EngineCapabilityMissTests.cs @@ -322,6 +322,12 @@ public sealed class EngineCapabilityReadWiringTests private static readonly Regex CollectorConst = new( @"private const string (\w+) = ""([a-z_0-9]+)"";", RegexOptions.Compiled); + /* A private helper a tool body may delegate its miss path to (#3541 A12: the health-parser family's + shared EmptyAsync ladder). A wiring call inside one is attributed to every tool whose body calls the + helper — the read still asks the question, one method further down. */ + private static readonly Regex HelperDeclaration = new( + @"private static async Task (\w+)(?:<\w+>)?\(", RegexOptions.Compiled); + /// /// Every collector name a shipped read asks the capability question about, across both SKUs. Exposed so /// @@ -353,6 +359,8 @@ internal static SortedDictionary> WiredReads(string mc var consts = CollectorConst.Matches(source) .ToDictionary(m => m.Groups[1].Value, m => m.Groups[2].Value, StringComparer.Ordinal); var marks = ToolMark.Matches(source); + var helpers = HelperDeclaration.Matches(source); + var viaHelper = new Dictionary>(StringComparer.Ordinal); foreach (Match call in WiringCall.Matches(source)) { @@ -360,8 +368,22 @@ internal static SortedDictionary> WiredReads(string mc ? call.Groups[1].Value : consts.TryGetValue(call.Groups[2].Value, out var resolved) ? resolved : call.Groups[2].Value; - /* The enclosing tool is the last McpServerTool mark before the call. */ + /* The enclosing tool is the last McpServerTool mark before the call — unless a private + helper is declared between that mark and the call, in which case the call belongs to the + helper and reaches every tool that calls it (#3541 A12). */ var owner = marks.Where(m => m.Index < call.Index).LastOrDefault(); + var helper = helpers.Where(h => h.Index < call.Index && (owner is null || h.Index > owner.Index)).LastOrDefault(); + if (helper is not null) + { + if (!viaHelper.TryGetValue(helper.Groups[1].Value, out var helperCollectors)) + { + viaHelper[helper.Groups[1].Value] = helperCollectors = new SortedSet(StringComparer.Ordinal); + } + + helperCollectors.Add(collector); + continue; + } + Assert.True(owner is not null, $"{Path.GetFileName(file)}: a capability call sits outside any MCP tool"); if (!wired.TryGetValue(owner!.Groups[1].Value, out var collectors)) @@ -371,6 +393,39 @@ internal static SortedDictionary> WiredReads(string mc collectors.Add(collector); } + + /* Each tool body (from its mark to the next) that calls a wired helper asks the helper's question. */ + for (var i = 0; i < marks.Count; i++) + { + var end = i + 1 < marks.Count ? marks[i + 1].Index : source.Length; + var body = source[marks[i].Index..end]; + foreach (var (helperName, helperCollectors) in viaHelper) + { + if (!Regex.IsMatch(body, $@"\b{Regex.Escape(helperName)}\(")) + { + continue; + } + + if (!wired.TryGetValue(marks[i].Groups[1].Value, out var collectors)) + { + wired[marks[i].Groups[1].Value] = collectors = new SortedSet(StringComparer.Ordinal); + } + + collectors.UnionWith(helperCollectors); + } + } + + /* A helper nobody calls would let the question go unasked while this scan still counted it. */ + foreach (var helperName in viaHelper.Keys) + { + Assert.True( + Enumerable.Range(0, marks.Count).Any(i => + { + var end = i + 1 < marks.Count ? marks[i + 1].Index : source.Length; + return Regex.IsMatch(source[marks[i].Index..end], $@"\b{Regex.Escape(helperName)}\("); + }), + $"{Path.GetFileName(file)}: helper {helperName} asks the capability question but no tool calls it"); + } } return wired; @@ -594,8 +649,13 @@ tempdb_stats stops being a permanent gap. Picking it as the example here would t Assert.Equal("not_collected", DarlingMcpTestData.StatusOf(azureTrace)); Assert.Contains("default_trace_events", azureTrace, StringComparison.Ordinal); - /* ── An Enterprise box, same empty store: every one of them keeps its own miss ── */ - Assert.Equal("empty", DarlingMcpTestData.StatusOf(await DarlingMcpHealthParserTools.GetSystemHealth(postgres, BoxServerName))); + /* ── An Enterprise box, same empty store: every one of them keeps its own miss. For the + health-parser family that own miss is "unavailable" since #3541 A12 (a server whose + system_health session has never been read into the store is not a clean bill), the answer + significant_waits alone used to give and the other eight now share. ── */ + var boxHealth = await DarlingMcpHealthParserTools.GetSystemHealth(postgres, BoxServerName); + Assert.Equal("unavailable", DarlingMcpTestData.StatusOf(boxHealth)); + Assert.Contains("system_health session is started", boxHealth, StringComparison.Ordinal); var boxWaits = await DarlingMcpHealthParserTools.GetSignificantWaits(postgres, BoxServerName); Assert.Equal("unavailable", DarlingMcpTestData.StatusOf(boxWaits)); @@ -703,8 +763,10 @@ off on no engine edition at all — so nothing but the kind axis could ever make Assert.Contains("runs PostgreSQL.", stockFlags, StringComparison.Ordinal); Assert.DoesNotContain("Aurora", stockFlags, StringComparison.Ordinal); - /* ── And the server nobody has probed keeps every one of its old misses ── */ - Assert.Equal("empty", DarlingMcpTestData.StatusOf(await DarlingMcpHealthParserTools.GetSystemHealth(postgres, UnprobedServerName))); + /* ── And the server nobody has probed keeps every one of its old misses — for the health-parser + family that own miss is "unavailable" since #3541 A12 (a never-read session is not a clean + bill); the point here is that it is NOT "not_collected": unknown is not never. ── */ + Assert.Equal("unavailable", DarlingMcpTestData.StatusOf(await DarlingMcpHealthParserTools.GetSystemHealth(postgres, UnprobedServerName))); Assert.Equal("empty", DarlingMcpTestData.StatusOf(await DarlingMcpConfigTools.GetTraceFlags(postgres, UnprobedServerName))); var unprobedWaits = await DarlingMcpHealthParserTools.GetSignificantWaits(postgres, UnprobedServerName); @@ -743,7 +805,9 @@ public async Task AServerWithNoProbedEdition_KeepsItsOldMiss() { await RegisterAsync(connection, ct, BoxServerId, BoxServerName, engineEdition: null); - Assert.Equal("empty", DarlingMcpTestData.StatusOf(await DarlingMcpHealthParserTools.GetSystemHealth(postgres, BoxServerName))); + /* The health parsers' own miss for a never-read session is "unavailable" (#3541 A12) — the + claim under test is that an unprobed edition does not turn it into "not_collected". */ + Assert.Equal("unavailable", DarlingMcpTestData.StatusOf(await DarlingMcpHealthParserTools.GetSystemHealth(postgres, BoxServerName))); Assert.Equal("empty", DarlingMcpTestData.StatusOf(await DarlingMcpDefaultTraceTools.GetDefaultTraceEvents(postgres, BoxServerName))); bodySucceeded = true; diff --git a/Darling/Darling.Tests/McpZeroIsAMeasurementTests.cs b/Darling/Darling.Tests/McpZeroIsAMeasurementTests.cs new file mode 100644 index 000000000..1071c1725 --- /dev/null +++ b/Darling/Darling.Tests/McpZeroIsAMeasurementTests.cs @@ -0,0 +1,767 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.ComponentModel; +using System.Linq; +using System.Reflection; +using System.Text.Json; +using System.Text.RegularExpressions; +using System.Threading; +using System.Threading.Tasks; +using ModelContextProtocol.Server; +using Npgsql; +using PerformanceMonitor.Collectors; +using PerformanceMonitor.Common; +using PerformanceMonitor.Darling.Service; +using PerformanceMonitor.Darling.Service.Mcp; +using PerformanceMonitor.Darling.Storage; +using Xunit; +using static Darling.Tests.RepoFile; + +namespace Darling.Tests; + +/// +/// #3541 A12 — contract rule 5, "zero is a measurement": a zero the product publishes must have been measured, +/// and an absence must be null / unknown WITH its reason. Six sites published a 0, an empty, or a +/// nominal-window label where nothing had been measured: +/// +/// +/// eight of the nine get_health_parser_* tools answered a dead system_health session (or a +/// collector that never ran) with the same empty a healthy quiet hour earns; +/// get_query_store_regressions coerced a NULL percent (baseline side 0 → no denominator) to 0, +/// publishing the largest possible I/O regression as "no change"; +/// the LAG-differenced duration trends rated the window's first collection 0.0 — a fabricated quiet +/// instant — on both SKUs and on the Query Store rollup route (#3540 A8's "first point of every differenced +/// series"); +/// get_pg_xmin_horizon divided a holder's wins by its OWN rows, so 2 wins in 2 holder-bearing +/// collections out of 288 captures read as 100% chronic; +/// get_pvs_stats published a measured 0 MB and an unmeasured NULL identically (both +/// pct_of_database: null); +/// get_table_index_sizes folded a missing 30-day baseline onto the 7-day one (and that onto the +/// oldest, and that onto current → growth 0) and labelled the result with the window asked for. +/// +/// +/// This file is the census: the discriminators are witnessed against the defect shapes as literals (a +/// matcher that quietly stopped matching reports a clean bill), the fixed shapes are pinned on BOTH SKUs from +/// source (Lite's assembly is not referenced here, the arrangement), the +/// SQL consts are pinned for the new columns, and the pure derivations (growth, PVS reasons) are executed. +/// runs the tools against live Postgres. +/// +public sealed class McpZeroIsAMeasurementTests +{ + private const string DarlingMcp = "Darling/PerformanceMonitor.Darling.Service/Mcp"; + private const string LiteMcp = "Lite/Mcp"; + + private static readonly string[] HealthParserTools = + [ + "get_health_parser_cpu_tasks", + "get_health_parser_io_issues", + "get_health_parser_memory_broker", + "get_health_parser_memory_conditions", + "get_health_parser_memory_node_oom", + "get_health_parser_scheduler_issues", + "get_health_parser_severe_errors", + "get_health_parser_significant_waits", + "get_health_parser_system_health", + ]; + + /* ───────────────────────── 1. the health-parser source witness ───────────────────────── */ + + /// + /// Every one of the nine, on both SKUs, publishes the witness on its data envelope and routes its + /// zero-row case through the shared four-rung ladder — a tool that kept a private + /// Status("empty", …) would be the defect returning under one name. + /// + [Theory] + [InlineData(DarlingMcp + "/DarlingMcpHealthParserTools.cs")] + [InlineData(LiteMcp + "/McpHealthParserTools.cs")] + public void EveryHealthParserTool_PublishesTheSourceWitness_AndClimbsTheSharedLadder(string file) + { + var source = ReadRepoFile(file.Split('/')); + + foreach (var tool in HealthParserTools) + { + var body = Strip(ToolBody(source, tool)); + Assert.Contains("source_observed = true", body, StringComparison.Ordinal); + Assert.Contains("last_captured_at = Stamp(", body, StringComparison.Ordinal); + Assert.Contains("EmptyAsync(", body, StringComparison.Ordinal); + Assert.DoesNotContain("McpHelpers.Status(\"empty\"", body, StringComparison.Ordinal); + + var description = DescriptionOf(source, tool); + Assert.Contains("source_observed", description, StringComparison.Ordinal); + Assert.Contains("last_captured_at", description, StringComparison.Ordinal); + } + } + + /// + /// The ladder itself: four rungs, three of them empty and exactly one unavailable — the + /// nothing-of-any-type-ever rung, with source_observed: false and the sentence the + /// EngineCapabilityMissTests pin (system_health session is started). Rung 3 — this type never, the + /// session alive — must be empty: a memory-node OOM that never happened is the healthy measurement. + /// + [Theory] + [InlineData(DarlingMcp + "/DarlingMcpHealthParserTools.cs")] + [InlineData(LiteMcp + "/McpHealthParserTools.cs")] + public void TheEmptyLadder_HasFourRungs_AndOnlyTheDeadSessionIsUnavailable(string file) + { + var source = ReadRepoFile(file.Split('/')); + var start = source.IndexOf("private static async Task EmptyAsync", StringComparison.Ordinal); + Assert.True(start > 0, $"{file} has no shared EmptyAsync ladder"); + var end = source.IndexOf("private static string WitnessStatus", start, StringComparison.Ordinal); + var ladder = Strip(source[start..end]); + + Assert.Equal(3, Regex.Matches(ladder, "WitnessStatus\\(\\s*\"empty\"").Count); + Assert.Single(Regex.Matches(ladder, "WitnessStatus\\(\\s*\"unavailable\"")); + Assert.Single(Regex.Matches(ladder, "sourceObserved: false")); + Assert.Contains("NOT an all-clear", ladder, StringComparison.Ordinal); + Assert.Contains("system_health session is started", ladder, StringComparison.Ordinal); + /* The engine-capability probe goes FIRST on the dead rung — the stronger claim. */ + Assert.True( + ladder.IndexOf("NotCollectedStatusAsync", StringComparison.Ordinal) < ladder.IndexOf("\"unavailable\"", StringComparison.Ordinal)); + /* The healthy rungs never reach for the dead rung's word, so a caller keying on it cannot be misled. */ + var rung3 = ladder[..ladder.IndexOf("NotCollectedStatusAsync", StringComparison.Ordinal)]; + Assert.DoesNotContain("EVER", rung3, StringComparison.Ordinal); + } + + /// + /// The witness reads the SAME view the tools read and is not the collection log — the log records a + /// SUCCESS for a run that read a dead session and stored nothing, which is the shape being fixed. + /// + [Fact] + public void TheWitness_IsTheEventsView_NotTheCollectionLog_OnBothSkus() + { + Assert.Contains("FROM v_system_health_events", DarlingSystemHealthReader.LastCaptureSql, StringComparison.Ordinal); + Assert.DoesNotContain("collection_log", DarlingSystemHealthReader.LastCaptureSql, StringComparison.Ordinal); + + var lite = ReadRepoFile("Lite", "Services", "LocalDataService.SystemEvents.cs"); + var probe = lite[lite.IndexOf("GetLastSystemHealthCaptureAsync", StringComparison.Ordinal)..]; + probe = probe[..probe.IndexOf("GetLastSystemHealthCaptureOfTypeAsync", StringComparison.Ordinal)]; + Assert.Contains("SELECT MAX(collection_time)", probe, StringComparison.Ordinal); + Assert.Contains("FROM v_system_health_events", probe, StringComparison.Ordinal); + Assert.DoesNotContain("collection_log", probe, StringComparison.Ordinal); + } + + /* ───────────────────────── 2. NULL percents stay NULL ───────────────────────── */ + + private static readonly Regex NullPercentCoerced = new(@"RegressionPercent = reader\.IsDBNull\(\d+\) \? 0 ", RegexOptions.Compiled); + + [Fact] + public void TheRegressionReaders_NeverCoerceANullPercentToZero_OnEitherSku() + { + /* Witness: the defect as it shipped on Lite. */ + Assert.Matches(NullPercentCoerced, "DurationRegressionPercent = reader.IsDBNull(4) ? 0 : ToDouble(reader.GetValue(4)),"); + + var lite = ReadRepoFile("Lite", "Services", "LocalDataService.QueryStoreRegressions.cs"); + Assert.DoesNotMatch(NullPercentCoerced, lite); + Assert.Equal(3, Regex.Matches(lite, @"RegressionPercent = reader\.IsDBNull\(\d+\) \? null ").Count); + Assert.Equal(3, Regex.Matches(lite, @"public double\? \w+RegressionPercent").Count); + + /* Darling's reader is positional; the three percent columns are 4, 7 and 10. */ + var darling = ReadRepoFile(DarlingMcp.Split('/').Append("DarlingQueryStoreRegressionReader.cs").ToArray()); + foreach (var column in new[] { 4, 7, 10 }) + { + Assert.Contains($"reader.IsDBNull({column}) ? null : Convert.ToDouble(reader.GetValue({column}))", darling, StringComparison.Ordinal); + } + + var percents = typeof(DarlingQueryStoreRegressionReader.RegressionRow).GetProperties() + .Where(p => p.Name.EndsWith("RegressionPercent", StringComparison.Ordinal)) + .ToArray(); + Assert.Equal(3, percents.Length); + Assert.All(percents, p => Assert.Equal(typeof(double?), p.PropertyType)); + } + + /// Both tools publish the reason beside the null, and a null duration ratio nulls the + /// severity banded from it rather than letting the TVF's ELSE 'LOW' stand. + [Theory] + [InlineData(DarlingMcp + "/DarlingMcpQueryStoreRegressionTools.cs")] + [InlineData(LiteMcp + "/McpQueryTools.cs")] + public void TheRegressionTool_SaysWhyAPercentIsNull_AndDoesNotBandAMissingRatio(string file) + { + var body = Strip(ToolBody(ReadRepoFile(file.Split('/')), "get_query_store_regressions")); + Assert.Contains("undefined_percents = UndefinedPercentNotes(r)", body, StringComparison.Ordinal); + Assert.Contains("severity = r.DurationRegressionPercent is null ? null : r.Severity", body, StringComparison.Ordinal); + + var description = DescriptionOf(ReadRepoFile(file.Split('/')), "get_query_store_regressions"); + Assert.Contains("undefined_percents", description, StringComparison.Ordinal); + Assert.Contains("null percent never sorts as 0", description, StringComparison.Ordinal); + } + + /* ───────────────────────── 3. differenced trends: the first point is unrated ───────────────────────── */ + + private static readonly Regex FabricatedFirstPoint = new(@"ELSE 0 END AS \w+_per_second", RegexOptions.Compiled); + + private static IEnumerable<(string Name, string Sql)> DifferencedTrendSql() + { + yield return (nameof(DarlingTrendReader.QueryDurationTrendSql), DarlingTrendReader.QueryDurationTrendSql); + yield return (nameof(DarlingTrendReader.ProcedureDurationTrendSql), DarlingTrendReader.ProcedureDurationTrendSql); + yield return (nameof(DarlingTrendReader.QueryStoreDurationTrendSql), DarlingTrendReader.QueryStoreDurationTrendSql); + yield return ("BuildRollupTrendSql(false)", QueryStoreTrendRouting.BuildRollupTrendSql(withDatabaseFilter: false)); + yield return ("BuildRollupTrendSql(true)", QueryStoreTrendRouting.BuildRollupTrendSql(withDatabaseFilter: true)); + } + + [Fact] + public void EveryDifferencedTrend_LeavesTheFirstPointUnrated_NeverZero() + { + /* Witness: the shipped shape. */ + Assert.Matches(FabricatedFirstPoint, "CASE WHEN interval_seconds > 0 THEN total_elapsed_ms / interval_seconds ELSE 0 END AS elapsed_ms_per_second,"); + + foreach (var (name, sql) in DifferencedTrendSql()) + { + Assert.Contains("LAG(", sql, StringComparison.Ordinal); + Assert.DoesNotMatch(FabricatedFirstPoint, sql); + /* Every rate column is a CASE with no ELSE — NULL where the denominator does not exist. */ + Assert.True( + Regex.Matches(sql, @"CASE WHEN interval_seconds > 0 THEN [^\n]*? END AS \w+_per_second").Count >= 2, + $"{name} no longer rates through a no-ELSE CASE"); + /* And the row is KEPT, not filtered: a lone collection is "no rate yet", not an empty window. */ + Assert.DoesNotContain("WHERE interval_seconds > 0", sql, StringComparison.Ordinal); + } + + /* Lite's four differenced trends, read from source: every `AS interval_seconds` statement rates + through a no-ELSE CASE and none carries the fabricated 0. */ + foreach (var file in new[] { "LocalDataService.QueryStats.cs", "LocalDataService.QueryStore.cs" }) + { + var lite = ReadRepoFile("Lite", "Services", file); + Assert.DoesNotMatch(FabricatedFirstPoint, lite); + var lagged = Regex.Matches(lite, @"\)\)\) AS interval_seconds").Count; + var rated = Regex.Matches(lite, @"CASE WHEN interval_seconds > 0 THEN [^\n]*? END AS \w+_per_second").Count; + Assert.True(lagged >= 1, $"{file}: the LAG idiom is gone, so this pin is looking at nothing"); + Assert.True(rated >= lagged, $"{file}: {lagged} differenced statement(s) but only {rated} no-ELSE rate column(s)"); + } + + /* The readers carry the null through instead of re-fabricating it. */ + var point = typeof(DarlingTrendReader.QueryDurationTrendPoint); + Assert.Equal(typeof(double?), point.GetProperty("Value")!.PropertyType); + Assert.Equal(typeof(long?), point.GetProperty("ExecutionCount")!.PropertyType); + Assert.Equal(typeof(double?), point.GetProperty("ExecutionsPerSecond")!.PropertyType); + var litePoint = ReadRepoFile("Lite", "Services", "LocalDataService.QueryStats.cs"); + Assert.Contains("public double? Value { get; set; }", litePoint, StringComparison.Ordinal); + Assert.Contains("public long? ExecutionCount { get; set; }", litePoint, StringComparison.Ordinal); + Assert.Contains("public double? ExecutionsPerSecond { get; set; }", litePoint, StringComparison.Ordinal); + } + + /// The three trend tools on both SKUs count and explain their unrated points with one sentence. + [Theory] + [InlineData(DarlingMcp + "/DarlingMcpTrendTools.cs")] + [InlineData(LiteMcp + "/McpQueryTools.cs")] + public void TheTrendTools_CountAndExplainUnratedPoints(string file) + { + var source = Strip(ReadRepoFile(file.Split('/'))); + Assert.Contains("envelope[\"unrated_points\"] = unrated;", source, StringComparison.Ordinal); + Assert.Contains("Unknowable is not 0", source, StringComparison.Ordinal); + foreach (var tool in new[] { "get_query_duration_trend", "get_procedure_duration_trend", "get_query_store_duration_trend" }) + { + Assert.Contains("unrated_points", DescriptionOf(ReadRepoFile(file.Split('/')), tool), StringComparison.Ordinal); + } + } + + /// + /// The Darling viewer shares the rollup builder, so its chart reader must tolerate the NULL first bucket — + /// by skipping it, since a chart has nowhere to draw "unknown" and coercing it to 0 would be the defect + /// plotted. (The viewer's OWN raw-route SQL copies still carry ELSE 0; that is the viewer lane's + /// A2 residual, out of this change's boundary, and not what this pin asserts.) + /// + [Fact] + public void TheViewerRollupReader_SkipsTheUnratedBucket_RatherThanPlottingZero() + { + var viewer = ReadRepoFile("Darling", "PerformanceMonitor.Darling.Viewer", "ViewerDataService.QueryTrends.cs"); + var start = viewer.IndexOf("QueryStoreDurationTrendRollupSql);", StringComparison.Ordinal); + var rollupReader = viewer[start..viewer.IndexOf("return items;", start, StringComparison.Ordinal)]; + Assert.Contains("if (reader.IsDBNull(1))", rollupReader, StringComparison.Ordinal); + Assert.Contains("continue;", rollupReader, StringComparison.Ordinal); + Assert.DoesNotContain("IsDBNull(1) ? 0", rollupReader, StringComparison.Ordinal); + + /* Lite's charts do the same with the nullable point. */ + var charts = ReadRepoFile("Lite", "Controls", "ServerTab.Charts.cs"); + Assert.Equal(4, Regex.Matches(charts, @"var rated = data\.Where\(d => d\.HasRate\)\.ToList\(\);").Count); + } + + /* ───────────────────────── 4. xmin: the window's denominator ───────────────────────── */ + + /// + /// The MCP share and the alert evaluator's horizon arm fraction over the SAME denominator: the + /// collector's own SUCCESS rows in collection_log, every time it looked, held or not. + /// + [Fact] + public void TheXminShare_DividesByEveryCapture_AndAgreesWithTheAlertEvaluator() + { + foreach (var sql in new[] { DarlingPgXminReader.XminCapturesInWindowSql, DarlingPostgresAlertReadAdapter.XminSql }) + { + Assert.Contains("FROM collection_log", sql, StringComparison.Ordinal); + Assert.Contains("collector_name = 'pg_xmin_horizon'", sql, StringComparison.Ordinal); + Assert.Contains("status = 'SUCCESS'", sql, StringComparison.Ordinal); + } + + /* And it is not the holder table: that table has no rows for an unheld capture. */ + var captures = DarlingPgXminReader.XminCapturesInWindowSql; + Assert.DoesNotContain("pg_xmin_horizon\n", captures, StringComparison.Ordinal); + Assert.Contains("COUNT(*) AS captures_in_window", captures, StringComparison.Ordinal); + + var body = Strip(ToolBody(ReadRepoFile(DarlingMcp.Split('/').Append("DarlingMcpPgXminTools.cs").ToArray()), "get_pg_xmin_horizon")); + Assert.Contains("/ capturesInWindow * 100", body, StringComparison.Ordinal); + Assert.DoesNotContain("/ r.Samples", body, StringComparison.Ordinal); + Assert.Contains("captures_in_window = capturesInWindow", body, StringComparison.Ordinal); + /* No captures and no holders is not an all-clear. */ + Assert.Contains("if (capturesInWindow == 0)", body, StringComparison.Ordinal); + Assert.Contains("status = \"unavailable\"", body, StringComparison.Ordinal); + /* Null, not 0, when there is nothing to divide by. */ + Assert.Contains(": (double?)null", body, StringComparison.Ordinal); + } + + /* ───────────────────────── 5. PVS: a measured zero is a measurement ───────────────────────── */ + + [Theory] + [InlineData(DarlingMcp + "/DarlingMcpPvsTools.cs")] + [InlineData(LiteMcp + "/McpPvsTools.cs")] + public void ThePvsShare_DividesAnyMeasuredSize_AndSaysWhyWhenItCannot(string file) + { + var body = Strip(ToolBody(ReadRepoFile(file.Split('/')), "get_pvs_stats")); + /* The defect: only a POSITIVE size divided, so 0 MB fell through to the same null as unmeasured. */ + Assert.DoesNotContain("PvsSizeMb is > 0 &&", body, StringComparison.Ordinal); + Assert.Contains("r.PvsSizeMb is { } pvsMb && r.DatabaseDataSizeMb is > 0", body, StringComparison.Ordinal); + Assert.Contains("pvs_measured = r.PvsSizeMb.HasValue", body, StringComparison.Ordinal); + Assert.Contains("pct_of_database_reason = PctReason(", body, StringComparison.Ordinal); + } + + [Fact] + public void ThePvsReason_IsNullOnlyWhenTheShareIsDefined_IncludingADefinedZero() + { + Assert.Null(DarlingMcpPvsTools.PctReason(pvsMeasured: true, databaseDataSizeMb: 1280)); + Assert.Contains("unknown — not zero", DarlingMcpPvsTools.PctReason(pvsMeasured: false, databaseDataSizeMb: 1280), StringComparison.Ordinal); + Assert.Contains("no denominator", DarlingMcpPvsTools.PctReason(pvsMeasured: true, databaseDataSizeMb: null), StringComparison.Ordinal); + Assert.Contains("no denominator", DarlingMcpPvsTools.PctReason(pvsMeasured: true, databaseDataSizeMb: 0), StringComparison.Ordinal); + } + + /* ───────────────────────── 6. growth: only over history the store holds ───────────────────────── */ + + private static readonly Regex FoldedBaseline = new(@"COALESCE\(p30\.reserved_mb, p7\.reserved_mb|COALESCE\(p7\.reserved_mb, o\.reserved_mb", RegexOptions.Compiled); + + [Fact] + public void TheGrowthRead_NeverFoldsAMissingBaselineOntoANearerOne_OnEitherSku() + { + /* Witness: the shipped chain. */ + Assert.Matches(FoldedBaseline, "l.current_reserved_mb - COALESCE(p30.reserved_mb, p7.reserved_mb, o.reserved_mb, l.current_reserved_mb) AS growth_30d_mb,"); + + var darling = DarlingObjectStatsReader.ObjectSizeGrowthSql; + Assert.DoesNotMatch(FoldedBaseline, darling); + Assert.Contains("MAX(collection_time) FILTER (WHERE collection_time <= $2) AS snapshot_7d_time", darling, StringComparison.Ordinal); + Assert.Contains("MAX(collection_time) FILTER (WHERE collection_time <= $3) AS snapshot_30d_time", darling, StringComparison.Ordinal); + foreach (var column in new[] { "reserved_mb_7d_ago", "reserved_mb_30d_ago", "reserved_mb_oldest", "b.snapshot_7d_time", "b.snapshot_30d_time", "b.earliest_time", "b.latest_time", "b.days_of_data" }) + { + Assert.Contains(column, darling, StringComparison.Ordinal); + } + Assert.DoesNotContain("growth_7d_mb", darling, StringComparison.Ordinal); + + var lite = ReadRepoFile("Lite", "Services", "LocalDataService.FinOps.IndexObjects.cs"); + var read = lite[lite.IndexOf("GetObjectSizeGrowthAsync(int serverId", StringComparison.Ordinal)..]; + read = read[..read.IndexOf("return items;", StringComparison.Ordinal)]; + Assert.DoesNotMatch(FoldedBaseline, read); + Assert.Contains("MAX(collection_time) FILTER (WHERE collection_time <= $2) AS snapshot_7d_time", read, StringComparison.Ordinal); + Assert.Contains("ReservedMb30dAgo = reader.IsDBNull(8) ? null", read, StringComparison.Ordinal); + + foreach (var file in new[] { DarlingMcp + "/DarlingMcpObjectStatsTools.cs", LiteMcp + "/McpObjectStatsTools.cs" }) + { + var body = Strip(ToolBody(ReadRepoFile(file.Split('/')), "get_table_index_sizes")); + foreach (var key in new[] { "history_days_available", "covers_7d", "covers_30d", "growth_over_available_history_mb", "growth_over_available_history_pct", "growth_window_days", "growth_note = GrowthNote(r)", "tables_returned = page.Count", "truncated," }) + { + Assert.Contains(key, body, StringComparison.Ordinal); + } + /* Truncation observed by over-fetch, never inferred from a full page (A3's rule). */ + Assert.Contains("TableSizesTop + 1", body, StringComparison.Ordinal); + Assert.Contains("var truncated = rows.Count > TableSizesTop;", body, StringComparison.Ordinal); + } + } + + /// The derivations, executed: each figure comes from exactly the baseline it names or is null. + [Fact] + public void TheGrowthRow_RefusesEveryFigureItsBaselineCannotSupport() + { + var earliest = new DateTime(2026, 3, 1, 3, 0, 0); + var latest = new DateTime(2026, 3, 4, 3, 0, 0); + + /* Three days of history: no 7-day or 30-day snapshot; the oldest holds the table at 100 MB. */ + var threeDays = new DarlingObjectStatsReader.ObjectSizeGrowthRow( + "db", "dbo", "Posts", CurrentReservedMb: 160, CurrentUsedMb: 150, TotalRows: 10, IndexCount: 2, + ReservedMb7dAgo: null, ReservedMb30dAgo: null, ReservedMbOldest: 100, + Snapshot7dTime: null, Snapshot30dTime: null, EarliestSnapshotTime: earliest, LatestSnapshotTime: latest, DaysOfData: 3); + Assert.Null(threeDays.Growth7dMb); + Assert.Null(threeDays.Growth30dMb); + Assert.Null(threeDays.GrowthPct30d); + Assert.Equal(60, threeDays.GrowthOverAvailableHistoryMb); + Assert.Equal(60.0, threeDays.GrowthOverAvailableHistoryPct!.Value, 9); + Assert.Equal(20.0, threeDays.DailyGrowthRateMb!.Value, 9); + var note = DarlingMcpObjectStatsTools.GrowthNote(threeDays)!; + Assert.Contains("no snapshot 7+ days old exists", note, StringComparison.Ordinal); + Assert.Contains("no snapshot 30+ days old exists", note, StringComparison.Ordinal); + + /* The shipped chain would have said growth_30d = 60 (labelled 30d, measured over 3). Now the 30-day + figure is null and the 3-day figure carries its own name and span. */ + + /* A table created since the earliest snapshot: nothing but growth, and the shipped chain said 0. */ + var newTable = threeDays with { ReservedMbOldest = null }; + Assert.Null(newTable.GrowthOverAvailableHistoryMb); + Assert.Null(newTable.DailyGrowthRateMb); + Assert.Contains("not in the earliest snapshot", DarlingMcpObjectStatsTools.GrowthNote(newTable)!, StringComparison.Ordinal); + + /* A single day of snapshots: no span, so no growth is knowable — every figure null. */ + var oneDay = threeDays with { DaysOfData = 0, EarliestSnapshotTime = latest }; + Assert.Null(oneDay.GrowthOverAvailableHistoryMb); + Assert.Null(oneDay.DailyGrowthRateMb); + Assert.Contains("no growth is knowable yet", DarlingMcpObjectStatsTools.GrowthNote(oneDay)!, StringComparison.Ordinal); + + /* Full history, every baseline present: every figure defined and NO note. */ + var full = threeDays with + { + ReservedMb7dAgo = 140, ReservedMb30dAgo = 80, ReservedMbOldest = 50, + Snapshot7dTime = latest.AddDays(-7), Snapshot30dTime = latest.AddDays(-30), EarliestSnapshotTime = latest.AddDays(-40), DaysOfData = 40, + }; + Assert.Equal(20, full.Growth7dMb); + Assert.Equal(80, full.Growth30dMb); + Assert.Equal(100.0, full.GrowthPct30d!.Value, 9); + Assert.Equal(110, full.GrowthOverAvailableHistoryMb); + Assert.Null(DarlingMcpObjectStatsTools.GrowthNote(full)); + + /* A 0 baseline has no ratio — the absolute stands, the percent is null with its reason. */ + var wasEmpty = full with { ReservedMb30dAgo = 0 }; + Assert.Equal(160, wasEmpty.Growth30dMb); + Assert.Null(wasEmpty.GrowthPct30d); + Assert.Contains("no denominator", DarlingMcpObjectStatsTools.GrowthNote(wasEmpty)!, StringComparison.Ordinal); + } + + /* ───────────────────────── helpers ───────────────────────── */ + + private static string ToolBody(string source, string toolName) + { + var marker = $"[McpServerTool(Name = \"{toolName}\")"; + var start = source.IndexOf(marker, StringComparison.Ordinal); + Assert.True(start >= 0, $"no tool named {toolName} in the source"); + var next = source.IndexOf("[McpServerTool(", start + marker.Length, StringComparison.Ordinal); + return next < 0 ? source[start..] : source[start..next]; + } + + /// The Description literal of one tool, whichever side of a line break it sits on (Lite's + /// get_pvs_stats opens its string on the next line). + private static string DescriptionOf(string source, string toolName) + { + var body = ToolBody(source, toolName); + var match = Regex.Match(body, @"Description\(\s*""((?:[^""\\]|\\.)*)""", RegexOptions.Singleline); + Assert.True(match.Success, $"{toolName} has no Description literal"); + return match.Groups[1].Value; + } + + private static string Strip(string source) => + Regex.Replace(Regex.Replace(source, @"/\*.*?\*/", string.Empty, RegexOptions.Singleline), @"//[^\n]*", string.Empty); +} + +/// +/// Gated (DARLING_TEST_PG) live round-trips for #3541 A12, through the real tool methods: the Query Store +/// regression whose baseline read nothing, the xmin holder that won 2 of 6 captures, the PVS row measured at +/// 0 MB beside one not measured at all, and the growth read over three days of history. +/// +[Collection("live-postgres")] +public sealed class McpZeroIsAMeasurementLivePostgresTests +{ + private const string ServerName = "zero-is-a-measurement-e2e"; + private static readonly int ServerId = ServerIdHelper.GetDeterministicHashCode(ServerName); + private const string Db = "ZeroDb"; + private static string? ConnectionString => Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + + [Fact] + public async Task ARegressionWithNoBaselineReads_StaysNull_AndSaysWhy() + { + var cs = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(cs), "Set DARLING_TEST_PG to a Postgres connection string to run the live #3541 A12 regression test."); + + var ct = TestContext.Current.CancellationToken; + using var connection = new NpgsqlConnection(cs); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + await DeleteRowsAsync(connection, ct); + await using var postgres = NpgsqlDataSource.Create(cs!); + + var bodySucceeded = false; + try + { + await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); + + /* Query 1: baseline read NOTHING (0 logical reads), recent reads 50,000 — the largest possible + I/O regression, and the row the shipped reader published as io_regression_percent 0. CPU + regressed 100% so it passes the gate. Query 2: every side defined. */ + await SeedQueryStoreAsync(connection, ct, HoursAgo(30), queryId: 1, intervalId: 1, executions: 10, avgDurationUs: 1_000, avgCpuUs: 500, avgReads: 0); + await SeedQueryStoreAsync(connection, ct, MinutesAgo(30), queryId: 1, intervalId: 2, executions: 10, avgDurationUs: 2_000, avgCpuUs: 1_000, avgReads: 50_000); + await SeedQueryStoreAsync(connection, ct, HoursAgo(30), queryId: 2, intervalId: 3, executions: 10, avgDurationUs: 1_000, avgCpuUs: 500, avgReads: 100); + await SeedQueryStoreAsync(connection, ct, MinutesAgo(30), queryId: 2, intervalId: 4, executions: 10, avgDurationUs: 3_000, avgCpuUs: 1_000, avgReads: 200); + + var root = JsonDocument.Parse(await DarlingMcpQueryStoreRegressionTools.GetQueryStoreRegressions(postgres, ServerName, hours_back: 24)).RootElement; + Assert.Equal(2, root.GetProperty("regression_count").GetInt32()); + + var rows = root.GetProperty("regressions").EnumerateArray().ToDictionary(r => r.GetProperty("query_id").GetInt64()); + + var noBaselineReads = rows[1]; + Assert.Equal(JsonValueKind.Null, noBaselineReads.GetProperty("io_regression_percent").ValueKind); + Assert.Equal(0, noBaselineReads.GetProperty("baseline_reads").GetDouble()); + Assert.Equal(50_000, noBaselineReads.GetProperty("recent_reads").GetDouble()); + var notes = noBaselineReads.GetProperty("undefined_percents").EnumerateArray().Select(n => n.GetString()!).ToArray(); + Assert.Single(notes); + Assert.Contains("io_regression_percent is null: no_baseline", notes[0], StringComparison.Ordinal); + Assert.Contains("NOT 0% change", notes[0], StringComparison.Ordinal); + /* The other two ratios exist and are unaffected; severity is banded from the duration ratio. */ + Assert.Equal(100.0, noBaselineReads.GetProperty("duration_regression_percent").GetDouble(), 6); + Assert.Equal(100.0, noBaselineReads.GetProperty("cpu_regression_percent").GetDouble(), 6); + Assert.Equal("HIGH", noBaselineReads.GetProperty("severity").GetString()); /* 100% duration: > 50, not > 100 */ + + var defined = rows[2]; + Assert.Equal(100.0, defined.GetProperty("io_regression_percent").GetDouble(), 6); + Assert.Equal(JsonValueKind.Null, defined.GetProperty("undefined_percents").ValueKind); + + /* The ranking key is the absolute delta, defined for both — query 2's 20 ms × 10 outranks + query 1's 10 ms × 10, whatever their ratios. */ + var order = root.GetProperty("regressions").EnumerateArray().Select(r => r.GetProperty("query_id").GetInt64()).ToArray(); + Assert.Equal(new long[] { 2, 1 }, order); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, async (cleanup, cleanupCt) => await DeleteRowsAsync(cleanup, cleanupCt)); + } + } + + [Fact] + public async Task TheXminShare_IsOverEveryCapture_AndAgreesWithTheAlertAdapter() + { + var cs = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(cs), "Set DARLING_TEST_PG to a Postgres connection string to run the live #3541 A12 xmin test."); + + var ct = TestContext.Current.CancellationToken; + using var connection = new NpgsqlConnection(cs); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + await DeleteRowsAsync(connection, ct); + await using var postgres = NpgsqlDataSource.Create(cs!); + + var bodySucceeded = false; + try + { + await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); + await DarlingMcpTestData.ExecAsync(connection, ct, "UPDATE servers SET engine_kind = $2 WHERE server_id = $1", ServerId, MonitoredEngineKind.AuroraPostgres); + + /* No holders and no captures: the collector did not look, so this is NOT "nothing holds the + horizon". */ + var blind = JsonDocument.Parse(await DarlingMcpPgXminTools.GetPgXminHorizon(postgres, ServerName, hours_back: 4)).RootElement; + Assert.Equal("unavailable", blind.GetProperty("status").GetString()); + Assert.Equal(0, blind.GetProperty("captures_in_window").GetInt32()); + + /* Six successful captures, one failed one (does not count), two of which recorded pid 104 as the + winner. The shipped payload divided 2 by the source's OWN 2 rows: 100%, "chronic". */ + for (var i = 1; i <= 6; i++) + await SeedLogAsync(connection, ct, MinutesAgo(i * 10), "SUCCESS"); + await SeedLogAsync(connection, ct, MinutesAgo(70), "ERROR"); + + /* Captures only, no holder: the healthy answer, and it names the denominator it rests on. */ + var clear = JsonDocument.Parse(await DarlingMcpPgXminTools.GetPgXminHorizon(postgres, ServerName, hours_back: 4)).RootElement; + Assert.Equal("no_holder", clear.GetProperty("status").GetString()); + Assert.Equal(6, clear.GetProperty("captures_in_window").GetInt32()); + Assert.Contains("captured 6 time(s)", clear.GetProperty("finding").GetString()!, StringComparison.Ordinal); + + await SeedHolderAsync(connection, ct, MinutesAgo(20), "session", 80_000_000, "104", "state=idle in transaction", isWinner: true); + await SeedHolderAsync(connection, ct, MinutesAgo(10), "session", 81_000_000, "104", "state=idle in transaction", isWinner: true); + + var held = JsonDocument.Parse(await DarlingMcpPgXminTools.GetPgXminHorizon(postgres, ServerName, hours_back: 4)).RootElement; + Assert.Equal("holder_present", held.GetProperty("status").GetString()); + Assert.Equal(6, held.GetProperty("captures_in_window").GetInt32()); + var session = held.GetProperty("holders").EnumerateArray().Single(h => h.GetProperty("source").GetString() == "session"); + Assert.Equal(2, session.GetProperty("samples_as_winner").GetInt32()); + Assert.Equal(2, session.GetProperty("captures_recording_this_source").GetInt32()); + Assert.Equal(33.3, session.GetProperty("pct_of_window_winning").GetDouble(), 1); + + /* The alert evaluator's horizon arm counts the same six captures — one denominator, two surfaces. */ + var adapter = new DarlingPostgresAlertReadAdapter(postgres); + var info = await adapter.GetXminHorizonAsync(ServerId, ct); + Assert.NotNull(info); + Assert.Equal(6, info!.CapturesInWindow); + Assert.Equal(6L, await DarlingPgXminReader.GetXminCapturesInWindowAsync(postgres, ServerId, HoursAgo(4), DateTime.UtcNow, ct)); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, async (cleanup, cleanupCt) => await DeleteRowsAsync(cleanup, cleanupCt)); + } + } + + [Fact] + public async Task AMeasuredZeroPvs_IsAMeasurement_AndAnUnmeasuredOneSaysSo() + { + var cs = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(cs), "Set DARLING_TEST_PG to a Postgres connection string to run the live #3541 A12 PVS test."); + + var ct = TestContext.Current.CancellationToken; + using var connection = new NpgsqlConnection(cs); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + await DeleteRowsAsync(connection, ct); + await using var postgres = NpgsqlDataSource.Create(cs!); + + var bodySucceeded = false; + try + { + await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); + var t = MinutesAgo(5); + await SeedPvsAsync(connection, ct, t, "CleanDb", pvsSizeMb: 0m, dataSizeMb: 1280m); + await SeedPvsAsync(connection, ct, t, "UnreadDb", pvsSizeMb: null, dataSizeMb: 1280m); + await SeedPvsAsync(connection, ct, t, "BusyDb", pvsSizeMb: 912.82m, dataSizeMb: 1280m); + + var root = JsonDocument.Parse(await DarlingMcpPvsTools.GetPvsStats(postgres, ServerName)).RootElement; + var byDb = root.GetProperty("databases").EnumerateArray().ToDictionary(d => d.GetProperty("database_name").GetString()!); + + /* 0 MB of 1,280 MB is 0.00% — measured, and said so. Before: pct_of_database null, same as unread. */ + var clean = byDb["CleanDb"]; + Assert.True(clean.GetProperty("pvs_measured").GetBoolean()); + Assert.Equal(0, clean.GetProperty("pvs_size_mb").GetDouble()); + Assert.Equal(0.0, clean.GetProperty("pct_of_database").GetDouble()); + Assert.Equal(JsonValueKind.Null, clean.GetProperty("pct_of_database_reason").ValueKind); + + var unread = byDb["UnreadDb"]; + Assert.False(unread.GetProperty("pvs_measured").GetBoolean()); + Assert.Equal(JsonValueKind.Null, unread.GetProperty("pvs_size_mb").ValueKind); + Assert.Equal(JsonValueKind.Null, unread.GetProperty("pct_of_database").ValueKind); + Assert.Contains("not zero", unread.GetProperty("pct_of_database_reason").GetString()!, StringComparison.Ordinal); + + var busy = byDb["BusyDb"]; + Assert.True(busy.GetProperty("pvs_measured").GetBoolean()); + Assert.Equal(71.31, busy.GetProperty("pct_of_database").GetDouble(), 2); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, async (cleanup, cleanupCt) => await DeleteRowsAsync(cleanup, cleanupCt)); + } + } + + [Fact] + public async Task GrowthOverThreeDaysOfHistory_IsLabelledAsThreeDays_NotThirty() + { + var cs = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(cs), "Set DARLING_TEST_PG to a Postgres connection string to run the live #3541 A12 growth test."); + + var ct = TestContext.Current.CancellationToken; + using var connection = new NpgsqlConnection(cs); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + await DeleteRowsAsync(connection, ct); + await using var postgres = NpgsqlDataSource.Create(cs!); + + var bodySucceeded = false; + try + { + await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); + var now = MinutesAgo(5); + var threeDaysAgo = now.AddDays(-3); + + /* Posts: 100 MB three days ago, 160 MB now. Comments: created since — only in the latest snapshot. */ + await SeedIndexAsync(connection, ct, threeDaysAgo, "Posts", reservedMb: 100m); + await SeedIndexAsync(connection, ct, now, "Posts", reservedMb: 160m); + await SeedIndexAsync(connection, ct, now, "Comments", reservedMb: 40m); + + var root = JsonDocument.Parse(await DarlingMcpObjectStatsTools.GetTableIndexSizes(postgres, ServerName)).RootElement; + + var history = root.GetProperty("history"); + Assert.Equal(3, history.GetProperty("history_days_available").GetInt32()); + Assert.False(history.GetProperty("covers_7d").GetBoolean()); + Assert.False(history.GetProperty("covers_30d").GetBoolean()); + Assert.Contains("3 day(s) of index snapshots", history.GetProperty("note").GetString()!, StringComparison.Ordinal); + Assert.Equal(2, root.GetProperty("tables_returned").GetInt32()); + Assert.False(root.GetProperty("truncated").GetBoolean()); + + var byTable = root.GetProperty("tables").EnumerateArray().ToDictionary(t => t.GetProperty("table_name").GetString()!); + + /* The shipped read said growth_30d_mb = 60 for Posts — measured over three days, labelled thirty. */ + var posts = byTable["Posts"]; + Assert.Equal(JsonValueKind.Null, posts.GetProperty("growth_7d_mb").ValueKind); + Assert.Equal(JsonValueKind.Null, posts.GetProperty("growth_30d_mb").ValueKind); + Assert.Equal(JsonValueKind.Null, posts.GetProperty("growth_pct_30d").ValueKind); + Assert.Equal(60, posts.GetProperty("growth_over_available_history_mb").GetDouble()); + Assert.Equal(60.0, posts.GetProperty("growth_over_available_history_pct").GetDouble(), 6); + Assert.Equal(3, posts.GetProperty("growth_window_days").GetInt32()); + Assert.Equal(20.0, posts.GetProperty("daily_growth_rate_mb").GetDouble(), 6); + Assert.Contains("no snapshot 30+ days old exists", posts.GetProperty("growth_note").GetString()!, StringComparison.Ordinal); + + /* And the shipped read said Comments grew 0 MB — for a table that is nothing but growth. */ + var comments = byTable["Comments"]; + Assert.Equal(JsonValueKind.Null, comments.GetProperty("growth_over_available_history_mb").ValueKind); + Assert.Equal(JsonValueKind.Null, comments.GetProperty("daily_growth_rate_mb").ValueKind); + Assert.Contains("not in the earliest snapshot", comments.GetProperty("growth_note").GetString()!, StringComparison.Ordinal); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, async (cleanup, cleanupCt) => await DeleteRowsAsync(cleanup, cleanupCt)); + } + } + + /* ───────────────────────── seeds ───────────────────────── */ + + private static DateTime MinutesAgo(int minutes) => DarlingMcpTestData.TruncateToSeconds(DateTime.UtcNow.AddMinutes(-minutes)); + private static DateTime HoursAgo(int hours) => DarlingMcpTestData.TruncateToSeconds(DateTime.UtcNow.AddHours(-hours)); + + private static Task SeedQueryStoreAsync( + NpgsqlConnection connection, CancellationToken ct, DateTime collectionTime, long queryId, long intervalId, + long executions, long avgDurationUs, long avgCpuUs, long avgReads) => + DarlingMcpTestData.ExecAsync(connection, ct, @" +INSERT INTO query_store_stats + (collection_id, collection_time, server_id, server_name, database_name, query_id, plan_id, + execution_type_desc, execution_count, avg_duration_us, avg_cpu_time_us, avg_logical_io_reads, + runtime_stats_interval_id, query_text, last_execution_time) +VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11, $12, $13, $14, $15)", + CollectionIdGenerator.Next(), collectionTime, ServerId, ServerName, Db, queryId, 9L, "Regular", + executions, avgDurationUs, avgCpuUs, avgReads, intervalId, $"SELECT {queryId}", collectionTime); + + private static Task SeedHolderAsync( + NpgsqlConnection connection, CancellationToken ct, DateTime collectionTime, string source, long xminAge, string holder, string? detail, bool isWinner) => + DarlingMcpTestData.ExecAsync(connection, ct, @" +INSERT INTO pg_xmin_horizon + (collection_id, collection_time, server_id, server_name, source, xmin_age, holder, detail, is_winner) +VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9)", + CollectionIdGenerator.Next(), collectionTime, ServerId, ServerName, source, xminAge, holder, detail, isWinner); + + private static Task SeedLogAsync(NpgsqlConnection connection, CancellationToken ct, DateTime collectionTime, string status) => + DarlingMcpTestData.ExecAsync(connection, ct, @" +INSERT INTO collection_log + (log_id, server_id, server_name, collector_name, collection_time, duration_ms, status, error_message, rows_collected, sql_duration_ms, duckdb_duration_ms) +VALUES ($1, $2, $3, 'pg_xmin_horizon', $4, 50, $5, NULL, 0, 40, 10)", + CollectionIdGenerator.Next(), ServerId, ServerName, collectionTime, status); + + private static Task SeedPvsAsync(NpgsqlConnection connection, CancellationToken ct, DateTime collectionTime, string database, decimal? pvsSizeMb, decimal dataSizeMb) => + DarlingMcpTestData.ExecAsync(connection, ct, @" +INSERT INTO collect.pvs_stats + (collection_id, collection_time, server_id, server_name, database_name, database_id, is_accelerated_database_recovery_on, + persistent_version_store_size_mb, online_index_version_store_size_mb, database_data_size_mb, current_aborted_transaction_count) +VALUES ($1, $2, $3, $4, $5, $6, TRUE, $7, 0, $8, 0)", + CollectionIdGenerator.Next(), collectionTime, ServerId, ServerName, database, 7, pvsSizeMb, dataSizeMb); + + private static Task SeedIndexAsync(NpgsqlConnection connection, CancellationToken ct, DateTime collectionTime, string table, decimal reservedMb) => + DarlingMcpTestData.ExecAsync(connection, ct, @" +INSERT INTO index_object_stats + (collection_id, collection_time, server_id, server_name, database_name, schema_name, object_id, table_name, index_id, index_name, index_type_desc, reserved_mb, used_mb, total_rows) +VALUES ($1, $2, $3, $4, $5, 'dbo', $6, $7, 1, $8, 'CLUSTERED', $9, $9, 1000)", + CollectionIdGenerator.Next(), collectionTime, ServerId, ServerName, Db, table.GetHashCode(StringComparison.Ordinal), table, "PK_" + table, reservedMb); + + private static async Task DeleteRowsAsync(NpgsqlConnection connection, CancellationToken ct) + { + using var cleanup = new NpgsqlCommand( + $"DELETE FROM query_store_stats WHERE server_id = {ServerId}; DELETE FROM pg_xmin_horizon WHERE server_id = {ServerId}; " + + $"DELETE FROM collection_log WHERE server_id = {ServerId}; DELETE FROM collect.pvs_stats WHERE server_id = {ServerId}; " + + $"DELETE FROM index_object_stats WHERE server_id = {ServerId}; DELETE FROM servers WHERE server_id = {ServerId};", connection); + await cleanup.ExecuteNonQueryAsync(ct); + } +} diff --git a/Darling/Darling.Tests/QueryStoreTrendRoutingLiveTests.cs b/Darling/Darling.Tests/QueryStoreTrendRoutingLiveTests.cs index 25f8df1e1..d1963d331 100644 --- a/Darling/Darling.Tests/QueryStoreTrendRoutingLiveTests.cs +++ b/Darling/Darling.Tests/QueryStoreTrendRoutingLiveTests.cs @@ -130,18 +130,20 @@ await SeedSnapshotsAsync(connection, intervalId: 3104, queryId: 64, intervalStar /* 10:00 — rollup bucket: interval P only (21, once). Interval M ran at 10:00 but was FETCHED at 11:00, so the rollup charges it to 11:00 — the collection-hour placement the payload discloses. */ Assert.Equal(hour10, points[0].CollectionTime); + /* The first united point has no predecessor to difference against: null, not 0 (#3541 A12). */ + Assert.False(points[0].HasRate); /* 11:00 — rollup bucket: M's final snapshot (40) + N's (25) = 65 executions over the 3,600 seconds since the previous point. Un-deduped this hour would be 10+40+5+25 = 80 — the rank the rollup already did. */ Assert.Equal(hour11, points[1].CollectionTime); - Assert.Equal(65d / 3600d, points[1].ExecutionsPerSecond, 6); - Assert.Equal(((40d * 100d + 25d * 200d) / 1000d) / 3600d, points[1].Value, 6); + Assert.Equal(65d / 3600d, points[1].ExecutionsPerSecond!.Value, 6); + Assert.Equal(((40d * 100d + 25d * 200d) / 1000d) / 3600d, points[1].Value!.Value, 6); /* 12:00 — the raw tail: interval T deduped to its final snapshot (9, not 3+9), placed at its interval start, rated over the seam to the last rollup bucket. */ Assert.Equal(hour12, points[2].CollectionTime); - Assert.Equal(9d / 3600d, points[2].ExecutionsPerSecond, 6); + Assert.Equal(9d / 3600d, points[2].ExecutionsPerSecond!.Value, 6); /* ── the raw-only route on the SAME fixture: the estimator this replaced. Interval M lands at its interval START (10:00 — so that hour reads 21+40=61) and the 11:00 point carries only N. This @@ -152,9 +154,11 @@ interval START (10:00 — so that hour reads 21+40=61) and the 11:00 point carri Assert.Equal(3, rawPoints.Count); Assert.Equal(hour10, rawPoints[0].CollectionTime); + /* The first united point has no predecessor to difference against: null, not 0 (#3541 A12). */ + Assert.False(rawPoints[0].HasRate); Assert.Equal(hour11, rawPoints[1].CollectionTime); - Assert.Equal(25d / 3600d, rawPoints[1].ExecutionsPerSecond, 6); - Assert.Equal(9d / 3600d, rawPoints[2].ExecutionsPerSecond, 6); + Assert.Equal(25d / 3600d, rawPoints[1].ExecutionsPerSecond!.Value, 6); + Assert.Equal(9d / 3600d, rawPoints[2].ExecutionsPerSecond!.Value, 6); /* ── the MCP payload discloses the routing: which relation served which region, and that the window's head reaches below the rollup's floor ── */ @@ -223,9 +227,13 @@ names the mechanism and the remedy instead of advising a wider window — and ca Assert.Equal("rollup+raw", tailOnly.GetProperty("source").GetString()); Assert.StartsWith("2026-03-04T12:00:00", tailOnly.GetProperty("effective_start").GetString()!, StringComparison.Ordinal); Assert.False(tailOnly.GetProperty("truncated").GetBoolean()); - Assert.Equal( - tailOnly.GetProperty("trend")[0].GetProperty("value").GetDouble(), - tailOnly.GetProperty("trend")[0].GetProperty("elapsed_ms_per_second").GetDouble()); + /* #3541 A12: a lone point has nothing to difference against, so it is UNRATED — `value` and its named + twin are both null (this assertion used to compare two fabricated zeros), the point is still + there (so effective_start above is truthful), and the envelope says why. */ + Assert.Equal(JsonValueKind.Null, tailOnly.GetProperty("trend")[0].GetProperty("value").ValueKind); + Assert.Equal(JsonValueKind.Null, tailOnly.GetProperty("trend")[0].GetProperty("elapsed_ms_per_second").ValueKind); + Assert.Equal(1, tailOnly.GetProperty("unrated_points").GetInt32()); + Assert.Contains("Unknowable is not 0", tailOnly.GetProperty("unrated_note").GetString()!, StringComparison.Ordinal); } /// diff --git a/Darling/Darling.Tests/TsqlConventionGuardTests.cs b/Darling/Darling.Tests/TsqlConventionGuardTests.cs index 64229a016..55d88ff6c 100644 --- a/Darling/Darling.Tests/TsqlConventionGuardTests.cs +++ b/Darling/Darling.Tests/TsqlConventionGuardTests.cs @@ -1444,6 +1444,14 @@ where the bound has nothing to compare against. */ Neither is T-SQL and neither is a tempdb label, so no census reads a site of that kind here. */ "Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpStoreMetricsTools.cs Stamp", "Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpStoreMetricsTools.cs Window", + /* #3541 A12: four expression-bodied derivations on the growth row, each `Baseline is { } b ? … : null` + — the property pattern's braces are where the walk stops. Every one is arithmetic over the row's + own fields and strands NO string literal at all, so no census reads a site of that kind here. The + Lite twin (LocalDataService.FinOps.IndexObjects.cs, below) derives the same four the same way. */ + "Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingObjectStatsReader.cs DailyGrowthRateMb", + "Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingObjectStatsReader.cs Growth30dMb", + "Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingObjectStatsReader.cs Growth7dMb", + "Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingObjectStatsReader.cs GrowthOverAvailableHistoryMb", "Darling/PerformanceMonitor.Darling.Service/Targets/PostgresTargetProvider.cs WithDatabase", "Darling/PerformanceMonitor.Darling.Service/Targets/SqlServerTargetProvider.cs WithDatabase", "Darling/PerformanceMonitor.Darling.Viewer/MainWindow.ServerManagement.cs SelectedTabCollectorScope", @@ -1460,6 +1468,12 @@ where the bound has nothing to compare against. */ "Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.SystemEvents.cs Local", "Darling/PerformanceMonitor.Darling.Viewer/ViewerPostgresDisplay.cs Timestamp", "Lite/Services/LocalDataService.CollectionHealth.cs OutputFinding", + /* #3541 A12: the Lite twin of the four DarlingObjectStatsReader growth derivations above — the same + `is { } b ? … : null` shape, the same absence of any string literal. */ + "Lite/Services/LocalDataService.FinOps.IndexObjects.cs DailyGrowthRateMb", + "Lite/Services/LocalDataService.FinOps.IndexObjects.cs Growth30dMb", + "Lite/Services/LocalDataService.FinOps.IndexObjects.cs Growth7dMb", + "Lite/Services/LocalDataService.FinOps.IndexObjects.cs GrowthOverAvailableHistoryMb", ]; /* ───────────────────────── the resolver, pinned on arranged source ───────────────────────── */ diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpHealthParserTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpHealthParserTools.cs index 7c288e3cb..49284156f 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpHealthParserTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpHealthParserTools.cs @@ -42,6 +42,15 @@ namespace PerformanceMonitor.Darling.Service.Mcp; /// parsed-table architecture) have no analog in the parse-on-read record and are omitted. Severe-error /// database_name is resolved from the collected size-stats mapping (the DB-free shred left it null). /// +/// +/// +/// Every one of the nine publishes its SOURCE WITNESS (#3541 A12): source_observed — whether the +/// collector has ever stored a system_health event of any type for this server, i.e. whether the ring buffer +/// has ever been read into the store — and last_captured_at, the collector's newest capture. A zero-row +/// window is then one of four nothings () and says which; a server whose session has +/// never been read answers unavailable, never empty. Before this, eight of the nine answered a +/// dead session with the same word a healthy quiet hour earns. +/// /// [McpServerToolType] public sealed class DarlingMcpHealthParserTools @@ -54,7 +63,7 @@ public sealed class DarlingMcpHealthParserTools /// private const string SystemHealthCollectorName = "system_health_events"; - [McpServerTool(Name = "get_health_parser_system_health"), Description("Gets parsed system_health extended event data: overall health indicators captured by sp_HealthParser.")] + [McpServerTool(Name = "get_health_parser_system_health"), Description("Gets parsed system_health extended event data: overall health indicators captured by sp_HealthParser. Every answer carries source_observed (whether this server's system_health session has EVER been read into the store) and last_captured_at (the collector's newest capture): an empty window on a server whose session was never read is status unavailable, not a clean bill; an empty window on one that has been read says whether the category was captured and gated out, captured before this window, or never recorded by the engine.")] public static async Task GetSystemHealth( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -72,13 +81,15 @@ GetSystemHealthAsync keeps every SYSTEM snapshot that has a timestamp. */ r => r.EventTime.HasValue); if (c.EarlyReturn != null) return c.EarlyReturn; if (c.Rows.Count == 0) - return await DarlingEngineCapability.NotCollectedStatusAsync(postgres, c.ServerId, c.ServerName, SystemHealthCollectorName) - ?? McpHelpers.Status("empty", "No system health data found in the requested time range."); + return await EmptyAsync(postgres, c, hours_back, SystemHealthParser.SpServerDiagnosticsEvent, + "none carried a SYSTEM component result with a timestamp (the other four sp_server_diagnostics components feed the sibling reads)"); return JsonSerializer.Serialize(new { server = c.ServerName, hours_back, + source_observed = true, + last_captured_at = Stamp(c.LastCapturedAt), total_entries = c.Rows.Count, shown = Math.Min(c.Rows.Count, limit), entries = c.Rows.Take(limit).Select(r => new @@ -105,7 +116,7 @@ GetSystemHealthAsync keeps every SYSTEM snapshot that has a timestamp. */ catch (Exception ex) { return McpHelpers.FormatError("get_health_parser_system_health", ex); } } - [McpServerTool(Name = "get_health_parser_severe_errors"), Description("Gets severe errors from system_health: stack dumps, non-yielding schedulers, and other critical SQL Server events.")] + [McpServerTool(Name = "get_health_parser_severe_errors"), Description("Gets severe errors from system_health: stack dumps, non-yielding schedulers, and other critical SQL Server events. Every answer carries source_observed (whether this server's system_health session has EVER been read into the store) and last_captured_at (the collector's newest capture): an empty window on a server whose session was never read is status unavailable, not a clean bill; an empty window on one that has been read says whether the category was captured and gated out, captured before this window, or never recorded by the engine.")] public static async Task GetSevereErrors( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -127,6 +138,7 @@ public static async Task GetSevereErrors( var xmls = await DarlingSystemHealthReader.ReadEventXmlAsync( postgres, resolved.ServerId, now.AddHours(-hours_back), now, SystemHealthParser.ErrorReportedEvent); var map = await mapTask; + var lastCapturedAt = await DarlingSystemHealthReader.GetLastCaptureAsync(postgres, resolved.ServerId); var rows = xmls .Select(SystemHealthParser.ParseSevereError) @@ -134,13 +146,17 @@ public static async Task GetSevereErrors( .Select(r => r!) .ToList(); if (rows.Count == 0) - return await DarlingEngineCapability.NotCollectedStatusAsync(postgres, resolved.ServerId, resolved.ServerName, SystemHealthCollectorName) - ?? McpHelpers.Status("empty", "No severe errors found in the requested time range."); + return await EmptyAsync( + postgres, new Collected(null, resolved.ServerId, resolved.ServerName, rows, xmls.Count, lastCapturedAt), + hours_back, SystemHealthParser.ErrorReportedEvent, + $"none was a significant severe error (severity {SystemHealthSignificance.SevereErrorMinSeverity}+ and off the benign connection-reset list)"); return JsonSerializer.Serialize(new { server = resolved.ServerName, hours_back, + source_observed = true, + last_captured_at = Stamp(lastCapturedAt), error_count = rows.Count, shown = Math.Min(rows.Count, limit), errors = rows.Take(limit).Select(r => new @@ -158,7 +174,7 @@ public static async Task GetSevereErrors( catch (Exception ex) { return McpHelpers.FormatError("get_health_parser_severe_errors", ex); } } - [McpServerTool(Name = "get_health_parser_io_issues"), Description("Gets I/O-related issues from system_health: 15-second I/O warnings, long I/O requests, and stalled I/O subsystems.")] + [McpServerTool(Name = "get_health_parser_io_issues"), Description("Gets I/O-related issues from system_health: 15-second I/O warnings, long I/O requests, and stalled I/O subsystems. Every answer carries source_observed (whether this server's system_health session has EVER been read into the store) and last_captured_at (the collector's newest capture): an empty window on a server whose session was never read is status unavailable, not a clean bill; an empty window on one that has been read says whether the category was captured and gated out, captured before this window, or never recorded by the engine.")] public static async Task GetIOIssues( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -175,13 +191,15 @@ public static async Task GetIOIssues( SystemHealthSignificance.IsSignificant); if (c.EarlyReturn != null) return c.EarlyReturn; if (c.Rows.Count == 0) - return await DarlingEngineCapability.NotCollectedStatusAsync(postgres, c.ServerId, c.ServerName, SystemHealthCollectorName) - ?? McpHelpers.Status("empty", "No I/O issues found in the requested time range."); + return await EmptyAsync(postgres, c, hours_back, SystemHealthParser.SpServerDiagnosticsEvent, + "none was an IO_SUBSYSTEM component result in the WARNING state"); return JsonSerializer.Serialize(new { server = c.ServerName, hours_back, + source_observed = true, + last_captured_at = Stamp(c.LastCapturedAt), issue_count = c.Rows.Count, shown = Math.Min(c.Rows.Count, limit), issues = c.Rows.Take(limit).Select(r => new @@ -199,7 +217,7 @@ public static async Task GetIOIssues( catch (Exception ex) { return McpHelpers.FormatError("get_health_parser_io_issues", ex); } } - [McpServerTool(Name = "get_health_parser_scheduler_issues"), Description("Gets scheduler issues from system_health: non-yielding schedulers, deadlocked schedulers, and scheduler monitor events.")] + [McpServerTool(Name = "get_health_parser_scheduler_issues"), Description("Gets scheduler issues from system_health: non-yielding schedulers, deadlocked schedulers, and scheduler monitor events. Every answer carries source_observed (whether this server's system_health session has EVER been read into the store) and last_captured_at (the collector's newest capture): an empty window on a server whose session was never read is status unavailable, not a clean bill; an empty window on one that has been read says whether the category was captured and gated out, captured before this window, or never recorded by the engine.")] public static async Task GetSchedulerIssues( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -215,13 +233,15 @@ public static async Task GetSchedulerIssues( SystemHealthSignificance.IsSignificant); if (c.EarlyReturn != null) return c.EarlyReturn; if (c.Rows.Count == 0) - return await DarlingEngineCapability.NotCollectedStatusAsync(postgres, c.ServerId, c.ServerName, SystemHealthCollectorName) - ?? McpHelpers.Status("empty", "No scheduler issues found in the requested time range."); + return await EmptyAsync(postgres, c, hours_back, SystemHealthParser.SchedulerMonitorEvent, + "none was a scheduler-monitor record in the WARNING state"); return JsonSerializer.Serialize(new { server = c.ServerName, hours_back, + source_observed = true, + last_captured_at = Stamp(c.LastCapturedAt), issue_count = c.Rows.Count, shown = Math.Min(c.Rows.Count, limit), issues = c.Rows.Take(limit).Select(r => new @@ -241,7 +261,7 @@ public static async Task GetSchedulerIssues( catch (Exception ex) { return McpHelpers.FormatError("get_health_parser_scheduler_issues", ex); } } - [McpServerTool(Name = "get_health_parser_memory_conditions"), Description("Gets memory condition events from system_health: low memory notifications, memory broker adjustments, and memory pressure indicators.")] + [McpServerTool(Name = "get_health_parser_memory_conditions"), Description("Gets memory condition events from system_health: low memory notifications, memory broker adjustments, and memory pressure indicators. Every answer carries source_observed (whether this server's system_health session has EVER been read into the store) and last_captured_at (the collector's newest capture): an empty window on a server whose session was never read is status unavailable, not a clean bill; an empty window on one that has been read says whether the category was captured and gated out, captured before this window, or never recorded by the engine.")] public static async Task GetMemoryConditions( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -257,13 +277,15 @@ public static async Task GetMemoryConditions( SystemHealthSignificance.IsSignificant); if (c.EarlyReturn != null) return c.EarlyReturn; if (c.Rows.Count == 0) - return await DarlingEngineCapability.NotCollectedStatusAsync(postgres, c.ServerId, c.ServerName, SystemHealthCollectorName) - ?? McpHelpers.Status("empty", "No memory condition events found in the requested time range."); + return await EmptyAsync(postgres, c, hours_back, SystemHealthParser.SpServerDiagnosticsEvent, + "none was a RESOURCE component result carrying a low-memory (RESOURCE_MEMPHYSICAL_LOW) notification"); return JsonSerializer.Serialize(new { server = c.ServerName, hours_back, + source_observed = true, + last_captured_at = Stamp(c.LastCapturedAt), event_count = c.Rows.Count, shown = Math.Min(c.Rows.Count, limit), events = c.Rows.Take(limit).Select(r => new @@ -306,7 +328,7 @@ public static async Task GetMemoryConditions( catch (Exception ex) { return McpHelpers.FormatError("get_health_parser_memory_conditions", ex); } } - [McpServerTool(Name = "get_health_parser_cpu_tasks"), Description("Gets CPU task events from system_health: long-running CPU-bound tasks, high CPU worker threads, and process utilization snapshots.")] + [McpServerTool(Name = "get_health_parser_cpu_tasks"), Description("Gets CPU task events from system_health: long-running CPU-bound tasks, high CPU worker threads, and process utilization snapshots. Every answer carries source_observed (whether this server's system_health session has EVER been read into the store) and last_captured_at (the collector's newest capture): an empty window on a server whose session was never read is status unavailable, not a clean bill; an empty window on one that has been read says whether the category was captured and gated out, captured before this window, or never recorded by the engine.")] public static async Task GetCPUTasks( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -322,13 +344,15 @@ public static async Task GetCPUTasks( SystemHealthSignificance.IsSignificant); if (c.EarlyReturn != null) return c.EarlyReturn; if (c.Rows.Count == 0) - return await DarlingEngineCapability.NotCollectedStatusAsync(postgres, c.ServerId, c.ServerName, SystemHealthCollectorName) - ?? McpHelpers.Status("empty", "No CPU task events found in the requested time range."); + return await EmptyAsync(postgres, c, hours_back, SystemHealthParser.SpServerDiagnosticsEvent, + $"none was a QUERY_PROCESSING component result in the WARNING state with at least {SystemHealthSignificance.CpuTaskMinPendingTasks} pending tasks"); return JsonSerializer.Serialize(new { server = c.ServerName, hours_back, + source_observed = true, + last_captured_at = Stamp(c.LastCapturedAt), event_count = c.Rows.Count, shown = Math.Min(c.Rows.Count, limit), events = c.Rows.Take(limit).Select(r => new @@ -350,7 +374,7 @@ public static async Task GetCPUTasks( catch (Exception ex) { return McpHelpers.FormatError("get_health_parser_cpu_tasks", ex); } } - [McpServerTool(Name = "get_health_parser_memory_broker"), Description("Gets memory broker events from system_health: cache shrink/grow notifications, memory clerk adjustments, and broker-mediated memory redistribution.")] + [McpServerTool(Name = "get_health_parser_memory_broker"), Description("Gets memory broker events from system_health: cache shrink/grow notifications, memory clerk adjustments, and broker-mediated memory redistribution. Every answer carries source_observed (whether this server's system_health session has EVER been read into the store) and last_captured_at (the collector's newest capture): an empty window on a server whose session was never read is status unavailable, not a clean bill; an empty window on one that has been read says whether the category was captured and gated out, captured before this window, or never recorded by the engine.")] public static async Task GetMemoryBroker( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -366,13 +390,15 @@ public static async Task GetMemoryBroker( SystemHealthSignificance.IsSignificant); if (c.EarlyReturn != null) return c.EarlyReturn; if (c.Rows.Count == 0) - return await DarlingEngineCapability.NotCollectedStatusAsync(postgres, c.ServerId, c.ServerName, SystemHealthCollectorName) - ?? McpHelpers.Status("empty", "No memory broker events found in the requested time range."); + return await EmptyAsync(postgres, c, hours_back, SystemHealthParser.MemoryBrokerEvent, + "none carried a low-memory notification (broker adjustments that are not a shrink under pressure are routine)"); return JsonSerializer.Serialize(new { server = c.ServerName, hours_back, + source_observed = true, + last_captured_at = Stamp(c.LastCapturedAt), event_count = c.Rows.Count, shown = Math.Min(c.Rows.Count, limit), events = c.Rows.Take(limit).Select(r => new @@ -396,7 +422,7 @@ public static async Task GetMemoryBroker( catch (Exception ex) { return McpHelpers.FormatError("get_health_parser_memory_broker", ex); } } - [McpServerTool(Name = "get_health_parser_memory_node_oom"), Description("Gets memory node OOM events from system_health: out-of-memory conditions on specific NUMA nodes.")] + [McpServerTool(Name = "get_health_parser_memory_node_oom"), Description("Gets memory node OOM events from system_health: out-of-memory conditions on specific NUMA nodes. Every answer carries source_observed (whether this server's system_health session has EVER been read into the store) and last_captured_at (the collector's newest capture): an empty window on a server whose session was never read is status unavailable, not a clean bill; an empty window on one that has been read says whether the category was captured and gated out, captured before this window, or never recorded by the engine.")] public static async Task GetMemoryNodeOOM( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -414,13 +440,15 @@ public static async Task GetMemoryNodeOOM( SystemHealthSignificance.IsSignificant); if (c.EarlyReturn != null) return c.EarlyReturn; if (c.Rows.Count == 0) - return await DarlingEngineCapability.NotCollectedStatusAsync(postgres, c.ServerId, c.ServerName, SystemHealthCollectorName) - ?? McpHelpers.Status("empty", "No memory node OOM events found in the requested time range."); + return await EmptyAsync(postgres, c, hours_back, SystemHealthParser.MemoryNodeOomEvent, + "none shredded to a memory-node OOM record (this category is ungated, so a captured OOM event that parsed would be here)"); return JsonSerializer.Serialize(new { server = c.ServerName, hours_back, + source_observed = true, + last_captured_at = Stamp(c.LastCapturedAt), event_count = c.Rows.Count, shown = Math.Min(c.Rows.Count, limit), events = c.Rows.Take(limit).Select(r => new @@ -459,7 +487,7 @@ public static async Task GetMemoryNodeOOM( catch (Exception ex) { return McpHelpers.FormatError("get_health_parser_memory_node_oom", ex); } } - [McpServerTool(Name = "get_health_parser_significant_waits"), Description("Gets significant individual waits from system_health: one row per wait_info event where a real session's non-BACKUP statement waited at least 500 ms on a wait type that is not idle/background — the wait type, total and signal duration, the wait resource, the session id and the waiting statement. get_wait_stats gives the instance-wide totals and can never name the statement that paid them; this is the individual waits, with their SQL text.")] + [McpServerTool(Name = "get_health_parser_significant_waits"), Description("Gets significant individual waits from system_health: one row per wait_info event where a real session's non-BACKUP statement waited at least 500 ms on a wait type that is not idle/background — the wait type, total and signal duration, the wait resource, the session id and the waiting statement. get_wait_stats gives the instance-wide totals and can never name the statement that paid them; this is the individual waits, with their SQL text. Every answer carries source_observed (whether this server's system_health session has EVER been read into the store) and last_captured_at (the collector's newest capture): an empty window on a server whose session was never read is status unavailable, not a clean bill; an empty window on one that has been read says whether the category was captured and gated out, captured before this window, or never recorded by the engine.")] public static async Task GetSignificantWaits( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -484,6 +512,7 @@ so the two would arrive indistinguishable. var now = windowEnd; var xmls = await DarlingSystemHealthReader.ReadEventXmlAsync( postgres, resolved.ServerId, now.AddHours(-hours_back), now, SystemHealthParser.WaitInfoEvent); + var lastCapturedAt = await DarlingSystemHealthReader.GetLastCaptureAsync(postgres, resolved.ServerId); var rows = xmls .Select(SystemHealthParser.ParseSignificantWait) @@ -494,46 +523,26 @@ so the two would arrive indistinguishable. if (rows.Count == 0) { /* - Three different nothings, and only one of them is good news. Events captured but none - significant is the healthy state and costs no extra query -- we already counted them. - Nothing captured in the window needs the probe to tell a quiet window from a server + The read this family's empty ladder was modelled on (#2484): events captured but none + significant is the healthy state and costs no extra query -- we already counted them; + nothing captured in the window needs the probe to tell a quiet window from a server whose wait_info has never been collected, because "no significant waits" is exactly - what an operator wants to hear and a caller who believes it stops looking. - */ - if (xmls.Count > 0) - { - return McpHelpers.Status( - "empty", - $"{xmls.Count} wait_info event(s) were captured for {resolved.ServerName} in the last {hours_back} hour(s) and none was significant (needs a real session, a non-BACKUP statement, at least {SystemHealthSignificance.SignificantWaitMinDurationMs} ms, and a wait type off the idle list). Events ARE being captured, so this is the healthy answer for this read rather than missing data."); - } - - var everCaptured = await DarlingSystemHealthReader.HasAnyEventOfTypeAsync( - postgres, resolved.ServerId, SystemHealthParser.WaitInfoEvent); - if (everCaptured) - { - return McpHelpers.Status( - "empty", - $"No wait_info events were captured for {resolved.ServerName} in the last {hours_back} hour(s). This server HAS captured them before, so the window is genuinely quiet rather than blind — widen hours_back to reach the most recent events."); - } - - /* - #2511 adds a FOURTH nothing, and it is the one that was being mis-explained. On an engine - whose system_health collector is gated off there is no session to start and no collection - to check, so the advice below is advice about something that cannot exist. The engine - answer goes first because it is the stronger claim; the text after it stays exactly right - for every engine that DOES collect this. + what an operator wants to hear and a caller who believes it stops looking. Since #3541 + A12 the ladder lives in EmptyAsync and all nine reads climb it; only the gate's own + description (the four conditions) is this tool's to word. */ - return await DarlingEngineCapability.NotCollectedStatusAsync( - postgres, resolved.ServerId, resolved.ServerName, SystemHealthCollectorName) - ?? McpHelpers.Status( - "unavailable", - $"No wait_info events have EVER been captured for {resolved.ServerName}, so this is NOT an all-clear — there is nothing here to be clear about. This read is served from the collected system_health ring buffer: check that collection is running for this server and that its system_health session is started before concluding nothing was waiting."); + return await EmptyAsync( + postgres, new Collected(null, resolved.ServerId, resolved.ServerName, rows, xmls.Count, lastCapturedAt), + hours_back, SystemHealthParser.WaitInfoEvent, + $"none was significant (needs a real session, a non-BACKUP statement, at least {SystemHealthSignificance.SignificantWaitMinDurationMs} ms, and a wait type off the idle list)"); } return JsonSerializer.Serialize(new { server = resolved.ServerName, hours_back, + source_observed = true, + last_captured_at = Stamp(lastCapturedAt), wait_count = rows.Count, shown = Math.Min(rows.Count, limit), waits = rows.Take(limit).Select(r => new @@ -561,7 +570,8 @@ signal close to the total is CPU pressure wearing a wait type's name. */ /// . The id rides along for the #2511 engine-capability probe on the zero-row path — /// re-resolving the name there would be a second chance to match a DIFFERENT server, since resolution is /// first-wins over a partial. - private readonly record struct Collected(string? EarlyReturn, int ServerId, string ServerName, List Rows); + private readonly record struct Collected( + string? EarlyReturn, int ServerId, string ServerName, List Rows, int RawEventCount, DateTime? LastCapturedAt); /// /// Resolves the server, validates hours_back + as_of + limit, reads the raw event_xml for @@ -570,20 +580,26 @@ signal close to the total is CPU pressure wearing a wait type's name. */ /// accepts. The seven gated categories pass their /// predicate; System Health passes an EventTime-present /// predicate (ungated, matching the viewer's chart read). + /// Also carries the two facts the payload owes under contract rule 5 (#3541 A12): how many raw + /// events of the type the window held BEFORE the gate (so zero survivors out of a hundred captured + /// reads as healthy, and zero out of zero does not), and the collector's newest capture of any type + /// (so a caller can see whether the source was ever observed at all). The witness is one index-walk + /// per call — see . /// private static async Task> CollectAsync( NpgsqlDataSource postgres, string? serverName, int hoursBack, int limit, string? asOf, string eventType, Func> shred, Func significant) where T : class { var (resolved, error) = await DarlingServerResolver.ResolveOrErrorAsync(postgres, serverName); - if (error != null) return new Collected(error, 0, "", new List()); + if (error != null) return new Collected(error, 0, "", new List(), 0, null); var validation = McpHelpers.ValidateWindow(hoursBack, asOf, out var windowEnd) ?? McpHelpers.ValidateTop(limit); - if (validation != null) return new Collected(validation, 0, "", new List()); + if (validation != null) return new Collected(validation, 0, "", new List(), 0, null); var now = windowEnd; var xmls = await DarlingSystemHealthReader.ReadEventXmlAsync( postgres, resolved.ServerId, now.AddHours(-hoursBack), now, eventType); + var lastCapturedAt = await DarlingSystemHealthReader.GetLastCaptureAsync(postgres, resolved.ServerId); var rows = new List(); foreach (var xml in xmls) @@ -595,9 +611,96 @@ private static async Task> CollectAsync( } } - return new Collected(null, resolved.ServerId, resolved.ServerName, rows); + return new Collected(null, resolved.ServerId, resolved.ServerName, rows, xmls.Count, lastCapturedAt); } + /* ─────────────────────────── the four nothings (#3541 A12) ─────────────────────────── */ + + /// + /// What zero rows means for one system_health category, which is four different things — and only the + /// first two are good news. Modelled on get_health_parser_significant_waits' three-way ladder (#2484), + /// which was the ONE read of the nine that refused to call a never-read session a clean bill; the other + /// eight answered empty to everything, so a dead system_health session, a collector that + /// never ran, and a healthy quiet hour all read as "no severe errors". Contract rule 5: zero is a + /// measurement, and an absence must say what it is an absence OF. + /// + /// Rung 1 — captured and gated out. Events of the type WERE stored in the window; the + /// shred + significance gate kept none. Healthy, and free: the raw count was taken on the data read. + /// Rung 2 — captured before, not in this window. Quiet window; widening reaches the most recent + /// events, and the message says when the last one was stored so the caller knows how far. + /// Rung 3 — this type never, but the session IS being read. Other categories have been stored, so + /// the ring buffer is reachable and the engine has simply never recorded one of these — for a + /// memory-node OOM or a severe error that is the healthy measurement, not a blind spot, and it must not + /// be called unavailable. Rung 4 — nothing of any type, ever. A dead session or a collector that + /// never ran: unavailable, the #3524 shape, never empty. The #2511 engine-capability probe + /// goes first on this rung because it is the stronger claim (an Azure SQL Database has no session to + /// start), and its text stays exactly right for every engine that does collect this. + /// + /// Every rung carries the same two witness keys the data envelope carries + /// (source_observed, last_captured_at) plus the rung's own evidence, at the top level + /// beside status — the trend family's precedent (#3541 A2): a caller reads the witness without + /// first checking which branch answered. The type-scoped probe runs only on the empty path, so + /// the healthy data path costs one witness query, not two. + /// + private static async Task EmptyAsync( + NpgsqlDataSource postgres, Collected c, int hoursBack, string eventType, string noneQualifiedBecause) + { + /* The type-scoped probe runs on every rung: on rung 1 the type exists in the window so the backward + index walk stops at its first row, and the stamp it returns is THIS type's newest capture rather + than the server-level witness standing in for it. */ + var lastOfType = await DarlingSystemHealthReader.GetLastCaptureOfTypeAsync(postgres, c.ServerId, eventType); + if (c.RawEventCount > 0) + { + return WitnessStatus( + "empty", + $"{c.RawEventCount} {eventType} event(s) were captured for {c.ServerName} in the last {hoursBack} hour(s) and {noneQualifiedBecause}. Events ARE being captured, so this is the healthy answer for this read rather than missing data.", + sourceObserved: true, c.LastCapturedAt, lastCapturedOfTypeAt: lastOfType, eventsInWindow: c.RawEventCount); + } + + if (lastOfType is DateTime seen) + { + return WitnessStatus( + "empty", + $"No {eventType} events were captured for {c.ServerName} in the last {hoursBack} hour(s). This server HAS captured them before (the newest was stored at {Stamp(seen)}), so the window is genuinely quiet rather than blind — widen hours_back to reach the most recent events.", + sourceObserved: true, c.LastCapturedAt, lastCapturedOfTypeAt: seen, eventsInWindow: 0); + } + + if (c.LastCapturedAt is DateTime alive) + { + return WitnessStatus( + "empty", + $"No {eventType} events have been captured for {c.ServerName} at any time, but its system_health session IS being read — the collector last stored an event of another type at {Stamp(alive)} — so for this category the absence is a measurement: the engine has not recorded one. Not a blind spot, and a wider window would not change it.", + sourceObserved: true, alive, lastCapturedOfTypeAt: null, eventsInWindow: 0); + } + + return await DarlingEngineCapability.NotCollectedStatusAsync(postgres, c.ServerId, c.ServerName, SystemHealthCollectorName) + ?? WitnessStatus( + "unavailable", + $"No system_health events of ANY type have EVER been captured for {c.ServerName}, so this is NOT an all-clear — there is nothing here to be clear about. This read is served from the collected system_health ring buffer: check that collection is running for this server and that its system_health session is started before concluding nothing happened.", + sourceObserved: false, lastCapturedAt: null, lastCapturedOfTypeAt: null, eventsInWindow: 0); + } + + /// + /// with the source witness beside status and message: the + /// same source_observed / last_captured_at pair the data envelope carries, plus what this + /// rung measured (last_captured_of_type_at, events_in_window). Top-level rather than under + /// hints so the keys sit in one place whichever branch answered. + /// + private static string WitnessStatus( + string status, string message, bool sourceObserved, DateTime? lastCapturedAt, DateTime? lastCapturedOfTypeAt, int eventsInWindow) + => JsonSerializer.Serialize(new + { + status, + message, + source_observed = sourceObserved, + last_captured_at = Stamp(lastCapturedAt), + last_captured_of_type_at = Stamp(lastCapturedOfTypeAt), + events_in_window = eventsInWindow, + }, McpHelpers.JsonOptions); + + /// The store's naive-UTC stamp in the same ISO shape the rows' event_time uses; null stays null. + private static string? Stamp(DateTime? stamp) => stamp?.ToString("o"); + /// Wraps a single-record shred (0-or-1) as the 0..n sequence expects. private static IEnumerable One(T? record) where T : class => record is null ? Enumerable.Empty() : new[] { record }; diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs index 436e32aae..5c467a40f 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs @@ -129,7 +129,7 @@ public static string Build(DarlingPeerDirectory.Snapshot peers) | `get_top_procedures_by_cpu` | Most expensive stored procedures by total CPU, with the same `cpu_attribution` disclosure | `server_name`, `hours_back` (default 24), `top` (default 20), `database_name`, `as_of` | | `get_query_store_top` | Expensive queries from Query Store with query_id / plan_id (survives restarts) | `server_name`, `hours_back` (default 24), `top` (default 20), `database_name`, `as_of` | | `get_query_heatmap` | The desktop viewer's Query Heatmap as a TABLE: how many DISTINCT queries fell into each (time bin x log-magnitude bucket) cell, with the most-executed query in each cell. The only query read with a TIME axis — `get_top_queries_by_cpu` ranks a whole window and cannot show that the window had a quiet half and a bad half, which is the first question about an incident that has already ended. Bins are **5 minutes** wide by default because that is exactly what the desktop viewer uses, so both surfaces draw the same picture; raise `bucket_minutes` to cover a longer window in fewer cells (it is the lever to reach for before the cap). The seven magnitude buckets are the viewer's, in the metric's own unit, and the labels come back with the result. `limit` caps CELLS, and truncation drops the OLDEST bins rather than the least interesting cells — `first_time_bin` / `last_time_bin` say which slice came back. Zero cells is THREE states and the read says which: never collected (`unavailable`, nobody looked), nothing collected in the window (`empty`, widen it), or collected and genuinely idle — captures exist and every one recorded zero executions (`empty`) | `server_name`, `hours_back` (default 24), `metric` (default `duration`), `database_name`, `bucket_minutes` (default 5), `limit` (default 500), `as_of` | - | `get_query_store_regressions` | Queries whose Query Store performance got WORSE: each (database, query_id) group's averages inside the recent window vs its BASELINE — every capture collected BEFORE that window. Baseline vs recent duration / CPU / logical reads with a regression percent each, the execution-count-weighted `additional_duration_ms` (the ranking key: a 5 ms regression run a million times outranks a 5-second one run twice), the plan counts on both sides, and a severity band. `get_query_store_top` answers what is EXPENSIVE and the costliest query is usually the one that always was; this answers what CHANGED. Kept only where average CPU regressed > 25%. Zero rows is FOUR states and the read says which: never collected (`unavailable`), no BASELINE because all history falls inside the window (`unavailable`, and NOT a clean bill of health — shorten hours_back), nothing collected in the window (`empty`, widen it), or a genuine all-clear (`empty`) | `server_name`, `hours_back` (default 24), `database_name`, `limit` (default 50), `as_of` | + | `get_query_store_regressions` | Queries whose Query Store performance got WORSE: each (database, query_id) group's averages inside the recent window vs its BASELINE — every capture collected BEFORE that window. Baseline vs recent duration / CPU / logical reads with a regression percent each, the execution-count-weighted `additional_duration_ms` (the ranking key: a 5 ms regression run a million times outranks a 5-second one run twice), the plan counts on both sides, and a severity band. `get_query_store_top` answers what is EXPENSIVE and the costliest query is usually the one that always was; this answers what CHANGED. Kept only where average CPU regressed > 25%. Zero rows is FOUR states and the read says which: never collected (`unavailable`), no BASELINE because all history falls inside the window (`unavailable`, and NOT a clean bill of health — shorten hours_back), nothing collected in the window (`empty`, widen it), or a genuine all-clear (`empty`) A percent whose baseline side is 0 (e.g. no logical reads before the window, 50k inside it) has no denominator and is null with the reason under `undefined_percents` — never 0, which would read as no change; the ranking key is an absolute delta and exists for every row, and `severity` is null when the duration percent is. | `server_name`, `hours_back` (default 24), `database_name`, `limit` (default 50), `as_of` | | `list_servers` | All monitored servers with collection-freshness status and last collection time. Each row names its engine: `engine_kind` is the registry token (`sqlserver` / `postgres` / `aurora-postgres`, null before any connect stamps the row) and `engine_version` the engine-aware version label ("SQL Server 2022", "PostgreSQL 18"); `sql_version` is a DEPRECATED alias of `engine_version` kept for existing consumers — do not read an engine from its name. Plus `peer_fleets` — the declared SIBLING Darling stores and what each covers (disclosure only; this server cannot read them) and `peer_note`, which says what an EMPTY `peer_fleets` does and does not prove | none | | `get_collection_health` | Per-collector health (running / failing / stale) over the last 7 days, plus the server's sweep_pressure block: a `verdict` for SUSTAINED demand (a SATURATED body collects at a multiple of its configured cadence with every collector healthy) and a separate `peak_cycle_risk` for a SINGLE sweep (BODY_OVERRUN means one scheduled body cannot fit the budget even when the verdict reads OK, the signature of one infrequent heavy collector; `peak_collector` names it). Per-collector rows carry `avg_duration_ms`, `p95_duration_ms` and `max_duration_ms`: a mean far below the p95 means the collector's runs come in two sizes and the mean describes neither | `server_name` | | `get_collection_log` | The RAW per-run collection log behind that rollup: one row per collector run with total duration split into time on the monitored server and time on the store, rows collected, status and any error. Reach for it when the rollup reads HEALTHY and collection still looks wrong, or to see what a collector was doing during a specific window. Newest first, or SLOWEST first when `min_duration_ms` is supplied — a duration floor under newest-first ordering cannot reach the tail, so the two are one decision and `order` names which you got. Both filters are applied in SQL BEFORE the cap, so `run_count` and `truncated` describe the MATCHING rows. `hours_back` is the span you asked for; `oldest_returned_collection_time` and `newest_returned_collection_time` bound the PAGE you got (under the default ordering that is also the reach; under a duration floor the page is a cost-ranked sample and its oldest row says nothing about reach), and the cap can make those differ by orders of magnitude — read them before concluding anything from the rows. An empty result distinguishes THREE states: filters that matched nothing (`empty`, and it says nothing about the window as a whole), a quiet window (`empty`, widen it), and a server that has never collected (`unavailable`, collection is not running) | `server_name`, `hours_back`, `limit`, `as_of`, `collector_name`, `min_duration_ms` | @@ -162,7 +162,7 @@ public static string Build(DarlingPeerDirectory.Snapshot peers) | `get_server_config` | CURRENT sys.configurations (latest snapshot) — what CTFP / MAXDOP / max memory are set to as of `captured_at` (captured on connect — can be days old) | `server_name` | | `get_database_config` | CURRENT per-database settings (latest snapshot) — recovery model, RCSI, Query Store, ... as of `captured_at` (captured on connect) | `server_name`, `database_name` | | `get_trace_flags` | CURRENT active trace flags (latest snapshot) — flag number, enabled, global/session, as of `captured_at` (captured on connect) | `server_name` | - | `get_table_index_sizes` | Largest tables with size + growth (7d/30d/daily) from the latest daily snapshot | `server_name` | + | `get_table_index_sizes` | The 100 largest tables with size + growth from the latest daily snapshot. Growth spans only history the store holds: `history` says how many days exist and whether the 7d/30d baselines are reachable; `growth_7d_mb` / `growth_30d_mb` / `growth_pct_30d` are null (reason in `growth_note`) when their baseline does not exist — never re-measured over a shorter span under the same name — and `growth_over_available_history_*` spans exactly `growth_window_days` | `server_name` | | `get_index_usage` | Per-index usage classified Unused / Write-only / Active | `server_name` | | `get_object_locking` | Per-index lock/latch contention, most contended first | `server_name` | | `get_database_sizes` | Per-file database sizes, space usage, and volume free space | `server_name` | @@ -192,7 +192,7 @@ public static string Build(DarlingPeerDirectory.Snapshot peers) | `get_perfmon_trend` | A single performance counter's value + delta over time (summed across instances) | `counter_name` (required), `server_name`, `hours_back` (default 24), `as_of` | | `get_file_io_trend` | Per-database file I/O read/write latency over time (top-10 busiest files). An empty result distinguishes a quiet window (`empty`, widen `hours_back`) from a server nothing has ever been collected for (`unavailable`, collection is not running) | `server_name`, `hours_back` (default 24), `as_of` | | `get_query_trend` | One query's per-collection history (deltas, avg cpu/elapsed, DOP) by query_hash | `query_hash` (required), `database_name` (required), `server_name`, `hours_back` (default 24), `as_of` | - | `get_query_duration_trend` | Overall query elapsed-ms/sec + executions/sec across all queries over time, from the PLAN CACHE. Each point carries `value` (ms/sec), `execution_count` and `executions_per_second` — the last two are the same quantity, and `execution_count` is truncated to an integer, so read `executions_per_second` on a quiet server where the rate is below 1. Each point also carries `elapsed_ms_per_second`, the same quantity as `value` with its unit in the name — read that one. Raw `query_stats` rows are dropped at 4 days on a TimescaleDB store, so a window reaching past that is served from the hourly rollup: `source` says which tier answered (`raw` or `hourly`), `bucket` its grain (`per-collection` or `1 hour`), `effective_start` / `effective_hours_back` where the served series actually begins, `truncated` whether that head sits later than asked, and `aggregate_note` what the hourly tier trades (each point is the hour's work over 3,600 s, so a partly-collected hour reads low; the rollup trails the clock by up to two hours). An empty result carries the same block and distinguishes a quiet window (`empty`, widen `hours_back`) from a server nothing has ever been collected for (`unavailable`, collection is not running) — and from a window whose head the tier does not hold (`empty`, but the message says the rows were dropped or not yet materialized and that widening cannot help; `--backfill-rollups` is the remedy) | `server_name`, `hours_back` (default 24), `as_of` | + | `get_query_duration_trend` | Overall query elapsed-ms/sec + executions/sec across all queries over time, from the PLAN CACHE. Each point carries `value` (ms/sec), `execution_count` and `executions_per_second` — the last two are the same quantity, and `execution_count` is truncated to an integer, so read `executions_per_second` on a quiet server where the rate is below 1. Each point also carries `elapsed_ms_per_second`, the same quantity as `value` with its unit in the name — read that one. Raw `query_stats` rows are dropped at 4 days on a TimescaleDB store, so a window reaching past that is served from the hourly rollup: `source` says which tier answered (`raw` or `hourly`), `bucket` its grain (`per-collection` or `1 hour`), `effective_start` / `effective_hours_back` where the served series actually begins, `truncated` whether that head sits later than asked, and `aggregate_note` what the hourly tier trades (each point is the hour's work over 3,600 s, so a partly-collected hour reads low; the rollup trails the clock by up to two hours). An empty result carries the same block and distinguishes a quiet window (`empty`, widen `hours_back`) from a server nothing has ever been collected for (`unavailable`, collection is not running) — and from a window whose head the tier does not hold (`empty`, but the message says the rows were dropped or not yet materialized and that widening cannot help; `--backfill-rollups` is the remedy) On the raw route each point is a rate over the gap since the PREVIOUS collection, so the window's first collection carries null rates (`unrated_points` / `unrated_note`): unknowable, never reported as 0 (the hourly route divides by the bucket width and has no such point). | `server_name`, `hours_back` (default 24), `as_of` | | `get_procedure_duration_trend` | The same series over `procedure_stats`. NOT a duplicate of the above: query_stats attributes a procedure's work to the individual statements inside it, so a procedure that got slower is smeared across however many statements it runs — this charges the whole call to the procedure. Read the two together to tell an ad-hoc regression from a procedure regression. Same tier routing and the same `source` / `effective_start` / `truncated` / `bucket` / `aggregate_note` block as `get_query_duration_trend`, over `procedure_stats` / `procedure_stats_hourly` | `server_name`, `hours_back` (default 24), `as_of` | | `get_query_store_duration_trend` | The same series over Query Store. The plan-cache trends lose everything an eviction or a restart takes with them; Query Store persists per interval, so this is the series that survives a failover and the one to reach for when a regression is older than the cache. Each interval is counted once, at the hour the work RAN. Its `unavailable` names the cause the other two do not have: Query Store may simply be OFF on every database. Carries the same `source` / `effective_start` / `truncated` / `bucket` / `aggregate_note` block as its siblings; where the store has the corrected Query Store rollup, `source` is `rollup+raw` and a `routing` block names the rollup, `raw_from` (the seam: 1-hour buckets before it, per-interval points from it) and, when the window reaches below what the rollup has materialized, `unserved_before` — points before that instant are missing, not zero | `server_name`, `hours_back` (default 24), `as_of` | @@ -200,6 +200,8 @@ public static string Build(DarlingPeerDirectory.Snapshot peers) The `get_health_parser_*` family the Dashboard exposes, over Darling's raw `system_health_events`. Where the Dashboard reads its server-side-parsed `collect.HealthParser_*` tables, these shred the raw extended-event XML ON READ with the shared SystemHealthParser and return the same SIGNIFICANT warning set (sp_HealthParser at `@warnings_only = 1`) — the exception is `get_health_parser_system_health`, whose corruption/contention counter series is UNGATED (every snapshot). Each returns the full sp_HealthParser column set per row keyed on the event's `event_time`; the tools window on `event_time` (the event's real time), so "last 24 hours" means events that happened in the last 24 hours. + Every one of the nine carries a SOURCE WITNESS: `source_observed` (whether this server's system_health session has EVER been read into the store — any event of any type) and `last_captured_at` (the collector's newest capture). Zero rows is four different answers and the read says which: captured in the window and gated out (`empty`, `events_in_window` > 0 — the healthy one), captured before but not in this window (`empty`, `last_captured_of_type_at` says when — widen), never captured for THIS category while the session is being read (`empty`, `source_observed` true — for a rare category such as a memory-node OOM this is the measurement, not a gap), and nothing of any type ever (`unavailable`, `source_observed` false — a dead session or a collector that never ran, and NOT an all-clear). Zero is a measurement only when the source was observed. + | Tool | Purpose | Key Parameters | |------|---------|----------------| | `get_health_parser_system_health` | SYSTEM-component snapshots: corruption (bad pages / dumps / access violations) + contention (non-yielding / latch / sick spinlock / CPU) counters | `server_name`, `hours_back` (default 24), `limit` (default 50), `as_of` | diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpObjectStatsTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpObjectStatsTools.cs index b8737f097..5a95bc997 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpObjectStatsTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpObjectStatsTools.cs @@ -7,6 +7,7 @@ */ using System; +using System.Collections.Generic; using System.ComponentModel; using System.Linq; using System.Text.Json; @@ -35,7 +36,7 @@ public sealed class DarlingMcpObjectStatsTools private const int IndexUsageTop = 200; private const int ObjectLockingTop = 200; - [McpServerTool(Name = "get_table_index_sizes"), Description("Gets the largest tables with per-table size, growth (7d/30d/daily rate), and row counts from the latest daily snapshot. Indexes are rolled up per table. Use to find storage hot-spots and fast-growing tables for capacity planning.")] + [McpServerTool(Name = "get_table_index_sizes"), Description("Gets the 100 largest tables with per-table size, growth (7d/30d/daily rate), and row counts from the latest daily snapshot. Indexes are rolled up per table. Use to find storage hot-spots and fast-growing tables for capacity planning. Growth is measured only over history the store actually holds: the history block says how many days of snapshots exist and whether the 7-day and 30-day baselines are reachable; growth_7d_mb / growth_30d_mb / growth_pct_30d are null (with the reason in growth_note) when their baseline does not exist, never re-labelled from a nearer one, and growth_over_available_history_* always spans exactly growth_window_days. A table absent from a baseline snapshot (created since) reports null growth for that window, not 0. tables_returned and truncated bound the page.")] public static async Task GetTableIndexSizes( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null) @@ -46,13 +47,22 @@ public static async Task GetTableIndexSizes( try { var now = DateTime.UtcNow; + /* Over-fetch by one so truncation is observed, not inferred from a full page (#3541 A3's rule). */ var rows = await DarlingObjectStatsReader.GetObjectSizeGrowthAsync( - postgres, resolved.ServerId, now.AddDays(-7), now.AddDays(-30), TableSizesTop); + postgres, resolved.ServerId, now.AddDays(-7), now.AddDays(-30), TableSizesTop + 1); if (rows.Count == 0) return await DarlingEngineCapability.NotCollectedStatusAsync(postgres, resolved.ServerId, resolved.ServerName, "index_object_stats") ?? McpHelpers.Status("unavailable", "No object size data available. Index/object stats are collected daily."); - var result = rows.Select(r => new + var truncated = rows.Count > TableSizesTop; + var page = rows.Take(TableSizesTop).ToList(); + + /* The store's span is one fact for every row (the boundaries CTE), so it is published once. */ + var span = page[0]; + var covers7d = span.Snapshot7dTime is not null; + var covers30d = span.Snapshot30dTime is not null; + + var result = page.Select(r => new { database_name = r.DatabaseName, schema_name = r.SchemaName, @@ -61,15 +71,39 @@ public static async Task GetTableIndexSizes( used_mb = r.CurrentUsedMb, total_rows = r.TotalRows, index_count = r.IndexCount, + /* Each nominal-window figure comes from exactly the baseline it names, or is null (#3541 + A12). The SQL this replaced folded a missing 30-day baseline onto the 7-day one and a + missing 7-day one onto the oldest, and labelled the result with the window asked for. */ growth_7d_mb = r.Growth7dMb, growth_30d_mb = r.Growth30dMb, + growth_pct_30d = r.GrowthPct30d, + /* The figure that is always honest: growth from the store's earliest snapshot of this table + to its latest, over exactly growth_window_days. Null only when there is no span at all. */ + growth_over_available_history_mb = r.GrowthOverAvailableHistoryMb, + growth_over_available_history_pct = r.GrowthOverAvailableHistoryPct, + growth_window_days = r.DaysOfData, daily_growth_rate_mb = r.DailyGrowthRateMb, - growth_pct_30d = r.GrowthPct30d + growth_note = GrowthNote(r), }); return JsonSerializer.Serialize(new { server = resolved.ServerName, + history = new + { + earliest_snapshot = span.EarliestSnapshotTime.ToString("o"), + latest_snapshot = span.LatestSnapshotTime.ToString("o"), + history_days_available = span.DaysOfData, + covers_7d = covers7d, + covers_30d = covers30d, + note = covers30d + ? null + : $"The store holds {span.DaysOfData} day(s) of index snapshots for this server, so the " + + (covers7d ? "30-day baseline does not exist: growth_30d_mb and growth_pct_30d are null" : "7-day and 30-day baselines do not exist: growth_7d_mb, growth_30d_mb and growth_pct_30d are null") + + " rather than re-measured over a shorter span under the same name. Read growth_over_available_history_* — it spans exactly growth_window_days.", + }, + tables_returned = page.Count, + truncated, tables = result }, McpHelpers.JsonOptions); } @@ -79,6 +113,34 @@ public static async Task GetTableIndexSizes( } } + /// + /// Why a row's growth figures are null, when they are (#3541 A12): the store has no snapshot old enough + /// for the window, or the snapshot exists but this table was not in it (created since), or there is no + /// span at all. Null when every figure is defined, so the common row carries no note. Lite's twin words + /// it identically. + /// + internal static string? GrowthNote(DarlingObjectStatsReader.ObjectSizeGrowthRow r) + { + var notes = new List(); + if (r.DaysOfData < 1) + notes.Add("the store holds a single day of snapshots for this server, so no growth is knowable yet — every growth figure is null, not 0"); + if (r.Snapshot7dTime is null) + notes.Add("no snapshot 7+ days old exists, so growth_7d_mb is null"); + else if (r.ReservedMb7dAgo is null) + notes.Add($"this table was not in the {r.Snapshot7dTime:o} snapshot (created since), so growth_7d_mb is null — its whole current size is newer than 7 days"); + if (r.Snapshot30dTime is null) + notes.Add("no snapshot 30+ days old exists, so growth_30d_mb and growth_pct_30d are null"); + else if (r.ReservedMb30dAgo is null) + notes.Add($"this table was not in the {r.Snapshot30dTime:o} snapshot (created since), so growth_30d_mb and growth_pct_30d are null"); + else if (r.ReservedMb30dAgo <= 0) + notes.Add("the table was empty 30 days ago, so growth_pct_30d has no denominator and is null (growth_30d_mb carries the absolute)"); + if (r.DaysOfData >= 1 && r.ReservedMbOldest is null) + notes.Add($"this table was not in the earliest snapshot ({r.EarliestSnapshotTime:o}), so growth_over_available_history_* and daily_growth_rate_mb are null"); + else if (r.DaysOfData >= 1 && r.ReservedMbOldest <= 0) + notes.Add("the table was empty at the earliest snapshot, so growth_over_available_history_pct has no denominator and is null"); + return notes.Count == 0 ? null : string.Join("; ", notes) + "."; + } + [McpServerTool(Name = "get_index_usage"), Description("Gets per-index usage (seeks, scans, lookups, updates) from the latest daily snapshot, classifying each index as Unused, Write-only, or Active. Unused and write-only indexes are listed FIRST because they are drop candidates - which means that on a server with many unused indexes the row limit can be filled entirely by one database's unused indexes, hiding every Active index elsewhere. Pass database_name to ask about one database, which is almost always what you want; the response carries matching_index_count and truncated so a short answer is never mistaken for an absent one. Counters are cumulative since the last instance restart. last_user_access is UTC - the underlying sys.dm_db_index_usage_stats columns are in the monitored server's local clock and this read de-skews them - so it compares directly against get_collection_log and list_servers.")] public static async Task GetIndexUsage( NpgsqlDataSource postgres, diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgXminTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgXminTools.cs index 1448af572..a4e28ef48 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgXminTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPgXminTools.cs @@ -56,7 +56,7 @@ public sealed class DarlingMcpPgXminTools _ => "Unrecognized holder source.", }; - [McpServerTool(Name = "get_pg_xmin_horizon"), Description("Gets what is holding back the PostgreSQL xmin horizon, attributed by cause. Use this whenever dead tuples or table bloat are growing while autovacuum appears to be running normally - that symptom has four unrelated causes which look identical from the outside, and each needs a completely different fix: a long-running or idle-in-transaction session, an abandoned replication slot, a logical slot holding catalog_xmin, a standby feeding back its xmin, or an orphaned prepared transaction. Reports the oldest holder for each source, which one is currently winning, and how persistent each has been across the window, so a chronic holder can be told apart from a query that merely ran long. Also relevant to wraparound risk: a pinned horizon blocks freezing, so an unattended holder here is an upstream cause of the risk get_pg_wraparound_risk measures. Works on any PostgreSQL target.")] + [McpServerTool(Name = "get_pg_xmin_horizon"), Description("Gets what is holding back the PostgreSQL xmin horizon, attributed by cause. Use this whenever dead tuples or table bloat are growing while autovacuum appears to be running normally - that symptom has four unrelated causes which look identical from the outside, and each needs a completely different fix: a long-running or idle-in-transaction session, an abandoned replication slot, a logical slot holding catalog_xmin, a standby feeding back its xmin, or an orphaned prepared transaction. Reports the oldest holder for each source, which one is currently winning, and how persistent each has been across the window, so a chronic holder can be told apart from a query that merely ran long: pct_of_window_winning is the share of EVERY capture in the window (captures_in_window, from the collector's own log - unheld captures included, the same denominator the vacuum-horizon alert uses), not the share of the captures that happened to record this source. Also relevant to wraparound risk: a pinned horizon blocks freezing, so an unattended holder here is an upstream cause of the risk get_pg_wraparound_risk measures. Works on any PostgreSQL target. Zero holders is reported as no_holder only when the collector captured in the window; no holders AND no captures is unavailable, not an all-clear.")] public static async Task GetPgXminHorizon( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -74,6 +74,8 @@ public static async Task GetPgXminHorizon( var now = windowEnd; var rows = await DarlingPgXminReader.GetPgXminHorizonAsync( postgres, resolved.ServerId, now.AddHours(-hours_back), now); + var capturesInWindow = await DarlingPgXminReader.GetXminCapturesInWindowAsync( + postgres, resolved.ServerId, now.AddHours(-hours_back), now); /* Nothing holding the horizon is the HEALTHY answer, and saying so plainly matters more here than for most tools: an operator arrives at this tool BECAUSE bloat is growing, so "no @@ -90,12 +92,29 @@ holding back the xmin horizon" is a confident all-clear about a mechanism that d return gated; } + /* Zero holders is a measurement only if the collector looked (#3541 A12): the collector + stores nothing on an unheld capture, so an empty holder table is ALSO what a collector + that never ran in this window leaves behind. The capture count is the witness. */ + if (capturesInWindow == 0) + { + return JsonSerializer.Serialize(new + { + server = resolved.ServerName, + hours_back, + status = "unavailable", + captures_in_window = 0, + message = $"No holder rows AND no successful pg_xmin_horizon captures are logged for {resolved.ServerName} in the last {hours_back} hour(s), so this is NOT a report that nothing holds the horizon — the collector did not look (or its collection_log rows are missing). Check get_collection_health for this server before reading the absence as clear.", + }, McpHelpers.JsonOptions); + } + return JsonSerializer.Serialize(new { server = resolved.ServerName, hours_back, status = "no_holder", - finding = "Nothing is holding back the xmin horizon in this window. Vacuum is free to " + captures_in_window = capturesInWindow, + finding = $"Nothing is holding back the xmin horizon in this window: the collector captured " + + $"{capturesInWindow} time(s) and recorded no holder. Vacuum is free to " + "reclaim dead rows, so bloat growth has a different cause — look at whether " + "autovacuum is being triggered at all (per-table thresholds and dead-tuple " + "counts) rather than at whether it is being blocked.", @@ -114,10 +133,18 @@ holding back the xmin horizon" is a confident all-clear about a mechanism that d /* Persistence, not just presence. A source that won nearly every sample is a standing problem someone must own; one that won twice was a query that ran long and finished. */ samples_as_winner = r.SamplesAsWinner, - samples = r.Samples, - pct_of_window_winning = r.Samples > 0 - ? Math.Round((double)r.SamplesAsWinner / r.Samples * 100, 1) - : 0, + /* Captures in which THIS source recorded a holder — its own rows, not the window. Named so it + cannot be read as the window's capture count, which is captures_in_window above. */ + captures_recording_this_source = r.Samples, + /* Over EVERY capture in the window (#3541 A12), not over this source's own rows: the collector + stores nothing on an unheld capture, so dividing by the source's rows made 2 wins in 2 rows + out of 288 captures read as 100% chronic. The same denominator the alert evaluator's + horizon arm fractions over (DarlingPostgresAlertReadAdapter.XminSql). Null, never 0, when + the log holds no captures to divide by; unclamped, so an undercounting log (a skipped + failure-isolated write) shows as a share above 100 rather than being rounded into a lie. */ + pct_of_window_winning = capturesInWindow > 0 + ? Math.Round((double)r.SamplesAsWinner / capturesInWindow * 100, 1) + : (double?)null, remedy = RemedyFor(r.Source), }).ToList(); @@ -128,6 +155,12 @@ holding back the xmin horizon" is a confident all-clear about a mechanism that d server = resolved.ServerName, hours_back, status = "holder_present", + /* The window's denominator: successful pg_xmin_horizon runs logged in the window, from + collection_log — every time the collector LOOKED, held or not. */ + captures_in_window = capturesInWindow, + pct_denominator = capturesInWindow > 0 + ? "pct_of_window_winning = samples_as_winner / captures_in_window; captures_in_window counts the collector's SUCCESS rows in collection_log for this window, so it includes captures that found no holder. It can undercount if a log write was skipped, in which case a share can exceed 100 — it is not clamped." + : "pct_of_window_winning is null: collection_log holds no successful pg_xmin_horizon capture in this window to divide by, though holder rows exist — read samples_as_winner as a count, not a share.", /* Lead with the actionable pair: which cause, and what to do about that cause. */ winning_source = winner.source, winning_holder = winner.holder, diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPvsTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPvsTools.cs index 4a798cc0d..4b8be6c16 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPvsTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpPvsTools.cs @@ -31,7 +31,7 @@ namespace PerformanceMonitor.Darling.Service.Mcp; public sealed class DarlingMcpPvsTools { [McpServerTool(Name = "get_pvs_stats"), Description( - "Gets the Accelerated Database Recovery (ADR) persistent version store state per database: PVS size and percent-of-database, online-index version store size, aborted transaction count, version-cleaner run state (a start time without an end time means the cleaner is mid-run), and the oldest active/aborted transaction ids. Use when a database's size is growing without table growth, when ADR cleanup looks stuck, or alongside the PVS pressure alert. A large PVS is pinned by long-running or aborted transactions; the id gap shows how far cleanup is behind. Optionally returns the size trend for the top-5 databases over a window. Every timestamp here is UTC, the four cleaner times included - the DMV reports those in the monitored server's local clock and this read de-skews them - so a cleaner time compares directly against as_of.")] + "Gets the Accelerated Database Recovery (ADR) persistent version store state per database: PVS size and percent-of-database, online-index version store size, aborted transaction count, version-cleaner run state (a start time without an end time means the cleaner is mid-run), and the oldest active/aborted transaction ids. Use when a database's size is growing without table growth, when ADR cleanup looks stuck, or alongside the PVS pressure alert. A large PVS is pinned by long-running or aborted transactions; the id gap shows how far cleanup is behind. Optionally returns the size trend for the top-5 databases over a window. Every timestamp here is UTC, the four cleaner times included - the DMV reports those in the monitored server's local clock and this read de-skews them - so a cleaner time compares directly against as_of. pvs_measured says whether the DMV reported a size for that database at all; a measured 0 MB is published as pvs_size_mb 0 and pct_of_database 0.00 (the healthy, fully-cleaned state), and pct_of_database is null only when the numerator was not measured or the denominator is absent, with pct_of_database_reason saying which.")] public static async Task GetPvsStats( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -62,10 +62,17 @@ public static async Task GetPvsStats( database_name = r.DatabaseName, is_adr_on = r.IsAdrOn, pvs_size_mb = r.PvsSizeMb, - /* The SAME denominator the FinOps grid and the pressure alert use, so no surface disagrees. */ - pct_of_database = r.PvsSizeMb is > 0 && r.DatabaseDataSizeMb is > 0 - ? Math.Round(r.PvsSizeMb.Value / r.DatabaseDataSizeMb.Value * 100.0, 2) + /* Whether the DMV reported a size at all (#3541 A12, contract rule 5). A measured 0 MB — the + healthy, fully-cleaned state — used to be indistinguishable from a NULL the collector + could not read: both fell through to pct_of_database = null. */ + pvs_measured = r.PvsSizeMb.HasValue, + /* The SAME denominator the FinOps grid and the pressure alert use, so no surface disagrees. + Any MEASURED size divides — 0 MB of a 100 GB database is 0.00%, a measurement — and only an + unmeasured numerator or an absent/zero denominator yields null, with the reason beside it. */ + pct_of_database = r.PvsSizeMb is { } pvsMb && r.DatabaseDataSizeMb is > 0 + ? Math.Round(pvsMb / r.DatabaseDataSizeMb.Value * 100.0, 2) : (double?)null, + pct_of_database_reason = PctReason(r.PvsSizeMb.HasValue, r.DatabaseDataSizeMb), online_index_version_store_mb = r.OnlineIndexVersionStoreMb, database_data_size_mb = r.DatabaseDataSizeMb, aborted_transaction_count = r.AbortedTransactionCount, @@ -116,4 +123,20 @@ public static async Task GetPvsStats( return McpHelpers.FormatError("get_pvs_stats", ex); } } + + /// + /// Why pct_of_database is null, when it is (#3541 A12): the numerator was not measured, or the + /// denominator was absent or zero. Null when the percent is defined — including a defined 0.00 — so the + /// healthy row carries no note. Lite's twin words it identically. + /// + internal static string? PctReason(bool pvsMeasured, double? databaseDataSizeMb) + { + if (!pvsMeasured) + return "pvs_size_mb was not reported by sys.dm_tran_persistent_version_store_stats in this capture, so the share is unknown — not zero."; + if (databaseDataSizeMb is null) + return "database_data_size_mb was not captured for this database, so there is no denominator — the share is unknown, not zero."; + if (databaseDataSizeMb <= 0) + return "database_data_size_mb is 0, so the share has no denominator — the share is unknown, not zero."; + return null; + } } diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpQueryStoreRegressionTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpQueryStoreRegressionTools.cs index 572e16c34..794b9a9e2 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpQueryStoreRegressionTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpQueryStoreRegressionTools.cs @@ -7,6 +7,7 @@ */ using System; +using System.Collections.Generic; using System.ComponentModel; using System.Linq; using System.Text.Json; @@ -33,7 +34,7 @@ namespace PerformanceMonitor.Darling.Service.Mcp; [McpServerToolType] public sealed class DarlingMcpQueryStoreRegressionTools { - [McpServerTool(Name = "get_query_store_regressions"), Description("Finds queries whose Query Store performance got WORSE, by comparing each (database, query_id) group's averages inside a recent window against its baseline - every capture BEFORE that window. Returns baseline vs recent duration, CPU and logical reads with the regression percent for each, the execution-count-weighted extra duration (the ranking key: a 5 ms regression executed a million times outranks a 5-second one executed twice), the plan counts on both sides, and a duration-driven severity band. get_query_store_top answers what is EXPENSIVE; the most expensive query is usually the one that always was. This answers what CHANGED. Rows are kept only where average CPU regressed by more than 25%.")] + [McpServerTool(Name = "get_query_store_regressions"), Description("Finds queries whose Query Store performance got WORSE, by comparing each (database, query_id) group's averages inside a recent window against its baseline - every capture BEFORE that window. Returns baseline vs recent duration, CPU and logical reads with the regression percent for each, the execution-count-weighted extra duration (the ranking key: a 5 ms regression executed a million times outranks a 5-second one executed twice), the plan counts on both sides, and a duration-driven severity band. get_query_store_top answers what is EXPENSIVE; the most expensive query is usually the one that always was. This answers what CHANGED. Rows are kept only where average CPU regressed by more than 25%. A regression percent whose BASELINE side is 0 has no denominator and is returned as null, with the reason under undefined_percents - never as 0, which would read as no change when the truth is the largest possible one; compare the two absolute figures instead. The ranking key is the absolute, execution-weighted duration delta, which exists whether or not a ratio does, so a null percent never sorts as 0. severity is banded from the duration percent and is null when that percent is.")] public static async Task GetQueryStoreRegressions( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -87,7 +88,10 @@ the recent window bigger AND the baseline shorter. { database_name = r.DatabaseName, query_id = r.QueryId, - severity = r.Severity, + /* Banded from the duration percent by the TVF's CASE, whose ELSE is 'LOW' — which for a + row with NO duration ratio is a verdict about a number that does not exist. Null there + (#3541 A12); the SQL's band is kept verbatim for the viewer it is shared with. */ + severity = r.DurationRegressionPercent is null ? null : r.Severity, baseline_duration_ms = r.BaselineDurationMs, recent_duration_ms = r.RecentDurationMs, duration_regression_percent = r.DurationRegressionPercent, @@ -97,8 +101,12 @@ the recent window bigger AND the baseline shorter. baseline_reads = r.BaselineReads, recent_reads = r.RecentReads, io_regression_percent = r.IoRegressionPercent, + /* Null percents, and why (#3541 A12): a 0 baseline has no ratio, and the reader used to + publish that as 0 — "no change" — for the row that changed the most. */ + undefined_percents = UndefinedPercentNotes(r), /* The ranking key, and the one number that says whether this regression MATTERS: a - 5 ms regression executed a million times outranks a 5-second one executed twice. */ + 5 ms regression executed a million times outranks a 5-second one executed twice. It is + an absolute delta, so it exists for every row and a null ratio never sorts as 0. */ additional_duration_ms = r.AdditionalDurationMs, baseline_exec_count = r.BaselineExecCount, recent_exec_count = r.RecentExecCount, @@ -117,6 +125,27 @@ 5 ms regression executed a million times outranks a 5-second one executed twice. } } + + /// + /// Which of a row's three regression percents are undefined, and why (#3541 A12, contract rule 5). Each + /// percent divides through NULLIF(baseline, 0), so a NULL means the baseline side was 0 — there is + /// no denominator, not no change — and the caller is pointed at the absolute pair it can still compare. + /// Null when every percent is defined, so the common row carries no noise. Lite's twin builds the same + /// sentences. + /// + private static List? UndefinedPercentNotes(DarlingQueryStoreRegressionReader.RegressionRow r) + { + List? notes = null; + void Note(string field, string baseline, string recent) + => (notes ??= new List()).Add( + $"{field} is null: no_baseline — {baseline} is 0, so the ratio has no denominator; this is NOT 0% change. Compare {baseline} to {recent} directly."); + + if (r.DurationRegressionPercent is null) Note("duration_regression_percent", "baseline_duration_ms", "recent_duration_ms"); + if (r.CpuRegressionPercent is null) Note("cpu_regression_percent", "baseline_cpu_ms", "recent_cpu_ms"); + if (r.IoRegressionPercent is null) Note("io_regression_percent", "baseline_reads", "recent_reads"); + return notes; + } + /// /// What zero regressions actually means, which is four different things. /// Only ONE of them is good news, and the other three all look identical to it in a bare empty diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTrendTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTrendTools.cs index 3f3871093..d512e9b66 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTrendTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTrendTools.cs @@ -438,7 +438,7 @@ concluding the query did not run. */ } } - [McpServerTool(Name = "get_query_duration_trend"), Description("Gets a time-series of average query duration over time. Useful for spotting overall performance degradation or improvement trends across all queries.")] + [McpServerTool(Name = "get_query_duration_trend"), Description("Gets a time-series of average query duration over time. Useful for spotting overall performance degradation or improvement trends across all queries. On the per-collection (raw) route every point is a rate over the gap since the PREVIOUS collection, so the window's first collection - which has no previous one to difference against - carries null rates: unknowable, never reported as 0 (unrated_points counts them, unrated_note says why). The hourly rollup route divides by the bucket width and has no such point.")] public static async Task GetQueryDurationTrend( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -499,7 +499,7 @@ is not a server nothing was ever stored for. */ } } - [McpServerTool(Name = "get_procedure_duration_trend"), Description("Gets a time-series of stored-procedure elapsed time per second and executions per second over time, summed across every procedure. The sibling of get_query_duration_trend, and NOT a duplicate of it: query_stats attributes a procedure's work to the individual statements inside it, so a procedure that got slower is smeared across however many statements it runs. This charges the whole call to the procedure. Read the two together to tell an ad-hoc SQL regression from a procedure regression.")] + [McpServerTool(Name = "get_procedure_duration_trend"), Description("Gets a time-series of stored-procedure elapsed time per second and executions per second over time, summed across every procedure. The sibling of get_query_duration_trend, and NOT a duplicate of it: query_stats attributes a procedure's work to the individual statements inside it, so a procedure that got slower is smeared across however many statements it runs. This charges the whole call to the procedure. Read the two together to tell an ad-hoc SQL regression from a procedure regression. On the per-collection (raw) route every point is a rate over the gap since the PREVIOUS collection, so the window's first collection - which has no previous one to difference against - carries null rates: unknowable, never reported as 0 (unrated_points counts them, unrated_note says why). The hourly rollup route divides by the bucket width and has no such point.")] public static async Task GetProcedureDurationTrend( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -544,7 +544,7 @@ public static async Task GetProcedureDurationTrend( } } - [McpServerTool(Name = "get_query_store_duration_trend"), Description("Gets a time-series of Query Store duration per second and executions per second over time, summed across every query. Where get_query_duration_trend reads the plan cache and loses everything an eviction or a restart takes with it, this reads Query Store, which persists per interval - so it is the series that survives a failover and the one to reach for when a regression is older than the cache. Each interval is counted once, at the hour the work ran.")] + [McpServerTool(Name = "get_query_store_duration_trend"), Description("Gets a time-series of Query Store duration per second and executions per second over time, summed across every query. Where get_query_duration_trend reads the plan cache and loses everything an eviction or a restart takes with it, this reads Query Store, which persists per interval - so it is the series that survives a failover and the one to reach for when a regression is older than the cache. Each interval is counted once, at the hour the work ran. Every point is a rate over the gap since the PREVIOUS point (interval start or rollup bucket), so the window's first point - which has no previous one to difference against - carries null rates: unknowable, never reported as 0 (unrated_points counts them, unrated_note says why).")] public static async Task GetQueryStoreDurationTrend( NpgsqlDataSource postgres, [Description("Server name or display name.")] string? server_name = null, @@ -757,6 +757,14 @@ private static string SerializeTrend( ["hours_back"] = hours_back, }; disclosure.WriteTo(envelope); + /* #3541 A12: a point with no rate is published as null, never as 0, and the envelope says how many + and why. On the raw route the window's first collection has no previous one to difference + against; the hourly route divides by the bucket width and produces none. */ + var unrated = points.Count(p => !p.HasRate); + envelope["unrated_points"] = unrated; + envelope["unrated_note"] = unrated == 0 + ? null + : $"{unrated} point(s) carry null rates: a per-collection rate is the work since the PREVIOUS collection divided by the seconds between them, and the window's first collection has no previous one inside the window (a collection landing in the same second as its predecessor has no denominator either). Unknowable is not 0 — the point is kept so effective_start is the first collection the store held, and its rates are null."; envelope["trend"] = points.Select(p => new { time = p.CollectionTime.ToString("o"), diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingObjectStatsReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingObjectStatsReader.cs index 4da3a897b..c8c849e5d 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingObjectStatsReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingObjectStatsReader.cs @@ -37,10 +37,58 @@ internal static class DarlingObjectStatsReader { /* ─────────────────────────── result rows ─────────────────────────── */ - /// One per-table size + growth row (indexes rolled up per table). + /// + /// One per-table size + growth row (indexes rolled up per table), carrying the raw baselines the growth + /// figures are derived from rather than the derived figures alone (#3541 A12, contract rule 5). + /// The SQL this replaced computed growth_7d / growth_30d / growth_pct_30d through a + /// COALESCE(p30, p7, oldest, current) chain, which is three lies in one expression: with ten days + /// of history "30-day growth" was growth since the SEVEN-day snapshot; with two days it was growth since + /// the oldest snapshot, still labelled 30d; and a table absent from every baseline (created this week) + /// fell through to current - current = 0, "not growing", for the one table that is nothing BUT + /// growth. A nominal window the store cannot reach is not a smaller window — it is no measurement, and + /// the payload has to say so. So the row carries each baseline as the store holds it (null where the + /// snapshot exists but the table was not in it, or where no snapshot old enough exists) plus the + /// store's span, and the derivations live in the properties below where each can refuse. + /// + /// The table's reserved MB at the newest snapshot at or before the 7-day cutoff; null when + /// no such snapshot exists or the table was not in it. + /// Same for the 30-day cutoff. + /// The table's reserved MB at the store's EARLIEST snapshot; null when the table was not + /// in it (created since). + /// The snapshot the 7-day baseline was read from; null when the store holds nothing that old. + /// Same for 30 days. + /// The store's oldest index_object_stats capture for this server. + /// The store's newest — the snapshot every current_* figure is read from. + /// Whole calendar days between the earliest and latest snapshots — how much history the + /// growth figures can honestly span. 0 means one day of snapshots: no growth is knowable. public sealed record ObjectSizeGrowthRow( string DatabaseName, string SchemaName, string TableName, double CurrentReservedMb, double CurrentUsedMb, - long TotalRows, int IndexCount, double Growth7dMb, double Growth30dMb, double DailyGrowthRateMb, double GrowthPct30d); + long TotalRows, int IndexCount, + double? ReservedMb7dAgo, double? ReservedMb30dAgo, double? ReservedMbOldest, + DateTime? Snapshot7dTime, DateTime? Snapshot30dTime, DateTime EarliestSnapshotTime, DateTime LatestSnapshotTime, int DaysOfData) + { + /// Growth since the 7-day baseline; null when there is no such baseline for this table. + public double? Growth7dMb => ReservedMb7dAgo is { } b ? CurrentReservedMb - b : null; + + /// Growth since the 30-day baseline; null when there is no such baseline for this table. + public double? Growth30dMb => ReservedMb30dAgo is { } b ? CurrentReservedMb - b : null; + + /// Percent growth over the 30-day baseline; null without a baseline, and null when the baseline + /// is 0 (no denominator — a table that was empty 30 days ago has no ratio, not an infinite one). + public double? GrowthPct30d => ReservedMb30dAgo is > 0 ? (CurrentReservedMb - ReservedMb30dAgo.Value) * 100.0 / ReservedMb30dAgo.Value : null; + + /// Growth since the store's earliest snapshot — the honest figure when the nominal windows + /// are out of reach. Null when the store holds a single day (no span) or the table was not in the + /// earliest snapshot. + public double? GrowthOverAvailableHistoryMb => DaysOfData >= 1 && ReservedMbOldest is { } o ? CurrentReservedMb - o : null; + + /// Percent form of ; null on a 0 baseline. + public double? GrowthOverAvailableHistoryPct => + DaysOfData >= 1 && ReservedMbOldest is > 0 ? (CurrentReservedMb - ReservedMbOldest.Value) * 100.0 / ReservedMbOldest.Value : null; + + /// MB per day over the available span; null when there is no span to divide by. + public double? DailyGrowthRateMb => GrowthOverAvailableHistoryMb is { } g ? g / DaysOfData : null; + } /// One per-index usage row with its Unused / Write-only / Active classification. public sealed record IndexUsageRow( @@ -63,17 +111,26 @@ public sealed record DatabaseSizeRow( /// /// Per-table size + growth over the daily snapshots — Lite's GetObjectSizeGrowthAsync ported to - /// Postgres: roll indexes up per (database, schema, table) at the latest snapshot, compare against the - /// newest snapshot at/older-than the 7-day ($2) and 30-day ($3) cutoffs (and the earliest snapshot as a - /// fallback), and derive the daily rate from the span of collected data. Ranks by current reserved size - /// descending, cap $4. $1 server_id. + /// Postgres: roll indexes up per (database, schema, table) at the latest snapshot, and read the same + /// table's reserved size at the newest snapshot at/older-than the 7-day ($2) and 30-day ($3) cutoffs and + /// at the store's earliest snapshot. Ranks by current reserved size descending, cap $4. $1 server_id. + /// Baselines are projected RAW, not folded (#3541 A12). The previous shape derived the growth + /// columns in SQL through COALESCE(p30, p7, oldest, current), so a baseline the store did not hold + /// was silently replaced by a nearer one and labelled with the farther window's name — and a table in no + /// baseline at all read as growth 0. Each baseline now comes back as its own nullable column, beside the + /// snapshot time it was read from and the store's span, and derives + /// each growth figure from exactly the baseline it names or declines to. The two cutoff snapshots are + /// resolved once in boundaries with FILTER so the baseline CTEs and the projected snapshot + /// times cannot disagree about which capture was used. /// public const string ObjectSizeGrowthSql = """ WITH boundaries AS ( SELECT MAX(collection_time) AS latest_time, MIN(collection_time) AS earliest_time, - CAST(MAX(collection_time) AS date) - CAST(MIN(collection_time) AS date) AS days_of_data + CAST(MAX(collection_time) AS date) - CAST(MIN(collection_time) AS date) AS days_of_data, + MAX(collection_time) FILTER (WHERE collection_time <= $2) AS snapshot_7d_time, + MAX(collection_time) FILTER (WHERE collection_time <= $3) AS snapshot_30d_time FROM v_index_object_stats WHERE server_id = $1 ), @@ -90,15 +147,13 @@ FROM v_index_object_stats past_7d AS ( SELECT database_name, schema_name, table_name, SUM(reserved_mb) AS reserved_mb FROM v_index_object_stats - WHERE server_id = $1 AND collection_time = ( - SELECT MAX(collection_time) FROM v_index_object_stats WHERE server_id = $1 AND collection_time <= $2) + WHERE server_id = $1 AND collection_time = (SELECT snapshot_7d_time FROM boundaries) GROUP BY database_name, schema_name, table_name ), past_30d AS ( SELECT database_name, schema_name, table_name, SUM(reserved_mb) AS reserved_mb FROM v_index_object_stats - WHERE server_id = $1 AND collection_time = ( - SELECT MAX(collection_time) FROM v_index_object_stats WHERE server_id = $1 AND collection_time <= $3) + WHERE server_id = $1 AND collection_time = (SELECT snapshot_30d_time FROM boundaries) GROUP BY database_name, schema_name, table_name ), oldest AS ( @@ -115,15 +170,14 @@ FROM v_index_object_stats CAST(l.current_used_mb AS double precision) AS current_used_mb, l.total_rows, l.index_count, - CAST(l.current_reserved_mb - COALESCE(p7.reserved_mb, o.reserved_mb, l.current_reserved_mb) AS double precision) AS growth_7d_mb, - CAST(l.current_reserved_mb - COALESCE(p30.reserved_mb, p7.reserved_mb, o.reserved_mb, l.current_reserved_mb) AS double precision) AS growth_30d_mb, - CASE WHEN b.days_of_data >= 1 - THEN CAST(l.current_reserved_mb - COALESCE(o.reserved_mb, l.current_reserved_mb) AS double precision) / CAST(b.days_of_data AS double precision) - ELSE 0 END AS daily_growth_rate_mb, - CASE WHEN COALESCE(p30.reserved_mb, p7.reserved_mb, o.reserved_mb) > 0 - THEN CAST(l.current_reserved_mb - COALESCE(p30.reserved_mb, p7.reserved_mb, o.reserved_mb) AS double precision) * 100.0 - / CAST(COALESCE(p30.reserved_mb, p7.reserved_mb, o.reserved_mb) AS double precision) - ELSE 0 END AS growth_pct_30d + CAST(p7.reserved_mb AS double precision) AS reserved_mb_7d_ago, + CAST(p30.reserved_mb AS double precision) AS reserved_mb_30d_ago, + CAST(o.reserved_mb AS double precision) AS reserved_mb_oldest, + b.snapshot_7d_time, + b.snapshot_30d_time, + b.earliest_time, + b.latest_time, + b.days_of_data FROM latest l CROSS JOIN boundaries b LEFT JOIN past_7d p7 ON p7.database_name = l.database_name AND p7.schema_name = l.schema_name AND p7.table_name = l.table_name @@ -154,10 +208,15 @@ public static async Task> GetObjectSizeGrowthAsync( reader.IsDBNull(4) ? 0 : reader.GetDouble(4), reader.IsDBNull(5) ? 0 : reader.GetInt64(5), reader.IsDBNull(6) ? 0 : Convert.ToInt32(reader.GetValue(6)), - reader.IsDBNull(7) ? 0 : reader.GetDouble(7), - reader.IsDBNull(8) ? 0 : reader.GetDouble(8), - reader.IsDBNull(9) ? 0 : reader.GetDouble(9), - reader.IsDBNull(10) ? 0 : reader.GetDouble(10))); + /* The baselines stay NULL when the store has none — a missing baseline is not a 0 baseline. */ + reader.IsDBNull(7) ? null : reader.GetDouble(7), + reader.IsDBNull(8) ? null : reader.GetDouble(8), + reader.IsDBNull(9) ? null : reader.GetDouble(9), + reader.IsDBNull(10) ? null : reader.GetDateTime(10), + reader.IsDBNull(11) ? null : reader.GetDateTime(11), + reader.GetDateTime(12), + reader.GetDateTime(13), + Convert.ToInt32(reader.GetValue(14)))); } return rows; diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingQueryStoreRegressionReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingQueryStoreRegressionReader.cs index dd481e38a..c43e404f0 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingQueryStoreRegressionReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingQueryStoreRegressionReader.cs @@ -38,19 +38,26 @@ internal static class DarlingQueryStoreRegressionReader { /// One regression row - the viewer's ViewerQueryStoreRegressionRow, without the /// display-formatting members. Durations and CPU are ms (converted from the stored microseconds); - /// reads are raw pages; the percents are plain deltas. + /// reads are raw pages; the percents are plain deltas. + /// The three percents are NULLABLE (#3541 A12, contract rule 5). Each divides through + /// NULLIF(baseline, 0), so a query whose baseline side is 0 — no logical reads in every capture + /// before the window, then 50,000 per execution inside it — has no ratio: the SQL returns NULL, and the + /// reader used to coerce it to 0, which published the most dramatic possible I/O regression as + /// io_regression_percent: 0, "no change". NULL stays NULL here and the tool says why. The CPU + /// percent cannot actually arrive NULL (the WHERE gate > 25 drops a NULL comparison), but it is + /// typed like its siblings so the three cannot drift in how they treat a missing denominator. public sealed record RegressionRow( string DatabaseName, long QueryId, double BaselineDurationMs, double RecentDurationMs, - double DurationRegressionPercent, + double? DurationRegressionPercent, double BaselineCpuMs, double RecentCpuMs, - double CpuRegressionPercent, + double? CpuRegressionPercent, double BaselineReads, double RecentReads, - double IoRegressionPercent, + double? IoRegressionPercent, double AdditionalDurationMs, long BaselineExecCount, long RecentExecCount, @@ -233,13 +240,13 @@ public static async Task> GetQueryStoreRegressionsAsync( reader.IsDBNull(1) ? 0 : reader.GetInt64(1), reader.IsDBNull(2) ? 0 : Convert.ToDouble(reader.GetValue(2)), reader.IsDBNull(3) ? 0 : Convert.ToDouble(reader.GetValue(3)), - reader.IsDBNull(4) ? 0 : Convert.ToDouble(reader.GetValue(4)), + reader.IsDBNull(4) ? null : Convert.ToDouble(reader.GetValue(4)), reader.IsDBNull(5) ? 0 : Convert.ToDouble(reader.GetValue(5)), reader.IsDBNull(6) ? 0 : Convert.ToDouble(reader.GetValue(6)), - reader.IsDBNull(7) ? 0 : Convert.ToDouble(reader.GetValue(7)), + reader.IsDBNull(7) ? null : Convert.ToDouble(reader.GetValue(7)), reader.IsDBNull(8) ? 0 : Convert.ToDouble(reader.GetValue(8)), reader.IsDBNull(9) ? 0 : Convert.ToDouble(reader.GetValue(9)), - reader.IsDBNull(10) ? 0 : Convert.ToDouble(reader.GetValue(10)), + reader.IsDBNull(10) ? null : Convert.ToDouble(reader.GetValue(10)), reader.IsDBNull(11) ? 0 : Convert.ToDouble(reader.GetValue(11)), reader.IsDBNull(12) ? 0 : reader.GetInt64(12), reader.IsDBNull(13) ? 0 : reader.GetInt64(13), diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingSystemHealthReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingSystemHealthReader.cs index 07ac30db9..3bce88bba 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingSystemHealthReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingSystemHealthReader.cs @@ -93,35 +93,78 @@ public static async Task> ReadEventXmlAsync( } /// - /// Whether this server has EVER recorded a system_health event of one type, ignoring any window. - /// Lets an empty parse-on-read result say WHICH kind of nothing it found. Zero significant rows - /// is true both of a healthy window and of a server whose system_health events were never collected, - /// and the two want opposite responses -- widen the window, versus go find out why nothing is being - /// captured. Probes v_system_health_events, the SAME source - /// reads, so it cannot report a server as captured for rows - /// the read itself can never see. Scoped to the event_type because that is the granularity the caller - /// asked about: a server capturing sp_server_diagnostics but no wait_info has not been sampled for - /// waits, whatever its other categories hold. LIMIT 1, so it stops at the first row. - /// $1 server_id, $2 event_type. + /// The newest collection_time at which the system_health collector stored ANY event for this + /// server — the source witness every one of the nine parse-on-read tools publishes (#3541 A12, contract + /// rule 5: zero is a measurement). + /// Eight of the nine tools answered a dead system_health session, or a collector that had + /// never run, with the same empty a healthy quiet window earns, and "no severe errors" from a + /// server nothing was ever read from is a clean bill of health nobody issued. The witness is the + /// events view itself, NOT collection_log: the log records a SUCCESS for a run that read a dead + /// session and stored nothing, which is exactly the shape being mis-reported, whereas a stored event is + /// proof the session was alive and the collector reached it. Windowless and type-less on purpose — it + /// answers "has this server's ring buffer ever been read into the store", which is the question a + /// category with no rows in the window needs answered first; the type-scoped question is + /// . Reads the SAME view the tools read, for the #2484 reason: a + /// probe on another relation could report a source as observed for rows the read itself can never + /// see. + /// Cheap by shape: MAX(collection_time) under server_id = $1 is a backward walk of the + /// (server_id, collection_time) index that stops at the first row, so it rides on the data path + /// of every call and not only on the empty one. Anchored by construction — it names no clock; it is a + /// fact about the store, and a caller anchored in the past receives the store's newest capture, + /// which may be later than its window and is labelled as the collector's, not the window's. + /// $1 server_id. /// - public const string HasAnyEventOfTypeSql = """ - SELECT 1 + public const string LastCaptureSql = """ + SELECT MAX(collection_time) + FROM v_system_health_events + WHERE server_id = $1 + AND event_xml IS NOT NULL + """; + + /// + /// The newest collection_time at which an event of ONE type was stored for this server — the + /// type-scoped half of the witness, run only when a window came back with no events of that type. + /// Separates "this category has fired before, the window is quiet" (widen) from "this category + /// has never fired here while the session IS being read" — which for a rare category (a memory-node + /// OOM, a severe error) is the healthy measurement, not a blind spot. Same view as the read, same + /// event_xml IS NOT NULL guard, same backward index walk with a type filter — it stops at the + /// first match for a type that exists and walks the server's rows for one that never did, the cost the + /// #2484 HasAnyEventOfTypeSql probe this replaces already paid on the same path (that probe + /// answered only yes/no; this one also says WHEN, which is what the message needs). + /// $1 server_id, $2 event_type. + /// + public const string LastCaptureOfTypeSql = """ + SELECT MAX(collection_time) FROM v_system_health_events WHERE server_id = $1 AND event_type = $2 AND event_xml IS NOT NULL - LIMIT 1 """; - /// Runs . - public static async Task HasAnyEventOfTypeAsync( + /// Runs : null when no system_health event of any type has ever + /// been stored for the server. + public static Task GetLastCaptureAsync( + NpgsqlDataSource postgres, int serverId, CancellationToken cancellationToken = default) + => ReadNullableTimestampAsync(postgres, LastCaptureSql, serverId, eventType: null, cancellationToken); + + /// Runs : null when no event of + /// has ever been stored for the server. + public static Task GetLastCaptureOfTypeAsync( NpgsqlDataSource postgres, int serverId, string eventType, CancellationToken cancellationToken = default) + => ReadNullableTimestampAsync(postgres, LastCaptureOfTypeSql, serverId, eventType, cancellationToken); + + private static async Task ReadNullableTimestampAsync( + NpgsqlDataSource postgres, string sql, int serverId, string? eventType, CancellationToken cancellationToken) { - await using var command = postgres.CreateCommand(HasAnyEventOfTypeSql); + await using var command = postgres.CreateCommand(sql); command.CommandTimeout = McpCommandDeadlines.ReadSeconds; DarlingMcpReadParameters.AddInt(command, serverId); - DarlingMcpReadParameters.AddText(command, eventType); - return await command.ExecuteScalarAsync(cancellationToken) is not null; + if (eventType is not null) + DarlingMcpReadParameters.AddText(command, eventType); + /* MAX over zero rows is one row holding SQL NULL, which Npgsql surfaces as DBNull — the aggregate + never returns no rows, so the null check is on the value rather than on the row. */ + var value = await command.ExecuteScalarAsync(cancellationToken); + return value is DateTime stamp ? stamp : null; } /// Loads the server's latest database_id → database_name map for Severe Errors DB resolution. diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingTrendReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingTrendReader.cs index 5b1b8689f..0d9986840 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingTrendReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingTrendReader.cs @@ -84,7 +84,15 @@ public sealed record FileIoLatencyTrendPoint( /// break; new readers should take ExecutionsPerSecond. /// public sealed record QueryDurationTrendPoint( - DateTime CollectionTime, double Value, long ExecutionCount, double ExecutionsPerSecond); + DateTime CollectionTime, double? Value, long? ExecutionCount, double? ExecutionsPerSecond) + { + /// + /// Whether this point carries a rate at all (#3541 A12). False for the window's first differenced + /// collection — no previous collection to difference against — and for a collection landing in the + /// same second as its predecessor; both have no denominator, and neither is 0. + /// + public bool HasRate => Value.HasValue; + } /// One point of a single query's per-collection history (Lite's QueryStatsHistoryRow, /// the columns get_query_trend surfaces): the interval deltas + DOP spread + the plan hash. Time metrics @@ -382,11 +390,24 @@ public static async Task> GetFileIoLatencyTrendAsy /// GetQueryDurationTrendAsync): per collection, the summed delta_elapsed_time (→ ms) and /// delta_execution_count divided by the seconds since the previous collection (the truncate-then- /// diff LAG epoch idiom proven value-identical DuckDB↔Postgres) for an elapsed-ms/sec + executions/sec - /// rate. The first row's LAG is NULL → interval NULL → the CASE yields 0 (Lite's behaviour). Reads the - /// base query_stats table because it projects no text — a read that wanted query_text or - /// query_plan_xml would have to go through v_query_stats to resolve the #1767 payload - /// dimensions. Summed bigints come back as numeric, so the reads Convert tolerantly. $1 server_id, - /// $2/$3 window (naive UTC). + /// rate. Reads the base query_stats table because it projects no text — a read that wanted + /// query_text or query_plan_xml would have to go through v_query_stats to resolve the + /// #1767 payload dimensions. Summed bigints come back as numeric, so the reads Convert tolerantly. + /// $1 server_id, $2/$3 window (naive UTC). + /// + /// The first collection in the window has no rate (#3541 A12, #3540 A8). Its LAG is NULL + /// — there is no previous collection inside the window to difference against — so its rate is + /// unknowable, and the shape this replaced (CASE ... ELSE 0 END, Lite's original behaviour) + /// published that unknowable as a measured 0.0: every duration series began with a fabricated quiet + /// instant, which an agent charting the window read as "idle, then busy" and which dragged every + /// first-bucket average toward zero. Contract rule 5 — zero is a measurement — so the CASE has no ELSE + /// and the rate columns are NULL for that row (and for the degenerate two-collections-in-one-second + /// case, whose denominator is 0 and whose rate is equally undefined). The row is KEPT rather than + /// filtered, deliberately: the collection happened, effective_start is truthfully its instant, + /// and a window holding exactly one collection is "one collection, no rate yet" rather than an empty + /// series the empty ladder would mis-describe as a quiet window. The reader carries the nulls through + /// () and the tool publishes them with the reason. The hourly + /// twin below has no such row: its denominator is the bucket width, known for every bucket. /// public const string QueryDurationTrendSql = """ WITH raw AS @@ -404,8 +425,8 @@ GROUP BY collection_time ) SELECT collection_time, - CASE WHEN interval_seconds > 0 THEN total_elapsed_ms / interval_seconds ELSE 0 END AS elapsed_ms_per_second, - CASE WHEN interval_seconds > 0 THEN CAST(total_executions AS DOUBLE PRECISION) / interval_seconds ELSE 0 END AS executions_per_second + CASE WHEN interval_seconds > 0 THEN total_elapsed_ms / interval_seconds END AS elapsed_ms_per_second, + CASE WHEN interval_seconds > 0 THEN CAST(total_executions AS DOUBLE PRECISION) / interval_seconds END AS executions_per_second FROM raw ORDER BY collection_time """; @@ -430,7 +451,8 @@ ORDER BY collection_time /// from the gap to the previous collection because a per-sweep row carries no shared interval (the /// measurement lane's A11a); an hour bucket's width is KNOWN, so this read divides by /// and every bucket has a real denominator — the LAG idiom's first - /// point, whose interval is NULL and whose rate is therefore a fabricated 0, does not exist here. The + /// collection, whose interval is NULL and whose rate is therefore unknowable (published as NULL by the raw + /// read since #3541 A12; as a fabricated 0 before it), does not exist here. The /// trade is stated in the payload rather than hidden: an hour the collector covered only partly (a /// service restart mid-hour, a gap past the delta policy) reads LOW, never high, because its summed work is /// still spread over the full 3,600 seconds. @@ -597,10 +619,11 @@ public static Task GetQueryDurationTrendAsync( /// #3540 (V128): the interval is the collection's STORED one where the rows have it — MAX /// over the collection's rows, because a plan first seen in an otherwise steady pass carries 0 beside /// its siblings' real interval and contributes 0 to the sums; MAX is 0 only when EVERY row was - /// unknowable (a restart), and that 0 becomes NULL through NULLIF so the rates are NULL and the - /// reader drops the point rather than rendering 0.00 ms/sec. NULL (a pre-V128 collection) falls back to - /// the LAG this read always used. No ELSE 0. Verbatim from the viewer's copy apart from the - /// database filter, as before. + /// unknowable (a restart), and that 0 becomes NULL through NULLIF so the rates are NULL — an + /// UNRATED point the reader keeps rather than rendering 0.00 ms/sec (#3541 A12; see + /// for why the row stays). NULL (a pre-V128 collection) falls back to + /// the LAG this read always used, whose first row is likewise unrated, never a fabricated 0. No + /// ELSE 0. Verbatim from the viewer's copy apart from the database filter, as before. /// public const string ProcedureDurationTrendSql = """ WITH raw AS @@ -638,7 +661,9 @@ ORDER BY collection_time /// interval start exists for them and none can be reconstructed. The arms split on /// interval_start_time_utc IS NULL, so they partition the rows with no overlap and no gap. /// Rewriting either arm here would make the browser and the desktop viewer disagree about the same - /// hour. $1 server_id, $2/$3 window (naive UTC). + /// hour. The first placed interval in the window carries NULL rates, not 0 — see + /// (#3541 A12); the rollup route's builder applies the same rule to + /// its first bucket. $1 server_id, $2/$3 window (naive UTC). /// #2736: this is now the FALLBACK, not the read. The rank-over-raw below costs the whole /// slab regardless of the window, which exceeds the mcp role's statement_timeout on a large store — /// so on stores with a materialized query_store_stats_corrected_hourly the tool routes through @@ -707,8 +732,8 @@ GROUP BY point_time ) SELECT point_time AS collection_time, - CASE WHEN interval_seconds > 0 THEN total_duration_ms / interval_seconds ELSE 0 END AS duration_ms_per_second, - CASE WHEN interval_seconds > 0 THEN CAST(total_executions AS DOUBLE PRECISION) / interval_seconds ELSE 0 END AS executions_per_second + CASE WHEN interval_seconds > 0 THEN total_duration_ms / interval_seconds END AS duration_ms_per_second, + CASE WHEN interval_seconds > 0 THEN CAST(total_executions AS DOUBLE PRECISION) / interval_seconds END AS executions_per_second FROM raw ORDER BY point_time """; @@ -806,19 +831,14 @@ private static async Task> ReadDurationPointsAsync await using var reader = await command.ExecuteReaderAsync(cancellationToken); while (await reader.ReadAsync(cancellationToken)) { - /* A NULL rate is an unknowable interval (#3540, V128): the point is dropped, not read as 0. Only - the procedure trend emits one today (its query-stats and Query Store siblings keep ELSE 0), so - this is a no-op for them and the missing-sample posture for it. */ - if (reader.IsDBNull(1)) - { - continue; - } - - var executionsPerSecond = reader.IsDBNull(2) ? 0 : Convert.ToDouble(reader.GetValue(2)); + /* NULL stays NULL (#3541 A12): no rate for the window's first collection, or for a collection whose + stored interval was unknowable (a restart pass, #3540 V128) — either way the point is kept as + UNRATED rather than dropped or coerced to the fabricated quiet the SQL stopped producing. */ + var executionsPerSecond = reader.IsDBNull(2) ? (double?)null : Convert.ToDouble(reader.GetValue(2)); items.Add(new QueryDurationTrendPoint( reader.GetDateTime(0), - Convert.ToDouble(reader.GetValue(1)), - (long)executionsPerSecond, + reader.IsDBNull(1) ? null : Convert.ToDouble(reader.GetValue(1)), + executionsPerSecond is { } eps ? (long)eps : null, executionsPerSecond)); } diff --git a/Darling/PerformanceMonitor.Darling.Storage/DarlingPgXminReader.cs b/Darling/PerformanceMonitor.Darling.Storage/DarlingPgXminReader.cs index b611e15ea..c4ea096a9 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/DarlingPgXminReader.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/DarlingPgXminReader.cs @@ -78,6 +78,52 @@ FROM latest AS l ORDER BY l.xmin_age DESC """; + /// + /// How many times the xmin collector actually CAPTURED in the window — the honest denominator for + /// "what share of the window was this source winning" (#3541 A12, contract rule 5). + /// 's samples counts a source's OWN rows, and the collector + /// writes a row only when something holds the horizon — an unheld capture stores nothing. So a source + /// that held the horizon in 2 of the window's 288 captures had samples = 2, + /// samples_as_winner = 2, and read as winning 100% of the window: a two-minute query rendered as + /// a chronic holder. The denominator has to be every time the collector LOOKED, and only + /// collection_log has that: one row per run INCLUDING the zero-row (healthy, unheld) runs, behind + /// its (server_id, collection_time) index, counting the exact collector whose captures are being + /// fractioned. Runs that stored nothing because they could not look (ERROR / ABANDONED / PERMISSIONS / + /// YIELDED) are excluded: a cycle that did not look is not evidence the horizon was clear. + /// This is the SAME denominator the alert evaluator's horizon arm uses + /// (DarlingPostgresAlertReadAdapter.XminSql's captures CTE, #3537): same table, same + /// collector name, same SUCCESS filter — pinned equal by DarlingPgXminReaderTests so the MCP payload and + /// the alert can never fraction the same window over different denominators. A separate statement rather + /// than a CROSS JOIN onto the holder rows because the viewer renders field for + /// field and this is a window fact, not a row fact. The log write is failure-isolated and can skip a + /// row, so the count may UNDERCOUNT — the payload says so rather than clamping the share. + /// $1 server_id, $2/$3 window (naive UTC). + /// + public const string XminCapturesInWindowSql = """ + SELECT COUNT(*) AS captures_in_window + FROM collection_log + WHERE server_id = $1 + AND collector_name = 'pg_xmin_horizon' + AND collection_time >= $2 + AND collection_time <= $3 + AND status = 'SUCCESS' + """; + + /// Runs . + public static async Task GetXminCapturesInWindowAsync( + NpgsqlDataSource postgres, int serverId, DateTime startUtc, DateTime endUtc, + CancellationToken cancellationToken = default) + { + await using var command = postgres.CreateCommand(XminCapturesInWindowSql); + command.CommandTimeout = StorageCommandDeadlines.McpReadSeconds; + command.Parameters.AddWithValue(serverId); + /* Kind-Unspecified at the bind, for the reason GetPgXminHorizonAsync states. */ + command.Parameters.AddWithValue(DateTime.SpecifyKind(startUtc, DateTimeKind.Unspecified)); + command.Parameters.AddWithValue(DateTime.SpecifyKind(endUtc, DateTimeKind.Unspecified)); + var value = await command.ExecuteScalarAsync(cancellationToken); + return value is long count ? count : Convert.ToInt64(value); + } + public static async Task> GetPgXminHorizonAsync( NpgsqlDataSource postgres, int serverId, DateTime startUtc, DateTime endUtc, CancellationToken cancellationToken = default) diff --git a/Darling/PerformanceMonitor.Darling.Storage/QueryStoreTrendRouting.cs b/Darling/PerformanceMonitor.Darling.Storage/QueryStoreTrendRouting.cs index 4ac3166f0..dc3293087 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/QueryStoreTrendRouting.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/QueryStoreTrendRouting.cs @@ -285,8 +285,12 @@ FROM united ) SELECT point_time AS collection_time, - CASE WHEN interval_seconds > 0 THEN total_duration_ms / interval_seconds ELSE 0 END AS duration_ms_per_second, - CASE WHEN interval_seconds > 0 THEN CAST(total_executions AS DOUBLE PRECISION) / interval_seconds ELSE 0 END AS executions_per_second + /* No ELSE: the first united point's LAG is NULL and its rate unknowable, so the rate is NULL — never a + fabricated 0 (#3541 A12). Shared by the MCP reader and the viewer, so both surfaces see the same + first bucket the same way: the MCP payload publishes it as an unrated point, the viewer's chart + reader skips it (a chart has nowhere to draw "unknown"). */ + CASE WHEN interval_seconds > 0 THEN total_duration_ms / interval_seconds END AS duration_ms_per_second, + CASE WHEN interval_seconds > 0 THEN CAST(total_executions AS DOUBLE PRECISION) / interval_seconds END AS executions_per_second FROM rated ORDER BY point_time """; diff --git a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.QueryTrends.cs b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.QueryTrends.cs index 6e03e7ac5..4cb214856 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.QueryTrends.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.QueryTrends.cs @@ -286,10 +286,16 @@ public async Task> GetQueryStoreDurationTrendAsync( await using var reader = await command.ExecuteReaderAsync(cancellationToken); while (await reader.ReadAsync(cancellationToken)) { + /* #3541 A12: the shared builder returns NULL rates for the window's first united point (its LAG + has nothing to difference against). A chart has nowhere to draw "unknown", so the point is + skipped here rather than coerced to the 0 it used to be plotted as. */ + if (reader.IsDBNull(1)) + continue; + items.Add(new QueryTrendPoint { CollectionTime = reader.GetDateTime(0), - Value = reader.IsDBNull(1) ? 0 : Convert.ToDouble(reader.GetValue(1)), + Value = Convert.ToDouble(reader.GetValue(1)), ExecutionCount = reader.IsDBNull(2) ? 0 : (long)Convert.ToDouble(reader.GetValue(2)), }); } diff --git a/Lite.Tests/DeltaFamilyUnknowableRowReadTests.cs b/Lite.Tests/DeltaFamilyUnknowableRowReadTests.cs index 2d0dada51..2dae155de 100644 --- a/Lite.Tests/DeltaFamilyUnknowableRowReadTests.cs +++ b/Lite.Tests/DeltaFamilyUnknowableRowReadTests.cs @@ -241,13 +241,15 @@ public async Task LatchAndSpinlockTrends_DropTheUnknowableRow_PreferTheStoredInt /// /// #3540 (v61): the procedure duration trend, the read that LAG-divided procedure_stats' fabricated /// zero into a confident 0.00 ms/sec. Four collections five minutes apart: t1/t2 are pre-v61 collections - /// (NULL interval) — t1 has no prior and is not a point, t2 divides by the LAG's 300 s. t3 is a restart: - /// every row stores 0, so MAX is 0 and the collection is absent. t4 is a steady pass with a plan the - /// TOP (150) just readmitted (its row stores 0 beside a 0 delta) beside a measured row (120 s), so MAX is - /// 120 — the stored interval wins over the LAG's 300 — and the readmitted plan adds nothing to the sums. + /// (NULL interval) — t1 has no prior and is UNRATED (a point with null rates, #3541 A12: kept rather than + /// dropped so a lone collection is never an empty series and the MCP payload's effective_start is the + /// first collection the store held), t2 divides by the LAG's 300 s. t3 is a restart: every row stores 0, + /// so MAX is 0 and the collection is likewise unrated — never 0.00 ms/sec. t4 is a steady pass with a plan + /// the TOP (150) just readmitted (its row stores 0 beside a 0 delta) beside a measured row (120 s), so MAX + /// is 120 — the stored interval wins over the LAG's 300 — and the readmitted plan adds nothing to the sums. /// [Fact] - public async Task ProcedureDurationTrend_DropsTheUnknowableCollection_PrefersTheStoredInterval_KeepsPreV61History() + public async Task ProcedureDurationTrend_LeavesTheUnknowableCollectionUnrated_PrefersTheStoredInterval_KeepsPreV61History() { var t1 = Truncate(DateTime.UtcNow.AddHours(-2)); var t2 = t1.AddMinutes(5); @@ -262,15 +264,23 @@ public async Task ProcedureDurationTrend_DropsTheUnknowableCollection_PrefersThe var points = await _dataService.GetProcedureDurationTrendAsync(ServerId, hoursBack: 3); - Assert.Equal(new[] { t2, t4 }, points.Select(p => p.CollectionTime).ToArray()); + Assert.Equal(new[] { t1, t2, t3, t4 }, points.Select(p => p.CollectionTime).ToArray()); + + /* t1 (no prior) and t3 (restart marker): present, unrated — null, never 0. */ + Assert.False(points[0].HasRate); + Assert.Null(points[0].Value); + Assert.Null(points[0].ExecutionsPerSecond); + Assert.False(points[2].HasRate); + Assert.Null(points[2].Value); + Assert.Null(points[2].ExecutionCount); /* t2 (pre-v61): 600 ms / 300 s = 2.0 ms/sec; 30 / 300 = 0.1 executions/sec. */ - Assert.Equal(2.0, points[0].Value, precision: 6); - Assert.Equal(0.1, points[0].ExecutionsPerSecond, precision: 6); + Assert.Equal(2.0, points[1].Value!.Value, precision: 6); + Assert.Equal(0.1, points[1].ExecutionsPerSecond!.Value, precision: 6); /* t4: the STORED 120 s — 1200 / 120 = 10.0, not the LAG's 1200 / 300 = 4.0; 24 / 120 = 0.2. */ - Assert.Equal(10.0, points[1].Value, precision: 6); - Assert.Equal(0.2, points[1].ExecutionsPerSecond, precision: 6); + Assert.Equal(10.0, points[3].Value!.Value, precision: 6); + Assert.Equal(0.2, points[3].ExecutionsPerSecond!.Value, precision: 6); } /// diff --git a/Lite.Tests/EngineCapabilityMissTests.cs b/Lite.Tests/EngineCapabilityMissTests.cs index 82d3b2166..8aafdc5d9 100644 --- a/Lite.Tests/EngineCapabilityMissTests.cs +++ b/Lite.Tests/EngineCapabilityMissTests.cs @@ -108,8 +108,13 @@ quoted. It must no longer tell an Azure caller to start a session that cannot ex Assert.Equal("not_collected", StatusOf(azureTrace)); Assert.Contains("default_trace_events", azureTrace, StringComparison.Ordinal); - /* ── The box, same empty store: every one of them keeps the answer it gave before. ── */ - Assert.Equal("empty", StatusOf(await McpHealthParserTools.GetSystemHealth(service, _serverManager, BoxServerName))); + /* ── The box, same empty store: every one of them keeps its own miss — the ENGINE answer must not + have become a blanket rule. For the health-parser family that own miss is "unavailable" since + #3541 A12 (a server whose system_health session has never been read into the store is not a + clean bill), the answer significant_waits alone used to give and the other eight now share. ── */ + var boxHealth = await McpHealthParserTools.GetSystemHealth(service, _serverManager, BoxServerName); + Assert.Equal("unavailable", StatusOf(boxHealth)); + Assert.Contains("system_health session is started", boxHealth, StringComparison.Ordinal); var boxWaits = await McpHealthParserTools.GetSignificantWaits(service, _serverManager, BoxServerName); Assert.Equal("unavailable", StatusOf(boxWaits)); @@ -136,7 +141,9 @@ public async Task AServerWithNoProbedEdition_KeepsItsOldMiss() var service = new LocalDataService(_duckDb); - Assert.Equal("empty", StatusOf(await McpHealthParserTools.GetSystemHealth(service, _serverManager, BoxServerName))); + /* "Old miss" is each family's own: for the health parsers a never-read session is "unavailable" + (#3541 A12), and the point here is that it is NOT "not_collected" — unknown is not never. */ + Assert.Equal("unavailable", StatusOf(await McpHealthParserTools.GetSystemHealth(service, _serverManager, BoxServerName))); Assert.Equal("empty", StatusOf(await McpDefaultTraceTools.GetDefaultTraceEvents(service, _serverManager, BoxServerName))); Assert.Equal("empty", StatusOf(await McpConfigTools.GetTraceFlags(service, _serverManager, BoxServerName))); } @@ -152,7 +159,7 @@ public async Task AServerWithNoRegistryRow_KeepsItsOldMiss() var service = new LocalDataService(_duckDb); Assert.Equal(CollectorEngineCapability.UnknownEngineEdition, await service.GetSqlEngineEditionAsync(_azureServerId)); - Assert.Equal("empty", StatusOf(await McpHealthParserTools.GetSystemHealth(service, _serverManager, AzureServerName))); + Assert.Equal("unavailable", StatusOf(await McpHealthParserTools.GetSystemHealth(service, _serverManager, AzureServerName))); } private static string StatusOf(string json) => diff --git a/Lite.Tests/IndexObjectStatsTests.cs b/Lite.Tests/IndexObjectStatsTests.cs index e0ab206e5..57eecf307 100644 --- a/Lite.Tests/IndexObjectStatsTests.cs +++ b/Lite.Tests/IndexObjectStatsTests.cs @@ -115,8 +115,19 @@ public async Task ObjectSizeGrowth_ComputesDelta() var big = rows.FirstOrDefault(r => r.TableName == "BigTable"); Assert.NotNull(big); Assert.Equal(600m, big!.CurrentReservedMb); - Assert.Equal(400m, big.Growth30dMb); - Assert.True(big.GrowthPct30d >= 199 && big.GrowthPct30d <= 201); + + /* #3541 A12: one day of history. The 400 MB / 200% this used to assert as Growth30dMb was growth over + ONE day labelled thirty — the store has no 30-day (or 7-day) snapshot, so those figures are null, + and the same delta carries its own name and its real span. */ + Assert.Null(big.Snapshot7dTime); + Assert.Null(big.Snapshot30dTime); + Assert.Null(big.Growth7dMb); + Assert.Null(big.Growth30dMb); + Assert.Null(big.GrowthPct30d); + Assert.Equal(1, big.DaysOfData); + Assert.Equal(400m, big.GrowthOverAvailableHistoryMb); + Assert.Equal(200m, big.GrowthOverAvailableHistoryPct); + Assert.Equal(400m, big.DailyGrowthRateMb); } [Fact] diff --git a/Lite.Tests/McpMissMessageParityPinTests.cs b/Lite.Tests/McpMissMessageParityPinTests.cs index d9ce6b23c..4bdac774f 100644 --- a/Lite.Tests/McpMissMessageParityPinTests.cs +++ b/Lite.Tests/McpMissMessageParityPinTests.cs @@ -188,6 +188,28 @@ a different next step. The negative-span refusal is built by McpHelpers.Validate "): the per-signal tables the health band reads (deadlocks, blocking, CPU, memory, waits) have been purged for that day, so no health verdict is possible and the counts would be zeros by construction, not by measurement. Longer-lived sources may still record the day — collection_runs and alert_count below are real where non-zero.", ". The filter was applied in SQL over the whole window, so this is the window's answer rather than a page artefact — drop parallel_only / min_dop to see the unfiltered ranking, or confirm current parallelism with analyze_query_plan.", ". The filters were applied in SQL over the whole window, so unfiltered snapshots may well exist — drop them to see what the window holds.", + + /* #3541 A12 — the health-parser family's four-rung empty ladder, now one EmptyAsync per SKU that all + nine reads climb. The rung sentences sit between interpolation holes (server, hours, event type, + stamp), so what is compared is the literal text a caller receives. The dead rung keeps the + "system_health session is started" sentence EngineCapabilityMissTests pins on both SKUs. */ + ". Events ARE being captured, so this is the healthy answer for this read rather than missing data.", + "), so the window is genuinely quiet rather than blind — widen hours_back to reach the most recent events.", + " — so for this category the absence is a measurement: the engine has not recorded one. Not a blind spot, and a wider window would not change it.", + "No system_health events of ANY type have EVER been captured for ", + ", so this is NOT an all-clear — there is nothing here to be clear about. This read is served from the collected system_health ring buffer: check that collection is running for this server and that its system_health session is started before concluding nothing happened.", + + /* #3541 A12 — get_query_store_regressions' per-percent reason (a 0 baseline has no ratio), the + trend trio's unrated-point note, get_pvs_stats' three share reasons, and get_table_index_sizes' + history note. Each is emitted by a helper or envelope builder that lives once per SKU. */ + " is null: no_baseline — ", + " is 0, so the ratio has no denominator; this is NOT 0% change. Compare ", + " point(s) carry null rates: a per-collection rate is the work since the PREVIOUS collection divided by the seconds between them, and the window's first collection has no previous one inside the window (a collection landing in the same second as its predecessor has no denominator either). Unknowable is not 0 — the point is kept so effective_start is the first collection the store held, and its rates are null.", + "pvs_size_mb was not reported by sys.dm_tran_persistent_version_store_stats in this capture, so the share is unknown — not zero.", + "database_data_size_mb was not captured for this database, so there is no denominator — the share is unknown, not zero.", + "database_data_size_mb is 0, so the share has no denominator — the share is unknown, not zero.", + " rather than re-measured over a shorter span under the same name. Read growth_over_available_history_* — it spans exactly growth_window_days.", + "the store holds a single day of snapshots for this server, so no growth is knowable yet — every growth figure is null, not 0", }; [Theory] diff --git a/Lite.Tests/McpZeroIsAMeasurementTests.cs b/Lite.Tests/McpZeroIsAMeasurementTests.cs new file mode 100644 index 000000000..d6141ad65 --- /dev/null +++ b/Lite.Tests/McpZeroIsAMeasurementTests.cs @@ -0,0 +1,160 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.IO; +using System.Linq; +using System.Text.Json; +using System.Threading.Tasks; +using DuckDB.NET.Data; +using PerformanceMonitor.Common; +using PerformanceMonitorLite.Database; +using PerformanceMonitorLite.Mcp; +using PerformanceMonitorLite.Models; +using PerformanceMonitorLite.Services; +using Xunit; + +namespace PerformanceMonitorLite.Tests; + +/// +/// Lite's half of #3541 A12 ("zero is a measurement"), against a real DuckDB, through the real tool methods — +/// the twin of Darling's McpZeroIsAMeasurementLivePostgresTests. The source-level census of both SKUs +/// lives in Darling.Tests (which reads Lite's files); this file is the one that EXECUTES Lite's ladder. +/// +/// The health-parser family is the weight-bearing case: eight of its nine reads answered a +/// never-read system_health session with the same empty a healthy quiet hour earns. The four +/// rungs are walked on get_health_parser_memory_node_oom — the rarest category, and therefore the one +/// where "never captured while the session IS being read" is most obviously the healthy measurement and most +/// obviously NOT unavailable. +/// +public sealed class McpZeroIsAMeasurementTests : IClassFixture, IDisposable +{ + private const string ServerName = "ZeroMeasureSrv"; + + /* Lite derives the server id from the storage name (see SignificantWaitsToolTests), so seeded rows must + be written under the same derived value the tool resolves to. */ + private readonly int _serverId; + + private readonly DuckDbInitializer _duckDb; + private readonly string _configDir; + private readonly ServerManager _serverManager; + private DuckDBConnection? _seedConn; + private long _nextId = 1; + + public McpZeroIsAMeasurementTests(SharedDuckDbFixture fixture) + { + fixture.ResetData(); + _duckDb = fixture.DuckDb; + + _configDir = Path.Combine(Path.GetTempPath(), "pmlite-zeromeasure-" + Guid.NewGuid().ToString("N")); + Directory.CreateDirectory(_configDir); + _serverManager = new ServerManager(_configDir); + + var server = new ServerConnection { Id = Guid.NewGuid().ToString(), ServerName = ServerName, IsEnabled = true }; + _serverManager.AddServer(server); + _serverId = RemoteCollectorService.GetDeterministicHashCode(RemoteCollectorService.GetServerNameForStorage(server)); + } + + public void Dispose() + { + _seedConn?.Dispose(); + try { Directory.Delete(_configDir, recursive: true); } catch (IOException) { /* temp dir */ } + } + + private static string LoadFixture(string name) => + File.ReadAllText(Path.Combine(AppContext.BaseDirectory, "Fixtures", "SystemHealth", name)); + + [Fact] + public async Task TheHealthParserLadder_FourRungs_OnlyTheDeadSessionIsUnavailable() + { + var service = new LocalDataService(_duckDb); + + /* Rung 4 — nothing of any type, ever: NOT a clean bill. This is the rung the other eight tools never + had; before #3541 A12 this read answered "empty" here. */ + var dead = JsonDocument.Parse(await McpHealthParserTools.GetMemoryNodeOOM(service, _serverManager, ServerName, 24, 50)).RootElement; + Assert.Equal("unavailable", dead.GetProperty("status").GetString()); + Assert.False(dead.GetProperty("source_observed").GetBoolean()); + Assert.Equal(JsonValueKind.Null, dead.GetProperty("last_captured_at").ValueKind); + Assert.Equal(0, dead.GetProperty("events_in_window").GetInt32()); + Assert.Contains("NOT an all-clear", dead.GetProperty("message").GetString()!, StringComparison.Ordinal); + Assert.Contains("system_health session is started", dead.GetProperty("message").GetString()!, StringComparison.Ordinal); + + /* Rung 3 — the session IS being read (another category was stored), this category never: the + healthy measurement. An OOM that never happened is not a blind spot. */ + var stored = Truncate(DateTime.UtcNow.AddMinutes(-30)); + await SeedEventAsync(SystemHealthParser.SpServerDiagnosticsEvent, LoadFixture("sp_server_diagnostics_system.xml"), Truncate(DateTime.UtcNow.AddMinutes(-31)), stored); + + var never = JsonDocument.Parse(await McpHealthParserTools.GetMemoryNodeOOM(service, _serverManager, ServerName, 24, 50)).RootElement; + Assert.Equal("empty", never.GetProperty("status").GetString()); + Assert.True(never.GetProperty("source_observed").GetBoolean()); + Assert.Equal(stored.ToString("o"), never.GetProperty("last_captured_at").GetString()); + Assert.Equal(JsonValueKind.Null, never.GetProperty("last_captured_of_type_at").ValueKind); + Assert.Contains("the absence is a measurement", never.GetProperty("message").GetString()!, StringComparison.Ordinal); + Assert.DoesNotContain("EVER", never.GetProperty("message").GetString()!, StringComparison.Ordinal); + Assert.DoesNotContain("widen", never.GetProperty("message").GetString()!, StringComparison.Ordinal); + + /* Rung 2 — captured before, outside this window: quiet, and widening reaches it. */ + var oldOom = Truncate(DateTime.UtcNow.AddHours(-48)); + await SeedEventAsync(SystemHealthParser.MemoryNodeOomEvent, LoadFixture("memory_node_oom.xml"), oldOom, oldOom); + + var quiet = JsonDocument.Parse(await McpHealthParserTools.GetMemoryNodeOOM(service, _serverManager, ServerName, 1, 50)).RootElement; + Assert.Equal("empty", quiet.GetProperty("status").GetString()); + Assert.True(quiet.GetProperty("source_observed").GetBoolean()); + Assert.Equal(oldOom.ToString("o"), quiet.GetProperty("last_captured_of_type_at").GetString()); + Assert.Contains("widen hours_back", quiet.GetProperty("message").GetString()!, StringComparison.Ordinal); + Assert.DoesNotContain("EVER", quiet.GetProperty("message").GetString()!, StringComparison.Ordinal); + + /* The data envelope carries the same witness pair beside its rows. */ + var hit = JsonDocument.Parse(await McpHealthParserTools.GetMemoryNodeOOM(service, _serverManager, ServerName, 72, 50)).RootElement; + Assert.Equal(ServerName, hit.GetProperty("server").GetString()); + Assert.True(hit.GetProperty("source_observed").GetBoolean()); + Assert.Equal(stored.ToString("o"), hit.GetProperty("last_captured_at").GetString()); + Assert.Equal(1, hit.GetProperty("event_count").GetInt32()); + + /* Rung 1 — captured IN the window and gated out: the memory-conditions read over the same + sp_server_diagnostics event, whose SYSTEM component is not a low-memory notification. Healthy, and + the count of what WAS captured is published so the caller can see the gate did the work. */ + var gated = JsonDocument.Parse(await McpHealthParserTools.GetMemoryConditions(service, _serverManager, ServerName, 1, 50)).RootElement; + Assert.Equal("empty", gated.GetProperty("status").GetString()); + Assert.Equal(1, gated.GetProperty("events_in_window").GetInt32()); + Assert.Contains("Events ARE being captured", gated.GetProperty("message").GetString()!, StringComparison.Ordinal); + Assert.DoesNotContain("widen", gated.GetProperty("message").GetString()!, StringComparison.Ordinal); + } + + private static DateTime Truncate(DateTime value) => + DateTime.SpecifyKind(new DateTime(value.Ticks - (value.Ticks % TimeSpan.TicksPerSecond)), DateTimeKind.Unspecified); + + private async Task SeedConnectionAsync() + { + if (_seedConn is null) + { + _seedConn = _duckDb.CreateConnection(); + await _seedConn.OpenAsync(); + } + return _seedConn; + } + + private async Task SeedEventAsync(string eventType, string eventXml, DateTime eventTimeUtc, DateTime collectionTimeUtc) + { + using var readLock = _duckDb.AcquireReadLock(); + var connection = await SeedConnectionAsync(); + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +INSERT INTO system_health_events + (system_health_event_id, collection_time, server_id, server_name, event_time, event_type, event_xml) +VALUES ($1, $2, $3, $4, $5, $6, $7)"; + cmd.Parameters.Add(new DuckDBParameter { Value = _nextId++ }); + cmd.Parameters.Add(new DuckDBParameter { Value = collectionTimeUtc }); + cmd.Parameters.Add(new DuckDBParameter { Value = _serverId }); + cmd.Parameters.Add(new DuckDBParameter { Value = ServerName }); + cmd.Parameters.Add(new DuckDBParameter { Value = eventTimeUtc }); + cmd.Parameters.Add(new DuckDBParameter { Value = eventType }); + cmd.Parameters.Add(new DuckDBParameter { Value = eventXml }); + await cmd.ExecuteNonQueryAsync(); + } +} diff --git a/Lite.Tests/PerformanceTrendsToolTests.cs b/Lite.Tests/PerformanceTrendsToolTests.cs index c957ffc26..b515b7e64 100644 --- a/Lite.Tests/PerformanceTrendsToolTests.cs +++ b/Lite.Tests/PerformanceTrendsToolTests.cs @@ -136,10 +136,11 @@ public async Task TheExecutionRate_SurvivesBeingBelowOnePerSecond() var service = new LocalDataService(_duckDb); /* Two snapshots five minutes apart, two executions between them: 0.0067/sec. The rows carry no - sample_interval_seconds (the pre-v61 shape), so the read LAG-derives the interval — and since v61 - (#3540) the FIRST snapshot, which has nothing to LAG against, is absent rather than a fabricated - 0.0 point (the correction v60 made for the wait trends). One point comes back: the second snapshot, - whose rate is the thing under test. */ + sample_interval_seconds (the pre-v61 shape), so the read LAG-derives the interval — and the FIRST + snapshot, which has nothing to LAG against, is UNRATED: two points come back, the first with null + rates (#3541 A12 — kept rather than dropped, so a lone collection is never an empty series and + effective_start is the first collection the store held; never the fabricated 0.0 it was before + #3540), the envelope counting it and saying why, and the second carrying the rate under test. */ var baseNow = Truncate(DateTime.UtcNow); await SeedProcedureAsync(baseNow.AddMinutes(-20), executions: 0, elapsedUs: 0); await SeedProcedureAsync(baseNow.AddMinutes(-15), executions: 2, elapsedUs: 600_000); @@ -147,9 +148,19 @@ whose rate is the thing under test. */ var hit = await McpQueryTools.GetProcedureDurationTrend(service, _serverManager, ServerName, 4); var root = JsonDocument.Parse(hit).RootElement; var trend = root.GetProperty("trend"); - Assert.Equal(1, trend.GetArrayLength()); + Assert.Equal(2, trend.GetArrayLength()); - var second = trend[0]; + /* #3541 A12: the first snapshot has nothing to difference against, so its rates are null — not the + 0 this series used to fabricate — and the envelope counts it and says why. Same keys as Darling. */ + var first = trend[0]; + Assert.Equal(JsonValueKind.Null, first.GetProperty("value").ValueKind); + Assert.Equal(JsonValueKind.Null, first.GetProperty("elapsed_ms_per_second").ValueKind); + Assert.Equal(JsonValueKind.Null, first.GetProperty("execution_count").ValueKind); + Assert.Equal(JsonValueKind.Null, first.GetProperty("executions_per_second").ValueKind); + Assert.Equal(1, root.GetProperty("unrated_points").GetInt32()); + Assert.Contains("no previous one inside the window", root.GetProperty("unrated_note").GetString()!, StringComparison.Ordinal); + + var second = trend[1]; Assert.True(second.GetProperty("value").GetDouble() > 0, "elapsed ms/sec must be a real rate"); /* The shipped integer field rounds this to an idle server. The double is why it is here. */ @@ -164,10 +175,10 @@ execution_count precedent above. */ /* #3541 A2: the disclosure block, with Lite's truth. One tier (raw, per-collection, no aggregate - note), and the series the read SERVED begins at its first point — the 15-minutes-ago seed, since - v61 dropped the prior-less first snapshot — effective_start says so, and because that head sits - three-plus hours past the requested 4-hour start, `truncated` is true. The label describes the - data, not the request; that is the whole contract. + note), and the series the store held begins at the 20-minutes-ago seed — the unrated first + collection, kept since #3541 A12 exactly so effective_start can say so — and because that head + sits three-plus hours past the requested 4-hour start, `truncated` is true. The label describes + the data, not the request; that is the whole contract. */ AssertDisclosureBlock(root); Assert.Equal("raw", root.GetProperty("source").GetString()); @@ -197,9 +208,11 @@ public async Task ASeriesThatBeginsWhereItWasAskedTo_IsNotTruncated() var root = JsonDocument.Parse(await McpQueryTools.GetProcedureDurationTrend(service, _serverManager, ServerName, 1)).RootElement; - /* One point, not two: these pre-v61 rows LAG-derive, and the prior-less first snapshot is absent since - v61 (#3540). The head is the 50-minutes-ago point, inside the slack — the property under test. */ - Assert.Equal(1, root.GetProperty("trend").GetArrayLength()); + /* Two points: these pre-v61 rows LAG-derive, and the prior-less first snapshot is present but UNRATED + (#3541 A12), so the head is the 55-minutes-ago collection — inside the slack, the property under + test — and effective_start names the first collection the store held rather than the first rate. */ + Assert.Equal(2, root.GetProperty("trend").GetArrayLength()); + Assert.Equal(JsonValueKind.Null, root.GetProperty("trend")[0].GetProperty("value").ValueKind); Assert.False(root.GetProperty("truncated").GetBoolean()); Assert.Equal(root.GetProperty("trend")[0].GetProperty("time").GetString(), root.GetProperty("effective_start").GetString()); Assert.InRange(root.GetProperty("effective_hours_back").GetDouble(), 0.8, 1.0); @@ -227,6 +240,8 @@ cumulative count. Charging every fetch to its collection time would give four po var trend = JsonDocument.Parse(hit).RootElement.GetProperty("trend"); Assert.Equal(2, trend.GetArrayLength()); + /* The first interval has no predecessor: null rates, not 0 (#3541 A12). */ + Assert.Equal(JsonValueKind.Null, trend[0].GetProperty("executions_per_second").ValueKind); /* The surviving snapshot is the FINAL one (25 executions over the 3600 seconds between the two diff --git a/Lite.Tests/QueryStoreDedupReadTests.cs b/Lite.Tests/QueryStoreDedupReadTests.cs index e4a24c3bf..6c771de01 100644 --- a/Lite.Tests/QueryStoreDedupReadTests.cs +++ b/Lite.Tests/QueryStoreDedupReadTests.cs @@ -320,14 +320,17 @@ await SeedAsync(h1.AddHours(1).AddMinutes(10), queryId: 2, planId: 22, FirstExec Assert.Equal(h0, points[0].CollectionTime); Assert.Equal(h1, points[1].CollectionTime); - /* The first point has no predecessor, so its rate is 0 — the same convention the query_stats and - procedure_stats trends have always used, not a Query Store quirk. */ - Assert.Equal(0d, points[0].Value); + /* The first point has no predecessor, so its rate is UNKNOWABLE — null, not the 0 the query_stats and + procedure_stats trends (and this one) used to fabricate (#3541 A12). The point itself is kept: + the interval ran, only its rate is undefined. */ + Assert.False(points[0].HasRate); + Assert.Null(points[0].Value); + Assert.Null(points[0].ExecutionsPerSecond); /* Interval 2's FINAL snapshot only: 9 executions x 2,000us = 18 ms of work, over the 3,600s between interval starts. Un-deduped this point would also carry interval 2's earlier 3x1,000us restatement AND interval 1's rows that were collected in this hour. */ - Assert.Equal(18.0 / 3600.0, points[1].Value, precision: 9); + Assert.Equal(18.0 / 3600.0, points[1].Value!.Value, precision: 9); } [Fact] diff --git a/Lite.Tests/TrendEmptyParityToolTests.cs b/Lite.Tests/TrendEmptyParityToolTests.cs index 493ed36fa..67d1fe51a 100644 --- a/Lite.Tests/TrendEmptyParityToolTests.cs +++ b/Lite.Tests/TrendEmptyParityToolTests.cs @@ -132,9 +132,14 @@ branches and the first point on the data one. */ var data = JsonDocument.Parse(payload).RootElement; Assert.Equal(data.GetProperty("trend")[0].GetProperty("time").GetString(), data.GetProperty("effective_start").GetString()); - Assert.Equal( - data.GetProperty("trend")[0].GetProperty("value").GetDouble(), - data.GetProperty("trend")[0].GetProperty("elapsed_ms_per_second").GetDouble()); + /* #3541 A12: one collection inside the 4-hour window (the other seed is 48 hours back) — a lone + collection has nothing to difference against, so the point is present but UNRATED: `value` and its + named twin are both null (this assertion used to compare two fabricated zeros), the envelope says + so, and the data branch is still a data branch rather than an empty one. */ + Assert.Equal(JsonValueKind.Null, data.GetProperty("trend")[0].GetProperty("value").ValueKind); + Assert.Equal(JsonValueKind.Null, data.GetProperty("trend")[0].GetProperty("elapsed_ms_per_second").ValueKind); + Assert.Equal(1, data.GetProperty("unrated_points").GetInt32()); + Assert.False(data.TryGetProperty("status", out _)); /* And get_query_trend over the same seeded query carries the block too — the last tool where the two SKUs' envelopes disagreed (Darling's has had it since #2353). */ diff --git a/Lite/Controls/ServerTab.Charts.cs b/Lite/Controls/ServerTab.Charts.cs index e69e4120d..775903bc9 100644 --- a/Lite/Controls/ServerTab.Charts.cs +++ b/Lite/Controls/ServerTab.Charts.cs @@ -1106,15 +1106,18 @@ private void UpdateQueryDurationTrendChart(List data, int hours ClearChart(QueryDurationTrendChart); ApplyTheme(QueryDurationTrendChart); - if (data.Count == 0) { RefreshEmptyChart(QueryDurationTrendChart, "Query Duration", "Duration (ms/sec)"); return; } + /* #3541 A12: the window's first collection carries a null rate (nothing to difference against) and + a chart has nowhere to draw "unknown" — it is skipped, not plotted as the 0 it used to be. */ + var rated = data.Where(d => d.HasRate).ToList(); + if (rated.Count == 0) { RefreshEmptyChart(QueryDurationTrendChart, "Query Duration", "Duration (ms/sec)"); return; } DateTime rangeEnd = toDate ?? DateTime.UtcNow.AddMinutes(UtcOffsetMinutes); DateTime rangeStart = fromDate ?? rangeEnd.AddHours(-hoursBack); double xMin = rangeStart.ToOADate(); double xMax = rangeEnd.ToOADate(); - var times = data.Select(d => d.CollectionTime.AddMinutes(UtcOffsetMinutes).ToOADate()).ToArray(); - var values = data.Select(d => d.Value).ToArray(); + var times = rated.Select(d => d.CollectionTime.AddMinutes(UtcOffsetMinutes).ToOADate()).ToArray(); + var values = rated.Select(d => d.Value!.Value).ToArray(); _queryDurationTrendHover?.Clear(); var plot = QueryDurationTrendChart.Plot.Add.TimeSeries(times, values); @@ -1137,15 +1140,18 @@ private void UpdateProcDurationTrendChart(List data, int hoursB ClearChart(ProcDurationTrendChart); ApplyTheme(ProcDurationTrendChart); - if (data.Count == 0) { RefreshEmptyChart(ProcDurationTrendChart, "Procedure Duration", "Duration (ms/sec)"); return; } + /* #3541 A12: the window's first collection carries a null rate (nothing to difference against) and + a chart has nowhere to draw "unknown" — it is skipped, not plotted as the 0 it used to be. */ + var rated = data.Where(d => d.HasRate).ToList(); + if (rated.Count == 0) { RefreshEmptyChart(ProcDurationTrendChart, "Procedure Duration", "Duration (ms/sec)"); return; } DateTime rangeEnd = toDate ?? DateTime.UtcNow.AddMinutes(UtcOffsetMinutes); DateTime rangeStart = fromDate ?? rangeEnd.AddHours(-hoursBack); double xMin = rangeStart.ToOADate(); double xMax = rangeEnd.ToOADate(); - var times = data.Select(d => d.CollectionTime.AddMinutes(UtcOffsetMinutes).ToOADate()).ToArray(); - var values = data.Select(d => d.Value).ToArray(); + var times = rated.Select(d => d.CollectionTime.AddMinutes(UtcOffsetMinutes).ToOADate()).ToArray(); + var values = rated.Select(d => d.Value!.Value).ToArray(); _procDurationTrendHover?.Clear(); var plot = ProcDurationTrendChart.Plot.Add.TimeSeries(times, values); @@ -1168,15 +1174,18 @@ private void UpdateQueryStoreDurationTrendChart(List data, int ClearChart(QueryStoreDurationTrendChart); ApplyTheme(QueryStoreDurationTrendChart); - if (data.Count == 0) { RefreshEmptyChart(QueryStoreDurationTrendChart, "Query Store Duration", "Duration (ms/sec)"); return; } + /* #3541 A12: the window's first collection carries a null rate (nothing to difference against) and + a chart has nowhere to draw "unknown" — it is skipped, not plotted as the 0 it used to be. */ + var rated = data.Where(d => d.HasRate).ToList(); + if (rated.Count == 0) { RefreshEmptyChart(QueryStoreDurationTrendChart, "Query Store Duration", "Duration (ms/sec)"); return; } DateTime rangeEnd = toDate ?? DateTime.UtcNow.AddMinutes(UtcOffsetMinutes); DateTime rangeStart = fromDate ?? rangeEnd.AddHours(-hoursBack); double xMin = rangeStart.ToOADate(); double xMax = rangeEnd.ToOADate(); - var times = data.Select(d => d.CollectionTime.AddMinutes(UtcOffsetMinutes).ToOADate()).ToArray(); - var values = data.Select(d => d.Value).ToArray(); + var times = rated.Select(d => d.CollectionTime.AddMinutes(UtcOffsetMinutes).ToOADate()).ToArray(); + var values = rated.Select(d => d.Value!.Value).ToArray(); _queryStoreDurationTrendHover?.Clear(); var plot = QueryStoreDurationTrendChart.Plot.Add.TimeSeries(times, values); @@ -1199,15 +1208,18 @@ private void UpdateExecutionCountTrendChart(List data, int hour ClearChart(ExecutionCountTrendChart); ApplyTheme(ExecutionCountTrendChart); - if (data.Count == 0) { RefreshEmptyChart(ExecutionCountTrendChart, "Executions", "Executions/sec"); return; } + /* #3541 A12: the window's first collection carries a null rate (nothing to difference against) and + a chart has nowhere to draw "unknown" — it is skipped, not plotted as the 0 it used to be. */ + var rated = data.Where(d => d.HasRate).ToList(); + if (rated.Count == 0) { RefreshEmptyChart(ExecutionCountTrendChart, "Executions", "Executions/sec"); return; } DateTime rangeEnd = toDate ?? DateTime.UtcNow.AddMinutes(UtcOffsetMinutes); DateTime rangeStart = fromDate ?? rangeEnd.AddHours(-hoursBack); double xMin = rangeStart.ToOADate(); double xMax = rangeEnd.ToOADate(); - var times = data.Select(d => d.CollectionTime.AddMinutes(UtcOffsetMinutes).ToOADate()).ToArray(); - var values = data.Select(d => d.Value).ToArray(); + var times = rated.Select(d => d.CollectionTime.AddMinutes(UtcOffsetMinutes).ToOADate()).ToArray(); + var values = rated.Select(d => d.Value!.Value).ToArray(); _executionCountTrendHover?.Clear(); var plot = ExecutionCountTrendChart.Plot.Add.TimeSeries(times, values); diff --git a/Lite/Mcp/McpHealthParserTools.cs b/Lite/Mcp/McpHealthParserTools.cs index db84cc169..68a3868ea 100644 --- a/Lite/Mcp/McpHealthParserTools.cs +++ b/Lite/Mcp/McpHealthParserTools.cs @@ -15,6 +15,15 @@ namespace PerformanceMonitorLite.Mcp; /// SystemHealthSignificance (the SAME significant set the viewer's System Events tab shows). System Health /// is the one UNGATED category (its corruption/contention counter series returns every snapshot). STORED /// reads, no live monitored-server hit; windowed on the XE event_time. Each tool caps output at limit. +/// +/// +/// Every one of the nine publishes its SOURCE WITNESS (#3541 A12): source_observed — whether the +/// collector has ever stored a system_health event of any type for this server, i.e. whether the ring buffer +/// has ever been read into the store — and last_captured_at, the collector's newest capture. A zero-row +/// window is then one of four nothings () and says which; a server whose session has +/// never been read answers unavailable, never empty. Before this, eight of the nine answered a +/// dead session with the same word a healthy quiet hour earns. Darling's twin does the same. +/// /// [McpServerToolType] public sealed class McpHealthParserTools @@ -27,7 +36,7 @@ public sealed class McpHealthParserTools /// private const string SystemHealthCollectorName = "system_health_events"; - [McpServerTool(Name = "get_health_parser_system_health"), Description("Gets parsed system_health extended event data: overall health indicators (spinlock backoffs, sick spinlocks, latch warnings, dump requests, non-yielding tasks, SQL vs system CPU, bad pages) captured by sp_server_diagnostics.")] + [McpServerTool(Name = "get_health_parser_system_health"), Description("Gets parsed system_health extended event data: overall health indicators (spinlock backoffs, sick spinlocks, latch warnings, dump requests, non-yielding tasks, SQL vs system CPU, bad pages) captured by sp_server_diagnostics. Every answer carries source_observed (whether this server's system_health session has EVER been read into the store) and last_captured_at (the collector's newest capture): an empty window on a server whose session was never read is status unavailable, not a clean bill; an empty window on one that has been read says whether the category was captured and gated out, captured before this window, or never recorded by the engine.")] public static async Task GetSystemHealth( LocalDataService dataService, ServerManager serverManager, @@ -45,14 +54,17 @@ public static async Task GetSystemHealth( if (validation != null) return validation; var rows = await dataService.GetSystemHealthAsync(resolved.ServerId, hours_back, asOfUtc: windowEnd); + var lastCapturedAt = await dataService.GetLastSystemHealthCaptureAsync(resolved.ServerId); if (rows.Count == 0) - return await McpEngineCapability.NotCollectedStatusAsync(dataService, resolved.ServerId, resolved.ServerName, SystemHealthCollectorName) - ?? McpHelpers.Status("empty", "No system health data found in the requested time range."); + return await EmptyAsync(dataService, resolved.ServerId, resolved.ServerName, hours_back, windowEnd, SystemHealthParser.SpServerDiagnosticsEvent, + "none carried a SYSTEM component result with a timestamp (the other four sp_server_diagnostics components feed the sibling reads)", capturedInWindow: null, lastCapturedAt); return JsonSerializer.Serialize(new { server = resolved.ServerName, hours_back, + source_observed = true, + last_captured_at = Stamp(lastCapturedAt), total_entries = rows.Count, shown = Math.Min(rows.Count, limit), entries = rows.Take(limit).Select(r => new @@ -79,7 +91,7 @@ public static async Task GetSystemHealth( catch (Exception ex) { return McpHelpers.FormatError("get_health_parser_system_health", ex); } } - [McpServerTool(Name = "get_health_parser_severe_errors"), Description("Gets severe errors from system_health (severity >= 19, benign connection-reset numbers excluded): error number, severity, state, database, and message. These are critical SQL Server events (stack dumps, fatal errors).")] + [McpServerTool(Name = "get_health_parser_severe_errors"), Description("Gets severe errors from system_health (severity >= 19, benign connection-reset numbers excluded): error number, severity, state, database, and message. These are critical SQL Server events (stack dumps, fatal errors). Every answer carries source_observed (whether this server's system_health session has EVER been read into the store) and last_captured_at (the collector's newest capture): an empty window on a server whose session was never read is status unavailable, not a clean bill; an empty window on one that has been read says whether the category was captured and gated out, captured before this window, or never recorded by the engine.")] public static async Task GetSevereErrors( LocalDataService dataService, ServerManager serverManager, @@ -97,14 +109,17 @@ public static async Task GetSevereErrors( if (validation != null) return validation; var rows = await dataService.GetSevereErrorsAsync(resolved.ServerId, hours_back, asOfUtc: windowEnd); + var lastCapturedAt = await dataService.GetLastSystemHealthCaptureAsync(resolved.ServerId); if (rows.Count == 0) - return await McpEngineCapability.NotCollectedStatusAsync(dataService, resolved.ServerId, resolved.ServerName, SystemHealthCollectorName) - ?? McpHelpers.Status("empty", "No severe errors found in the requested time range."); + return await EmptyAsync(dataService, resolved.ServerId, resolved.ServerName, hours_back, windowEnd, SystemHealthParser.ErrorReportedEvent, + $"none was a significant severe error (severity {SystemHealthSignificance.SevereErrorMinSeverity}+ and off the benign connection-reset list)", capturedInWindow: null, lastCapturedAt); return JsonSerializer.Serialize(new { server = resolved.ServerName, hours_back, + source_observed = true, + last_captured_at = Stamp(lastCapturedAt), error_count = rows.Count, shown = Math.Min(rows.Count, limit), errors = rows.Take(limit).Select(r => new @@ -122,7 +137,7 @@ public static async Task GetSevereErrors( catch (Exception ex) { return McpHelpers.FormatError("get_health_parser_severe_errors", ex); } } - [McpServerTool(Name = "get_health_parser_io_issues"), Description("Gets I/O-related issues from system_health (IO_SUBSYSTEM component): 15-second I/O warnings, long I/O request counts, and the longest pending request duration with its file path.")] + [McpServerTool(Name = "get_health_parser_io_issues"), Description("Gets I/O-related issues from system_health (IO_SUBSYSTEM component): 15-second I/O warnings, long I/O request counts, and the longest pending request duration with its file path. Every answer carries source_observed (whether this server's system_health session has EVER been read into the store) and last_captured_at (the collector's newest capture): an empty window on a server whose session was never read is status unavailable, not a clean bill; an empty window on one that has been read says whether the category was captured and gated out, captured before this window, or never recorded by the engine.")] public static async Task GetIOIssues( LocalDataService dataService, ServerManager serverManager, @@ -140,14 +155,17 @@ public static async Task GetIOIssues( if (validation != null) return validation; var rows = await dataService.GetIoIssuesAsync(resolved.ServerId, hours_back, asOfUtc: windowEnd); + var lastCapturedAt = await dataService.GetLastSystemHealthCaptureAsync(resolved.ServerId); if (rows.Count == 0) - return await McpEngineCapability.NotCollectedStatusAsync(dataService, resolved.ServerId, resolved.ServerName, SystemHealthCollectorName) - ?? McpHelpers.Status("empty", "No I/O issues found in the requested time range."); + return await EmptyAsync(dataService, resolved.ServerId, resolved.ServerName, hours_back, windowEnd, SystemHealthParser.SpServerDiagnosticsEvent, + "none was an IO_SUBSYSTEM component result in the WARNING state", capturedInWindow: null, lastCapturedAt); return JsonSerializer.Serialize(new { server = resolved.ServerName, hours_back, + source_observed = true, + last_captured_at = Stamp(lastCapturedAt), issue_count = rows.Count, shown = Math.Min(rows.Count, limit), issues = rows.Take(limit).Select(r => new @@ -165,7 +183,7 @@ public static async Task GetIOIssues( catch (Exception ex) { return McpHelpers.FormatError("get_health_parser_io_issues", ex); } } - [McpServerTool(Name = "get_health_parser_scheduler_issues"), Description("Gets scheduler issues from system_health: non-yielding schedulers and scheduler-monitor warnings, with the scheduler/cpu ids, online/runnable/running state, and non-yielding time.")] + [McpServerTool(Name = "get_health_parser_scheduler_issues"), Description("Gets scheduler issues from system_health: non-yielding schedulers and scheduler-monitor warnings, with the scheduler/cpu ids, online/runnable/running state, and non-yielding time. Every answer carries source_observed (whether this server's system_health session has EVER been read into the store) and last_captured_at (the collector's newest capture): an empty window on a server whose session was never read is status unavailable, not a clean bill; an empty window on one that has been read says whether the category was captured and gated out, captured before this window, or never recorded by the engine.")] public static async Task GetSchedulerIssues( LocalDataService dataService, ServerManager serverManager, @@ -183,14 +201,17 @@ public static async Task GetSchedulerIssues( if (validation != null) return validation; var rows = await dataService.GetSchedulerIssuesAsync(resolved.ServerId, hours_back, asOfUtc: windowEnd); + var lastCapturedAt = await dataService.GetLastSystemHealthCaptureAsync(resolved.ServerId); if (rows.Count == 0) - return await McpEngineCapability.NotCollectedStatusAsync(dataService, resolved.ServerId, resolved.ServerName, SystemHealthCollectorName) - ?? McpHelpers.Status("empty", "No scheduler issues found in the requested time range."); + return await EmptyAsync(dataService, resolved.ServerId, resolved.ServerName, hours_back, windowEnd, SystemHealthParser.SchedulerMonitorEvent, + "none was a scheduler-monitor record in the WARNING state", capturedInWindow: null, lastCapturedAt); return JsonSerializer.Serialize(new { server = resolved.ServerName, hours_back, + source_observed = true, + last_captured_at = Stamp(lastCapturedAt), issue_count = rows.Count, shown = Math.Min(rows.Count, limit), issues = rows.Take(limit).Select(r => new @@ -210,7 +231,7 @@ public static async Task GetSchedulerIssues( catch (Exception ex) { return McpHelpers.FormatError("get_health_parser_scheduler_issues", ex); } } - [McpServerTool(Name = "get_health_parser_memory_conditions"), Description("Gets memory condition snapshots from system_health (RESOURCE_MEMPHYSICAL_LOW): low-memory notifications, out-of-memory exceptions, and the memory-manager report (available physical/virtual/paging memory, working set, VM reserved/committed, pages, and the physical/virtual memory-low flags).")] + [McpServerTool(Name = "get_health_parser_memory_conditions"), Description("Gets memory condition snapshots from system_health (RESOURCE_MEMPHYSICAL_LOW): low-memory notifications, out-of-memory exceptions, and the memory-manager report (available physical/virtual/paging memory, working set, VM reserved/committed, pages, and the physical/virtual memory-low flags). Every answer carries source_observed (whether this server's system_health session has EVER been read into the store) and last_captured_at (the collector's newest capture): an empty window on a server whose session was never read is status unavailable, not a clean bill; an empty window on one that has been read says whether the category was captured and gated out, captured before this window, or never recorded by the engine.")] public static async Task GetMemoryConditions( LocalDataService dataService, ServerManager serverManager, @@ -228,14 +249,17 @@ public static async Task GetMemoryConditions( if (validation != null) return validation; var rows = await dataService.GetMemoryConditionsAsync(resolved.ServerId, hours_back, asOfUtc: windowEnd); + var lastCapturedAt = await dataService.GetLastSystemHealthCaptureAsync(resolved.ServerId); if (rows.Count == 0) - return await McpEngineCapability.NotCollectedStatusAsync(dataService, resolved.ServerId, resolved.ServerName, SystemHealthCollectorName) - ?? McpHelpers.Status("empty", "No memory condition events found in the requested time range."); + return await EmptyAsync(dataService, resolved.ServerId, resolved.ServerName, hours_back, windowEnd, SystemHealthParser.SpServerDiagnosticsEvent, + "none was a RESOURCE component result carrying a low-memory (RESOURCE_MEMPHYSICAL_LOW) notification", capturedInWindow: null, lastCapturedAt); return JsonSerializer.Serialize(new { server = resolved.ServerName, hours_back, + source_observed = true, + last_captured_at = Stamp(lastCapturedAt), event_count = rows.Count, shown = Math.Min(rows.Count, limit), events = rows.Take(limit).Select(r => new @@ -278,7 +302,7 @@ public static async Task GetMemoryConditions( catch (Exception ex) { return McpHelpers.FormatError("get_health_parser_memory_conditions", ex); } } - [McpServerTool(Name = "get_health_parser_cpu_tasks"), Description("Gets CPU task events from system_health (QUERY_PROCESSING component): worker thread counts (max/created/idle), tasks completed within the interval, pending tasks and oldest pending task wait time, plus deadlock/blocking flags.")] + [McpServerTool(Name = "get_health_parser_cpu_tasks"), Description("Gets CPU task events from system_health (QUERY_PROCESSING component): worker thread counts (max/created/idle), tasks completed within the interval, pending tasks and oldest pending task wait time, plus deadlock/blocking flags. Every answer carries source_observed (whether this server's system_health session has EVER been read into the store) and last_captured_at (the collector's newest capture): an empty window on a server whose session was never read is status unavailable, not a clean bill; an empty window on one that has been read says whether the category was captured and gated out, captured before this window, or never recorded by the engine.")] public static async Task GetCPUTasks( LocalDataService dataService, ServerManager serverManager, @@ -296,14 +320,17 @@ public static async Task GetCPUTasks( if (validation != null) return validation; var rows = await dataService.GetCpuTasksAsync(resolved.ServerId, hours_back, asOfUtc: windowEnd); + var lastCapturedAt = await dataService.GetLastSystemHealthCaptureAsync(resolved.ServerId); if (rows.Count == 0) - return await McpEngineCapability.NotCollectedStatusAsync(dataService, resolved.ServerId, resolved.ServerName, SystemHealthCollectorName) - ?? McpHelpers.Status("empty", "No CPU task events found in the requested time range."); + return await EmptyAsync(dataService, resolved.ServerId, resolved.ServerName, hours_back, windowEnd, SystemHealthParser.SpServerDiagnosticsEvent, + $"none was a QUERY_PROCESSING component result in the WARNING state with at least {SystemHealthSignificance.CpuTaskMinPendingTasks} pending tasks", capturedInWindow: null, lastCapturedAt); return JsonSerializer.Serialize(new { server = resolved.ServerName, hours_back, + source_observed = true, + last_captured_at = Stamp(lastCapturedAt), event_count = rows.Count, shown = Math.Min(rows.Count, limit), events = rows.Take(limit).Select(r => new @@ -325,7 +352,7 @@ public static async Task GetCPUTasks( catch (Exception ex) { return McpHelpers.FormatError("get_health_parser_cpu_tasks", ex); } } - [McpServerTool(Name = "get_health_parser_memory_broker"), Description("Gets memory broker events from system_health: broker ratio changes and target adjustments (currently predicated / allocated / previously allocated), the broker name, and the notification (RESOURCE_MEMPHYSICAL_HIGH/LOW).")] + [McpServerTool(Name = "get_health_parser_memory_broker"), Description("Gets memory broker events from system_health: broker ratio changes and target adjustments (currently predicated / allocated / previously allocated), the broker name, and the notification (RESOURCE_MEMPHYSICAL_HIGH/LOW). Every answer carries source_observed (whether this server's system_health session has EVER been read into the store) and last_captured_at (the collector's newest capture): an empty window on a server whose session was never read is status unavailable, not a clean bill; an empty window on one that has been read says whether the category was captured and gated out, captured before this window, or never recorded by the engine.")] public static async Task GetMemoryBroker( LocalDataService dataService, ServerManager serverManager, @@ -343,14 +370,17 @@ public static async Task GetMemoryBroker( if (validation != null) return validation; var rows = await dataService.GetMemoryBrokerAsync(resolved.ServerId, hours_back, asOfUtc: windowEnd); + var lastCapturedAt = await dataService.GetLastSystemHealthCaptureAsync(resolved.ServerId); if (rows.Count == 0) - return await McpEngineCapability.NotCollectedStatusAsync(dataService, resolved.ServerId, resolved.ServerName, SystemHealthCollectorName) - ?? McpHelpers.Status("empty", "No memory broker events found in the requested time range."); + return await EmptyAsync(dataService, resolved.ServerId, resolved.ServerName, hours_back, windowEnd, SystemHealthParser.MemoryBrokerEvent, + "none carried a low-memory notification (broker adjustments that are not a shrink under pressure are routine)", capturedInWindow: null, lastCapturedAt); return JsonSerializer.Serialize(new { server = resolved.ServerName, hours_back, + source_observed = true, + last_captured_at = Stamp(lastCapturedAt), event_count = rows.Count, shown = Math.Min(rows.Count, limit), events = rows.Take(limit).Select(r => new @@ -374,7 +404,7 @@ public static async Task GetMemoryBroker( catch (Exception ex) { return McpHelpers.FormatError("get_health_parser_memory_broker", ex); } } - [McpServerTool(Name = "get_health_parser_memory_node_oom"), Description("Gets memory node OOM events from system_health: out-of-memory conditions on specific NUMA nodes, with the node's physical/virtual/page-file memory, target/reserved/committed KB, the failure type, and the memory-low flags. Never gated — every recorded OOM is returned.")] + [McpServerTool(Name = "get_health_parser_memory_node_oom"), Description("Gets memory node OOM events from system_health: out-of-memory conditions on specific NUMA nodes, with the node's physical/virtual/page-file memory, target/reserved/committed KB, the failure type, and the memory-low flags. Never gated — every recorded OOM is returned. Every answer carries source_observed (whether this server's system_health session has EVER been read into the store) and last_captured_at (the collector's newest capture): an empty window on a server whose session was never read is status unavailable, not a clean bill; an empty window on one that has been read says whether the category was captured and gated out, captured before this window, or never recorded by the engine.")] public static async Task GetMemoryNodeOOM( LocalDataService dataService, ServerManager serverManager, @@ -392,14 +422,17 @@ public static async Task GetMemoryNodeOOM( if (validation != null) return validation; var rows = await dataService.GetMemoryNodeOomAsync(resolved.ServerId, hours_back, asOfUtc: windowEnd); + var lastCapturedAt = await dataService.GetLastSystemHealthCaptureAsync(resolved.ServerId); if (rows.Count == 0) - return await McpEngineCapability.NotCollectedStatusAsync(dataService, resolved.ServerId, resolved.ServerName, SystemHealthCollectorName) - ?? McpHelpers.Status("empty", "No memory node OOM events found in the requested time range."); + return await EmptyAsync(dataService, resolved.ServerId, resolved.ServerName, hours_back, windowEnd, SystemHealthParser.MemoryNodeOomEvent, + "none shredded to a memory-node OOM record (this category is ungated, so a captured OOM event that parsed would be here)", capturedInWindow: null, lastCapturedAt); return JsonSerializer.Serialize(new { server = resolved.ServerName, hours_back, + source_observed = true, + last_captured_at = Stamp(lastCapturedAt), event_count = rows.Count, shown = Math.Min(rows.Count, limit), events = rows.Take(limit).Select(r => new @@ -438,7 +471,7 @@ public static async Task GetMemoryNodeOOM( catch (Exception ex) { return McpHelpers.FormatError("get_health_parser_memory_node_oom", ex); } } - [McpServerTool(Name = "get_health_parser_significant_waits"), Description("Gets significant individual waits from system_health: one row per wait_info event where a real session's non-BACKUP statement waited at least 500 ms on a wait type that is not idle/background - the wait type, total and signal duration, the wait resource, the session id and the waiting statement. get_wait_stats gives the instance-wide totals and can never name the statement that paid them; this is the individual waits, with their SQL text.")] + [McpServerTool(Name = "get_health_parser_significant_waits"), Description("Gets significant individual waits from system_health: one row per wait_info event where a real session's non-BACKUP statement waited at least 500 ms on a wait type that is not idle/background - the wait type, total and signal duration, the wait resource, the session id and the waiting statement. get_wait_stats gives the instance-wide totals and can never name the statement that paid them; this is the individual waits, with their SQL text. Every answer carries source_observed (whether this server's system_health session has EVER been read into the store) and last_captured_at (the collector's newest capture): an empty window on a server whose session was never read is status unavailable, not a clean bill; an empty window on one that has been read says whether the category was captured and gated out, captured before this window, or never recorded by the engine.")] public static async Task GetSignificantWaits( LocalDataService dataService, ServerManager serverManager, @@ -456,51 +489,31 @@ public static async Task GetSignificantWaits( if (validation != null) return validation; var (rows, captured) = await dataService.GetSignificantWaitsWithCaptureAsync(resolved.ServerId, hours_back, asOfUtc: windowEnd); + var lastCapturedAt = await dataService.GetLastSystemHealthCaptureAsync(resolved.ServerId); if (rows.Count == 0) { /* - Three different nothings, and only one of them is good news. Events captured but none + The read this family's empty ladder was modelled on (#2484): events captured but none significant is the healthy state and costs no extra query - the reader already counted - them. Nothing captured at all needs the probe to tell a quiet window from a server whose - wait_info has never been collected, because "no significant waits" is exactly what an - operator wants to hear and a caller who believes it stops looking. Darling's twin makes - the same three distinctions in the same words. + them; nothing captured in the window needs the probe to tell a quiet window from a server + whose wait_info has never been collected, because "no significant waits" is exactly what + an operator wants to hear and a caller who believes it stops looking. Since #3541 A12 the + ladder lives in EmptyAsync and all nine reads climb it; only the gate's own description + (the four conditions) is this tool's to word. Darling's twin climbs the same ladder in the + same words. */ - if (captured > 0) - { - return McpHelpers.Status( - "empty", - $"{captured} wait_info event(s) were captured for {resolved.ServerName} in the last {hours_back} hour(s) and none was significant (needs a real session, a non-BACKUP statement, at least {SystemHealthSignificance.SignificantWaitMinDurationMs} ms, and a wait type off the idle list). Events ARE being captured, so this is the healthy answer for this read rather than missing data."); - } - - var everCaptured = await dataService.HasAnySystemHealthEventOfTypeAsync( - resolved.ServerId, SystemHealthParser.WaitInfoEvent); - if (everCaptured) - { - return McpHelpers.Status( - "empty", - $"No wait_info events were captured for {resolved.ServerName} in the last {hours_back} hour(s). This server HAS captured them before, so the window is genuinely quiet rather than blind — widen hours_back to reach the most recent events."); - } - - /* - #2511 adds a FOURTH nothing, and it is the one that was being mis-explained. On an engine - whose system_health collector is gated off there is no session to start and no collection - to check, so the advice below is advice about something that cannot exist. The engine - answer goes first because it is the stronger claim; the text after it stays exactly right - for every engine that DOES collect this. - */ - return await McpEngineCapability.NotCollectedStatusAsync( - dataService, resolved.ServerId, resolved.ServerName, SystemHealthCollectorName) - ?? McpHelpers.Status( - "unavailable", - $"No wait_info events have EVER been captured for {resolved.ServerName}, so this is NOT an all-clear — there is nothing here to be clear about. This read is served from the collected system_health ring buffer: check that collection is running for this server and that its system_health session is started before concluding nothing was waiting."); + return await EmptyAsync(dataService, resolved.ServerId, resolved.ServerName, hours_back, windowEnd, SystemHealthParser.WaitInfoEvent, + $"none was significant (needs a real session, a non-BACKUP statement, at least {SystemHealthSignificance.SignificantWaitMinDurationMs} ms, and a wait type off the idle list)", + capturedInWindow: captured, lastCapturedAt); } return JsonSerializer.Serialize(new { server = resolved.ServerName, hours_back, + source_observed = true, + last_captured_at = Stamp(lastCapturedAt), wait_count = rows.Count, shown = Math.Min(rows.Count, limit), waits = rows.Take(limit).Select(r => new @@ -519,4 +532,95 @@ signal close to the total is CPU pressure wearing a wait type's name. */ } catch (Exception ex) { return McpHelpers.FormatError("get_health_parser_significant_waits", ex); } } + + /* ─────────────────────────── the four nothings (#3541 A12) ─────────────────────────── */ + + /// + /// What zero rows means for one system_health category, which is four different things — and only the + /// first two are good news. Modelled on get_health_parser_significant_waits' three-way ladder (#2484), + /// which was the ONE read of the nine that refused to call a never-read session a clean bill; the other + /// eight answered empty to everything, so a dead system_health session, a collector that + /// never ran, and a healthy quiet hour all read as "no severe errors". Contract rule 5: zero is a + /// measurement, and an absence must say what it is an absence OF. + /// + /// Rung 1 — captured and gated out. Events of the type WERE stored in the window; the + /// shred + significance gate kept none. Healthy. The waits reader returns the count with its rows; the + /// other eight return survivors only, so the count is a bounded second read over the same window, + /// taken here only once the rows came back empty. Rung 2 — captured before, not in this window. + /// Quiet window; widening reaches the most recent events, and the message says when the last one was + /// stored so the caller knows how far. Rung 3 — this type never, but the session IS being read. + /// Other categories have been stored, so the ring buffer is reachable and the engine has simply never + /// recorded one of these — for a memory-node OOM or a severe error that is the healthy measurement, not a + /// blind spot, and it must not be called unavailable. Rung 4 — nothing of any type, ever. A dead + /// session or a collector that never ran: unavailable, the #3524 shape, never empty. The + /// #2511 engine-capability probe goes first on this rung because it is the stronger claim (an Azure SQL + /// Database has no session to start), and its text stays exactly right for every engine that does + /// collect this. + /// + /// Every rung carries the same two witness keys the data envelope carries + /// (source_observed, last_captured_at) plus the rung's own evidence, at the top level + /// beside status — the trend family's precedent (#3541 A2): a caller reads the witness without + /// first checking which branch answered. Darling's DarlingMcpHealthParserTools.EmptyAsync is the + /// twin, sentence for sentence; McpMissMessageParityPinTests holds the shared ones. + /// + private static async Task EmptyAsync( + LocalDataService dataService, int serverId, string serverName, int hoursBack, DateTime windowEnd, string eventType, + string noneQualifiedBecause, int? capturedInWindow, DateTime? lastCapturedAt) + { + var captured = capturedInWindow + ?? await dataService.CountSystemHealthEventsAsync(serverId, eventType, hoursBack, asOfUtc: windowEnd); + /* The type-scoped probe runs on every rung: on rung 1 the type exists in the window, and the stamp it + returns is THIS type's newest capture rather than the server-level witness standing in for it. */ + var lastOfType = await dataService.GetLastSystemHealthCaptureOfTypeAsync(serverId, eventType); + if (captured > 0) + { + return WitnessStatus( + "empty", + $"{captured} {eventType} event(s) were captured for {serverName} in the last {hoursBack} hour(s) and {noneQualifiedBecause}. Events ARE being captured, so this is the healthy answer for this read rather than missing data.", + sourceObserved: true, lastCapturedAt, lastCapturedOfTypeAt: lastOfType, eventsInWindow: captured); + } + + if (lastOfType is DateTime seen) + { + return WitnessStatus( + "empty", + $"No {eventType} events were captured for {serverName} in the last {hoursBack} hour(s). This server HAS captured them before (the newest was stored at {Stamp(seen)}), so the window is genuinely quiet rather than blind — widen hours_back to reach the most recent events.", + sourceObserved: true, lastCapturedAt, lastCapturedOfTypeAt: seen, eventsInWindow: 0); + } + + if (lastCapturedAt is DateTime alive) + { + return WitnessStatus( + "empty", + $"No {eventType} events have been captured for {serverName} at any time, but its system_health session IS being read — the collector last stored an event of another type at {Stamp(alive)} — so for this category the absence is a measurement: the engine has not recorded one. Not a blind spot, and a wider window would not change it.", + sourceObserved: true, alive, lastCapturedOfTypeAt: null, eventsInWindow: 0); + } + + return await McpEngineCapability.NotCollectedStatusAsync(dataService, serverId, serverName, SystemHealthCollectorName) + ?? WitnessStatus( + "unavailable", + $"No system_health events of ANY type have EVER been captured for {serverName}, so this is NOT an all-clear — there is nothing here to be clear about. This read is served from the collected system_health ring buffer: check that collection is running for this server and that its system_health session is started before concluding nothing happened.", + sourceObserved: false, lastCapturedAt: null, lastCapturedOfTypeAt: null, eventsInWindow: 0); + } + + /// + /// with the source witness beside status and message: the + /// same source_observed / last_captured_at pair the data envelope carries, plus what this + /// rung measured (last_captured_of_type_at, events_in_window). Top-level rather than under + /// hints so the keys sit in one place whichever branch answered. + /// + private static string WitnessStatus( + string status, string message, bool sourceObserved, DateTime? lastCapturedAt, DateTime? lastCapturedOfTypeAt, int eventsInWindow) + => JsonSerializer.Serialize(new + { + status, + message, + source_observed = sourceObserved, + last_captured_at = Stamp(lastCapturedAt), + last_captured_of_type_at = Stamp(lastCapturedOfTypeAt), + events_in_window = eventsInWindow, + }, McpHelpers.JsonOptions); + + /// The store's naive-UTC stamp in the same ISO shape the rows' event_time uses; null stays null. + private static string? Stamp(DateTime? stamp) => stamp?.ToString("o"); } diff --git a/Lite/Mcp/McpInstructions.cs b/Lite/Mcp/McpInstructions.cs index 336158085..2d9eee5b8 100644 --- a/Lite/Mcp/McpInstructions.cs +++ b/Lite/Mcp/McpInstructions.cs @@ -95,9 +95,9 @@ You are connected to a SQL Server performance monitoring tool via Performance Mo | `get_top_procedures_by_cpu` | Expensive stored procedures by CPU time, with the same `cpu_attribution` disclosure | `server_name`, `hours_back`, `top`, `database_name`, `as_of` | | `get_query_store_top` | Expensive queries from Query Store (persistent) | `server_name`, `hours_back`, `top`, `database_name`, `as_of` | | `get_query_heatmap` | The desktop viewer's Query Heatmap as a TABLE: how many DISTINCT queries fell into each (time bin x log-magnitude bucket) cell, with the most-executed query in each. The only query read with a TIME axis - the rankings above cannot show that a window had a quiet half and a bad half. Bins are 5 minutes by default, the viewer's own width, so both surfaces draw the same picture; raise `bucket_minutes` to cover a longer window in fewer cells. `limit` caps CELLS and truncation drops the OLDEST bins, so read `first_time_bin` / `last_time_bin`. Zero cells is three states and the read says which: never collected (`unavailable`), nothing in the window (`empty`, widen it), or collected and genuinely idle (`empty`) | `server_name`, `hours_back`, `metric`, `database_name`, `bucket_minutes`, `limit`, `as_of` | - | `get_query_store_regressions` | Queries whose Query Store performance got WORSE: the recent window's averages vs the BASELINE (every capture before it), with duration / CPU / reads regression percents, the execution-count-weighted extra duration (the ranking key), the plan counts on both sides and a severity band. `get_query_store_top` answers what is EXPENSIVE; this answers what CHANGED. Kept only where average CPU regressed > 25%. Zero rows is four states and the read says which — including "no baseline, so no regression is detectable", which is NOT a clean bill of health | `server_name`, `hours_back`, `database_name`, `limit`, `as_of` | + | `get_query_store_regressions` | Queries whose Query Store performance got WORSE: the recent window's averages vs the BASELINE (every capture before it), with duration / CPU / reads regression percents, the execution-count-weighted extra duration (the ranking key), the plan counts on both sides and a severity band. `get_query_store_top` answers what is EXPENSIVE; this answers what CHANGED. Kept only where average CPU regressed > 25%. Zero rows is four states and the read says which — including "no baseline, so no regression is detectable", which is NOT a clean bill of health A percent whose baseline side is 0 (e.g. no logical reads before the window, 50k inside it) has no denominator and is null with the reason under `undefined_percents` — never 0, which would read as no change; the ranking key is an absolute delta and exists for every row, and `severity` is null when the duration percent is. | `server_name`, `hours_back`, `database_name`, `limit`, `as_of` | | `get_query_trend` | Time-series for a specific query by query_hash. Carries the same `source` / `effective_start` / `truncated` / `bucket` block as the duration trends (always `raw`, `per-collection` on Lite) | `query_hash` (required), `database_name` (required), `server_name`, `hours_back`, `as_of` | - | `get_query_duration_trend` | Overall query elapsed-ms/sec + executions/sec over time, from the PLAN CACHE. `execution_count` and `executions_per_second` are the same quantity; the first is truncated to an integer, so read the second on a quiet server. Each point also carries `elapsed_ms_per_second`, the same quantity as `value` with its unit in the name — read that one. The payload says what it served: `source` (always `raw` on Lite — DuckDB keeps every collected row for the collector's `retention_days`, nothing rolls them up), `bucket` (`per-collection`), `effective_start` / `effective_hours_back` (where the series the store actually held begins) and `truncated` (true when that head sits later than asked — a server added mid-window, or a retention purge shorter than `hours_back`). Same field set as Darling's twin, whose `source` can also be `hourly`. An empty result carries the same block and distinguishes a quiet window (`empty`, widen `hours_back`) from a server nothing has ever been collected for (`unavailable`, collection is not running) | `server_name`, `hours_back`, `as_of` | + | `get_query_duration_trend` | Overall query elapsed-ms/sec + executions/sec over time, from the PLAN CACHE. `execution_count` and `executions_per_second` are the same quantity; the first is truncated to an integer, so read the second on a quiet server. Each point also carries `elapsed_ms_per_second`, the same quantity as `value` with its unit in the name — read that one. The payload says what it served: `source` (always `raw` on Lite — DuckDB keeps every collected row for the collector's `retention_days`, nothing rolls them up), `bucket` (`per-collection`), `effective_start` / `effective_hours_back` (where the series the store actually held begins) and `truncated` (true when that head sits later than asked — a server added mid-window, or a retention purge shorter than `hours_back`). Same field set as Darling's twin, whose `source` can also be `hourly`. An empty result carries the same block and distinguishes a quiet window (`empty`, widen `hours_back`) from a server nothing has ever been collected for (`unavailable`, collection is not running) Each point is a rate over the gap since the PREVIOUS collection, so the window's first collection carries null rates (`unrated_points` / `unrated_note`): unknowable, never reported as 0. | `server_name`, `hours_back`, `as_of` | | `get_procedure_duration_trend` | The same series over procedure_stats. NOT a duplicate: query_stats smears a procedure's work across the statements inside it, this charges the whole call to the procedure. Read both to tell an ad-hoc regression from a procedure regression. Same `source` / `effective_start` / `truncated` / `bucket` block as `get_query_duration_trend` | `server_name`, `hours_back`, `as_of` | | `get_query_store_duration_trend` | The same series over Query Store, which persists per interval and survives a plan-cache eviction or a restart. Each interval is counted once, at the hour the work RAN. Its `unavailable` names the cause the plan-cache trends do not have: Query Store may be OFF on every database. Same `source` / `effective_start` / `truncated` / `bucket` block as its siblings, with `bucket` `per-interval` | `server_name`, `hours_back`, `as_of` | @@ -137,7 +137,7 @@ You are connected to a SQL Server performance monitoring tool via Performance Mo ### Storage & Index Tools | Tool | Purpose | Key Parameters | |------|---------|----------------| - | `get_table_index_sizes` | Largest tables with size, growth (7d/30d/daily), and row counts | `server_name` | + | `get_table_index_sizes` | The 100 largest tables with size, growth and row counts from the latest daily snapshot. Growth spans only history the store holds: `history` says how many days exist and whether the 7d/30d baselines are reachable; `growth_7d_mb` / `growth_30d_mb` / `growth_pct_30d` are null (reason in `growth_note`) when their baseline does not exist — never re-measured over a shorter span under the same name — and `growth_over_available_history_*` spans exactly `growth_window_days` | `server_name` | | `get_index_usage` | Per-index seeks/scans/lookups/updates with Unused/Write-only/Active classification (drop candidates first) | `server_name` | | `get_object_locking` | Per-index lock/latch waits and lock escalations, top contended objects | `server_name` | @@ -185,6 +185,8 @@ You are connected to a SQL Server performance monitoring tool via Performance Mo | `get_health_parser_memory_node_oom` | Per-NUMA-node out-of-memory events | `server_name`, `hours_back`, `limit`, `as_of` | | `get_health_parser_significant_waits` | Individual wait_info events: a real session's non-BACKUP statement waited 500 ms+ on a non-idle wait type, with the wait type, duration and signal duration, resource, session id and the waiting SQL text. `get_wait_stats` gives the instance-wide totals and can never name the statement that paid them. An empty result says which nothing it is: events captured but none significant (the healthy answer), a quiet window (`empty`), or wait_info never captured (`unavailable`, NOT an all-clear) | `server_name`, `hours_back`, `limit`, `as_of` | + Every one of the nine carries a SOURCE WITNESS: `source_observed` (whether this server's system_health session has EVER been read into the store — any event of any type) and `last_captured_at` (the collector's newest capture). Zero rows is four different answers and the read says which: captured in the window and gated out (`empty`, `events_in_window` > 0 — the healthy one), captured before but not in this window (`empty`, `last_captured_of_type_at` says when — widen), never captured for THIS category while the session is being read (`empty`, `source_observed` true — for a rare category such as a memory-node OOM this is the measurement, not a gap), and nothing of any type ever (`unavailable`, `source_observed` false — a dead session or a collector that never ran, and NOT an all-clear). Zero is a measurement only when the source was observed. + ### Server Information Tools | Tool | Purpose | Key Parameters | |------|---------|----------------| diff --git a/Lite/Mcp/McpObjectStatsTools.cs b/Lite/Mcp/McpObjectStatsTools.cs index 647343024..8fe68ab07 100644 --- a/Lite/Mcp/McpObjectStatsTools.cs +++ b/Lite/Mcp/McpObjectStatsTools.cs @@ -9,7 +9,10 @@ namespace PerformanceMonitorLite.Mcp; [McpServerToolType] public sealed class McpObjectStatsTools { - [McpServerTool(Name = "get_table_index_sizes"), Description("Gets the largest tables with per-table size, growth (7d/30d/daily rate), and row counts from the latest daily snapshot. Indexes are rolled up per table. Use to find storage hot-spots and fast-growing tables for capacity planning.")] + /// Lite's default result cap for get_table_index_sizes (the tool takes no top parameter) — the same 100 Darling uses. + private const int TableSizesTop = 100; + + [McpServerTool(Name = "get_table_index_sizes"), Description("Gets the 100 largest tables with per-table size, growth (7d/30d/daily rate), and row counts from the latest daily snapshot. Indexes are rolled up per table. Use to find storage hot-spots and fast-growing tables for capacity planning. Growth is measured only over history the store actually holds: the history block says how many days of snapshots exist and whether the 7-day and 30-day baselines are reachable; growth_7d_mb / growth_30d_mb / growth_pct_30d are null (with the reason in growth_note) when their baseline does not exist, never re-labelled from a nearer one, and growth_over_available_history_* always spans exactly growth_window_days. A table absent from a baseline snapshot (created since) reports null growth for that window, not 0. tables_returned and truncated bound the page.")] public static async Task GetTableIndexSizes( LocalDataService dataService, ServerManager serverManager, @@ -20,14 +23,23 @@ public static async Task GetTableIndexSizes( try { - var rows = await dataService.GetObjectSizeGrowthAsync(resolved.ServerId); + /* Over-fetch by one so truncation is observed, not inferred from a full page (#3541 A3's rule). */ + var rows = await dataService.GetObjectSizeGrowthAsync(resolved.ServerId, TableSizesTop + 1); if (rows.Count == 0) { return await McpEngineCapability.NotCollectedStatusAsync(dataService, resolved.ServerId, resolved.ServerName, "index_object_stats") ?? McpHelpers.Status("unavailable", "No object size data available. Index/object stats are collected daily."); } - var result = rows.Select(r => new + var truncated = rows.Count > TableSizesTop; + var page = rows.Take(TableSizesTop).ToList(); + + /* The store's span is one fact for every row (the boundaries CTE), so it is published once. */ + var span = page[0]; + var covers7d = span.Snapshot7dTime is not null; + var covers30d = span.Snapshot30dTime is not null; + + var result = page.Select(r => new { database_name = r.DatabaseName, schema_name = r.SchemaName, @@ -36,15 +48,39 @@ public static async Task GetTableIndexSizes( used_mb = r.CurrentUsedMb, total_rows = r.TotalRows, index_count = r.IndexCount, + /* Each nominal-window figure comes from exactly the baseline it names, or is null (#3541 + A12). The SQL this replaced folded a missing 30-day baseline onto the 7-day one and a + missing 7-day one onto the oldest, and labelled the result with the window asked for. */ growth_7d_mb = r.Growth7dMb, growth_30d_mb = r.Growth30dMb, + growth_pct_30d = r.GrowthPct30d, + /* The figure that is always honest: growth from the store's earliest snapshot of this table + to its latest, over exactly growth_window_days. Null only when there is no span at all. */ + growth_over_available_history_mb = r.GrowthOverAvailableHistoryMb, + growth_over_available_history_pct = r.GrowthOverAvailableHistoryPct, + growth_window_days = r.DaysOfData, daily_growth_rate_mb = r.DailyGrowthRateMb, - growth_pct_30d = r.GrowthPct30d + growth_note = GrowthNote(r), }); return JsonSerializer.Serialize(new { server = resolved.ServerName, + history = new + { + earliest_snapshot = span.EarliestSnapshotTime.ToString("o"), + latest_snapshot = span.LatestSnapshotTime.ToString("o"), + history_days_available = span.DaysOfData, + covers_7d = covers7d, + covers_30d = covers30d, + note = covers30d + ? null + : $"The store holds {span.DaysOfData} day(s) of index snapshots for this server, so the " + + (covers7d ? "30-day baseline does not exist: growth_30d_mb and growth_pct_30d are null" : "7-day and 30-day baselines do not exist: growth_7d_mb, growth_30d_mb and growth_pct_30d are null") + + " rather than re-measured over a shorter span under the same name. Read growth_over_available_history_* — it spans exactly growth_window_days.", + }, + tables_returned = page.Count, + truncated, tables = result }, McpHelpers.JsonOptions); } @@ -54,6 +90,34 @@ public static async Task GetTableIndexSizes( } } + /// + /// Why a row's growth figures are null, when they are (#3541 A12): the store has no snapshot old enough + /// for the window, or the snapshot exists but this table was not in it (created since), or there is no + /// span at all. Null when every figure is defined, so the common row carries no note. Darling's twin words + /// it identically. + /// + internal static string? GrowthNote(ObjectSizeGrowthBaselineRow r) + { + var notes = new List(); + if (r.DaysOfData < 1) + notes.Add("the store holds a single day of snapshots for this server, so no growth is knowable yet — every growth figure is null, not 0"); + if (r.Snapshot7dTime is null) + notes.Add("no snapshot 7+ days old exists, so growth_7d_mb is null"); + else if (r.ReservedMb7dAgo is null) + notes.Add($"this table was not in the {r.Snapshot7dTime:o} snapshot (created since), so growth_7d_mb is null — its whole current size is newer than 7 days"); + if (r.Snapshot30dTime is null) + notes.Add("no snapshot 30+ days old exists, so growth_30d_mb and growth_pct_30d are null"); + else if (r.ReservedMb30dAgo is null) + notes.Add($"this table was not in the {r.Snapshot30dTime:o} snapshot (created since), so growth_30d_mb and growth_pct_30d are null"); + else if (r.ReservedMb30dAgo <= 0) + notes.Add("the table was empty 30 days ago, so growth_pct_30d has no denominator and is null (growth_30d_mb carries the absolute)"); + if (r.DaysOfData >= 1 && r.ReservedMbOldest is null) + notes.Add($"this table was not in the earliest snapshot ({r.EarliestSnapshotTime:o}), so growth_over_available_history_* and daily_growth_rate_mb are null"); + else if (r.DaysOfData >= 1 && r.ReservedMbOldest <= 0) + notes.Add("the table was empty at the earliest snapshot, so growth_over_available_history_pct has no denominator and is null"); + return notes.Count == 0 ? null : string.Join("; ", notes) + "."; + } + [McpServerTool(Name = "get_index_usage"), Description("Gets per-index usage (seeks, scans, lookups, updates) from the latest daily snapshot, classifying each index as Unused, Write-only, or Active. Unused and write-only indexes are listed first - these are drop candidates. Counters are cumulative since the last instance restart. last_user_access is UTC - the underlying sys.dm_db_index_usage_stats columns are in the monitored server's local clock and this read de-skews them - so it compares directly against get_collection_log and list_servers.")] public static async Task GetIndexUsage( LocalDataService dataService, diff --git a/Lite/Mcp/McpPvsTools.cs b/Lite/Mcp/McpPvsTools.cs index 7bc7f7464..d663e4469 100644 --- a/Lite/Mcp/McpPvsTools.cs +++ b/Lite/Mcp/McpPvsTools.cs @@ -23,7 +23,7 @@ namespace PerformanceMonitorLite.Mcp; public sealed class McpPvsTools { [McpServerTool(Name = "get_pvs_stats"), Description( - "Gets the Accelerated Database Recovery (ADR) persistent version store state per database: PVS size and percent-of-database, online-index version store size, aborted transaction count, version-cleaner run state (a start time without an end time means the cleaner is mid-run), and the oldest active/aborted transaction ids. Use when a database's size is growing without table growth, when ADR cleanup looks stuck, or alongside the PVS pressure alert. A large PVS is pinned by long-running or aborted transactions; the id gap shows how far cleanup is behind. Optionally returns the size trend for the top-5 databases over a window. Every timestamp here is UTC, the four cleaner times included - the DMV reports those in the monitored server's local clock and this read de-skews them - so a cleaner time compares directly against as_of.")] + "Gets the Accelerated Database Recovery (ADR) persistent version store state per database: PVS size and percent-of-database, online-index version store size, aborted transaction count, version-cleaner run state (a start time without an end time means the cleaner is mid-run), and the oldest active/aborted transaction ids. Use when a database's size is growing without table growth, when ADR cleanup looks stuck, or alongside the PVS pressure alert. A large PVS is pinned by long-running or aborted transactions; the id gap shows how far cleanup is behind. Optionally returns the size trend for the top-5 databases over a window. Every timestamp here is UTC, the four cleaner times included - the DMV reports those in the monitored server's local clock and this read de-skews them - so a cleaner time compares directly against as_of. pvs_measured says whether the DMV reported a size for that database at all; a measured 0 MB is published as pvs_size_mb 0 and pct_of_database 0.00 (the healthy, fully-cleaned state), and pct_of_database is null only when the numerator was not measured or the denominator is absent, with pct_of_database_reason saying which.")] public static async Task GetPvsStats( LocalDataService dataService, ServerManager serverManager, @@ -63,10 +63,17 @@ one by breaking the other. */ database_name = r.DatabaseName, is_adr_on = r.IsAdrOn, pvs_size_mb = r.PvsSizeMb, - /* The SAME denominator the FinOps grid and the pressure alert use, so no surface disagrees. */ - pct_of_database = r.PvsSizeMb is > 0 && r.DatabaseDataSizeMb is > 0 - ? Math.Round((double)(r.PvsSizeMb.Value / r.DatabaseDataSizeMb.Value) * 100.0, 2) + /* Whether the DMV reported a size at all (#3541 A12, contract rule 5). A measured 0 MB — the + healthy, fully-cleaned state — used to be indistinguishable from a NULL the collector + could not read: both fell through to pct_of_database = null. */ + pvs_measured = r.PvsSizeMb.HasValue, + /* The SAME denominator the FinOps grid and the pressure alert use, so no surface disagrees. + Any MEASURED size divides — 0 MB of a 100 GB database is 0.00%, a measurement — and only an + unmeasured numerator or an absent/zero denominator yields null, with the reason beside it. */ + pct_of_database = r.PvsSizeMb is { } pvsMb && r.DatabaseDataSizeMb is > 0 + ? Math.Round((double)(pvsMb / r.DatabaseDataSizeMb.Value) * 100.0, 2) : (double?)null, + pct_of_database_reason = PctReason(r.PvsSizeMb.HasValue, r.DatabaseDataSizeMb), online_index_version_store_mb = r.OnlineIndexVersionStoreMb, database_data_size_mb = r.DatabaseDataSizeMb, aborted_transaction_count = r.AbortedTransactionCount, @@ -115,4 +122,20 @@ one whole UTC offset stale on a value that is usually seconds old. */ return McpHelpers.FormatError("get_pvs_stats", ex); } } + + /// + /// Why pct_of_database is null, when it is (#3541 A12): the numerator was not measured, or the + /// denominator was absent or zero. Null when the percent is defined — including a defined 0.00 — so the + /// healthy row carries no note. Darling's twin words it identically. + /// + internal static string? PctReason(bool pvsMeasured, decimal? databaseDataSizeMb) + { + if (!pvsMeasured) + return "pvs_size_mb was not reported by sys.dm_tran_persistent_version_store_stats in this capture, so the share is unknown — not zero."; + if (databaseDataSizeMb is null) + return "database_data_size_mb was not captured for this database, so there is no denominator — the share is unknown, not zero."; + if (databaseDataSizeMb <= 0) + return "database_data_size_mb is 0, so the share has no denominator — the share is unknown, not zero."; + return null; + } } diff --git a/Lite/Mcp/McpQueryTools.cs b/Lite/Mcp/McpQueryTools.cs index 65a56bb4d..01d11ab1c 100644 --- a/Lite/Mcp/McpQueryTools.cs +++ b/Lite/Mcp/McpQueryTools.cs @@ -304,7 +304,7 @@ the thing that cannot read it. */ } } - [McpServerTool(Name = "get_query_store_regressions"), Description("Finds queries whose Query Store performance got WORSE, by comparing each (database, query_id) group's averages inside a recent window against its baseline - every capture BEFORE that window. Returns baseline vs recent duration, CPU and logical reads with the regression percent for each, the execution-count-weighted extra duration (the ranking key: a 5 ms regression executed a million times outranks a 5-second one executed twice), the plan counts on both sides, and a duration-driven severity band. get_query_store_top answers what is EXPENSIVE; the most expensive query is usually the one that always was. This answers what CHANGED. Rows are kept only where average CPU regressed by more than 25%.")] + [McpServerTool(Name = "get_query_store_regressions"), Description("Finds queries whose Query Store performance got WORSE, by comparing each (database, query_id) group's averages inside a recent window against its baseline - every capture BEFORE that window. Returns baseline vs recent duration, CPU and logical reads with the regression percent for each, the execution-count-weighted extra duration (the ranking key: a 5 ms regression executed a million times outranks a 5-second one executed twice), the plan counts on both sides, and a duration-driven severity band. get_query_store_top answers what is EXPENSIVE; the most expensive query is usually the one that always was. This answers what CHANGED. Rows are kept only where average CPU regressed by more than 25%. A regression percent whose BASELINE side is 0 has no denominator and is returned as null, with the reason under undefined_percents - never as 0, which would read as no change when the truth is the largest possible one; compare the two absolute figures instead. The ranking key is the absolute, execution-weighted duration delta, which exists whether or not a ratio does, so a null percent never sorts as 0. severity is banded from the duration percent and is null when that percent is.")] public static async Task GetQueryStoreRegressions( LocalDataService dataService, ServerManager serverManager, @@ -354,7 +354,10 @@ the recent window bigger AND the baseline shorter. { database_name = r.DatabaseName, query_id = r.QueryId, - severity = r.Severity, + /* Banded from the duration percent by the TVF's CASE, whose ELSE is 'LOW' — which for a + row with NO duration ratio is a verdict about a number that does not exist. Null there + (#3541 A12); the SQL's band is kept verbatim for the viewer it is shared with. */ + severity = r.DurationRegressionPercent is null ? null : r.Severity, baseline_duration_ms = r.BaselineDurationMs, recent_duration_ms = r.RecentDurationMs, duration_regression_percent = r.DurationRegressionPercent, @@ -364,7 +367,11 @@ the recent window bigger AND the baseline shorter. baseline_reads = r.BaselineReads, recent_reads = r.RecentReads, io_regression_percent = r.IoRegressionPercent, - /* The ranking key, and the one number that says whether this regression MATTERS. */ + /* Null percents, and why (#3541 A12): a 0 baseline has no ratio, and the reader used to + publish that as 0 — "no change" — for the row that changed the most. */ + undefined_percents = UndefinedPercentNotes(r), + /* The ranking key, and the one number that says whether this regression MATTERS. It is + an absolute delta, so it exists for every row and a null ratio never sorts as 0. */ additional_duration_ms = r.AdditionalDurationMs, baseline_exec_count = r.BaselineExecCount, recent_exec_count = r.RecentExecCount, @@ -383,6 +390,26 @@ the recent window bigger AND the baseline shorter. } } + /// + /// Which of a row's three regression percents are undefined, and why (#3541 A12, contract rule 5). Each + /// percent divides through NULLIF(baseline, 0), so a NULL means the baseline side was 0 — there is + /// no denominator, not no change — and the caller is pointed at the absolute pair it can still compare. + /// Null when every percent is defined, so the common row carries no noise. Darling's twin builds the same + /// sentences. + /// + private static List? UndefinedPercentNotes(QueryStoreRegressionRow r) + { + List? notes = null; + void Note(string field, string baseline, string recent) + => (notes ??= new List()).Add( + $"{field} is null: no_baseline — {baseline} is 0, so the ratio has no denominator; this is NOT 0% change. Compare {baseline} to {recent} directly."); + + if (r.DurationRegressionPercent is null) Note("duration_regression_percent", "baseline_duration_ms", "recent_duration_ms"); + if (r.CpuRegressionPercent is null) Note("cpu_regression_percent", "baseline_cpu_ms", "recent_cpu_ms"); + if (r.IoRegressionPercent is null) Note("io_regression_percent", "baseline_reads", "recent_reads"); + return notes; + } + /// /// What zero regressions actually means, which is four different things — word for word with Darling. /// Only ONE of them is good news, and the other three look identical to it in a bare empty array. @@ -601,7 +628,7 @@ private static async Task EmptyHeatmapAsync( private static string InvalidHeatmapMetric(string metric) => $"Invalid metric '{metric}'. Valid values: duration, cpu, logical_reads, logical_writes, execution_count."; - [McpServerTool(Name = "get_query_duration_trend"), Description("Gets a time-series of average query duration over time. Useful for spotting overall performance degradation or improvement trends across all queries.")] + [McpServerTool(Name = "get_query_duration_trend"), Description("Gets a time-series of average query duration over time. Useful for spotting overall performance degradation or improvement trends across all queries. Every point is a rate over the gap since the PREVIOUS point, so the window's first collection - which has no previous one to difference against - carries null rates: unknowable, never reported as 0 (unrated_points counts them, unrated_note says why).")] public static async Task GetQueryDurationTrend( LocalDataService dataService, ServerManager serverManager, @@ -652,7 +679,7 @@ public static async Task GetQueryDurationTrend( } } - [McpServerTool(Name = "get_procedure_duration_trend"), Description("Gets a time-series of stored-procedure elapsed time per second and executions per second over time, summed across every procedure. The sibling of get_query_duration_trend, and NOT a duplicate of it: query_stats attributes a procedure's work to the individual statements inside it, so a procedure that got slower is smeared across however many statements it runs. This charges the whole call to the procedure. Read the two together to tell an ad-hoc SQL regression from a procedure regression.")] + [McpServerTool(Name = "get_procedure_duration_trend"), Description("Gets a time-series of stored-procedure elapsed time per second and executions per second over time, summed across every procedure. The sibling of get_query_duration_trend, and NOT a duplicate of it: query_stats attributes a procedure's work to the individual statements inside it, so a procedure that got slower is smeared across however many statements it runs. This charges the whole call to the procedure. Read the two together to tell an ad-hoc SQL regression from a procedure regression. Every point is a rate over the gap since the PREVIOUS point, so the window's first collection - which has no previous one to difference against - carries null rates: unknowable, never reported as 0 (unrated_points counts them, unrated_note says why).")] public static async Task GetProcedureDurationTrend( LocalDataService dataService, ServerManager serverManager, @@ -693,7 +720,7 @@ public static async Task GetProcedureDurationTrend( } } - [McpServerTool(Name = "get_query_store_duration_trend"), Description("Gets a time-series of Query Store duration per second and executions per second over time, summed across every query. Where get_query_duration_trend reads the plan cache and loses everything an eviction or a restart takes with it, this reads Query Store, which persists per interval - so it is the series that survives a failover and the one to reach for when a regression is older than the cache. Each interval is counted once, at the hour the work ran.")] + [McpServerTool(Name = "get_query_store_duration_trend"), Description("Gets a time-series of Query Store duration per second and executions per second over time, summed across every query. Where get_query_duration_trend reads the plan cache and loses everything an eviction or a restart takes with it, this reads Query Store, which persists per interval - so it is the series that survives a failover and the one to reach for when a regression is older than the cache. Each interval is counted once, at the hour the work ran. Every point is a rate over the gap since the PREVIOUS point, so the window's first collection - which has no previous one to difference against - carries null rates: unknowable, never reported as 0 (unrated_points counts them, unrated_note says why).")] public static async Task GetQueryStoreDurationTrend( LocalDataService dataService, ServerManager serverManager, @@ -800,6 +827,14 @@ private static string SerializeTrend( ["hours_back"] = hours_back, }; WriteDisclosure(envelope, points.Count > 0 ? points[0].CollectionTime : null, startUtc, windowEndUtc, bucket); + /* #3541 A12: a point with no rate is published as null, never as 0, and the envelope says how many + and why — the window's first collection has no previous one to difference against. Same keys and + the same sentence as Darling's twin. */ + var unrated = points.Count(p => !p.HasRate); + envelope["unrated_points"] = unrated; + envelope["unrated_note"] = unrated == 0 + ? null + : $"{unrated} point(s) carry null rates: a per-collection rate is the work since the PREVIOUS collection divided by the seconds between them, and the window's first collection has no previous one inside the window (a collection landing in the same second as its predecessor has no denominator either). Unknowable is not 0 — the point is kept so effective_start is the first collection the store held, and its rates are null."; envelope["trend"] = points.Select(p => new { time = p.CollectionTime.ToString("o"), diff --git a/Lite/Services/LocalDataService.FinOps.IndexObjects.cs b/Lite/Services/LocalDataService.FinOps.IndexObjects.cs index 088aaf4d7..0fa549383 100644 --- a/Lite/Services/LocalDataService.FinOps.IndexObjects.cs +++ b/Lite/Services/LocalDataService.FinOps.IndexObjects.cs @@ -22,10 +22,20 @@ public partial class LocalDataService // ============================================ /// - /// Per-table size and growth (indexes rolled up per table) for a server, comparing the - /// latest snapshot to 7d/30d ago, with a daily growth rate over the available history. + /// Per-table size and growth (indexes rolled up per table) for a server: the latest snapshot's sizes + /// beside the same table's reserved size at the newest snapshot at/older than 7 and 30 days and at the + /// store's earliest snapshot, projected RAW (#3541 A12). The growth figures are derived on + /// , where each can refuse when its baseline does not exist. + /// The SQL this replaced derived them here through COALESCE(p30, p7, oldest, current), which + /// is three lies in one expression: with ten days of history "30-day growth" was growth since the + /// SEVEN-day snapshot; with two days it was growth since the oldest snapshot, still labelled 30d; and a + /// table absent from every baseline (created this week) fell through to current - current = 0, + /// "not growing", for the one table that is nothing BUT growth. The two cutoff snapshots are resolved + /// once in boundaries with FILTER so the baseline CTEs and the projected snapshot times + /// cannot disagree about which capture was used. Darling's DarlingObjectStatsReader.ObjectSizeGrowthSql + /// is the twin. /// - public async Task> GetObjectSizeGrowthAsync(int serverId, int topN = 100) + public async Task> GetObjectSizeGrowthAsync(int serverId, int topN = 100) { using var connection = await OpenConnectionAsync(); using var command = connection.CreateCommand(); @@ -39,7 +49,9 @@ WITH boundaries AS ( SELECT MAX(collection_time) AS latest_time, MIN(collection_time) AS earliest_time, - CAST(MAX(collection_time) AS DATE) - CAST(MIN(collection_time) AS DATE) AS days_of_data + CAST(MAX(collection_time) AS DATE) - CAST(MIN(collection_time) AS DATE) AS days_of_data, + MAX(collection_time) FILTER (WHERE collection_time <= $2) AS snapshot_7d_time, + MAX(collection_time) FILTER (WHERE collection_time <= $3) AS snapshot_30d_time FROM v_index_object_stats WHERE server_id = $1 ), @@ -56,15 +68,13 @@ FROM v_index_object_stats past_7d AS ( SELECT database_name, schema_name, table_name, SUM(reserved_mb) AS reserved_mb FROM v_index_object_stats - WHERE server_id = $1 AND collection_time = ( - SELECT MAX(collection_time) FROM v_index_object_stats WHERE server_id = $1 AND collection_time <= $2) + WHERE server_id = $1 AND collection_time = (SELECT snapshot_7d_time FROM boundaries) GROUP BY database_name, schema_name, table_name ), past_30d AS ( SELECT database_name, schema_name, table_name, SUM(reserved_mb) AS reserved_mb FROM v_index_object_stats - WHERE server_id = $1 AND collection_time = ( - SELECT MAX(collection_time) FROM v_index_object_stats WHERE server_id = $1 AND collection_time <= $3) + WHERE server_id = $1 AND collection_time = (SELECT snapshot_30d_time FROM boundaries) GROUP BY database_name, schema_name, table_name ), oldest AS ( @@ -81,15 +91,14 @@ FROM v_index_object_stats l.current_used_mb, l.total_rows, l.index_count, - l.current_reserved_mb - COALESCE(p7.reserved_mb, o.reserved_mb, l.current_reserved_mb) AS growth_7d_mb, - l.current_reserved_mb - COALESCE(p30.reserved_mb, p7.reserved_mb, o.reserved_mb, l.current_reserved_mb) AS growth_30d_mb, - CASE WHEN b.days_of_data >= 1 - THEN (l.current_reserved_mb - COALESCE(o.reserved_mb, l.current_reserved_mb)) / CAST(b.days_of_data AS DOUBLE PRECISION) - ELSE 0 END AS daily_growth_rate_mb, - CASE WHEN COALESCE(p30.reserved_mb, p7.reserved_mb, o.reserved_mb) > 0 - THEN (l.current_reserved_mb - COALESCE(p30.reserved_mb, p7.reserved_mb, o.reserved_mb)) * 100.0 - / COALESCE(p30.reserved_mb, p7.reserved_mb, o.reserved_mb) - ELSE 0 END AS growth_pct_30d + p7.reserved_mb AS reserved_mb_7d_ago, + p30.reserved_mb AS reserved_mb_30d_ago, + o.reserved_mb AS reserved_mb_oldest, + b.snapshot_7d_time, + b.snapshot_30d_time, + b.earliest_time, + b.latest_time, + b.days_of_data FROM latest l CROSS JOIN boundaries b LEFT JOIN past_7d p7 ON p7.database_name = l.database_name AND p7.schema_name = l.schema_name AND p7.table_name = l.table_name @@ -102,11 +111,11 @@ ORDER BY l.current_reserved_mb DESC command.Parameters.Add(new DuckDBParameter { Value = cutoff7d }); command.Parameters.Add(new DuckDBParameter { Value = cutoff30d }); - var items = new List(); + var items = new List(); using var reader = await command.ExecuteReaderAsync(); while (await reader.ReadAsync()) { - items.Add(new ObjectSizeGrowthRow + items.Add(new ObjectSizeGrowthBaselineRow { DatabaseName = reader.IsDBNull(0) ? "" : reader.GetString(0), SchemaName = reader.IsDBNull(1) ? "" : reader.GetString(1), @@ -115,10 +124,15 @@ ORDER BY l.current_reserved_mb DESC CurrentUsedMb = reader.IsDBNull(4) ? 0m : Convert.ToDecimal(reader.GetValue(4)), TotalRows = reader.IsDBNull(5) ? 0L : Convert.ToInt64(reader.GetValue(5)), IndexCount = reader.IsDBNull(6) ? 0 : Convert.ToInt32(reader.GetValue(6)), - Growth7dMb = reader.IsDBNull(7) ? 0m : Convert.ToDecimal(reader.GetValue(7)), - Growth30dMb = reader.IsDBNull(8) ? 0m : Convert.ToDecimal(reader.GetValue(8)), - DailyGrowthRateMb = reader.IsDBNull(9) ? 0m : Convert.ToDecimal(reader.GetValue(9)), - GrowthPct30d = reader.IsDBNull(10) ? 0m : Convert.ToDecimal(reader.GetValue(10)) + /* The baselines stay NULL when the store has none — a missing baseline is not a 0 baseline. */ + ReservedMb7dAgo = reader.IsDBNull(7) ? null : Convert.ToDecimal(reader.GetValue(7)), + ReservedMb30dAgo = reader.IsDBNull(8) ? null : Convert.ToDecimal(reader.GetValue(8)), + ReservedMbOldest = reader.IsDBNull(9) ? null : Convert.ToDecimal(reader.GetValue(9)), + Snapshot7dTime = reader.IsDBNull(10) ? null : reader.GetDateTime(10), + Snapshot30dTime = reader.IsDBNull(11) ? null : reader.GetDateTime(11), + EarliestSnapshotTime = reader.GetDateTime(12), + LatestSnapshotTime = reader.GetDateTime(13), + DaysOfData = Convert.ToInt32(reader.GetValue(14)), }); } return items; @@ -549,6 +563,61 @@ public class ObjectSizeGrowthRow public decimal GrowthPct30d { get; set; } } +/// +/// One per-table size + growth row for the MCP read, carrying the raw baselines the growth figures derive +/// from (#3541 A12, contract rule 5): a nominal window the store cannot reach is not a smaller window, it is +/// no measurement, and each derived property below refuses (null) rather than substituting a nearer baseline +/// or a 0. A separate class from because that one is the FinOps heatmap +/// drill's grid row (#1138), whose growth is a single window it sets directly. Darling's +/// DarlingObjectStatsReader.ObjectSizeGrowthRow derives the same figures by the same rules. +/// +public class ObjectSizeGrowthBaselineRow +{ + public string DatabaseName { get; set; } = ""; + public string SchemaName { get; set; } = ""; + public string TableName { get; set; } = ""; + public decimal CurrentReservedMb { get; set; } + public decimal CurrentUsedMb { get; set; } + public long TotalRows { get; set; } + public int IndexCount { get; set; } + + /// Reserved MB at the newest snapshot at/before the 7-day cutoff; null when no such snapshot exists or the table was not in it. + public decimal? ReservedMb7dAgo { get; set; } + /// Same for the 30-day cutoff. + public decimal? ReservedMb30dAgo { get; set; } + /// Reserved MB at the store's EARLIEST snapshot; null when the table was not in it (created since). + public decimal? ReservedMbOldest { get; set; } + /// The snapshot the 7-day baseline was read from; null when the store holds nothing that old. + public DateTime? Snapshot7dTime { get; set; } + /// Same for 30 days. + public DateTime? Snapshot30dTime { get; set; } + public DateTime EarliestSnapshotTime { get; set; } + public DateTime LatestSnapshotTime { get; set; } + /// Whole calendar days between the earliest and latest snapshots. 0 means one day of snapshots: no growth is knowable. + public int DaysOfData { get; set; } + + /// Growth since the 7-day baseline; null when there is no such baseline for this table. + public decimal? Growth7dMb => ReservedMb7dAgo is { } b ? CurrentReservedMb - b : null; + + /// Growth since the 30-day baseline; null when there is no such baseline for this table. + public decimal? Growth30dMb => ReservedMb30dAgo is { } b ? CurrentReservedMb - b : null; + + /// Percent growth over the 30-day baseline; null without a baseline, and null on a 0 baseline + /// (no denominator — a table that was empty 30 days ago has no ratio, not an infinite one). + public decimal? GrowthPct30d => ReservedMb30dAgo is > 0 ? (CurrentReservedMb - ReservedMb30dAgo.Value) * 100m / ReservedMb30dAgo.Value : null; + + /// Growth since the store's earliest snapshot — the honest figure when the nominal windows are out + /// of reach. Null when the store holds a single day (no span) or the table was not in the earliest snapshot. + public decimal? GrowthOverAvailableHistoryMb => DaysOfData >= 1 && ReservedMbOldest is { } o ? CurrentReservedMb - o : null; + + /// Percent form of ; null on a 0 baseline. + public decimal? GrowthOverAvailableHistoryPct => + DaysOfData >= 1 && ReservedMbOldest is > 0 ? (CurrentReservedMb - ReservedMbOldest.Value) * 100m / ReservedMbOldest.Value : null; + + /// MB per day over the available span; null when there is no span to divide by. + public decimal? DailyGrowthRateMb => GrowthOverAvailableHistoryMb is { } g ? g / DaysOfData : null; +} + /// Per-index usage with unused/write-only classification. public class IndexUsageRow { diff --git a/Lite/Services/LocalDataService.QueryStats.cs b/Lite/Services/LocalDataService.QueryStats.cs index d120b4e2e..b03adfaff 100644 --- a/Lite/Services/LocalDataService.QueryStats.cs +++ b/Lite/Services/LocalDataService.QueryStats.cs @@ -1128,7 +1128,20 @@ LIMIT 1 } /// - /// Gets query duration trend — total elapsed time per collection snapshot. + /// Gets query duration trend — elapsed ms per second per collection snapshot, the summed + /// delta_elapsed_time divided by the seconds since the PREVIOUS collection (the LAG epoch idiom). + /// The first collection in the window has no rate (#3541 A12, #3540 A8). Its LAG is NULL — + /// no previous collection inside the window to difference against — so its rate is unknowable, and the + /// CASE ... ELSE 0 END this replaced published that unknowable as a measured 0.0: every trend chart + /// and every MCP duration series began with a fabricated quiet instant. Contract rule 5 — zero is a + /// measurement — so the CASE has no ELSE and the rate columns are NULL for that row (and for the + /// two-collections-in-one-second case, whose denominator is 0 and whose rate is equally undefined). The + /// row is KEPT rather than filtered: the collection happened, the MCP payload's effective_start is + /// truthfully its instant, and a window holding exactly one collection is "one collection, no rate yet" + /// rather than an empty series. carries the nulls; the MCP tool publishes + /// them with the reason and the charts skip them (a chart has nowhere to draw "unknown"). The three + /// sibling trends in this file and GetQueryStoreDurationTrendAsync apply the same rule; Darling's + /// DarlingTrendReader raw reads and its Query Store rollup builder are the twins. /// public async Task> GetQueryDurationTrendAsync(int serverId, int hoursBack = 24, DateTime? fromDate = null, DateTime? toDate = null, IReadOnlyList? databaseNames = null, DateTime? asOfUtc = null) { @@ -1154,8 +1167,10 @@ GROUP BY collection_time ) SELECT collection_time, - CASE WHEN interval_seconds > 0 THEN total_elapsed_ms / interval_seconds ELSE 0 END AS elapsed_ms_per_second, - CASE WHEN interval_seconds > 0 THEN CAST(total_executions AS DOUBLE PRECISION) / interval_seconds ELSE 0 END AS executions_per_second + /* No ELSE: the first collection's LAG is NULL and its rate unknowable, so the rate is NULL — never a + fabricated 0 (#3541 A12). */ + CASE WHEN interval_seconds > 0 THEN total_elapsed_ms / interval_seconds END AS elapsed_ms_per_second, + CASE WHEN interval_seconds > 0 THEN CAST(total_executions AS DOUBLE PRECISION) / interval_seconds END AS executions_per_second FROM raw ORDER BY collection_time"; @@ -1169,12 +1184,14 @@ FROM raw using var reader = await command.ExecuteReaderAsync(); while (await reader.ReadAsync()) { + /* NULL stays NULL (#3541 A12): no rate for the window's first collection, and coercing that to 0 + here would be the fabricated quiet the SQL stopped producing. */ items.Add(new QueryTrendPoint { CollectionTime = reader.GetDateTime(0), - Value = reader.IsDBNull(1) ? 0 : ToDouble(reader.GetValue(1)), - ExecutionCount = reader.IsDBNull(2) ? 0 : (long)ToDouble(reader.GetValue(2)), - ExecutionsPerSecond = reader.IsDBNull(2) ? 0 : ToDouble(reader.GetValue(2)) + Value = reader.IsDBNull(1) ? null : ToDouble(reader.GetValue(1)), + ExecutionCount = reader.IsDBNull(2) ? null : (long)ToDouble(reader.GetValue(2)), + ExecutionsPerSecond = reader.IsDBNull(2) ? null : ToDouble(reader.GetValue(2)) }); } return items; @@ -1211,10 +1228,11 @@ FROM v_query_stats /// over the collection's rows, because a plan first seen in an otherwise steady pass (a TOP (150) /// readmission) carries 0 beside its siblings' real interval and contributes 0 to the sums; MAX is 0 /// only when EVERY row was unknowable (a restart), and that 0 becomes NULL through NULLIF so the - /// rates are NULL and the point is dropped rather than rendered as 0.00 ms/sec. NULL (a pre-v61 - /// collection that never recorded one) falls back to the LAG over collection_time this read always - /// used, so history renders exactly as it did. No ELSE 0: the first row of a pre-v61 series is - /// absent rather than a fabricated 0.0, the same correction v60 made for the wait trends. + /// rates are NULL — an UNRATED point, kept rather than rendered as 0.00 ms/sec (#3541 A12: see + /// for why the row stays). NULL (a pre-v61 collection that never + /// recorded one) falls back to the LAG over collection_time this read always used, so history renders + /// exactly as it did. No ELSE 0: the first row of a pre-v61 series carries NULL rates rather than a + /// fabricated 0.0. /// public async Task> GetProcedureDurationTrendAsync(int serverId, int hoursBack = 24, DateTime? fromDate = null, DateTime? toDate = null, IReadOnlyList? databaseNames = null, DateTime? asOfUtc = null) { @@ -1243,6 +1261,8 @@ GROUP BY collection_time ) SELECT collection_time, + /* No ELSE: the first collection's LAG is NULL and its rate unknowable, so the rate is NULL — never a + fabricated 0 (#3541 A12). */ CASE WHEN interval_seconds > 0 THEN total_elapsed_ms / interval_seconds END AS elapsed_ms_per_second, CASE WHEN interval_seconds > 0 THEN CAST(total_executions AS DOUBLE PRECISION) / interval_seconds END AS executions_per_second FROM raw @@ -1258,25 +1278,23 @@ FROM raw using var reader = await command.ExecuteReaderAsync(); while (await reader.ReadAsync()) { - /* A NULL rate is an unknowable interval (#3540, v61): the point is dropped, not read as 0. */ - if (reader.IsDBNull(1)) - { - continue; - } - + /* NULL stays NULL (#3541 A12): no rate for the window's first collection, or for a collection whose + stored interval was unknowable (a restart pass, #3540 v61) — either way the point is kept as + UNRATED rather than coerced to the fabricated quiet the SQL stopped producing. */ items.Add(new QueryTrendPoint { CollectionTime = reader.GetDateTime(0), - Value = ToDouble(reader.GetValue(1)), - ExecutionCount = reader.IsDBNull(2) ? 0 : (long)ToDouble(reader.GetValue(2)), - ExecutionsPerSecond = reader.IsDBNull(2) ? 0 : ToDouble(reader.GetValue(2)) + Value = reader.IsDBNull(1) ? null : ToDouble(reader.GetValue(1)), + ExecutionCount = reader.IsDBNull(2) ? null : (long)ToDouble(reader.GetValue(2)), + ExecutionsPerSecond = reader.IsDBNull(2) ? null : ToDouble(reader.GetValue(2)) }); } return items; } /// - /// Gets execution count trend — executions per second per collection snapshot from query_stats. + /// Gets execution count trend — executions per second per collection snapshot from query_stats. The first + /// collection in the window carries a NULL rate, not 0 — see (#3541 A12). /// public async Task> GetExecutionCountTrendAsync(int serverId, int hoursBack = 24, DateTime? fromDate = null, DateTime? toDate = null, IReadOnlyList? databaseNames = null) { @@ -1301,7 +1319,9 @@ GROUP BY collection_time ) SELECT collection_time, - CASE WHEN interval_seconds > 0 THEN CAST(total_executions AS DOUBLE PRECISION) / interval_seconds ELSE 0 END AS executions_per_second + /* No ELSE: the first collection's LAG is NULL and its rate unknowable, so the rate is NULL — never a + fabricated 0 (#3541 A12). */ + CASE WHEN interval_seconds > 0 THEN CAST(total_executions AS DOUBLE PRECISION) / interval_seconds END AS executions_per_second FROM raw ORDER BY collection_time"; @@ -1315,10 +1335,11 @@ FROM raw using var reader = await command.ExecuteReaderAsync(); while (await reader.ReadAsync()) { + /* NULL stays NULL (#3541 A12) — see GetQueryDurationTrendAsync. */ items.Add(new QueryTrendPoint { CollectionTime = reader.GetDateTime(0), - Value = reader.IsDBNull(1) ? 0 : ToDouble(reader.GetValue(1)) + Value = reader.IsDBNull(1) ? null : ToDouble(reader.GetValue(1)) }); } return items; @@ -1491,11 +1512,17 @@ public class HeatmapResult public HeatmapCell[,] CellDetails { get; set; } = new HeatmapCell[0, 0]; } +/// +/// One point of a per-collection rate series. The rates are NULLABLE (#3541 A12): the window's first +/// collection has no previous one to difference against, so it has no rate — is false +/// and the three rate members are null, never 0. The MCP tool publishes such a point with the reason; the +/// charts skip it. Darling's twin is DarlingTrendReader.QueryDurationTrendPoint. +/// public class QueryTrendPoint { public DateTime CollectionTime { get; set; } - public double Value { get; set; } - public long ExecutionCount { get; set; } + public double? Value { get; set; } + public long? ExecutionCount { get; set; } /// /// The SAME quantity as - executions per second - without the truncation. @@ -1504,7 +1531,11 @@ public class QueryTrendPoint /// as an idle server rather than a slow one. Kept alongside rather than replacing it so nothing reading /// the long breaks. Darling's twin is QueryDurationTrendPoint.ExecutionsPerSecond. /// - public double ExecutionsPerSecond { get; set; } + public double? ExecutionsPerSecond { get; set; } + + /// Whether this point carries a rate at all — false for the window's first differenced + /// collection and for one landing in the same second as its predecessor (no denominator). + public bool HasRate => Value.HasValue; } public class QueryStatsRow diff --git a/Lite/Services/LocalDataService.QueryStore.cs b/Lite/Services/LocalDataService.QueryStore.cs index 365cbff97..d8df73abb 100644 --- a/Lite/Services/LocalDataService.QueryStore.cs +++ b/Lite/Services/LocalDataService.QueryStore.cs @@ -897,6 +897,8 @@ AND qsp.query_plan IS NOT NULL /// no row is counted twice and none is dropped. A window spanning the upgrade therefore renders a /// corrected recent section and an un-corrected older one, each behaving as its own generation /// always did, and the mixture resolves itself as the pre-upgrade rows age out of retention. + /// The first placed interval in the window carries NULL rates, not 0 — see + /// (#3541 A12). /// public async Task> GetQueryStoreDurationTrendAsync(int serverId, int hoursBack = 24, DateTime? fromDate = null, DateTime? toDate = null, IReadOnlyList? databaseNames = null, DateTime? asOfUtc = null) { @@ -972,8 +974,10 @@ GROUP BY point_time ) SELECT point_time AS collection_time, - CASE WHEN interval_seconds > 0 THEN total_duration_ms / interval_seconds ELSE 0 END AS duration_ms_per_second, - CASE WHEN interval_seconds > 0 THEN CAST(total_executions AS DOUBLE PRECISION) / interval_seconds ELSE 0 END AS executions_per_second + /* No ELSE: the first placed interval's LAG is NULL and its rate unknowable, so the rate is NULL — never a + fabricated 0 (#3541 A12). */ + CASE WHEN interval_seconds > 0 THEN total_duration_ms / interval_seconds END AS duration_ms_per_second, + CASE WHEN interval_seconds > 0 THEN CAST(total_executions AS DOUBLE PRECISION) / interval_seconds END AS executions_per_second FROM raw ORDER BY point_time"; @@ -987,12 +991,13 @@ FROM raw using var reader = await command.ExecuteReaderAsync(); while (await reader.ReadAsync()) { + /* NULL stays NULL (#3541 A12) — see GetQueryDurationTrendAsync. */ items.Add(new QueryTrendPoint { CollectionTime = reader.GetDateTime(0), - Value = reader.IsDBNull(1) ? 0 : ToDouble(reader.GetValue(1)), - ExecutionCount = reader.IsDBNull(2) ? 0 : (long)ToDouble(reader.GetValue(2)), - ExecutionsPerSecond = reader.IsDBNull(2) ? 0 : ToDouble(reader.GetValue(2)) + Value = reader.IsDBNull(1) ? null : ToDouble(reader.GetValue(1)), + ExecutionCount = reader.IsDBNull(2) ? null : (long)ToDouble(reader.GetValue(2)), + ExecutionsPerSecond = reader.IsDBNull(2) ? null : ToDouble(reader.GetValue(2)) }); } return items; diff --git a/Lite/Services/LocalDataService.QueryStoreRegressions.cs b/Lite/Services/LocalDataService.QueryStoreRegressions.cs index 183e20d0a..215c83da1 100644 --- a/Lite/Services/LocalDataService.QueryStoreRegressions.cs +++ b/Lite/Services/LocalDataService.QueryStoreRegressions.cs @@ -15,20 +15,27 @@ namespace PerformanceMonitorLite.Services; /// One Query Store regression row — the DuckDB twin of Darling's /// DarlingQueryStoreRegressionReader.RegressionRow. Durations and CPU are ms (converted from the -/// stored microseconds); reads are raw pages; the percents are plain deltas. +/// stored microseconds); reads are raw pages; the percents are plain deltas. +/// The three percents are NULLABLE (#3541 A12, contract rule 5). Each divides through +/// NULLIF(baseline, 0), so a query whose baseline side is 0 — no logical reads in every capture before +/// the window, then 50,000 per execution inside it — has no ratio: the SQL returns NULL, and the reader used +/// to coerce it to 0, which published the most dramatic possible I/O regression as +/// io_regression_percent: 0, "no change". NULL stays NULL here and the tool says why. The CPU percent +/// cannot actually arrive NULL (the WHERE gate > 25 drops a NULL comparison), but it is typed like +/// its siblings so the three cannot drift in how they treat a missing denominator. public sealed class QueryStoreRegressionRow { public string DatabaseName { get; set; } = ""; public long QueryId { get; set; } public double BaselineDurationMs { get; set; } public double RecentDurationMs { get; set; } - public double DurationRegressionPercent { get; set; } + public double? DurationRegressionPercent { get; set; } public double BaselineCpuMs { get; set; } public double RecentCpuMs { get; set; } - public double CpuRegressionPercent { get; set; } + public double? CpuRegressionPercent { get; set; } public double BaselineReads { get; set; } public double RecentReads { get; set; } - public double IoRegressionPercent { get; set; } + public double? IoRegressionPercent { get; set; } public double AdditionalDurationMs { get; set; } public long BaselineExecCount { get; set; } public long RecentExecCount { get; set; } @@ -193,13 +200,13 @@ ORDER BY additional_duration_ms DESC QueryId = reader.IsDBNull(1) ? 0 : Convert.ToInt64(reader.GetValue(1)), BaselineDurationMs = reader.IsDBNull(2) ? 0 : ToDouble(reader.GetValue(2)), RecentDurationMs = reader.IsDBNull(3) ? 0 : ToDouble(reader.GetValue(3)), - DurationRegressionPercent = reader.IsDBNull(4) ? 0 : ToDouble(reader.GetValue(4)), + DurationRegressionPercent = reader.IsDBNull(4) ? null : ToDouble(reader.GetValue(4)), BaselineCpuMs = reader.IsDBNull(5) ? 0 : ToDouble(reader.GetValue(5)), RecentCpuMs = reader.IsDBNull(6) ? 0 : ToDouble(reader.GetValue(6)), - CpuRegressionPercent = reader.IsDBNull(7) ? 0 : ToDouble(reader.GetValue(7)), + CpuRegressionPercent = reader.IsDBNull(7) ? null : ToDouble(reader.GetValue(7)), BaselineReads = reader.IsDBNull(8) ? 0 : ToDouble(reader.GetValue(8)), RecentReads = reader.IsDBNull(9) ? 0 : ToDouble(reader.GetValue(9)), - IoRegressionPercent = reader.IsDBNull(10) ? 0 : ToDouble(reader.GetValue(10)), + IoRegressionPercent = reader.IsDBNull(10) ? null : ToDouble(reader.GetValue(10)), AdditionalDurationMs = reader.IsDBNull(11) ? 0 : ToDouble(reader.GetValue(11)), BaselineExecCount = reader.IsDBNull(12) ? 0 : Convert.ToInt64(reader.GetValue(12)), RecentExecCount = reader.IsDBNull(13) ? 0 : Convert.ToInt64(reader.GetValue(13)), diff --git a/Lite/Services/LocalDataService.SystemEvents.cs b/Lite/Services/LocalDataService.SystemEvents.cs index 1e7154bfb..bb28b5e70 100644 --- a/Lite/Services/LocalDataService.SystemEvents.cs +++ b/Lite/Services/LocalDataService.SystemEvents.cs @@ -469,31 +469,89 @@ public async Task> GetSignificantWaitsAsync(int serverI } /// - /// Whether this server has EVER captured a system_health event of one type, ignoring any window. - /// Separates a quiet window from a blind one on the empty path. Reads - /// v_system_health_events - the SAME source uses, so - /// it cannot report a server as captured for rows the read itself can never see - and is scoped to the - /// event_type, because a server capturing sp_server_diagnostics but no wait_info has not been sampled - /// for waits whatever its other categories hold. Darling twin: - /// DarlingSystemHealthReader.HasAnyEventOfTypeAsync; the two must stay in step so a user moving + /// The newest collection_time at which the system_health collector stored ANY event for this + /// server - the source witness every one of the nine parse-on-read MCP tools publishes (#3541 A12, + /// contract rule 5: zero is a measurement). Null when nothing has ever been stored. + /// The witness is the events view itself, NOT collection_log: the log records a success for + /// a run that read a dead system_health session and stored nothing, which is exactly the shape + /// being mis-reported, whereas a stored event is proof the session was alive and the collector reached + /// it. Windowless and type-less on purpose - it answers "has this server's ring buffer ever been read + /// into the store"; the type-scoped question is . + /// Reads v_system_health_events, the SAME source + /// uses, so it cannot report a source as observed for rows the read itself can never see. Darling twin: + /// DarlingSystemHealthReader.GetLastCaptureAsync; the two must stay in step so a user moving /// between the SKUs is not told a different story about the same state. /// - public async Task HasAnySystemHealthEventOfTypeAsync(int serverId, string eventType) + public async Task GetLastSystemHealthCaptureAsync(int serverId) { using var connection = await OpenConnectionAsync(); using var command = connection.CreateCommand(); command.CommandText = @" -SELECT 1 +SELECT MAX(collection_time) +FROM v_system_health_events +WHERE server_id = $1 +AND event_xml IS NOT NULL"; + + command.Parameters.Add(new DuckDBParameter { Value = serverId }); + return await command.ExecuteScalarAsync() is DateTime stamp ? stamp : null; + } + + /// + /// The newest collection_time at which an event of ONE type was stored for this server - the + /// type-scoped half of the witness, run only when a window came back with no events of that type. + /// Separates "this category has fired before, the window is quiet" (widen) from "this category + /// has never fired here while the session IS being read" - which for a rare category (a memory-node + /// OOM, a severe error) is the healthy measurement rather than a blind spot. Succeeds the #2484 + /// HasAnySystemHealthEventOfTypeAsync yes/no probe, which could not say WHEN. Darling twin: + /// DarlingSystemHealthReader.GetLastCaptureOfTypeAsync. + /// + public async Task GetLastSystemHealthCaptureOfTypeAsync(int serverId, string eventType) + { + using var connection = await OpenConnectionAsync(); + using var command = connection.CreateCommand(); + + command.CommandText = @" +SELECT MAX(collection_time) FROM v_system_health_events WHERE server_id = $1 AND event_type = $2 -AND event_xml IS NOT NULL -LIMIT 1"; +AND event_xml IS NOT NULL"; + + command.Parameters.Add(new DuckDBParameter { Value = serverId }); + command.Parameters.Add(new DuckDBParameter { Value = eventType }); + return await command.ExecuteScalarAsync() is DateTime stamp ? stamp : null; + } + + /// + /// How many raw events of one type the window held BEFORE any shred or significance gate - the count that + /// lets a zero-row answer say "captured and gated out" (healthy) rather than "nothing here" (#3541 A12). + /// Run only on the empty path, and over the SAME window arithmetic the typed readers use + /// ( with the MCP anchor), so the count describes the rows the read just + /// looked at and not a neighbouring window. Darling gets this number for free from its shared collect + /// step; Lite's typed readers return only the surviving rows, so the count is a second, bounded read. + /// + public async Task CountSystemHealthEventsAsync(int serverId, string eventType, int hoursBack = 24, DateTime? asOfUtc = null) + { + var (startTime, endTime) = GetTimeRange(hoursBack, fromDate: null, toDate: null, asOfUtc, SelectedServerTabUtcOffsetMinutes); + + using var connection = await OpenConnectionAsync(); + using var command = connection.CreateCommand(); + + command.CommandText = @" +SELECT COUNT(*) +FROM v_system_health_events +WHERE server_id = $1 +AND event_time >= $2 +AND event_time <= $3 +AND event_type = $4 +AND event_xml IS NOT NULL"; command.Parameters.Add(new DuckDBParameter { Value = serverId }); + command.Parameters.Add(new DuckDBParameter { Value = startTime }); + command.Parameters.Add(new DuckDBParameter { Value = endTime }); command.Parameters.Add(new DuckDBParameter { Value = eventType }); - return await command.ExecuteScalarAsync() is not null and not DBNull; + return Convert.ToInt32(await command.ExecuteScalarAsync()); } // ── CPU Tasks ── From c39de4c8ec6adceb2fa078e30433ae991b039efe Mon Sep 17 00:00:00 2001 From: erikdarlingdata <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 18 Sep 2026 18:42:24 -0400 Subject: [PATCH 69/69] A Claude review that posted nothing no longer finishes green: read-only tools end the denial churn, the job fails when no verdict was submitted, and the transcript survives the runner (#3650) The prompt told the reviewer the branch was checked out and asked for a correctness, parity and security review, while --allowedTools permitted only four gh verbs and the inline-comment tool. Every Read, Grep, Glob and git call was a permission denial; the swallowed runs' result blocks read 50 turns / 18 denials, 39 / 21, 18 / 17, each ending subtype=success with nothing posted. Three PRs, about thirteen paid runs in one night. - --allowedTools gains Read, Grep, Glob, Bash(git diff:*), Bash(git log:*), Bash(git show:*). Nothing that writes. - A step after the action fails the job when claude[bot] submitted no non-empty-bodied review since the run's own start stamp, and prints the transcript's result block. A refusal-to-run (this file differing from the default branch's copy) is named separately. - claude-execution-output.json is uploaded as the claude-review-transcript artifact, 7 days. Lands on main and dev together: claude-code-action refuses to run whenever this file differs from the default branch's copy, so a dev-only edit would disable review on every PR until the next release. --- .github/workflows/claude-review.yml | 154 +++++++++++++++++++++++++++- 1 file changed, 151 insertions(+), 3 deletions(-) diff --git a/.github/workflows/claude-review.yml b/.github/workflows/claude-review.yml index 06430dc31..a700b3b9e 100644 --- a/.github/workflows/claude-review.yml +++ b/.github/workflows/claude-review.yml @@ -7,8 +7,10 @@ name: Claude Auto Review # enforcement lives in claude-review-guard.yml, whose verdict arm fails while the newest verdict # is changes-requested, so a substantive finding becomes a red check instead of a comment that # auto-merge outruns (the #3470-#3473 train shipped six findings in one night that way; one was -# real). It no-ops cleanly until the CLAUDE_CODE_OAUTH_TOKEN repo secret is set, and on fork PRs -# (which do not receive secrets), so neither case shows a failed check. +# real). Since #3650 the job also fails ITSELF when the run submitted no verdict, with the cause in +# its log and the transcript attached -- a green here means a verdict exists, and the guard is the +# second line. It no-ops cleanly until the CLAUDE_CODE_OAUTH_TOKEN repo secret is set, and on fork +# PRs (which do not receive secrets), so neither case shows a failed check. on: pull_request: types: [opened, synchronize, reopened] @@ -37,7 +39,19 @@ jobs: with: fetch-depth: 1 + # #3650: the verdict check at the bottom counts the bot's formal reviews submitted AFTER this + # instant, so a verdict left by an earlier run on the same PR cannot vouch for this one. Read + # once, here, before the action installs anything -- GitHub stamps submitted_at in the same + # ISO-8601 UTC shape (2026-09-18T22:00:00Z), so the comparison below is a plain string one. + - name: Open the verdict window + id: window + if: ${{ env.CLAUDE_CODE_OAUTH_TOKEN != '' }} + run: echo "start=$(date -u +%FT%TZ)" >> "$GITHUB_OUTPUT" + + # The step NAME is read by claude-review-guard.yml (REVIEW_STEP) to tell a clean no-op from a + # run that said nothing; renaming it blinds the guard. The id is for the steps below. - name: Claude review + id: review if: ${{ env.CLAUDE_CODE_OAUTH_TOKEN != '' }} uses: anthropics/claude-code-action@v1 with: @@ -78,5 +92,139 @@ jobs: does not enable, and no gate needs it. The guard's rule is newest-verdict-wins: on a re-review after new commits, review the NEW diff and submit a fresh verdict, and a clean fresh verdict clears an earlier changes-requested by itself. + # #3650: the prompt above says the branch is checked out and asks for a correctness, + # parity and security review -- an invitation to read code -- while the allowlist used to + # permit only the four gh verbs and the inline-comment tool. Every Read, Grep, Glob and + # git call the reviewer reached for was a permission denial, and the swallowed runs' own + # result blocks put the ratio on record: 50 turns / 18 denials, 39 / 21, 18 / 17, each + # ending subtype=success with NOTHING posted. After enough denials the session ends + # without ever reaching the verdict protocol, so a paid run leaves no trace (three PRs, + # about thirteen runs, one night). Read-only tools are enough: the reviewer needs to open + # the files the diff touches, find a symbol's other callers and read a parity twin, and + # nothing about reviewing needs a write. Nothing here can edit, commit, push or post + # outside the gh verbs already listed; the git verbs are the read-only three. claude_args: | - --allowedTools "mcp__github_inline_comment__create_inline_comment,Bash(gh pr comment:*),Bash(gh pr diff:*),Bash(gh pr view:*),Bash(gh pr review:*)" + --allowedTools "mcp__github_inline_comment__create_inline_comment,Bash(gh pr comment:*),Bash(gh pr diff:*),Bash(gh pr view:*),Bash(gh pr review:*),Read,Grep,Glob,Bash(git diff:*),Bash(git log:*),Bash(git show:*)" + + # #3650: a review that posted nothing used to finish GREEN. The prompt mandates one formal + # review per run -- changes-requested on a substantive finding, a "LGTM" comment review + # otherwise -- so on this repo a run that submitted no verdict has broken its contract every + # time; a clean review is never silent, and a zero here is never legitimate. This step turns + # that into the job's own colour, with the diagnosis in the log, so the guard (#2229/#3492, + # still the REQUIRED check, still enforcing newest-verdict-wins and the drift arm) becomes the + # second line rather than the first. Counted: reviews by the bot, submitted inside this run's + # window, with a NON-EMPTY body. The body test is load-bearing: the inline-comment tool files + # each comment inside a review of its own with an empty body (#3647 carried four bot reviews, + # two of them bodiless carriers), and both verdict shapes must carry one (gh and the API both + # refuse a bodiless comment/changes-requested review) -- so "posted inline notes, never a + # verdict" is the #3470 shape and fails here, as it should. + # Two runs end with zero verdicts and both are defects on this repo, told apart by whether + # Claude ran at all: the action exits SUCCESS in seconds without running when this file + # differs from the default branch's copy (cause 1 in the guard's header), and then it sets no + # execution_file and writes no transcript. Expected on a PR that edits this file; a repo-wide + # outage otherwise -- the guard grades which, because it can read the PR's file list from a + # workflow that is free to change. The step's own outcome is left to speak for itself when it + # is not success (a failed step already reds the job; a cancelled one is a superseded push). + # A gh lookup failure is NOT a verdict on the review (#2309): it warns and stands down rather + # than forcing a paid re-run of the whole job to clear a transient API error -- the guard's + # own tally, a separate and free-to-rerun workflow, stays the enforcement. + - name: Verify the review posted a verdict + if: ${{ always() && env.CLAUDE_CODE_OAUTH_TOKEN != '' }} + env: + GH_TOKEN: ${{ github.token }} + R: ${{ github.repository }} + PR: ${{ github.event.pull_request.number }} + SINCE: ${{ steps.window.outputs.start }} + REVIEW_OUTCOME: ${{ steps.review.outcome }} + # Set by the action only after Claude actually ran; empty when it refused at validation. + EXECUTION_FILE: ${{ steps.review.outputs.execution_file }} + # Author of every artifact the review leaves behind (the action's bot_name default). + REVIEW_BOT: claude[bot] + run: | + set -euo pipefail + summary() { echo "$*" >> "$GITHUB_STEP_SUMMARY"; } + + if [ "$REVIEW_OUTCOME" != "success" ]; then + echo "::notice title=Verdict check skipped::The review step ended '$REVIEW_OUTCOME'; its own"\ + "outcome carries the story, so this step has nothing to add." + exit 0 + fi + + transcript="${EXECUTION_FILE:-$RUNNER_TEMP/claude-execution-output.json}" + + errfile=$(mktemp 2>/dev/null || echo /dev/null) + if ! matched=$(gh api --paginate "repos/$R/pulls/$PR/reviews?per_page=100" \ + --jq ".[] | select(.user.login == env.REVIEW_BOT + and .submitted_at != null + and .submitted_at >= env.SINCE + and ((.body // \"\") | length) > 0) + | \"\\(.state)\\t\\(.submitted_at)\\t\\(.html_url)\"" 2>"$errfile"); then + err=$(cat "$errfile" 2>/dev/null || true) + [ "$errfile" != /dev/null ] && rm -f "$errfile" || true + echo "::warning title=Verdict check could not read the PR's reviews::gh api"\ + "repos/$R/pulls/$PR/reviews failed: ${err:-no stderr}. A lookup failure is not a"\ + "review verdict (#2309), so the review is UNCONFIRMED here; the guard's own tally"\ + "decides, or read the PR by eye." + summary "- Verdict check: lookup failed; review UNCONFIRMED (#2309)." + exit 0 + fi + [ "$errfile" != /dev/null ] && rm -f "$errfile" || true + + count=$(printf '%s' "$matched" | grep -c . || true) + echo "verdict reviews by $REVIEW_BOT on PR #$PR since $SINCE: $count" + if [ -n "$matched" ]; then printf '%s\n' "$matched"; fi + + if [ "$count" -gt 0 ]; then + summary "### Claude review posted $count verdict review(s) since $SINCE" + exit 0 + fi + + summary '### Claude review posted NO verdict' + + if [ -z "$EXECUTION_FILE" ] && [ ! -s "$transcript" ]; then + echo "::error title=Claude never ran::claude-code-action exited success without running"\ + "Claude -- no execution file, no transcript. That is what it does when this branch's"\ + ".github/workflows/claude-review.yml differs from the default branch's copy (#2229,"\ + "cause 1). Expected on a PR that edits this file; on any other PR it means the file"\ + "has drifted and EVERY PR in the repo is going unreviewed -- see the guard's verdict"\ + "on this PR. Either way, do not read this PR as reviewed." + summary '- Claude never ran (workflow validation refused). Not reviewed.' + exit 1 + fi + + echo "::error title=Review ran and posted no verdict::The review ran and posted no"\ + "verdict: $REVIEW_BOT submitted no formal review on PR #$PR since $SINCE, and the"\ + "prompt promises one every run even when clean. Real money was spent and the output"\ + "vanished (#3650; the #2229 failure shape). Do NOT read this PR as reviewed. The"\ + "transcript is attached to this run as the claude-review-transcript artifact; its"\ + "result block follows." + summary '- The review ran and submitted no formal review. Not reviewed (#3650).' + if [ -s "$transcript" ]; then + echo "--- result block of $transcript ---" + jq -c '.[-1] | {type, subtype, is_error, num_turns, permission_denials_count, + total_cost_usd, duration_ms, + denied: [.permission_denials[]? | .tool_name]}' "$transcript" \ + 2>/dev/null || true + echo "--- tail of $transcript ---" + tail -c 4000 "$transcript" || true + echo + else + echo "(no transcript at $transcript)" + fi + exit 1 + + # #3650: the per-turn transcript is what proves WHY a run said nothing, and it used to die + # with the runner -- the denial-ratio evidence above had to be inferred from result blocks in + # the step log. Seven days is long enough to diagnose the next swallowed run in one click and + # short enough that nothing accumulates. The reviewer's tools are read-only over a public + # tree and gh reads of public PR data, so the transcript holds nothing that is not already + # public; keep the allowlist that way, because this artifact is readable by anyone who can + # read the repo. Missing file (the action refused to run, or was skipped) is not an error. + - name: Retain the review transcript + if: ${{ always() && env.CLAUDE_CODE_OAUTH_TOKEN != '' }} + uses: actions/upload-artifact@v6 + with: + name: claude-review-transcript + path: ${{ steps.review.outputs.execution_file || format('{0}/claude-execution-output.json', runner.temp) }} + if-no-files-found: ignore + retention-days: 7