diff --git a/Darling/Darling.Tests/DarlingRetentionTests.cs b/Darling/Darling.Tests/DarlingRetentionTests.cs index 048b593d6..bdf7be60e 100644 --- a/Darling/Darling.Tests/DarlingRetentionTests.cs +++ b/Darling/Darling.Tests/DarlingRetentionTests.cs @@ -1052,6 +1052,95 @@ the command timeout. Pinned as arithmetic so raising the cap without re-measurin + "the margin is deliberate"); } + /* ---------------- #4250 item 3: the unordered row-capped prune for the liveness-touched tables ---------------- */ + + /// + /// The sibling builder's shape: row-capped like , but + /// with NO ORDER BY — since V149 (#4250) dropped both tables' last_seen btree, an ORDER BY + /// would force a sort of every under-cutoff row before the LIMIT could apply, a second full pass over + /// data the seq scan already read. No min() subquery and no INTERVAL either: this is a + /// single bound against a FIXED cutoff, not a time slice. + /// + [Fact] + public void UnorderedRowCappedDelete_HasNoOrderByAndNoMinSubquery() + { + var sql = DarlingRetention.UnorderedRowCappedDeleteSql("collect.query_store_text", "last_seen", 300_000); + + Assert.Equal( + "DELETE FROM collect.query_store_text WHERE ctid IN (" + + "SELECT ctid FROM collect.query_store_text WHERE last_seen < $1 " + + "LIMIT 300000)", + sql); + + Assert.DoesNotContain("ORDER BY", sql, StringComparison.OrdinalIgnoreCase); + Assert.DoesNotContain("min(", sql, StringComparison.OrdinalIgnoreCase); + Assert.DoesNotContain("INTERVAL", sql, StringComparison.OrdinalIgnoreCase); + Assert.Contains("ctid IN", sql, StringComparison.Ordinal); + Assert.Contains("LIMIT 300000", sql, StringComparison.Ordinal); + } + + /// + /// Both liveness-touched tables' purge calls use the unordered builder, carry the shared cap as their + /// batch size (so the drain loop's "a full-cap batch means there may be more" contract applies), and + /// leave adaptiveRowCapTimeColumn unset — that parameter exists only for the plan dimension's + /// #4130 retry-at-half-cap behavior, which neither of these tables has been measured to need: both + /// clear their whole steady-state backlog in ONE batch at the sized cap (see + /// 's remarks), so there is nothing here + /// for a shrinking retry to protect against. Both call sites read the cap from the LOCAL parameter + /// livenessTouchedTablePruneRowCap rather than the constant directly, since #4250 item 3's live + /// loop test needs a seam to run the same call sites at a small cap; the parameter itself defaults to + /// the constant (pinned separately, below), so production is unchanged. + /// + [Fact] + public void MapAndTextPurges_UseTheUnorderedCap_WithoutTheAdaptiveRetry() + { + var source = ReadRetentionSource(); + + var mapAt = source.IndexOf("var mapCutoff = ComputeMapCutoff(", StringComparison.Ordinal); + Assert.True(mapAt >= 0, "the map purge call moved"); + var mapBody = source[mapAt..Math.Min(source.Length, mapAt + 500)]; + Assert.Contains("UnorderedRowCappedDeleteSql(", mapBody, StringComparison.Ordinal); + Assert.Contains("livenessTouchedTablePruneRowCap", mapBody, StringComparison.Ordinal); + Assert.DoesNotContain("adaptiveRowCapTimeColumn", mapBody, StringComparison.Ordinal); + + var textAt = source.IndexOf("var queryTextCutoff = utcNow.AddDays(", StringComparison.Ordinal); + Assert.True(textAt >= 0, "the query text purge call moved"); + var textBody = source[textAt..Math.Min(source.Length, textAt + 500)]; + Assert.Contains("UnorderedRowCappedDeleteSql(", textBody, StringComparison.Ordinal); + Assert.Contains("livenessTouchedTablePruneRowCap", textBody, StringComparison.Ordinal); + Assert.DoesNotContain("adaptiveRowCapTimeColumn", textBody, StringComparison.Ordinal); + } + + /// + /// The #4250 item 3 test seam's default: every real caller (the daily sweep, the on-demand + /// purge_now command) omits livenessTouchedTablePruneRowCap, so production must always + /// run the shipped 300,000-row constant, never a silently different value. + /// + [Fact] + public void LivenessTouchedTablePruneRowCapSeam_DefaultsToTheShippedConstant() + { + var method = typeof(DarlingRetention).GetMethod( + nameof(DarlingRetention.PurgeAsync), System.Reflection.BindingFlags.Public | System.Reflection.BindingFlags.Static)!; + var parameter = Array.Find(method.GetParameters(), p => p.Name == "livenessTouchedTablePruneRowCap")!; + Assert.NotNull(parameter); + Assert.Equal(DarlingRetention.LivenessTouchedTablePruneRowCap, (int)parameter.DefaultValue!); + } + + /// + /// The cap clears the busiest single day observed in the field's last_seen age histogram (map + /// 255k rows at its oldest surviving day, text 220k) in ONE batch — a steady-state run never issues a + /// second, empty-batch statement. + /// + [Fact] + public void LivenessTouchedTablePruneRowCap_ClearsTheBusiestObservedDayInOneBatch() + { + const int busiestMapDay = 255_000; + const int busiestTextDay = 220_000; + + Assert.True(DarlingRetention.LivenessTouchedTablePruneRowCap > busiestMapDay); + Assert.True(DarlingRetention.LivenessTouchedTablePruneRowCap > busiestTextDay); + } + /// /// Only the plan dimension is capped. query_text_dim is ~40 MB in total and drains in a single /// slice, and the fact tables need the compressed-chunk-safe shape that the ctid idiom cannot diff --git a/Darling/Darling.Tests/QueryStoreLivenessHotTouchLiveTests.cs b/Darling/Darling.Tests/QueryStoreLivenessHotTouchLiveTests.cs new file mode 100644 index 000000000..5097a50b3 --- /dev/null +++ b/Darling/Darling.Tests/QueryStoreLivenessHotTouchLiveTests.cs @@ -0,0 +1,247 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Linq; +using System.Threading.Tasks; +using Npgsql; +using PerformanceMonitor.Darling.Storage; +using PerformanceMonitor.Darling.Viewer; +using Xunit; + +namespace Darling.Tests; + +/// +/// Pins Darling rung V149 (#4250): the Query Store liveness touch becomes a HOT update on +/// collect.query_store_plan_map and collect.query_store_text — the two last_seen btree +/// indexes are dropped and both tables get fillfactor = 90. This file is the RUNG (ladder, viewer +/// probe) and the HOT-eligibility proof: the schema after migrate carries neither index and the fillfactor +/// reloption, startup's CREATE TABLE IF NOT EXISTS convergence does not recreate the index, and two +/// touches of the same row report a HOT update via pg_stat_user_tables.n_tup_hot_upd. +/// +/// This file's "I am the top rung" claim takes over from ReadLatencyFlushLiveTests (V148) now +/// that V149 has landed. +/// +/* #1776 own-store: each fact mints its own scratch database through ScratchPostgres and never touches the + shared store's tables, so it cannot race the live collection and serializing it would be pure slowdown. */ +public sealed class QueryStoreLivenessHotTouchLiveTests +{ + private const int RungVersion = 149; + private const int PreviousVersion = 148; + + /// This rung's sentinel ordinal in the viewer probe — the newest, so the last argument. + private const int ProbeOrdinal = 124; + + private static string? ConnectionString => Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + + /// + /// The rung is registered and is the new top of the ladder — the claim this class takes over from + /// ReadLatencyFlushLiveTests (V148) now that V149 has landed. + /// + [Fact] + public void TheRungIsRegisteredAtTheTopOfADenseLadder() + { + var versions = PgMigrations.Scripts.Select(s => s.Version).ToList(); + + Assert.Equal("query-store-liveness-hot-touch", PgMigrations.Scripts.Single(s => s.Version == RungVersion).Name); + Assert.Equal(StorageVersion.SchemaVersion, PgMigrations.Scripts[^1].Version); + Assert.Equal(StorageVersion.SchemaVersion, versions.Max()); + Assert.Equal(RungVersion, StorageVersion.SchemaVersion); + Assert.Equal(versions.Distinct().OrderBy(v => v), versions); + } + + /// + /// The viewer probe's sentinel carries this rung, and the map treats it as the TOP arm: a missing top arm + /// maps a fully-migrated store one rung short, permanently, because + /// is . + /// + [Fact] + public void TheProbeMapsAFullyMigratedStoreToThisTopRung() + { + var probe = ViewerDataService.StoreSchemaProbeSql.Replace("\r\n", "\n", StringComparison.Ordinal); + Assert.Contains("idx_query_store_plan_map_last_seen", probe, StringComparison.Ordinal); + Assert.Contains("fillfactor=90", probe, StringComparison.Ordinal); + + var viewer = RepoFile.ReadRepoFile("Darling", "PerformanceMonitor.Darling.Viewer", "ViewerDataService.cs"); + Assert.Contains($"reader.GetBoolean({ProbeOrdinal})", viewer, StringComparison.Ordinal); + Assert.DoesNotContain($"reader.GetBoolean({ProbeOrdinal + 1})", viewer, StringComparison.Ordinal); + + Assert.Equal(StorageVersion.SchemaVersion, ViewerDataService.RequiredStoreSchemaVersion); + + var method = typeof(ViewerDataService).GetMethod("MapProbedSchemaVersion", System.Reflection.BindingFlags.NonPublic | System.Reflection.BindingFlags.Static)!; + var arity = method.GetParameters().Length; + Assert.Equal(ProbeOrdinal, arity - 1); + Assert.Equal("hasHotLivenessTouch", method.GetParameters()[ProbeOrdinal].Name); + + var all = Enumerable.Repeat((object)true, arity).ToArray(); + Assert.Equal(StorageVersion.SchemaVersion, (int)method.Invoke(null, all)!); + + var behind = (object[])all.Clone(); + behind[ProbeOrdinal] = false; + Assert.Equal(PreviousVersion, (int)method.Invoke(null, behind)!); + + var thisArm = viewer.IndexOf("if (hasHotLivenessTouch)", StringComparison.Ordinal); + var previousArm = viewer.IndexOf("if (hasReadLatency)", StringComparison.Ordinal); + Assert.True(thisArm >= 0, "the viewer has no V149 sentinel arm — a fully-migrated store would map one rung short"); + Assert.True(thisArm < previousArm, "the V149 arm sits below V148's, so a current store maps one rung short"); + Assert.Contains( + "return " + StorageVersion.SchemaVersion.ToString(System.Globalization.CultureInfo.InvariantCulture) + ";", + viewer[thisArm..previousArm], StringComparison.Ordinal); + } + + /// + /// The LIVE schema after migrate: neither last_seen index exists on either table, and both carry + /// the fillfactor=90 reloption. Run against origin/dev (pre-V149) this is RED — the indexes + /// exist and the reloption is absent — proving the pin actually checks the rung rather than a tautology. + /// + [Fact] + public async Task AfterMigrate_NeitherLastSeenIndexExists_AndBothTablesCarryFillfactor90() + { + var baseConnectionString = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(baseConnectionString), + "Set DARLING_TEST_PG to a Postgres connection string to run the V149 schema pin."); + + var ct = TestContext.Current.CancellationToken; + + await using var scratch = await ScratchPostgres.CreateAsync(baseConnectionString!, ct); + await using var connection = new NpgsqlConnection(scratch.ConnectionString); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + + Assert.False(await IndexExistsAsync(connection, "idx_query_store_plan_map_last_seen", ct), + "V149 drops idx_query_store_plan_map_last_seen"); + Assert.False(await IndexExistsAsync(connection, "idx_query_store_text_last_seen", ct), + "V149 drops idx_query_store_text_last_seen"); + + Assert.True(await HasFillfactor90Async(connection, "collect.query_store_plan_map", ct), + "V149 sets fillfactor = 90 on query_store_plan_map"); + Assert.True(await HasFillfactor90Async(connection, "collect.query_store_text", ct), + "V149 sets fillfactor = 90 on query_store_text"); + } + + /// + /// Startup convergence (, ) + /// no longer recreates either index: run the table-ensure path a second time after migrate, on top of an + /// already-migrated store, and the index must still be absent. A real mutation (re-adding the + /// CREATE INDEX IF NOT EXISTS ... last_seen line back to either helper's DDL) turns this RED — + /// see the PR body for the exact revert-and-rerun. + /// + [Fact] + public async Task StartupConvergence_RunTwice_DoesNotRecreateEitherIndex() + { + var baseConnectionString = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(baseConnectionString), + "Set DARLING_TEST_PG to a Postgres connection string to run the V149 convergence pin."); + + var ct = TestContext.Current.CancellationToken; + + await using var scratch = await ScratchPostgres.CreateAsync(baseConnectionString!, ct); + await using var connection = new NpgsqlConnection(scratch.ConnectionString); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + + for (var i = 0; i < 2; i++) + { + await using var mapCommand = new NpgsqlCommand(QueryStorePlanMap.CreateTableSql, connection); + await mapCommand.ExecuteNonQueryAsync(ct); + + await using var textCommand = new NpgsqlCommand(QueryStoreTextStore.CreateTableSql, connection); + await textCommand.ExecuteNonQueryAsync(ct); + } + + Assert.False(await IndexExistsAsync(connection, "idx_query_store_plan_map_last_seen", ct), + "convergence must not recreate idx_query_store_plan_map_last_seen"); + Assert.False(await IndexExistsAsync(connection, "idx_query_store_text_last_seen", ct), + "convergence must not recreate idx_query_store_text_last_seen"); + } + + /// + /// The payoff: two touches of the same map row after V149 report a HOT update. The first touch (row + /// starts fresh, no prior version) may or may not be HOT depending on page layout; the SECOND touch of + /// the same row — after the guard interval has re-elapsed — is the one this asserts, because it is the + /// steady-state case the field actually runs (every guard cycle touches rows that were touched last + /// cycle too). Uses pg_stat_force_next_flush() plus a short settle (as + /// QueryStoreIntervalWideGridLiveTests and StoreToastAndCheckpointerTests do) because + /// pg_stat_user_tables counters are backend-pending and throttled to once a second. + /// + [Fact] + public async Task TwoTouchesOfTheSameRow_TheSecondTouchReportsAHotUpdate() + { + var baseConnectionString = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(baseConnectionString), + "Set DARLING_TEST_PG to a Postgres connection string to run the V149 HOT pin."); + + var ct = TestContext.Current.CancellationToken; + + await using var scratch = await ScratchPostgres.CreateAsync(baseConnectionString!, ct); + await using var connection = new NpgsqlConnection(scratch.ConnectionString); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + + await using var seed = new NpgsqlCommand( + "INSERT INTO collect.query_store_plan_map (server_id, database_name, plan_id, digest, plan_hash, last_seen) " + + "VALUES (1, 'db', 1, '\\x01', 'h1', now() AT TIME ZONE 'UTC' - interval '1 day')", connection); + await seed.ExecuteNonQueryAsync(ct); + + var touch = "UPDATE collect.query_store_plan_map SET last_seen = now() AT TIME ZONE 'UTC' " + + "WHERE server_id = 1 AND database_name = 'db' AND plan_id = 1"; + + // First touch: may build a fresh heap version, not the steady-state case. + await using (var first = new NpgsqlCommand(touch, connection)) + { + await first.ExecuteNonQueryAsync(ct); + } + + await using (var flush1 = new NpgsqlCommand("SELECT pg_stat_force_next_flush()", connection)) + { + await flush1.ExecuteScalarAsync(ct); + } + + var before = await ReadHotUpdatesAsync(connection, ct); + + // Second touch: the steady-state case — the row already has a settled heap version with room. + await using (var second = new NpgsqlCommand(touch, connection)) + { + await second.ExecuteNonQueryAsync(ct); + } + + await using (var flush2 = new NpgsqlCommand("SELECT pg_stat_force_next_flush()", connection)) + { + await flush2.ExecuteScalarAsync(ct); + } + + var after = await ReadHotUpdatesAsync(connection, ct); + + Assert.True(after > before, + $"the second touch of an already-settled row must be HOT (n_tup_hot_upd {before} -> {after})"); + } + + private static async Task IndexExistsAsync(NpgsqlConnection connection, string indexName, System.Threading.CancellationToken ct) + { + await using var command = new NpgsqlCommand( + "SELECT EXISTS (SELECT 1 FROM pg_indexes WHERE schemaname = 'collect' AND indexname = $1)", connection); + command.Parameters.AddWithValue(indexName); + return (bool)(await command.ExecuteScalarAsync(ct))!; + } + + private static async Task HasFillfactor90Async(NpgsqlConnection connection, string tableName, System.Threading.CancellationToken ct) + { + await using var command = new NpgsqlCommand( + "SELECT coalesce((SELECT c.reloptions FROM pg_class c WHERE c.oid = $1::regclass) @> ARRAY['fillfactor=90'], false)", connection); + command.Parameters.AddWithValue(tableName); + return (bool)(await command.ExecuteScalarAsync(ct))!; + } + + private static async Task ReadHotUpdatesAsync(NpgsqlConnection connection, System.Threading.CancellationToken ct) + { + await using var command = new NpgsqlCommand( + "SELECT n_tup_hot_upd FROM pg_stat_user_tables WHERE schemaname = 'collect' AND relname = 'query_store_plan_map'", connection); + var scalar = await command.ExecuteScalarAsync(ct); + return Convert.ToInt64(scalar, System.Globalization.CultureInfo.InvariantCulture); + } +} diff --git a/Darling/Darling.Tests/QueryStoreLivenessTablePruneLiveTests.cs b/Darling/Darling.Tests/QueryStoreLivenessTablePruneLiveTests.cs new file mode 100644 index 000000000..fb5705d3f --- /dev/null +++ b/Darling/Darling.Tests/QueryStoreLivenessTablePruneLiveTests.cs @@ -0,0 +1,204 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Threading.Tasks; +using Npgsql; +using PerformanceMonitor.Darling.Service; +using PerformanceMonitor.Darling.Storage; +using Xunit; + +namespace Darling.Tests; + +/// +/// The live loop test for #4250 item 3's row-capped prune (), +/// through the real purge path () rather than calling the builder or +/// directly. Seeds collect.query_store_plan_map and +/// collect.query_store_text with more than 2x a SMALL test cap of expired rows plus a handful of +/// in-cutoff rows, runs the product's own sweep, and asserts every expired row is gone, every in-cutoff row +/// remains, and the batch count matches the drain contract (ceil(expired / cap) full batches, the drain +/// loop stops on the first batch that clears fewer than the cap). +/// +/// Uses the test seam (livenessTouchedTablePruneRowCap) +/// rather than seeding ~650k rows per table: the production call sites omit the parameter, so production +/// always runs the shipped 300,000 constant unchanged; only this test passes a small cap (1,000), pinned +/// separately below against the shipped default. +/// +/* #1776 own-store: mints its own scratch database through ScratchPostgres and never touches the shared + store's tables, so it cannot race the live collection. */ +public sealed class QueryStoreLivenessTablePruneLiveTests +{ + private const int TestCap = 1_000; + + private static string? ConnectionString => Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + + /// + /// The seam defaults to the shipped production constant — a test that forgets to pass a cap still + /// exercises the real 300,000-row shape, not a silently-different one. + /// + [Fact] + public void TheSeamDefaultsToTheShippedProductionCap() + { + var method = typeof(DarlingRetention).GetMethod( + nameof(DarlingRetention.PurgeAsync), System.Reflection.BindingFlags.Public | System.Reflection.BindingFlags.Static)!; + var parameter = Array.Find(method.GetParameters(), p => p.Name == "livenessTouchedTablePruneRowCap")!; + Assert.NotNull(parameter); + Assert.Equal(DarlingRetention.LivenessTouchedTablePruneRowCap, (int)parameter.DefaultValue!); + } + + /// + /// The loop test itself. Seeds * 2 + a short remainder of EXPIRED rows in each + /// table (so the drain needs 3 batches: two full at the cap, one short), plus 5 IN-CUTOFF rows that must + /// survive. Runs — the same product entry point the daily + /// sweep and the on-demand purge_now command call — with the small test cap, then asserts: + /// every expired row gone, every in-cutoff row kept, and the logged batch count for each table matches + /// the drain contract via the {Batches} field of 's captured + /// "Retention purge drained ... in {Batches} batch(es)" line (batchSize > 1 is required for that line + /// to be written — see PurgeOneAsync — which both callers satisfy). + /// + /// A stop-after-one-batch mutation at the drain loop for these callers (temporarily forcing + /// to return after its FIRST execution) turns this RED: + /// see the PR body for the exact revert-and-rerun. That mutation is not committed here — it is a + /// temporary product-code edit, verified and reverted by hand. + /// + [Fact] + public async Task PurgeAsync_DrainsBothLivenessTouchedTables_ClearingExpiredKeepingCurrent_InTheContractedBatchCount() + { + var baseConnectionString = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(baseConnectionString), + "Set DARLING_TEST_PG to a Postgres connection string to run the #4250 item 3 live loop test."); + + var ct = TestContext.Current.CancellationToken; + + await using var scratch = await ScratchPostgres.CreateAsync(baseConnectionString!, ct); + await using var connection = new NpgsqlConnection(scratch.ConnectionString); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + + var utcNow = DateTime.SpecifyKind(DateTime.UtcNow, DateTimeKind.Unspecified); + + /* Both cutoffs are widestFactRetentionDays (the widest SHARED retention of any dim-feeding + collector on this store, not a fixed 1-day floor — the scratch store's own catalog can widen it + well past 1 day) plus each table's own margin. Measured live above: the actual map cutoff landed + at 31 days back on this scratch store. 400 days back is comfortably past every horizon any + collector's shared RetentionDays can produce (the widest of them, plan_force_actions, is 365 + days), matching the margin the existing live retention E2Es already use for their own + "definitely expired" rows. A 1-hour-back row is comfortably inside every one of them. */ + var expiredStamp = utcNow.AddDays(-400); + var keptStamp = utcNow.AddHours(-1); + + const int expiredPerTable = TestCap * 2 + 137; // 2 full batches + 1 short batch = 3 batches, drain contract. + const int keptPerTable = 5; + + await SeedMapRowsAsync(connection, expiredPerTable, keptPerTable, expiredStamp, keptStamp, ct); + await SeedTextRowsAsync(connection, expiredPerTable, keptPerTable, expiredStamp, keptStamp, ct); + + await using var postgres = NpgsqlDataSource.Create(scratch.ConnectionString); + var purgeLog = new CapturingTestLogger(); + + await DarlingRetention.PurgeAsync( + postgres, timescaleAvailable: false, purgeLog, ct, + livenessTouchedTablePruneRowCap: TestCap); + + // Every expired row is gone; every in-cutoff row survives, on BOTH tables. + await AssertMapSurvivorsAsync(connection, keptPerTable, ct, purgeLog); + await AssertTextSurvivorsAsync(connection, keptPerTable, ct, purgeLog); + + // The drain contract: ceil(expiredPerTable / cap) = 3 batches for each table (2 full + 1 short). + var mapBatches = ExtractBatchCount(purgeLog, "query_store_plan_map"); + var textBatches = ExtractBatchCount(purgeLog, "query_store_text"); + Assert.Equal(3, mapBatches); + Assert.Equal(3, textBatches); + } + + private static async Task SeedMapRowsAsync( + NpgsqlConnection connection, int expiredCount, int keptCount, DateTime expiredStamp, DateTime keptStamp, System.Threading.CancellationToken ct) + { + await using var command = new NpgsqlCommand( + "INSERT INTO collect.query_store_plan_map (server_id, database_name, plan_id, digest, plan_hash, last_seen) " + + "SELECT 1, 'db', gs, ('\\x' || lpad(to_hex(gs), 8, '0'))::bytea, 'h' || gs, $2 " + + "FROM generate_series(1, $1) gs", connection); + command.Parameters.AddWithValue(expiredCount); + command.Parameters.AddWithValue(expiredStamp); + await command.ExecuteNonQueryAsync(ct); + + await using var keptCommand = new NpgsqlCommand( + "INSERT INTO collect.query_store_plan_map (server_id, database_name, plan_id, digest, plan_hash, last_seen) " + + "SELECT 2, 'db', gs, ('\\x' || lpad(to_hex(gs), 8, '0'))::bytea, 'h' || gs, $2 " + + "FROM generate_series(1, $1) gs", connection); + keptCommand.Parameters.AddWithValue(keptCount); + keptCommand.Parameters.AddWithValue(keptStamp); + await keptCommand.ExecuteNonQueryAsync(ct); + } + + private static async Task SeedTextRowsAsync( + NpgsqlConnection connection, int expiredCount, int keptCount, DateTime expiredStamp, DateTime keptStamp, System.Threading.CancellationToken ct) + { + await using var command = new NpgsqlCommand( + "INSERT INTO collect.query_store_text (server_id, database_name, query_id, query_sql_text, last_seen) " + + "SELECT 1, 'db', gs, 'SELECT ' || gs, $2 " + + "FROM generate_series(1, $1) gs", connection); + command.Parameters.AddWithValue(expiredCount); + command.Parameters.AddWithValue(expiredStamp); + await command.ExecuteNonQueryAsync(ct); + + await using var keptCommand = new NpgsqlCommand( + "INSERT INTO collect.query_store_text (server_id, database_name, query_id, query_sql_text, last_seen) " + + "SELECT 2, 'db', gs, 'SELECT ' || gs, $2 " + + "FROM generate_series(1, $1) gs", connection); + keptCommand.Parameters.AddWithValue(keptCount); + keptCommand.Parameters.AddWithValue(keptStamp); + await keptCommand.ExecuteNonQueryAsync(ct); + } + + private static async Task AssertMapSurvivorsAsync( + NpgsqlConnection connection, int expectedKept, System.Threading.CancellationToken ct, CapturingTestLogger purgeLog) + { + await using var expiredCount = new NpgsqlCommand( + "SELECT COUNT(*) FROM collect.query_store_plan_map WHERE server_id = 1", connection); + Assert.True(0L == (long)(await expiredCount.ExecuteScalarAsync(ct))!, purgeLog.Joined); + + await using var keptCount = new NpgsqlCommand( + "SELECT COUNT(*) FROM collect.query_store_plan_map WHERE server_id = 2", connection); + Assert.True((long)expectedKept == (long)(await keptCount.ExecuteScalarAsync(ct))!, purgeLog.Joined); + } + + private static async Task AssertTextSurvivorsAsync( + NpgsqlConnection connection, int expectedKept, System.Threading.CancellationToken ct, CapturingTestLogger purgeLog) + { + await using var expiredCount = new NpgsqlCommand( + "SELECT COUNT(*) FROM collect.query_store_text WHERE server_id = 1", connection); + Assert.True(0L == (long)(await expiredCount.ExecuteScalarAsync(ct))!, purgeLog.Joined); + + await using var keptCount = new NpgsqlCommand( + "SELECT COUNT(*) FROM collect.query_store_text WHERE server_id = 2", connection); + Assert.True((long)expectedKept == (long)(await keptCount.ExecuteScalarAsync(ct))!, purgeLog.Joined); + } + + /// + /// Pulls the batch count out of PurgeOneAsync's "Retention purge drained {Rows} row(s) from + /// {Table} in {Batches} batch(es) (cap {Cap}), cutoff ..." line for the given table's unqualified name. + /// + private static int ExtractBatchCount(CapturingTestLogger purgeLog, string tableSuffix) + { + foreach (var line in purgeLog.Lines) + { + if (line.Contains("Retention purge drained", StringComparison.Ordinal) + && line.Contains("from collect." + tableSuffix + " in", StringComparison.Ordinal)) + { + var marker = " in "; + var afterIn = line.IndexOf(marker, line.IndexOf("from collect." + tableSuffix, StringComparison.Ordinal), StringComparison.Ordinal) + marker.Length; + var spaceIndex = line.IndexOf(' ', afterIn); + return int.Parse(line[afterIn..spaceIndex]); + } + } + + Assert.Fail($"no drain-batch log line found for collect.{tableSuffix}; {purgeLog.Joined}"); + return -1; + } +} diff --git a/Darling/Darling.Tests/QueryStoreTextStoreTests.cs b/Darling/Darling.Tests/QueryStoreTextStoreTests.cs index bd56b8779..0d1022fd7 100644 --- a/Darling/Darling.Tests/QueryStoreTextStoreTests.cs +++ b/Darling/Darling.Tests/QueryStoreTextStoreTests.cs @@ -37,7 +37,11 @@ public sealed class QueryStoreTextStoreTests /// /// The rung and the helper's own DDL must agree, or a fresh store and an upgraded one get different /// tables — the same discipline the ladder diff enforces for collector tables, applied by hand because a - /// non-collector table is outside that generator. + /// non-collector table is outside that generator. V149 (#4250) drops the rung's last_seen index + /// and sets fillfactor 90 on the live table, so the helper's CreateTableSql (which a fresh store + /// never runs directly — runs the V74 rung, then V149) is now the + /// POST-V149 shape, not the V74 rung's byte-for-byte shape. This asserts the columns V74 and V149 both + /// agree on rather than the whole string. /// [Fact] public void TheRungMatchesTheHelpersCreateTableSql() @@ -45,7 +49,8 @@ public void TheRungMatchesTheHelpersCreateTableSql() var rung = PgMigrations.Scripts.Single(s => s.Version == 74); Assert.Equal("query-store-text", rung.Name); - Assert.Equal(Normalize(QueryStoreTextStore.CreateTableSql), Normalize(rung.Sql)); + Assert.Contains("PRIMARY KEY (server_id, database_name, query_id)", Normalize(rung.Sql), StringComparison.Ordinal); + Assert.Contains("PRIMARY KEY (server_id, database_name, query_id)", Normalize(QueryStoreTextStore.CreateTableSql), StringComparison.Ordinal); } /// @@ -122,13 +127,15 @@ public void TheProbeAsksForTheTable_AndTheThreePlacesAgree() /// /// The table's shape: keyed on query_id — which is already a stored fact column, so this rung - /// adds a table and touches nothing existing — and indexed on the column it is pruned by. + /// adds a table and touches nothing existing. Since V149 (#4250) the prune no longer has an index to + /// lean on — last_seen is still the pruned column, it is just no longer indexed, so the helper's + /// live DDL carries the column but not the (now-dropped) index. /// [Fact] - public void TheTableIsKeyedOnQueryIdAndIndexedForThePrune() + public void TheTableIsKeyedOnQueryIdAndPrunedOnLastSeen() { Assert.Contains("PRIMARY KEY (server_id, database_name, query_id)", QueryStoreTextStore.CreateTableSql, StringComparison.Ordinal); - Assert.Contains("idx_query_store_text_last_seen", QueryStoreTextStore.CreateTableSql, StringComparison.Ordinal); + Assert.DoesNotContain("idx_query_store_text_last_seen", QueryStoreTextStore.CreateTableSql, StringComparison.Ordinal); Assert.Contains(QueryStoreTextStore.LastSeenColumn, QueryStoreTextStore.CreateTableSql, StringComparison.Ordinal); } @@ -153,19 +160,9 @@ public void TheUpsertOverwritesTheTextAndOrdersByTheConflictKey() Assert.Contains("WHERE EXCLUDED.last_seen >= query_store_text.last_seen", QueryStoreTextStore.UpsertSql, StringComparison.Ordinal); } - /// - /// The prune is bounded to roughly one chunk-width of the OLDEST rows per call, so a single sweep cannot - /// take an unbounded row lock — the same shape as the plan map's. - /// - [Fact] - public void ThePruneIsBoundedToOneChunkWidth() - { - var sql = QueryStoreTextStore.PruneSql(7); - - Assert.StartsWith("DELETE FROM collect.query_store_text WHERE last_seen < $1", sql, StringComparison.Ordinal); - Assert.Contains("INTERVAL '7 days'", sql, StringComparison.Ordinal); - Assert.Contains("SELECT min(last_seen)", sql, StringComparison.Ordinal); - } + /* The prune's statement shape (row-capped, no ORDER BY, no min() subqueries) is pinned in + DarlingRetentionTests against DarlingRetention.UnorderedRowCappedDeleteSql, the shared builder + #4250 item 3 introduced for this table and query_store_plan_map. */ /// /// The retention margin is ADDED to the fact horizon, never subtracted. Text has to OUTLIVE the rows diff --git a/Darling/Darling.Tests/ReadLatencyFlushLiveTests.cs b/Darling/Darling.Tests/ReadLatencyFlushLiveTests.cs index 01ab9c849..f242c59b0 100644 --- a/Darling/Darling.Tests/ReadLatencyFlushLiveTests.cs +++ b/Darling/Darling.Tests/ReadLatencyFlushLiveTests.cs @@ -24,8 +24,9 @@ namespace Darling.Tests; /// probe) and the FLUSH (): a field-shaped record-and-flush /// round trip, a same-hour second flush, retention, and an empty flush. /// -/// This file's "I am the top rung" claim takes over from ComposeStatementTimeoutV147MigrationLiveTests -/// (V147) now that V148 has landed. +/// This file's "I am the top rung" claim moved to QueryStoreLivenessHotTouchLiveTests (V149) +/// now that V149 has landed; this file's own rung/probe facts below keep asserting what stays true forever +/// (present, in-order, gated behind the arm above it) rather than "is exactly the top". /// /* #1776 own-store: deliberately NOT [Collection("live-postgres")]. Each fact mints its own scratch database through ScratchPostgres and never touches the shared store's tables, so it cannot race the live collection @@ -45,15 +46,15 @@ public sealed class ReadLatencyFlushLiveTests /// ComposeStatementTimeoutV147MigrationLiveTests (V147) now that V148 has landed. /// [Fact] - public void TheRungIsRegisteredAtTheTopOfADenseLadder() + public void TheRungIsRegistered_AndTheLadderIsDenseAboveTheHistoricalGap() { var versions = PgMigrations.Scripts.Select(s => s.Version).ToList(); Assert.Equal("read-latency", PgMigrations.Scripts.Single(s => s.Version == RungVersion).Name); - Assert.Equal(StorageVersion.SchemaVersion, PgMigrations.Scripts[^1].Version); - Assert.Equal(StorageVersion.SchemaVersion, versions.Max()); - Assert.Equal(RungVersion, StorageVersion.SchemaVersion); Assert.Equal(versions.Distinct().OrderBy(v => v), versions); + + var above = versions.Where(v => v > 45).OrderBy(v => v).ToList(); + Assert.Equal(Enumerable.Range(above[0], above.Count), above); } /// @@ -62,7 +63,7 @@ public void TheRungIsRegisteredAtTheTopOfADenseLadder() /// is . /// [Fact] - public void TheProbeMapsAFullyMigratedStoreToThisTopRung() + public void TheProbeCarriesThisRungsSentinel_AndTheArmSitsBelowTheCurrentTop() { Assert.Contains( "table_name = 'read_latency'", @@ -70,29 +71,29 @@ public void TheProbeMapsAFullyMigratedStoreToThisTopRung() var viewer = RepoFile.ReadRepoFile("Darling", "PerformanceMonitor.Darling.Viewer", "ViewerDataService.cs"); Assert.Contains($"reader.GetBoolean({ProbeOrdinal})", viewer, StringComparison.Ordinal); - Assert.DoesNotContain($"reader.GetBoolean({ProbeOrdinal + 1})", viewer, StringComparison.Ordinal); - - Assert.Equal(StorageVersion.SchemaVersion, ViewerDataService.RequiredStoreSchemaVersion); var method = typeof(ViewerDataService).GetMethod("MapProbedSchemaVersion", System.Reflection.BindingFlags.NonPublic | System.Reflection.BindingFlags.Static)!; var arity = method.GetParameters().Length; - Assert.Equal(ProbeOrdinal, arity - 1); Assert.Equal("hasReadLatency", method.GetParameters()[ProbeOrdinal].Name); + /* Every rung above this one (V149's hasHotLivenessTouch) must also be false, or the map finds the + newer arm first and this assertion is checking the wrong rung's fallthrough. */ var all = Enumerable.Repeat((object)true, arity).ToArray(); - Assert.Equal(StorageVersion.SchemaVersion, (int)method.Invoke(null, all)!); - var behind = (object[])all.Clone(); behind[ProbeOrdinal] = false; + behind[arity - 1] = false; Assert.Equal(PreviousVersion, (int)method.Invoke(null, behind)!); + /* V149 (#4250) is now the top rung, so this arm no longer needs to be the LAST one — it only has + to sit below the current top's arm, which is what the ladder-dense invariant above already + guarantees is registered ahead of it. */ var thisArm = viewer.IndexOf("if (hasReadLatency)", StringComparison.Ordinal); - var previousArm = viewer.IndexOf("if (hasComposeTimeoutSixty)", StringComparison.Ordinal); + var topArm = viewer.IndexOf("if (hasHotLivenessTouch)", StringComparison.Ordinal); Assert.True(thisArm >= 0, "the viewer has no V148 sentinel arm \u2014 a fully-migrated store would map one rung short"); - Assert.True(thisArm < previousArm, "the V148 arm sits below V147's, so a current store maps one rung short"); + Assert.True(topArm >= 0 && topArm < thisArm, "the current top rung's arm must sit above the V148 arm"); Assert.Contains( - "return " + StorageVersion.SchemaVersion.ToString(System.Globalization.CultureInfo.InvariantCulture) + ";", - viewer[thisArm..previousArm], StringComparison.Ordinal); + "return " + RungVersion.ToString(System.Globalization.CultureInfo.InvariantCulture) + ";", + viewer[thisArm..(viewer.IndexOf("if (hasComposeTimeoutSixty)", StringComparison.Ordinal))], StringComparison.Ordinal); } [Fact] diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingRetention.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingRetention.cs index ef9137917..9093f4e2c 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingRetention.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingRetention.cs @@ -214,6 +214,30 @@ internal static int EffectivePurgeRetentionDays(string collectorName, int resolv /// internal const string TerminalCommandStatuses = "status IN ('succeeded', 'failed')"; + /// + /// The row cap for 's and 's prune + /// (#4250 item 3), shared by both tables rather than sized per table, because both clear it in ONE + /// batch at every measured rate and a single named constant is easier to keep correct than two. + /// + /// Sizing (read-only field histogram of last_seen age, collect.query_store_plan_map + /// / collect.query_store_text). Map rows retire only up to age 14 days, at roughly 180k-255k + /// rows/day; text rows retire up to age 32 days, at roughly 150k-220k rows/day. The purge runs twice a + /// day ('s cadence), so a normal, on-schedule run's due backlog + /// is about half a day's worth — map ≈90k-130k, text ≈75k-110k — and even ONE missed run (the whole + /// day's retirement landing on the next one) is at most the single busiest day observed: map 255k, text + /// 220k. 300,000 clears every one of those in a single batch (no second, empty-batch statement), while + /// still bounding a real backlog — a 10-day gap drains in roughly 7-9 batches per table (10 days × + /// ~200k ÷ 300k), each its own committed transaction, never one unbounded delete. + /// + /// Transaction size per batch, worst case (one missed run). Map: 255,000 rows × ≈175 B/row + /// (709 MB / 4.04 M rows, field-measured) ≈ 45 MB. Text: 220,000 rows × ≈1.4 KB/row (4,992 MB main + + /// 4,355 MB TOAST / 6.86 M rows, field-measured, TOAST included) ≈ 300 MB. Both are single-slice-sized + /// or smaller against the old shape's own one-day slice, which already deleted comparable row counts + /// per statement — this is not a new order of magnitude of transaction size, only a cheaper way to find + /// the same rows. + /// + internal const int LivenessTouchedTablePruneRowCap = 300_000; + /* Each one-day slice is bounded work (see TimeSlicedDeleteSql); the generous 300s per-slice command timeout (well above Npgsql's 30s default) is belt-and-suspenders for a slow disk. Slicing is also what keeps a large first purge from ever hitting a timeout at all — a single unbounded DELETE could roll @@ -261,9 +285,17 @@ keeps a large first purge from ever hitting a timeout at all — a single unboun /// before the GC may take it, independent of the fact-coupled horizon. 0 (the default here, for /// callers and tests that predate the knob) disables it — the fact-coupled horizon stands alone. /// + /// + /// TEST SEAM ONLY (#4250 item 3's live loop test): the row cap the map/text prune uses in place of + /// . Every real caller ('s daily + /// sweep and its on-demand purge_now command) omits this, so production always runs the shipped + /// 300,000 constant; a live test can pass a small cap (e.g. 1,000) to exercise a multi-batch drain + /// without seeding hundreds of thousands of rows. + /// public static async Task PurgeAsync( NpgsqlDataSource postgres, bool timescaleAvailable, ILogger? logger, CancellationToken cancellationToken, - Func? retentionDaysFor = null, int planContentRetentionDays = 0) + Func? retentionDaysFor = null, int planContentRetentionDays = 0, + int livenessTouchedTablePruneRowCap = LivenessTouchedTablePruneRowCap) { /* Clamp at the destructive sink, like retentionDaysFor's clamp below (review catch): the value arrives pre-clamped only when a store read succeeded and ApplyToConfig ran. On a @@ -577,8 +609,10 @@ dim cutoff overtakes this one and a live map row can point at deleted content var mapCutoff = ComputeMapCutoff(utcNow, widestFactRetentionDays, planContentRetentionDays); var mapDeleted = await PurgeOneAsync( postgres, QueryStorePlanMap.TableName, - QueryStorePlanMap.PruneSql(TimescaleSupport.ChunkIntervalDays), - mapCutoff, logger, cancellationToken); + UnorderedRowCappedDeleteSql( + QueryStorePlanMap.TableName, QueryStorePlanMap.LastSeenColumn, livenessTouchedTablePruneRowCap), + mapCutoff, logger, cancellationToken, + batchSize: livenessTouchedTablePruneRowCap); if (mapDeleted is not null) { tablesPurged++; @@ -595,8 +629,10 @@ statement that never had text rather than as one whose text expired. */ var queryTextCutoff = utcNow.AddDays(-(widestFactRetentionDays + QueryStoreTextStore.PruneMarginDays)); var queryTextDeleted = await PurgeOneAsync( postgres, QueryStoreTextStore.TableName, - QueryStoreTextStore.PruneSql(TimescaleSupport.ChunkIntervalDays), - queryTextCutoff, logger, cancellationToken); + UnorderedRowCappedDeleteSql( + QueryStoreTextStore.TableName, QueryStoreTextStore.LastSeenColumn, livenessTouchedTablePruneRowCap), + queryTextCutoff, logger, cancellationToken, + batchSize: livenessTouchedTablePruneRowCap); if (queryTextDeleted is not null) { tablesPurged++; @@ -1057,6 +1093,48 @@ internal static string RowCappedDeleteSql(string table, string timeColumn, int c + $"SELECT ctid FROM {table} WHERE {timeColumn} < $1 " + $"ORDER BY {timeColumn} LIMIT {cap})"; + /// + /// A row-capped purge with NO ORDER BY (#4250 item 3) — the sibling of + /// for the two tables whose V149 migration dropped their + /// last_seen btree (, ): with no + /// index, ORDER BY {timeColumn} forces a sort of every row under the cutoff before the + /// LIMIT can apply, which is a second full pass over exactly the rows the seq scan already read. + /// Dropping it removes that pass; the row cap and the drain contract are unchanged. + /// + /// Why dropping the ordering is still correct here, unlike the case + /// 's own doc warns against. That doc's "an unordered cap would + /// nibble arbitrary rows and leave the floor where it was" describes a MOVING cutoff recomputed from + /// the table's own remaining minimum each call — nibbling from the middle there really can strand old + /// rows below a floor that never advances. This builder's cutoff is FIXED: it is the caller's own + /// $1, computed once from wall time before the drain starts ( for + /// the map, the fact-horizon-plus-margin sum for text), not from what happened to survive the last + /// batch. Every row the predicate matches is expired by that same fixed cutoff regardless of which ones + /// a given batch happens to take, and each batch deletes rows the predicate matched, so the matching set + /// strictly shrinks every round. A drain that keeps taking rows from a strictly shrinking set reaches + /// empty in a bounded number of rounds — arbitrary selection changes WHICH rows go in which round, never + /// WHETHER the drain terminates or converges on the same fixed floor. "Leave the floor where it was" is + /// exactly what cannot happen: there is no floor to leave behind, only a set that only ever gets + /// smaller. + /// + /// The #2316 margin ordering (, the text-outlives-facts margin on + /// ) is about which CUTOFF each table's caller computes + /// relative to the others — map's cutoff stays strictly newer than the plan dim's, text's stays older + /// than the facts that reference it — never about the order rows are removed WITHIN one table's own + /// delete. Dropping the intra-table ordering touches none of that: the cutoffs the callers pass in are + /// unchanged, and every row this builder can touch in either table is already past its own table's + /// cutoff no matter which order the batch takes them in. + /// + /// The ctid idiom itself requires a plain table — reading ctid through + /// TimescaleDB's transparent decompression is unsupported (#1564) — and both + /// and are plain + /// tables, never hypertables (no create_hypertable call anywhere in their migrations), so that + /// constraint never applies to either. + /// + internal static string UnorderedRowCappedDeleteSql(string table, string timeColumn, int cap) => + $"DELETE FROM {table} WHERE ctid IN (" + + $"SELECT ctid FROM {table} WHERE {timeColumn} < $1 " + + $"LIMIT {cap})"; + /// /// The dimension GC's cutoff (#1795): the ASSUMED horizon (widest dim-feeding fact retention + /// drop_chunks granularity + 1 day for the diff --git a/Darling/PerformanceMonitor.Darling.Storage/PgMigrations.cs b/Darling/PerformanceMonitor.Darling.Storage/PgMigrations.cs index 30c1c85f2..cc3f9a328 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/PgMigrations.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/PgMigrations.cs @@ -228,6 +228,7 @@ holds the ordering. */ new Migration(146, "managed-conf-verdicts", V146Sql), new Migration(147, "compose-statement-timeout-sixty", V147Sql), new Migration(148, "read-latency", V148Sql), + new Migration(149, "query-store-liveness-hot-touch", V149Sql), }; /// @@ -2326,6 +2327,37 @@ PRIMARY KEY (metric_time, surface, route, outcome) CREATE INDEX IF NOT EXISTS idx_read_latency_time ON collect.read_latency(metric_time);"; + /// + /// V149 — the Query Store liveness touch becomes a HOT update on collect.query_store_plan_map and + /// collect.query_store_text (#4250): drops each table's last_seen btree index and sets + /// fillfactor = 90. TouchAndProbeSql on both tables (, + /// ) only ever writes last_seen and, conditionally, an + /// unindexed hash column — so once the index is gone, no indexed column the touch changes remains, and a + /// page with fillfactor headroom lets the new tuple stay on its old page: both conditions Postgres's HOT + /// optimization needs. Confirmed by git grep against the service and storage code: the only other + /// reader of either index was each table's own PruneSql time-sliced DELETE ... WHERE last_seen < + /// $1 — no reader orders, filters, or range-scans last_seen for anything else. This change + /// replaces that prune on both tables with : + /// one capped sequential pass per batch, with no ordering or subquery over last_seen to lose by + /// dropping the index. See that builder's summary for the shape and the measured cost. + /// + /// Plain DROP INDEX and ALTER TABLE ... SET (fillfactor = ...), not CONCURRENTLY: + /// MigrateAsync wraps every rung in a transaction, and CREATE/DROP INDEX CONCURRENTLY cannot + /// run inside one. Both operations here are metadata-only — the index drop does not touch the heap, and + /// the fillfactor change only affects pages written from here on — so the short ACCESS EXCLUSIVE + /// each takes is a catalog update, not a rewrite; nothing else in the migrate session holds a competing + /// lock on either table at that moment. + /// + /// Fillfactor 90 is prospective only: existing pages, packed at the old default of 100, do not gain + /// HOT headroom until they are rewritten by organic churn (the tables are continuously purged) or a + /// deliberate rewrite. No rewrite ships in this rung. + /// + private const string V149Sql = @" +DROP INDEX IF EXISTS collect.idx_query_store_plan_map_last_seen; +DROP INDEX IF EXISTS collect.idx_query_store_text_last_seen; +ALTER TABLE collect.query_store_plan_map SET (fillfactor = 90); +ALTER TABLE collect.query_store_text SET (fillfactor = 90);"; + /// /// V2 — the service's observability store: the servers registry (upserted on every /// successful connect) and the per-run collection_log. Column names deliberately mirror diff --git a/Darling/PerformanceMonitor.Darling.Storage/QueryStorePlanMap.cs b/Darling/PerformanceMonitor.Darling.Storage/QueryStorePlanMap.cs index bfedece25..3af1307dc 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/QueryStorePlanMap.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/QueryStorePlanMap.cs @@ -7,7 +7,6 @@ */ using System; -using System.Globalization; namespace PerformanceMonitor.Darling.Storage; @@ -53,9 +52,14 @@ public static class QueryStorePlanMap public const string LastSeenColumn = "last_seen"; /* This const mirrors the IMMUTABLE migration rung that created the table and stays byte-frozen with it. - The LIVE shape differs in one place: V77 relaxed digest to nullable for the #2312 content-less marker + The LIVE shape differs in two places: V77 relaxed digest to nullable for the #2312 content-less marker rows (a plan whose XML the engine cannot persist gets a map row with a NULL digest, so the probe reads - it as known instead of refetching it forever). */ + it as known instead of refetching it forever), and V149 (#4250) drops the last_seen btree index and + sets fillfactor 90, so the liveness touch's UPDATE (see TouchAndProbeSql below) can go HOT: last_seen + is the only indexed column that touch ever changed, so once nothing indexes it and the page has + fillfactor headroom, the touch stops writing a new index entry and (once the page has room) stops + moving the tuple. The prune (DarlingRetention.UnorderedRowCappedDeleteSql, item 3 of #4250) is the + only other reader of that index and keeps working off a row-capped sequential scan instead. */ public const string CreateTableSql = @"CREATE TABLE IF NOT EXISTS collect.query_store_plan_map ( server_id integer NOT NULL, database_name text NOT NULL, @@ -64,8 +68,7 @@ public static class QueryStorePlanMap plan_hash text, last_seen timestamp NOT NULL, PRIMARY KEY (server_id, database_name, plan_id) -); -CREATE INDEX IF NOT EXISTS idx_query_store_plan_map_last_seen ON collect.query_store_plan_map(last_seen);"; +);"; /* Deliberately NO index on digest. Nothing reads this table by digest: readers resolve (server_id, database_name, plan_id) -> digest through the primary key, the liveness touch joins on that @@ -254,23 +257,23 @@ LEFT JOIN collect.query_store_plan_map AS m public static bool MarginOrderingHolds(int chunkIntervalDays) => PruneMarginDays < chunkIntervalDays + 1; - /// - /// Retires map rows whose facts have all aged out: an index range scan on - /// against the fact horizon plus , time-sliced like every sibling purge. - /// - /// Timestamp-driven, NOT an existence check against query_store_stats. An anti-join against a - /// 43 GB hypertable per map row is exactly the cost this architecture avoids, and it is unnecessary here - /// because keeps last_seen current for anything live — the same argument the - /// dimension GC already rests on, applied to one more timestamped table. - /// - /// A plan whose query goes quiet needs no special handling: it stops being touched, its facts age out - /// within retention, and the margin ordering retires the map row before the dim row it points at. - /// - public static string PruneSql(int chunkIntervalDays) => - "DELETE FROM collect.query_store_plan_map WHERE " + LastSeenColumn + " < $1" + - " AND " + LastSeenColumn + " >= (SELECT min(" + LastSeenColumn + ") FROM collect.query_store_plan_map WHERE " + - LastSeenColumn + " < $1)" + - " AND " + LastSeenColumn + " < (SELECT min(" + LastSeenColumn + ") FROM collect.query_store_plan_map WHERE " + - LastSeenColumn + " < $1) + INTERVAL '" + - chunkIntervalDays.ToString(CultureInfo.InvariantCulture) + " days'"; + /* Retires map rows whose facts have all aged out. Since V149 (#4250) dropped LastSeenColumn's btree + index, the three-scan slice shape this table used to run here (two min() subqueries plus the + DELETE) cost a sequential scan three times over per statement — on this table's field scale + (4.04 M rows / 709 MB), a real but modest cost; on its sibling query_store_text (6.86 M rows / ~5 + GB main fork + ~4.35 GB TOAST) the same shape reads on the order of tens of GB per slice. #4250 + item 3 replaced it fleet-wide (both tables) with DarlingRetention.UnorderedRowCappedDeleteSql: a + single ctid-capped scan with no ORDER BY, which this table's PRIMARY KEY is not indexed by anyway + (LastSeenColumn has no index to sort through), removing the two extra scans entirely. See that + builder's summary for why dropping the ordering is correct against a FIXED cutoff, and + DarlingRetention.LivenessTouchedTablePruneRowCap for how the cap is sized from the field's + last_seen age histogram. + + Timestamp-driven, NOT an existence check against query_store_stats. An anti-join against a 43 GB + hypertable per map row is exactly the cost this architecture avoids, and it is unnecessary here + because TouchAndProbeSql keeps last_seen current for anything live — the same argument the + dimension GC already rests on, applied to one more timestamped table. + + A plan whose query goes quiet needs no special handling: it stops being touched, its facts age out + within retention, and the margin ordering retires the map row before the dim row it points at. */ } diff --git a/Darling/PerformanceMonitor.Darling.Storage/QueryStoreTextStore.cs b/Darling/PerformanceMonitor.Darling.Storage/QueryStoreTextStore.cs index 680685903..ec6e08f71 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/QueryStoreTextStore.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/QueryStoreTextStore.cs @@ -6,7 +6,6 @@ * Licensed under the MIT License. See LICENSE file in the project root for full license information. */ -using System.Globalization; namespace PerformanceMonitor.Darling.Storage; @@ -54,6 +53,12 @@ public static class QueryStoreTextStore /// public const int PruneMarginDays = 2; + /* This const mirrors the IMMUTABLE V74 migration rung and stays byte-frozen with it, EXCEPT for V149 + (#4250): the rung drops the last_seen btree index and sets fillfactor 90 on the live table so the + liveness touch's UPDATE (see TouchAndProbeSql below) can go HOT — last_seen is the only indexed + column that touch ever changed, and the text/digest payload columns it never rewrites stay untouched + either way. The prune (DarlingRetention.UnorderedRowCappedDeleteSql, item 3 of #4250) is the only + other reader of that index and keeps working off a row-capped sequential scan instead. */ public const string CreateTableSql = @"CREATE TABLE IF NOT EXISTS collect.query_store_text ( server_id integer NOT NULL, database_name text NOT NULL, @@ -61,9 +66,7 @@ public static class QueryStoreTextStore query_sql_text text, last_seen timestamp NOT NULL, PRIMARY KEY (server_id, database_name, query_id) -); -CREATE INDEX IF NOT EXISTS idx_query_store_text_last_seen - ON collect.query_store_text(last_seen);"; +);"; /// /// Records what a text fetch landed. @@ -147,21 +150,21 @@ LEFT JOIN collect.query_store_text AS t AND t.query_id = batch.query_id ORDER BY batch.server_id, batch.database_name, batch.query_id"; - /// - /// Retires text whose facts have all aged out, bounded to roughly one chunk-width of the oldest rows - /// per call so a single sweep cannot take an unbounded row lock — the same shape and the same reason as - /// . - /// - /// Safe to run against live data because last_seen is refreshed by every pass that - /// re-observes a statement: a row can only fall behind the cutoff once nothing has referenced it for - /// the retention window, and re-fetching text for a statement that comes back is one row through a - /// watermark that has already expired. - /// - public static string PruneSql(int chunkIntervalDays) => - "DELETE FROM collect.query_store_text WHERE " + LastSeenColumn + " < $1" + - " AND " + LastSeenColumn + " >= (SELECT min(" + LastSeenColumn + ") FROM collect.query_store_text WHERE " + - LastSeenColumn + " < $1)" + - " AND " + LastSeenColumn + " < (SELECT min(" + LastSeenColumn + ") FROM collect.query_store_text WHERE " + - LastSeenColumn + " < $1) + INTERVAL '" + - chunkIntervalDays.ToString(CultureInfo.InvariantCulture) + " days'"; + /* Retires text whose facts have all aged out. This table used to run the same three-scan slice shape + as query_store_plan_map (two min() subqueries plus the DELETE); since V149 (#4250) dropped + LastSeenColumn's btree index, that shape's cost on this table specifically was measured at field + scale (6.86 M rows, ~5 GB main fork plus ~4.35 GB TOAST) at roughly 29 GB of logical buffers PER + SLICE (three sequential scans of a TOAST-heavy table) — tens of GB/day at the drain loop's normal + cadence, the largest of the costs #4250 item 3 measured. That item replaced it fleet-wide (both + this table and query_store_plan_map) with DarlingRetention.UnorderedRowCappedDeleteSql: a single + ctid-capped scan with no ORDER BY (this table has no index over LastSeenColumn to sort through + either), cutting the read cost to roughly a third of the old shape by dropping the two extra + scans. See that builder's summary for why the missing ORDER BY is still correct against a FIXED + cutoff, and DarlingRetention.LivenessTouchedTablePruneRowCap for the cap's sizing from the field's + last_seen age histogram (this table's rows retire up to ~32 days old, ~150k-220k rows/day). + + Safe to run against live data because last_seen is refreshed by every pass that re-observes a + statement: a row can only fall behind the cutoff once nothing has referenced it for the retention + window, and re-fetching text for a statement that comes back is one row through a watermark that + has already expired. */ } diff --git a/Darling/PerformanceMonitor.Darling.Storage/StorageVersion.cs b/Darling/PerformanceMonitor.Darling.Storage/StorageVersion.cs index 4b8057d7c..0059394e9 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/StorageVersion.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/StorageVersion.cs @@ -16,5 +16,5 @@ namespace PerformanceMonitor.Darling.Storage; /// public static class StorageVersion { - public const int SchemaVersion = 148; + public const int SchemaVersion = 149; } diff --git a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.cs b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.cs index f93da57ce..99ee907b3 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.cs +++ b/Darling/PerformanceMonitor.Darling.Viewer/ViewerDataService.cs @@ -971,7 +971,14 @@ rather than a value that could be masked by an operator override. It is not yet /* V148 (#4442 scope 2) probes the new read-latency histogram table. It is not yet read by any viewer surface, so this gate rests on the standing invariant alone. Named only in this probe line, never in prose, per the V71 finding. */ - EXISTS (SELECT 1 FROM information_schema.tables WHERE table_name = 'read_latency')"; + EXISTS (SELECT 1 FROM information_schema.tables WHERE table_name = 'read_latency'), + /* V149 (#4250) drops query_store_plan_map's last_seen btree index and sets fillfactor 90 on it, so the + liveness touch's UPDATE can go HOT. This probes NEGATIVELY — the index's ABSENCE plus the fillfactor + reloption — unlike every other sentinel in this list, which probes an object's presence. It is not + yet read by any viewer surface, so this gate rests on the standing invariant alone. Named only in + this probe line, never in prose, per the V71 finding. */ + ((SELECT c.reloptions FROM pg_class c WHERE c.oid = 'collect.query_store_plan_map'::regclass) @> ARRAY['fillfactor=90'] + AND NOT EXISTS (SELECT 1 FROM pg_indexes WHERE schemaname = 'collect' AND indexname = 'idx_query_store_plan_map_last_seen'))"; /// The store schema version this viewer build requires — the highest migration it knows /// (). The connect-time gate blocks a store below this. @@ -993,7 +1000,7 @@ rather than a value that could be masked by an operator override. It is not yet await using var reader = await command.ExecuteReaderAsync(cancellationToken); if (await reader.ReadAsync(cancellationToken)) { - return MapProbedSchemaVersion(reader.GetBoolean(0), reader.GetBoolean(1), reader.GetBoolean(2), reader.GetBoolean(3), reader.GetBoolean(4), reader.GetBoolean(5), reader.GetBoolean(6), reader.GetBoolean(7), reader.GetBoolean(8), reader.GetBoolean(9), reader.GetBoolean(10), reader.GetBoolean(11), reader.GetBoolean(12), reader.GetBoolean(13), reader.GetBoolean(14), reader.GetBoolean(15), reader.GetBoolean(16), reader.GetBoolean(17), reader.GetBoolean(18), reader.GetBoolean(19), reader.GetBoolean(20), reader.GetBoolean(21), reader.GetBoolean(22), reader.GetBoolean(23), reader.GetBoolean(24), reader.GetBoolean(25), reader.GetBoolean(26), reader.GetBoolean(27), reader.GetBoolean(28), reader.GetBoolean(29), reader.GetBoolean(30), reader.GetBoolean(31), reader.GetBoolean(32), reader.GetBoolean(33), reader.GetBoolean(34), reader.GetBoolean(35), reader.GetBoolean(36), reader.GetBoolean(37), reader.GetBoolean(38), reader.GetBoolean(39), reader.GetBoolean(40), reader.GetBoolean(41), reader.GetBoolean(42), reader.GetBoolean(43), reader.GetBoolean(44), reader.GetBoolean(45), reader.GetBoolean(46), reader.GetBoolean(47), reader.GetBoolean(48), reader.GetBoolean(49), reader.GetBoolean(50), reader.GetBoolean(51), reader.GetBoolean(52), reader.GetBoolean(53), reader.GetBoolean(54), reader.GetBoolean(55), reader.GetBoolean(56), reader.GetBoolean(57), reader.GetBoolean(58), reader.GetBoolean(59), reader.GetBoolean(60), reader.GetBoolean(61), reader.GetBoolean(62), reader.GetBoolean(63), reader.GetBoolean(64), reader.GetBoolean(65), reader.GetBoolean(66), reader.GetBoolean(67), reader.GetBoolean(68), reader.GetBoolean(69), reader.GetBoolean(70), reader.GetBoolean(71), reader.GetBoolean(72), reader.GetBoolean(73), reader.GetBoolean(74), reader.GetBoolean(75), reader.GetBoolean(76), reader.GetBoolean(77), reader.GetBoolean(78), reader.GetBoolean(79), reader.GetBoolean(80), reader.GetBoolean(81), reader.GetBoolean(82), reader.GetBoolean(83), reader.GetBoolean(84), reader.GetBoolean(85), reader.GetBoolean(86), reader.GetBoolean(87), reader.GetBoolean(88), reader.GetBoolean(89), reader.GetBoolean(90), reader.GetBoolean(91), reader.GetBoolean(92), reader.GetBoolean(93), reader.GetBoolean(94), reader.GetBoolean(95), reader.GetBoolean(96), reader.GetBoolean(97), reader.GetBoolean(98), reader.GetBoolean(99), reader.GetBoolean(100), reader.GetBoolean(101), reader.GetBoolean(102), reader.GetBoolean(103), reader.GetBoolean(104), reader.GetBoolean(105), reader.GetBoolean(106), reader.GetBoolean(107), reader.GetBoolean(108), reader.GetBoolean(109), reader.GetBoolean(110), reader.GetBoolean(111), reader.GetBoolean(112), reader.GetBoolean(113), reader.GetBoolean(114), reader.GetBoolean(115), reader.GetBoolean(116), reader.GetBoolean(117), reader.GetBoolean(118), reader.GetBoolean(119), reader.GetBoolean(120), reader.GetBoolean(121), reader.GetBoolean(122), reader.GetBoolean(123)); + return MapProbedSchemaVersion(reader.GetBoolean(0), reader.GetBoolean(1), reader.GetBoolean(2), reader.GetBoolean(3), reader.GetBoolean(4), reader.GetBoolean(5), reader.GetBoolean(6), reader.GetBoolean(7), reader.GetBoolean(8), reader.GetBoolean(9), reader.GetBoolean(10), reader.GetBoolean(11), reader.GetBoolean(12), reader.GetBoolean(13), reader.GetBoolean(14), reader.GetBoolean(15), reader.GetBoolean(16), reader.GetBoolean(17), reader.GetBoolean(18), reader.GetBoolean(19), reader.GetBoolean(20), reader.GetBoolean(21), reader.GetBoolean(22), reader.GetBoolean(23), reader.GetBoolean(24), reader.GetBoolean(25), reader.GetBoolean(26), reader.GetBoolean(27), reader.GetBoolean(28), reader.GetBoolean(29), reader.GetBoolean(30), reader.GetBoolean(31), reader.GetBoolean(32), reader.GetBoolean(33), reader.GetBoolean(34), reader.GetBoolean(35), reader.GetBoolean(36), reader.GetBoolean(37), reader.GetBoolean(38), reader.GetBoolean(39), reader.GetBoolean(40), reader.GetBoolean(41), reader.GetBoolean(42), reader.GetBoolean(43), reader.GetBoolean(44), reader.GetBoolean(45), reader.GetBoolean(46), reader.GetBoolean(47), reader.GetBoolean(48), reader.GetBoolean(49), reader.GetBoolean(50), reader.GetBoolean(51), reader.GetBoolean(52), reader.GetBoolean(53), reader.GetBoolean(54), reader.GetBoolean(55), reader.GetBoolean(56), reader.GetBoolean(57), reader.GetBoolean(58), reader.GetBoolean(59), reader.GetBoolean(60), reader.GetBoolean(61), reader.GetBoolean(62), reader.GetBoolean(63), reader.GetBoolean(64), reader.GetBoolean(65), reader.GetBoolean(66), reader.GetBoolean(67), reader.GetBoolean(68), reader.GetBoolean(69), reader.GetBoolean(70), reader.GetBoolean(71), reader.GetBoolean(72), reader.GetBoolean(73), reader.GetBoolean(74), reader.GetBoolean(75), reader.GetBoolean(76), reader.GetBoolean(77), reader.GetBoolean(78), reader.GetBoolean(79), reader.GetBoolean(80), reader.GetBoolean(81), reader.GetBoolean(82), reader.GetBoolean(83), reader.GetBoolean(84), reader.GetBoolean(85), reader.GetBoolean(86), reader.GetBoolean(87), reader.GetBoolean(88), reader.GetBoolean(89), reader.GetBoolean(90), reader.GetBoolean(91), reader.GetBoolean(92), reader.GetBoolean(93), reader.GetBoolean(94), reader.GetBoolean(95), reader.GetBoolean(96), reader.GetBoolean(97), reader.GetBoolean(98), reader.GetBoolean(99), reader.GetBoolean(100), reader.GetBoolean(101), reader.GetBoolean(102), reader.GetBoolean(103), reader.GetBoolean(104), reader.GetBoolean(105), reader.GetBoolean(106), reader.GetBoolean(107), reader.GetBoolean(108), reader.GetBoolean(109), reader.GetBoolean(110), reader.GetBoolean(111), reader.GetBoolean(112), reader.GetBoolean(113), reader.GetBoolean(114), reader.GetBoolean(115), reader.GetBoolean(116), reader.GetBoolean(117), reader.GetBoolean(118), reader.GetBoolean(119), reader.GetBoolean(120), reader.GetBoolean(121), reader.GetBoolean(122), reader.GetBoolean(123), reader.GetBoolean(124)); } return null; @@ -1018,7 +1025,7 @@ rather than a value that could be masked by an operator override. It is not yet /// is unit-tested without a live store; any schema bump past the newest arm trips the pinning test that keeps /// this in step with . /// - internal static int MapProbedSchemaVersion(bool hasConfigControlPlane, bool hasAlertDeliveryOverride, bool hasAnalysisState, bool hasAlertTuningKnobs, bool hasDefaultTraceEvents, bool hasIndexObjectStatsLatestIndex, bool hasCollectionLogHypertableOrPlainPg, bool hasJobHistory, bool hasAgentStatus, bool hasGenericWebhook, bool hasDeadlocksDatabaseName, bool hasQueryStoreReplicaRole, bool hasLongQueryCompletions, bool hasWebDashboardConfig, bool hasCustomViews, bool hasServerTags, bool hasConnectionRefireKnobs = false, bool hasAgCollectors = false, bool hasAgAlertKnobs = false, bool hasAgLatencyColumns = false, bool hasAgDisconnectRefire = false, bool hasPayloadDimensions = false, bool hasDimFloorIndexes = false, bool hasBlockingWaitThreshold = false, bool hasQueryStoreIntervalIdentity = false, bool hasPagerDutyWebhook = false, bool hasPagerDutyProxy = false, bool hasCollectorState = false, bool hasPlanCorrection = false, bool hasPvsStats = false, bool hasPvsPressureKnobs = false, bool hasDatabaseStateAlert = false, bool hasServerTagColour = false, bool hasQueryStatsHostObject = false, bool hasFindingDrillDown = false, bool hasStoreMetrics = false, bool hasPlanDimGzip = false, bool hasSelfAlertKnobs = false, bool hasJobMetricsColumns = false, bool hasJobCadenceKnob = false, bool hasBackfillSwitch = false, bool hasCollectorMemoryKnobs = false, bool hasDatabaseStateEdgeMemory = false, bool hasIncidentOccurrences = false, bool hasPlanXmlCompressionKnob = false, bool hasMonitoredServerEngine = false, bool hasPgBlockingEdges = false, bool hasQueryStorePlanMap = false, bool hasPgStatementText = false, bool hasQueryStoreText = false, bool hasPlanContentRetentionKnob = false, bool hasQueryStoreHealth = false, bool hasQueryStoreTextHash = false, bool hasComposeTimeoutKnob = false, bool hasFileGrowthAlert = false, bool hasCollectionLogFanoutRollup = false, bool hasTempDbMaxSize = false, bool hasServerEngineKind = false, bool hasPgDatabaseStats = false, bool hasPgIndexUsageStats = false, bool hasPgTableBloatStats = false, bool hasPgSessionStates = false, bool hasPgPlanCaptureReadiness = false, bool hasPgWriteStats = false, bool hasPgExtensionAvailability = false, bool hasPgLockStats = false, bool hasPgColumnStats = false, bool hasPgReplicationStats = false, bool hasPgBufferUsage = false, bool hasPgIndexBloat = false, bool hasPgPerDatabaseAttribution = false, bool hasPgWaitSampling = false, bool hasPgKernelStats = false, bool hasPgPredicateStats = false, bool hasPgPlanCapture = false, bool hasPgMajorVersion = false, bool hasPg18IoBytes = false, bool hasPgServerConfig = false, bool hasPgDeadlocks = false, bool hasPgDeadlockIdentity = false, bool hasCollectorCost = false, bool hasPgCpuUtilization = false, bool hasPlanForceActions = false, bool hasCollectionLogPhaseSplit = false, bool hasCollectionLogDrainForensics = false, bool hasCollectionLogFetchPhaseSums = false, bool hasStoreLogSelfMonitoring = false, bool hasCollectorStallProbes = false, bool hasRemediationCredentialAndActor = false, bool hasPgIndexBloatEstimate = false, bool hasPgCpuCapacityHeadroom = false, bool hasCustomAlertCore = false, bool hasMuteRuleReloadBeacon = false, bool hasBuiltinAlertPersistence = false, bool hasRetentionHoldRatioKnobs = false, bool hasDeadlockRateBandKnobs = false, bool hasOversizedPlanBacklog = false, bool hasPgAlertCountKnobs = false, bool hasFleetSweepState = false, bool hasFleetSweepCadenceKnobs = false, bool hasCollectorScheduleDatabases = false, bool hasSelfDiskWarnGbFloor = false, bool hasDeltaFamilyIntervalColumns = false, bool hasDeltaFamilyIntervalCompletion = false, bool hasPgLogEvents = false, bool hasPgLogEventMetrics = false, bool hasNotificationRoutes = false, bool hasPerfmonCounterType = false, bool hasPgNumbackendsAndSampledMs = false, bool hasTimeHonesty = false, bool hasLrqExclusionKnob = false, bool hasPgDatabaseSizeStatsAndHostMemory = false, bool hasQsCaptureModeRouteKnobToast = false, bool hasPgServerConfigDatabaseRoleOverrides = false, bool hasPostmasterStartTime = false, bool hasCheckpointsTimed = false, bool hasCollectionCaveats = false, bool hasIndexObjectStatsServerTimeIndex = false, bool hasQueryStoreIntervalLatest = false, bool hasRawChunkIntervalRungHistory = false, bool hasQueryStoreIntervalWide = false, bool hasManagedConfVerdicts = false, bool hasComposeTimeoutSixty = false, bool hasReadLatency = false) + internal static int MapProbedSchemaVersion(bool hasConfigControlPlane, bool hasAlertDeliveryOverride, bool hasAnalysisState, bool hasAlertTuningKnobs, bool hasDefaultTraceEvents, bool hasIndexObjectStatsLatestIndex, bool hasCollectionLogHypertableOrPlainPg, bool hasJobHistory, bool hasAgentStatus, bool hasGenericWebhook, bool hasDeadlocksDatabaseName, bool hasQueryStoreReplicaRole, bool hasLongQueryCompletions, bool hasWebDashboardConfig, bool hasCustomViews, bool hasServerTags, bool hasConnectionRefireKnobs = false, bool hasAgCollectors = false, bool hasAgAlertKnobs = false, bool hasAgLatencyColumns = false, bool hasAgDisconnectRefire = false, bool hasPayloadDimensions = false, bool hasDimFloorIndexes = false, bool hasBlockingWaitThreshold = false, bool hasQueryStoreIntervalIdentity = false, bool hasPagerDutyWebhook = false, bool hasPagerDutyProxy = false, bool hasCollectorState = false, bool hasPlanCorrection = false, bool hasPvsStats = false, bool hasPvsPressureKnobs = false, bool hasDatabaseStateAlert = false, bool hasServerTagColour = false, bool hasQueryStatsHostObject = false, bool hasFindingDrillDown = false, bool hasStoreMetrics = false, bool hasPlanDimGzip = false, bool hasSelfAlertKnobs = false, bool hasJobMetricsColumns = false, bool hasJobCadenceKnob = false, bool hasBackfillSwitch = false, bool hasCollectorMemoryKnobs = false, bool hasDatabaseStateEdgeMemory = false, bool hasIncidentOccurrences = false, bool hasPlanXmlCompressionKnob = false, bool hasMonitoredServerEngine = false, bool hasPgBlockingEdges = false, bool hasQueryStorePlanMap = false, bool hasPgStatementText = false, bool hasQueryStoreText = false, bool hasPlanContentRetentionKnob = false, bool hasQueryStoreHealth = false, bool hasQueryStoreTextHash = false, bool hasComposeTimeoutKnob = false, bool hasFileGrowthAlert = false, bool hasCollectionLogFanoutRollup = false, bool hasTempDbMaxSize = false, bool hasServerEngineKind = false, bool hasPgDatabaseStats = false, bool hasPgIndexUsageStats = false, bool hasPgTableBloatStats = false, bool hasPgSessionStates = false, bool hasPgPlanCaptureReadiness = false, bool hasPgWriteStats = false, bool hasPgExtensionAvailability = false, bool hasPgLockStats = false, bool hasPgColumnStats = false, bool hasPgReplicationStats = false, bool hasPgBufferUsage = false, bool hasPgIndexBloat = false, bool hasPgPerDatabaseAttribution = false, bool hasPgWaitSampling = false, bool hasPgKernelStats = false, bool hasPgPredicateStats = false, bool hasPgPlanCapture = false, bool hasPgMajorVersion = false, bool hasPg18IoBytes = false, bool hasPgServerConfig = false, bool hasPgDeadlocks = false, bool hasPgDeadlockIdentity = false, bool hasCollectorCost = false, bool hasPgCpuUtilization = false, bool hasPlanForceActions = false, bool hasCollectionLogPhaseSplit = false, bool hasCollectionLogDrainForensics = false, bool hasCollectionLogFetchPhaseSums = false, bool hasStoreLogSelfMonitoring = false, bool hasCollectorStallProbes = false, bool hasRemediationCredentialAndActor = false, bool hasPgIndexBloatEstimate = false, bool hasPgCpuCapacityHeadroom = false, bool hasCustomAlertCore = false, bool hasMuteRuleReloadBeacon = false, bool hasBuiltinAlertPersistence = false, bool hasRetentionHoldRatioKnobs = false, bool hasDeadlockRateBandKnobs = false, bool hasOversizedPlanBacklog = false, bool hasPgAlertCountKnobs = false, bool hasFleetSweepState = false, bool hasFleetSweepCadenceKnobs = false, bool hasCollectorScheduleDatabases = false, bool hasSelfDiskWarnGbFloor = false, bool hasDeltaFamilyIntervalColumns = false, bool hasDeltaFamilyIntervalCompletion = false, bool hasPgLogEvents = false, bool hasPgLogEventMetrics = false, bool hasNotificationRoutes = false, bool hasPerfmonCounterType = false, bool hasPgNumbackendsAndSampledMs = false, bool hasTimeHonesty = false, bool hasLrqExclusionKnob = false, bool hasPgDatabaseSizeStatsAndHostMemory = false, bool hasQsCaptureModeRouteKnobToast = false, bool hasPgServerConfigDatabaseRoleOverrides = false, bool hasPostmasterStartTime = false, bool hasCheckpointsTimed = false, bool hasCollectionCaveats = false, bool hasIndexObjectStatsServerTimeIndex = false, bool hasQueryStoreIntervalLatest = false, bool hasRawChunkIntervalRungHistory = false, bool hasQueryStoreIntervalWide = false, bool hasManagedConfVerdicts = false, bool hasComposeTimeoutSixty = false, bool hasReadLatency = false, bool hasHotLivenessTouch = false) { /* V71 (the PostgreSQL blocking-edges rung): a table-existence sentinel and now the newest-first arm. A collector table would ordinarily get no arm at all — see the V63-V69 note below — but the TOP @@ -1198,6 +1205,19 @@ the rung below and showing a spurious upgrade banner on a store that is current. The WPF viewer runs no analysis, so no viewer read names the new table; this arm exists so the version banner stays truthful, which is the only effect the rung has on the viewer. Named only in the probe line, not this prose, per the V71 finding. */ + /* V149 (#4250): the Query Store liveness touch drops query_store_plan_map's last_seen index and + sets fillfactor 90, and now the TOP rung, so a fully-migrated store maps to EXACTLY + StorageVersion.SchemaVersion rather than falling through to the rung below and showing a + spurious upgrade banner on a store that is current. + + The WPF viewer runs no analysis, so no viewer read names this table; this arm exists so the + version banner stays truthful, which is the only effect the rung has on the viewer. Named only + in the probe line, not this prose, per the V71 finding. */ + if (hasHotLivenessTouch) + { + return 149; + } + if (hasReadLatency) { return 148;