diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index bd8e9924e..11c263ce9 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -275,19 +275,19 @@ jobs: # spin-up and its filter-drift risk. - name: Run Lite tests if: steps.filter.outputs.lite == 'true' || steps.filter.outputs.core == 'true' || steps.filter.outputs.root == 'true' || github.event_name == 'release' - run: dotnet test Lite.Tests/Lite.Tests.csproj -c Release --no-build --verbosity normal + run: dotnet run --project Lite.Tests/Lite.Tests.csproj -c Release --no-build - name: Run Installer tests if: steps.filter.outputs.installer == 'true' || steps.filter.outputs.installer_core == 'true' || steps.filter.outputs.root == 'true' || github.event_name == 'release' - run: dotnet test deprecated/Installer.Tests/Installer.Tests.csproj -c Release --no-build --verbosity normal --filter "FullyQualifiedName!~VersionDetectionTests&FullyQualifiedName!~IdempotencyTests&FullyQualifiedName!~AdversarialTests" + run: dotnet run --project deprecated/Installer.Tests/Installer.Tests.csproj -c Release --no-build -- -class- "Installer.Tests.VersionDetectionTests" -class- "Installer.Tests.IdempotencyTests" -class- "Installer.Tests.AdversarialTests" - name: Run Dashboard tests if: steps.filter.outputs.dashboard == 'true' || steps.filter.outputs.core == 'true' || steps.filter.outputs.root == 'true' || github.event_name == 'release' - run: dotnet test deprecated/Dashboard.Tests/Dashboard.Tests.csproj -c Release --no-build --verbosity normal + run: dotnet run --project deprecated/Dashboard.Tests/Dashboard.Tests.csproj -c Release --no-build - name: Run Darling tests if: steps.filter.outputs.darling == 'true' || steps.filter.outputs.core == 'true' || steps.filter.outputs.root == 'true' || github.event_name == 'release' - run: dotnet test Darling/Darling.Tests/Darling.Tests.csproj -c Release --no-build --verbosity normal + run: dotnet run --project Darling/Darling.Tests/Darling.Tests.csproj -c Release --no-build - name: Get version if: steps.fastpath.outputs.engaged != 'true' @@ -712,7 +712,7 @@ jobs: env: DARLING_TEST_PG: "Host=127.0.0.1;Port=5541;Username=darling;Database=darling" DARLING_TEST_PGRUNTIME: ${{ github.workspace }}\Darling\artifacts\pg-runtime - run: dotnet test Darling/Darling.Tests/Darling.Tests.csproj -c Release --no-build --verbosity normal --logger "trx;LogFileName=darling-pr.trx" --results-directory TestResults + run: dotnet run --project Darling/Darling.Tests/Darling.Tests.csproj -c Release --no-build -- -trx TestResults/darling-pr.trx - name: Stop PostgreSQL if: always() && steps.filter.outputs.darling == 'true' diff --git a/.github/workflows/nightly.yml b/.github/workflows/nightly.yml index dce91ea53..2ae97572a 100644 --- a/.github/workflows/nightly.yml +++ b/.github/workflows/nightly.yml @@ -117,7 +117,7 @@ jobs: dotnet restore Darling/PerformanceMonitor.Darling.Viewer/PerformanceMonitor.Darling.Viewer.csproj --locked-mode - name: Run tests - run: dotnet test Lite.Tests/Lite.Tests.csproj -c Release --verbosity normal + run: dotnet run --project Lite.Tests/Lite.Tests.csproj -c Release - name: Publish Lite run: dotnet publish Lite/PerformanceMonitorLite.csproj -c Release -o publish/Lite @@ -376,7 +376,7 @@ jobs: # sees a FRESH store, so nothing else can catch an upgrade path that breaks. DARLING_TEST_PGRUNTIME_OLD: ${{ github.workspace }}\Darling\artifacts\upgrade-fixture\old\pg-runtime DARLING_TEST_PGRUNTIME_NEWZIP: ${{ github.workspace }}\Darling\artifacts\pg-runtime.zip - run: dotnet test Darling/Darling.Tests/Darling.Tests.csproj -c Release --no-build --verbosity normal --logger "trx;LogFileName=darling-nightly.trx" --results-directory TestResults + run: dotnet run --project Darling/Darling.Tests/Darling.Tests.csproj -c Release --no-build -- -trx TestResults/darling-nightly.trx - name: Stop PostgreSQL if: always() diff --git a/CHANGELOG.md b/CHANGELOG.md index b73c1e085..d20e253d6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,11 +5,14 @@ All notable changes to this project will be documented in this file. The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). -## [Unreleased] - ## [3.5.0] - 2026-08-19 ### Added +- **Alert on database file SIZE growth, graded per server** ([#2349]) - between `tempdb Space`, whose denominator GROWS with autogrowth so its percentage falls as tempdb balloons, and `Volume Free Space`, which fires on the consequence and cannot attribute it to a file, sits a file that has grown large but has not yet filled its disk. Two gates, both graded per server so one global setting works across a heterogeneous fleet: a RISE gate (this file grew N MB in the window) which is primary because a level alone re-pages forever about a size that has been true since Tuesday, and a LEVEL gate (this file is N% of its VOLUME) which self-scales to each server's disk layout. Zero disables either gate independently. Ships OFF. Both SKUs. Requested by @gotqn. +- **`Database` and `LastEventUtc` on the incident projection** ([#2361]) - a consumer had to string-search `Details[]` for the database, which is exact only for deadlocks: every other fingerprinted alert appends a BARE Incident item beside its data item, so the fallback degraded to "any Database anywhere in the payload" - the WRONG value on a multi-incident alert spanning databases, with nothing marking it wrong. `LastEventUtc` is the counterpart to `IncidentStartedUtc`, projected from `IncidentOccurrenceState.LastObservedUtc`, which the accumulator already computed and persisted. Both trailing and optional, so contextJson written before they existed still round-trips; both in shared Notifications, so Lite and Darling get them by construction. Requested by @gotqn. +- **Every fingerprinted alert now carries a monotonic occurrence total, not just blocking and deadlocks** ([#2362]) - the #2216 accumulator had two call sites; Long-Running Query, Volume Free Space, Version Store (PVS), Long-Running Job and Failed Agent Job produced dedup-keyed incidents with no total, so a consumer had one code path that SETS an exact count and another that has to GET-and-add on every recurrence. The observation list is UNCAPPED while the card stays capped (3 or 5 depending on the alert): observing only the displayed subset would reset the total of any fingerprint that fell out of the top N, which is the undercount #2216 exists to fix, reintroduced by the fix for it. Reported by @gotqn. +- **`--harden-files`: re-apply the secret-file ACLs from an elevated prompt** ([#2352]) - the service already computes the correct DACL and already detects when the real one is wrong, but cannot apply it: re-ACLing a file it does not own needs WRITE_DAC and taking ownership needs a privilege a virtual service account is not granted, so it could only log the remedy. Until now the only thing that applied it to an existing install was `install-darling.ps1`, leaving anyone who registered the exe by hand - the README's own `sc create` path - typing three `icacls` lines out of a log message. Covers darling.json (the only target the interactive operator keeps read on), its `.bak-*` siblings, the store directory and the DPAPI credentials; VERIFIES each one afterwards rather than claiming success, and exits non-zero if anything is still readable so it is usable in a provisioning script. Hardens for the account the SERVICE is registered under (the SCM's `ObjectName`, so a re-homed domain account or gMSA is handled), never for whoever is running the verb - it exists BECAUSE the service cannot re-ACL a file it does not own, so its caller is never the service, and granting the caller would strip the service and break the install on its next start. Verification covers both directions for the same reason: an ACL that excludes ordinary users AND the service is maximally private and completely broken, so a locked-out target fails rather than reporting SECURED. Idempotent. +- **Declared peer stores: a Darling MCP server that names its siblings instead of answering "unknown server"** ([#2339]) - tier 1 of the multi-store fix, disclosure only. With the fleet split across several boxes (one store each: SQL Server primaries on one, their readable replicas on another, PostgreSQL on a third) every box's MCP server answered over ITS store alone, so a server monitored by a sibling resolved as not-found - indistinguishable from a server nobody monitors, which with a deliberately-split fleet is now the normal case rather than an edge. An optional `peers` block in darling.json (`thisStoreCovers`, plus per-peer `name` / `covers` / optional `matches` name-substrings) is disclosed at the three places an agent forms its picture of the fleet: the MCP instructions gain a Fleet Coverage section above the tool census, `list_servers` gains `this_store_covers` + `peer_fleets` + a `peer_note` (its empty-registry answer is prose rather than JSON, and carries the peer list too - a store with nothing registered is a fresh or just-restarted box, the worst place to drop the disclosure), and the server-resolution miss appends "not monitored HERE - matches the declared coverage of " to the existing available-servers listing. There is NO credential, NO address and NO connectivity behind it: the service never contacts a peer and says so in every message, and the publish itself REFUSES a peers block whose text looks like a connection string or credential, since all of it is sent verbatim to every connected MCP client. That guard lives in the publish rather than only in config validation because the MCP host loads its own config and deliberately never validates it (its fail-closed checks are host-local), so validation alone would have left the one path that actually broadcasts uncovered; a refusal publishes nothing at all rather than the valid subset. An empty `peer_fleets` carries its own note - "this may be the only store, or nobody declared the siblings, and this server cannot tell those apart" - so absence never reads as "you are looking at the whole fleet". With nothing declared the PROSE surfaces are byte-for-byte unchanged (the instructions, the resolution miss, and the empty-registry sentence), with ONE deliberate exception: `list_servers`' JSON envelope carries `this_store_covers` / `peer_fleets` / `peer_note` on every response, declared or not, so a client comparing that tool's exact shape sees three new keys on upgrade. That is the point rather than an oversight - an empty `peer_fleets` means either "only store" or "nobody declared the siblings", and a conditional note would say nothing in exactly the case that produces the wrong conclusion. Lite has no peers concept and gets no twin. Federated cross-store reads stay unbuilt on purpose - **`cpu_attribution` on the top-CPU rankings: what fraction of the box the ranking explains** ([#2320]) - the last unshipped item from #2235's wishlist. `get_top_queries_by_cpu` and `get_top_procedures_by_cpu` (both SKUs) now return the returned rows' summed CPU-seconds, the SQL process's measured CPU-seconds for the same window (avg `cpu_utilization` % x core count x window - both stores already collect every piece), and `attributed_cpu_ratio`. Pre-#2290 the reads explained ~10% of the box and nothing said so - a caller chased the visible tenth assuming it was everything; and the ratio catches impossible claims at a glance (an external comparison died the moment its worker_time sum divided out to 137% of the box's available CPU-seconds - above the process's own measured consumption the note now says to distrust the numbers rather than presenting them). Below half, a note explains where unattributable CPU goes (evictions between snapshots, rows outside the top-N, zero-cost rows, non-query CPU). The degrade rule is explicit: missing CPU series, missing core count, or a series covering under 90% of the window omits the ratio rather than inventing one. One computation in `PerformanceMonitor.Common` (`CpuAttribution`), pinned by the same decision table in both test projects; the denominator read windows on `collection_time` with the same bounds as the rankings, so numerator and denominator share collection gaps. - **`get_query_store_health`: the MCP read for the new collector, both SKUs** ([#2319]) - the promised follow-up to the `query_store_health` collector: one browsable tool (beside `get_database_scoped_config`, whose latest-snapshot shape it mirrors) returning per-database actual vs desired state with the mismatch pre-folded into `state_matches_desired`, `readonly_reason` both raw and decoded, storage used vs cap with `pct_of_cap`, cleanup mode/thresholds, and the runtime-stats interval length; also exposed as a `/api/read` web endpoint. The `readonly_reason` bit table now lives ONCE in `PerformanceMonitor.Common` (`QueryStoreReadonlyReason`) and both viewers' grids and both MCP servers decode through it - the labels were miswritten from memory once during #2319 review, so a single source is the fix. While counting the tools for the instructions doc, the census sentence turned out to have silently drifted (it said ninety tools while the server exposed one hundred); it is rewritten with accurate digit counts (101 total / 76 shared with Lite / 25 Darling-only) and a new cross-app pin test parses it against the scanned inventory so it can never drift again. - **Per-database Query Store health: a new `query_store_health` collector, both SKUs, both stores** ([#2319]) - `database_config` knows exactly one bit (`is_query_store_on = true`), which cannot answer the questions an investigation like #2312 needed: is Query Store actually WORKING (the classic silent failure is desired_state READ_WRITE with actual_state READ_ONLY after the storage cap hit - `readonly_reason` says why), how close to the cap is it, and what interval grain is it aggregating at. The new collector reads `sys.database_query_store_options` per database - the same proven enumeration idiom as `database_scoped_config` (list accessible ONLINE primaries, then `[db].sys.sp_executesql` per database), deliberately NOT filtered to QS-on databases: the options view answers one row even when Query Store is off, so OFF is recorded as OFF and an absent row can only mean "not collected". Hourly rather than the config family's on-load cadence, because unlike operator-changed knobs these values change BY THEMSELVES and the cap-hit transition is the whole point of collecting them. Every column exists on 2016+, so there are no version gates. Surfaced as a Query Store sub-tab on the Configuration tab in both apps (V76 store table + `v_query_store_health` passthrough keep the two viewers' SQL byte-identical; Lite's table and archive view generate from the catalog); a `get_query_store_health` MCP read follows separately. The issue asked for the fields on `database_config` itself; they land as a sibling enumerating collector instead because `database_config` is a single `sys.databases` scan and these fields need per-database context - bolting an enumeration onto it would change its execution model and failure isolation, and the codebase already has the per-database config member in `database_scoped_config` to mirror. @@ -74,6 +77,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - **The #2136 capacity model is now proven by a synthetic scale test, not asserted from one observation** - a live end-to-end drives a throwaway hypertable's compression job at 1x and then 4x row volume (parked policy, deterministic run_job - the #1888 discipline, so the scheduler can never race the measurement) and pins the whole loop: job runtime GROWS with volume (monotonicity, not a ratio - runner jitter owns the constant factor, the direction is the claim), the V56 telemetry series records both readings in order, and the Store Job Over Cadence alert fires its Critical tier from REAL store readings once the schedule interval is shrunk under the measured duration. ### Changed +- **The compose `statement_timeout` is a store setting instead of a hardcoded 15s** ([#2357]) - it is the hard backstop a composed query can never exceed, and 15 seconds is a judgement about store size and disk speed that the product cannot make for someone else's deployment. V78 adds `config_service.compose_statement_timeout_seconds`, defaulting to 15 so nothing changes for anyone who does not touch it, clamped to [5,600] on both read and use because a zero would remove the ceiling the whole design leans on. It could not be fixed by raising the constant: the value lives in role PROVISIONING DDL, so an existing install already has the old one baked in - but that DDL re-runs on every managed start, so a change reaches a running store on its next restart with no new machinery. Reported by @carlei1978, who hit it on a 30-day `get_query_store_top`. +- **MCP tool results serialize compact instead of pretty-printed** ([#2350]) - the only consumer of a tool result is a language model, and indentation buys a model nothing. One property on the shared `McpHelpers.JsonOptions` in Common, so both SKUs move together, plus the two readers that carry their own options for the `/api/*` twins. Saving is payload-shaped - 23% of the bytes on a 15-field record array, 36% on a narrow one - and the TOKEN saving is smaller than the byte saving, because BPE tokenizers pack runs of spaces efficiently. The config files people hand-edit (`servers.json`, profiles, schedules) keep indenting, and a test pins that boundary in both directions. +- **The test suites run under Microsoft.Testing.Platform instead of VSTest** ([#2347]) - the .NET 10 SDK dropped VSTest-mode support for MTP-based frameworks, so xunit.v3 4.0.0 could not be taken at all (blocking two Dependabot bumps). All four suites now build as their own executables and CI invokes them with `dotnet run`; `xunit.runner.visualstudio` and `Microsoft.NET.Test.Sdk` are gone entirely, since the first exists only to bridge xunit to VSTest and the second is the VSTest host. Deliberately done on the CURRENT xunit 3.2.2 so the runner migration and the version bump are separately reviewable. - **The query_store per-database log split now names the plan-XML and text fetches** ([#2312] investigation) - the per-item sql: stopwatch wraps the whole read, which since the separate fetches landed includes two more queries against the Query Store catalogs after the payload drain - so on a closed-only cycle that shipped ZERO rows, a 298-second bill (measured, ayr-01) had nowhere visible to live: drain silently absorbed it. Cycles that ran a separate fetch now log `... + plan_fetch:Nms + text_fetch:Nms`, the fetch phases come out of drain in the one shipped DrainMsFrom arithmetic (pinned like the #2164 split it extends), and every collector line without a separate fetch is byte-identical to before. Darling-only by construction - Lite runs no separate fetches and its zeros mean exactly that. - **Query Store statement text is now resolved from `collect.query_store_text`, and the separate fetch is ON** ([#2150]) - the flip of `FetchQueryTextSeparately` and the conversion of every reader that projects `query_text`, in one change, because they cannot land apart: flipping the flag nulls the payload's inline `query_sql_text`, so any reader still reading that column would show BLANK text for newly collected rows while looking perfectly healthy - no error, no empty result, just a grid with the statement missing. Six blocks across five files now resolve the side table first and fall back to the inline column: the stored-plan resolver behind Get Actual Plan, the MCP `get_query_store_top` read, the Viewer's Query Store grid, its current-vs-baseline comparison, its regressions grid, and the PLAN_REGRESSION drill-down. **The fallback is permanent, not a migration step**: rows collected before this carry their text inline and nothing backfills them, so removing the fallback later would blank all existing history - the two arms are the two populations, not an old way and a new way. **Where the resolution goes was chosen to keep the filters honest.** Three of these queries also FILTER on the text (the [#1565] `WAITFOR` self-exclusion, and the resolver's `IS NOT NULL`), and testing the raw column there would have excluded every post-cutover row - the whole set the change exists to serve - so each one resolves the text ONCE, inside the existing lateral or a derived table, and the filter tests the resolved value under its original name. That also keeps the diffs to one block each rather than a projection edit plus a filter edit plus a repeated `COALESCE` in the `WHERE`. The comparison read needed `query_id` projected through its dedup CTEs (it groups by `query_hash`, but text is stored per `query_id`) and both its arms converted, since the final projection coalesces current over baseline and converting one arm would leave a GONE row with nothing to fall back to. **Verified against the live 52-server store rather than by reading the SQL**, which is the only instrument that can see any of this - there is no CI job that executes these Postgres strings. All six shipped query bodies were extracted from source, planned with `EXPLAIN (GENERIC_PLAN)` (six plans, every one containing `query_store_text` nodes), then run against real data twice: with the side table EMPTY, each converted query returned a byte-identical md5 to its pre-change form over 330 rows (1 / 50 / 50 / 179 / 50 / 1), proving the fallback arm and proving no keyed join fans out; then with the post-cutover condition INDUCED by shadowing the table with distinctive rows, all 331 rows resolved from the side table at identical counts, proving the arm that will serve every row once this ships. `FetchRowsAsync` deliberately stays OFF - it hands rows straight to its caller and writes nothing, so nulling the inline column there would lose the text outright instead of relocating it - and Lite is untouched, its DuckDB store having no side table to resolve from. @@ -89,6 +95,16 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 - **The force-plan recommendation now warns when the regressed query is parameter-sensitive** ([#2138] gap 3) - each regressed_queries row carries a `parameter_sensitivity_cofired` flag, computed inside the drill-down with the PARAMETER_SENSITIVITY detector's own thresholds (one cached plan whose per-execution cost varies >= 10x across parameter values, same floors, same window) joined by query hash - so the flag can never claim evidence the detector would not report. A flagged target's force-plan preview gains a caution block naming the risk (forcing pins ONE shape for every parameter value; the population that preferred the other plan inherits the wrong one permanently, quietly, because a forced plan no longer recompiles away) and the gentler first levers (statistics updates; PSP optimization / Query Store hints on 2022+), and the copy-paste surface gets a compact two-line version of the same warning. Unflagged targets render byte-identically to before. This flag is also the standing gate for the future auto-force bot: a flagged target is never auto-forced. Both SKUs, pinned by live tests in both stores. ### Fixed +- **Server Inventory's `Last Updated` was a config-snapshot time wearing a freshness label** ([#2359]) - `server_properties` ships with `FrequencyMinutes 0`, which the schedule table defines as "collect once on server load only", so the column showed the last SERVICE START and every actively-monitored server on a long-running install looked days stale. Renamed to `Inventory As Of`, and a new `Last Collected` column carries the real heartbeat from `v_collection_log` - the same signal `list_servers` uses. Darling only: Lite reads live and stamps the value at read time, so its column was never misleading. Reported by @ghauan, who correctly pushed back when the first fix did not explain their case. +- **get_query_store_top reported a window it could not serve** ([#2364]) - it reads raw `query_store_stats`, which is dropped at 4 days on a store with the rollups armed, so a 30-day request returned at most four days with `hours_back: 720` echoed back unchanged. It cannot be routed to a rollup the way #2353 was: the corrected CAGGs carry no `query_id` or `plan_id`, and plan identity is the whole point of the tool. So the response now carries `effective_start`, `effective_hours_back`, `truncated` and a note naming why, and the empty path no longer blames Query Store configuration for a window it never read. Found while scoping #2357 with @carlei1978. +- **Server Inventory's Last Updated looked broken on decommissioned servers** ([#2359]) - the grid lists every REGISTERED server and never read `servers.is_enabled`, so a server removed from monitoring kept the `collection_time` it had when collection stopped. Accurate, and read by everyone as a stale freshness column. Measured on a 61-server fleet the split was total: 19 disabled servers last collected within five minutes of each other on the day they were removed, 42 enabled servers all fresh within a minute. A new Monitoring column now says Active or Stopped. The disabled rows are KEPT, not filtered - this is the FinOps tab and a decommissioned server's cost history is what it exists to show. Reported by @ghauan. +- **Extended-length paths no longer slip past the install-location guards** ([#2348]) - `\\?\UNC\server\share` is the long spelling of a REAL share and `\\?\C:\Users\bob` of a REAL profile, but the wholesale `\\?\` exclusion waved both through undiagnosed, because skipping a check is not the same as passing it. Both implementations now strip the prefix BEFORE classifying, so the long spelling gets the same verdict as the short one, and `\\?\C:\PerformanceMonitorDarling` - an ordinary local root written the long way - is still correctly left alone. The shared decision table and the cross-language parity test hold the service and `install-darling.ps1` to it together. +- **get_query_trend silently truncated any window past 4 days** ([#2353]) - it read only the raw `query_stats` table, whose raw tier is dropped at 4 days independently of the collector's much longer advertised retention, while echoing `hours_back: 168` back unchanged. Worse than a short array: the empty path asserted "No history found ... within the last 168 hours" over a span it never read, which for a query whose history had aged out is a false statement rather than an incomplete one. It now routes to the hourly rollup by the age of the window's OLDEST point measured from wall clock, and reports `source`, `effective_hours_back`, `bucket` and `truncated` so the response describes the data instead of the request. Columns the rollup does not carry come back NULL, never zero - on an aggregate row a zero would read as "none observed". Reported by @carlei1978. +- **The store's scale test no longer asserts that TimescaleDB compresses more rows in more time** ([#2266] item 1, measured on a rig) - `ScaleTest_JobDurationGrowsWithVolume_...` required `d10 > d1` between two sub-second job durations, and it has failed on PR after PR whose diffs cannot reach it (`d1=970/d10=863`, then `d1=689/d10=689`). Fifteen consecutive runs of the exact sequence against TimescaleDB 2.29/PG17 settle what no amount of reasoning from CI logs could: the chunks compress perfectly (counts go 1, 2, 3; per-day rows are exactly 2000/50000/500000 every single time), so the earlier suspicion that both runs were compressing nothing is **refuted** - but a 10x volume increase buys only ~3.2x the duration, about **85 ms** of absolute signal, because compression cost is largely fixed per run. CI's baseline for the same pair is 690-970 ms, roughly twenty times that fixed cost, so the volume-dependent component there is ~10% of the measurement's own magnitude and sits inside the run-to-run variance of launching a background worker on Windows. That is a benchmark of somebody else's compression engine on shared hardware, and no threshold, ratio or volume rescues it: at ~0.19 ms per thousand rows it would take millions of rows per chunk to clear a variance nobody has measured on the platform that actually fails. The byte-identical pair was never as improbable as it looked either, because that pair is only ever read when the test FAILS, which selects for differences already near zero. **It is replaced by something strictly stronger, not weaker**: each measured run must have compressed the chunk its own seed created, and that chunk must hold exactly the seeded row count - exact counts instead of two timings. A negative control proves the difference rather than assuming it. Seed the 10x rows into a chunk that is not yet compression-eligible and the old assertion fails and the new ones fail too, naming `compressed=2`; but seed them into the **1x chunk** and the old assertion **passes 3/3 with a 6-8x ratio** while the fixture has quietly stopped producing two chunks at two volumes, and only the new assertions catch it (`rows=[2000,550000]`, `total=2`). So the shipped assertion was not merely flaky, it was blind to the fixture defect it was supposed to be guarding. What the product owns is asserted and unchanged: a real duration is measured, the V56 series records both readings in order, and the real evaluator fires the [#2136] cadence alert from a real reading. One gap closed on the way past - `d10 > 0` was never asserted, and `ReadJobDurationMsAsync` maps a NULL duration to 0, so a 10x run whose duration was unmeasurable satisfied the telemetry check as `0 == 0` and passed. The test is renamed to stop claiming what it no longer measures. +- **The service now says WHY it cannot start when it is installed somewhere its own account cannot read** ([#2185]) - the last open half of the reported failure. #2186 decoded the loader status, #2197 stopped the missing-credential message from advising a restart that cannot help, and #2187 taught `install-darling.ps1` to refuse a user-profile or network install root - but the installer only guards installs that go through it, and the README's manual `sc create` path (or any hand-registered exe) bypasses it entirely. Those installs still reached the reporter's experience: an empty `Output:`, a bare exit code, then a missing `pg-admin-credential.dpapi` and advice to start the service once, which they had. The service now classifies its own install directory as the FIRST thing `ExecuteAsync` does - ahead of reading `darling.json`, which an unreadable tree also takes out, and long before the managed-Postgres bootstrap - and logs one critical line naming the offending path, the account it is actually running as, why a virtual service account cannot read that location, and where to move it. It diagnoses and continues rather than refusing to start, the same asymmetry the installer applies to an upgrade and for the same reason (#2187's rejected option 2: an operator may have granted the tree read + execute by hand, and stranding a deployment that runs today would be worse than the disease). Silent on a console run - an interactive run IS the profile owner, and test-driving the exe from a Desktop folder is something the README suggests - and unreachable on Linux/container hosts. The decision table is deliberately the installer's own (profile root from Windows rather than a hardcoded `C:\Users`, plus `%USERPROFILE%`; UNC excluding the `\\?\` long-path prefix; a drive letter whose type is network), and a new test runs BOTH implementations over ONE table under Windows PowerShell 5.1 so the two cannot drift apart silently. +- **The per-database watermark read was an unbounded MAX over every chunk in retention** ([#2344]) - with #2333's catalog walk gone, the per-database log split exposed the phase that does NOT subside after catch-up: `wm`, at 1-3 seconds per database per cycle. It is not a query against the monitored server at all - it is `SELECT MAX(last_execution_time) FROM query_store_stats WHERE server_id = ? AND database_name = ?` against OUR store, and because the hypertable partitions on `collection_time`, a MAX over a different timestamp with no time predicate touches every chunk that database has. Measured on the live 106 GB use1 store: **25,766 buffer reads plus temp spill cold, 228 ms warm, against 29 ms bounded** with five chunks excluded - and the unbounded cost scales with STORE SIZE and cache residency rather than with anything the monitored server is doing, so it degrades exactly where an operator is weakest (long-lived store, busier Query Store, slower disks). The read is now bounded on the partitioning column, which is provably free: every consumer ends at `max(stored, now - MaxCatchup)` because the clamp floors anything older and a null result falls back to the same 60-minute instant, so a row below the horizon cannot change the answer whether it is found or not. Bounded for query_store ONLY, on both hosts - a ring-buffer collector whose legitimate catch-up spans days must keep reading its full history, so the clamp and the bound travel together. +- **Aurora detection called a catalog lookup instead of the function, and silently disabled two collectors on real Aurora** ([#2340]) - the probe decided Aurora with `count(*) FROM pg_proc WHERE proname = 'aurora_version'`, and on a live Aurora PostgreSQL 17.7 cluster a `pg_monitor` role gets **0** from that lookup while `SELECT aurora_version()` returns `17.7.2`. Because `pg_wait_stats` AND `pg_statement_stats` both gate on `IsAurora`, one wrong boolean removed the two most valuable PostgreSQL reads from every Aurora target - `pg_stat_statements` was installed and available on the cluster, so that collector was skipped purely by the flag - and it did so with a healthy-looking log line and a pre-flight that just printed a smaller collector count. Detection now CALLS the function in its own statement, catching a stock-PostgreSQL 42883 as the expected negative and logging any other failure at debug so the question is answerable from the service log rather than requiring a live psql session against the target, which is what diagnosing this took. Existence-by-catalog-lookup and callability are different questions; the collectors care about the second. +- **query_store's plan/text fetch is activity-driven: the store is the watermark** ([#2312]) - the collector's invariant 40-110s-per-run bill on big catalogs had a named mechanism at last: the #2210 watermark walk's daily expiry was supposed to be replaced by a re-verify cursor that was built, tested, documented, and **never wired** - so catalogs whose full walk needs more than a day expired MID-walk, restarted from plan_id 0, and looped the full catalog fetch forever (Finding 4). `TouchSql`, the liveness refresh the dimension GC depends on, had the same story: designed as 'the whole of the protection,' zero callers (Finding 3) - latent only because the perpetual walk's re-upserts accidentally stood in for it. The reshape retires all of it: the cycle's collected rows name their plans/texts, one touch-and-probe round trip refreshes map/dim liveness AND answers what the store lacks (plus per-cycle in-place-rewrite and Query-Store-reset detection via the live hashes the payload already carries), and the fetch selects exactly the missing ids under the same byte-budget arithmetic. A caught-up database issues NO target query - the measured 23s-to-discover-nothing becomes nothing; a Query Store reset recovers as the normal path instead of a special arm; a dormant plan resuming execution is fetched the cycle it resumes instead of waiting on a refresh horizon. V77 carries the schema strokes (nullable map digest for the content-less NULL-XML marker, `query_store_text.query_hash`, and wholesale deletion of the orphaned `planwm:`/`textwm:` state rows). Budget-deferred ids carry over in memory so a plan referenced once cannot be starved; ids the target no longer serves are dropped only on a provably-uncut pass. This also bends #2316's plan-dimension growth going forward: plans never referenced by a collected row are no longer shipped or stored at all. - **Darling Viewer crash on Queries -> Query Store by Duration** ([#2181], [#2331]) - the same uncatchable crash class as Lite's #2114, on the OTHER SKU: the grid's inline View Plan button referenced `DarkButton`, a key that IS defined in the Viewer - in `MainWindow.xaml`'s window resources, a scope a UserControl's templates cannot see, because StaticResource resolves lexically at load rather than through the runtime tree. The miss inside a cell template stack-overflows the process the moment the grid renders a row, which is also why it survived dogfooding: an EMPTY Query Store grid never applies its cell template. #2181 reported this against the Darling Viewer and was closed as a duplicate of the Lite fix on a wrong premise; #2331 re-proved it on 3.4.0. The button uses default chrome now (Lite's exact fix), and the XAML hygiene test's model is widened from per-app to per-FILE resolution (own keys + merged dictionaries + App.xaml scope - WPF's actual lookup), which flags exactly this class and produced zero false positives across both apps. - **The store self-metrics sweep's ~5-a-day "Exception while reading from stream" ERRORs were command timeouts in a network-fault costume** ([#2317]) - the sweep's sizing queries (`hypertable_detailed_size` across every hypertable - its inner `hypertable_local_size` is the frame the server log names - and `pg_database_size` over the whole store) ran on Npgsql's default 30-second timeout, which they outgrew under load on a 141-object store with a 100+ GB plan dimension. Npgsql enforces its deadline by cancelling the statement (the store side logs `canceling statement due to user request` - confirmed at the exact failure timestamps in the managed server's own log) and the client is left holding a torn stream, so the ERROR read as "the network broke" - the same misdirection #2294 named on the baseline path, one layer over. Every sweep statement now carries a five-minute timeout, and the worker caps the WHOLE sweep at the same five minutes through a linked cancellation (the sweep is awaited on the main loop, so five sequential statement timeouts must not stack into a 25-minute stall of per-server dispatch), the statement count is pinned to the timeout count so a sixth statement cannot ride the default back in, and the worker's catch names a timeout as a timeout - a sweep that still cannot finish skips the tick and the series gains a self-healing one-hour gap, deliberately NOT retrying into the same load. - **Gapped charts no longer bury their neighbours under opaque black fill** ([#2324]) - the #1944 gap markers (a NaN Y injected mid-gap so lines break across an outage) shipped first in 3.4.0, and they collide with the gradient area fill: reproduced headlessly against ScottPlot 5.1.59, one NaN in a FillY + ColorPositions series renders the ribbon as opaque black polygons with straight chord edges crossing the gap - the fill path closes its contours through the break, and its fill paint under ColorPositions is hardcoded black with the gradient shader expected to paint over it, which a NaN-bearing series defeats. On the reporter's dark theme that black buried every other series on every tab whose data had a collection gap; the one healthy tab was the one with gapless data. A gap-marked series now renders line-only - the break stays visible, nothing is buried - and continuous series keep the full gradient ribbon, pinned in both directions so the fix cannot quietly repeal the fill feature. @@ -2784,7 +2800,6 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 [#2220]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2220 [#2228]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2228 [#2218]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2218 -[#2266]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2266 [#2235]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2235 [#2165]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2165 [#2255]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2255 @@ -2800,10 +2815,24 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 [#2158]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2158 [#2246]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2246 [#2319]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2319 +[#2340]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2340 +[#2344]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2344 +[#2349]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2349 +[#2357]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2357 +[#2361]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2361 +[#2362]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2362 +[#2364]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2364 +[#2347]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2347 +[#2359]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2359 +[#2348]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2348 +[#2350]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2350 +[#2353]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2353 +[#2352]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2352 [#2331]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2331 [#2181]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2181 [#2317]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2317 [#2320]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2320 +[#2339]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2339 [#2316]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2316 [#2324]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2324 [#2300]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/2300 diff --git a/Darling/Darling.Tests/ActivityDrivenPlanFetchStoreTests.cs b/Darling/Darling.Tests/ActivityDrivenPlanFetchStoreTests.cs new file mode 100644 index 000000000..2d3138dca --- /dev/null +++ b/Darling/Darling.Tests/ActivityDrivenPlanFetchStoreTests.cs @@ -0,0 +1,130 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Linq; +using PerformanceMonitor.Darling.Storage; +using PerformanceMonitor.Darling.Viewer; +using Xunit; + +namespace Darling.Tests; + +/// +/// The V77 rung (#2312 Finding 2) — the schema strokes behind the activity-driven plan/text fetch: the +/// plan map's digest goes nullable (the content-less marker for plans whose XML the engine cannot +/// persist), query_store_text gains query_hash (the Query Store reset detector), and the +/// retired planwm:/textwm: watermark state rows are deleted wholesale. These facts pin the +/// rung's place on the ladder, the viewer probe's newest-first arm, and the migration SQL's load-bearing +/// strokes. The fetch behavior itself is pinned in QueryStorePlanFetchTests and exercised live in +/// the gated Postgres suites. +/// +public sealed class ActivityDrivenPlanFetchStoreTests +{ + /* ---------------- the rung ---------------- */ + + [Fact] + public void TheRungIsTheTopOfADenseLadder() + { + var versions = PgMigrations.Scripts.Select(s => s.Version).ToList(); + + /* #2357 added V78, so this rung is no longer the maximum. What stays true: it is PRESENT, the + ladder is ordered and dense, and the build's schema version tracks the maximum. */ + Assert.Contains(77, versions); + Assert.Equal(StorageVersion.SchemaVersion, versions.Max()); + Assert.Equal(versions.Distinct().OrderBy(v => v), versions); + + /* Dense above the one sanctioned historical hole at V45. */ + var above = versions.Where(v => v > 45).OrderBy(v => v).ToList(); + Assert.Equal(Enumerable.Range(above[0], above.Count), above); + + Assert.Equal("activity-driven-plan-fetch", PgMigrations.Scripts.Single(s => s.Version == 77).Name); + } + + /// The three strokes, each load-bearing and none allowed to drift out of the rung: without + /// the nullable digest the NULL-XML marker cannot land, without query_hash the reset detector has no + /// stored baseline, and without the deletes the orphaned watermark rows live forever (collector_state + /// has no retention, and the prune set no longer owns those prefixes). + [Fact] + public void TheRungCarriesAllThreeStrokes() + { + var sql = PgMigrations.Scripts.Single(s => s.Version == 77).Sql; + + Assert.Contains("ALTER TABLE collect.query_store_plan_map ALTER COLUMN digest DROP NOT NULL", sql, StringComparison.Ordinal); + Assert.Contains("ALTER TABLE collect.query_store_text ADD COLUMN IF NOT EXISTS query_hash text", sql, StringComparison.Ordinal); + Assert.Contains("DELETE FROM collector_state WHERE collector_name = 'query_store_plan_xml' AND state_key LIKE 'planwm:%'", sql, StringComparison.Ordinal); + Assert.Contains("DELETE FROM collector_state WHERE collector_name = 'query_store_text' AND state_key LIKE 'textwm:%'", sql, StringComparison.Ordinal); + } + + /* ---------------- the viewer probe ---------------- */ + + [Fact] + public void TheProbeMapsAStoreAtExactly77To77() + { + /* #2357 added V78, so this rung is no longer the top — the "I am the top" claim moves to the newest + rung's own test (ComposeStatementTimeoutStoreTests). What stays true forever is the arm itself: a + store migrated to EXACTLY 77 must answer 77 rather than falling through to 76. */ + Assert.Equal(StorageVersion.SchemaVersion, ViewerDataService.RequiredStoreSchemaVersion); + + /* 52 positional sentinels, then this rung's own by name. Anything a LATER rung appends is padded + by InvokeMap from the method's arity, so this count stays fixed as the ladder grows. */ + var all = Enumerable.Repeat(true, 52).Cast().ToArray(); + + Assert.Equal(77, InvokeMap(all, hasQueryStoreTextHash: true)); + Assert.Equal(76, InvokeMap(all, hasQueryStoreTextHash: false)); + } + + [Fact] + public void TheProbeAsksForTheColumn_AndTheThreePlacesAgree() + { + Assert.Contains( + "table_name = 'query_store_text' AND column_name = 'query_hash'", + ViewerDataService.StoreSchemaProbeSql, StringComparison.Ordinal); + + var mapParameters = typeof(ViewerDataService) + .GetMethod("MapProbedSchemaVersion", System.Reflection.BindingFlags.NonPublic | System.Reflection.BindingFlags.Static)! + .GetParameters().Length; + + var viewerSource = ReadViewerSource(); + + /* The reader must hand over exactly one argument per map parameter: ordinals are 0-based, so the + highest is Count - 1, and the next one up must NOT appear. */ + Assert.Contains($"reader.GetBoolean({mapParameters - 1})", viewerSource, StringComparison.Ordinal); + Assert.DoesNotContain($"reader.GetBoolean({mapParameters})", viewerSource, StringComparison.Ordinal); + } + + /* ---------------- helpers ---------------- */ + + private static int InvokeMap(object[] leading, bool hasQueryStoreTextHash) + { + var method = typeof(ViewerDataService) + .GetMethod("MapProbedSchemaVersion", System.Reflection.BindingFlags.NonPublic | System.Reflection.BindingFlags.Static)!; + + /* Parameters appended by LATER rungs are padded FALSE, so this fact keeps exercising its own arm + rather than a newer one. Derived from the method's arity rather than named by hand: listing them + made every new rung break this file, which is exactly what V79 (#2349) did. */ + var args = leading.Concat(new object[] { hasQueryStoreTextHash }).ToArray(); + args = args + .Concat(Enumerable.Repeat((object)false, method.GetParameters().Length - args.Length)) + .ToArray(); + + return (int)method.Invoke(null, args)!; + } + + private static string ReadViewerSource([System.Runtime.CompilerServices.CallerFilePath] string thisFile = "") + { + var dir = System.IO.Path.GetDirectoryName(thisFile)!; + var relative = System.IO.Path.Combine("Darling", "PerformanceMonitor.Darling.Viewer", "ViewerDataService.cs"); + while (dir is not null && !System.IO.File.Exists(System.IO.Path.Combine(dir, relative))) + { + dir = System.IO.Path.GetDirectoryName(dir); + } + + Assert.NotNull(dir); + return System.IO.File.ReadAllText(System.IO.Path.Combine(dir!, relative)); + } +} diff --git a/Darling/Darling.Tests/AlertEngineTests.cs b/Darling/Darling.Tests/AlertEngineTests.cs index c8c12353a..8ad3b34e2 100644 --- a/Darling/Darling.Tests/AlertEngineTests.cs +++ b/Darling/Darling.Tests/AlertEngineTests.cs @@ -74,6 +74,12 @@ test switches on exactly the check it pins (a disabled check must not even fetch /* #1984: DarlingConfig defaults (40% / 1 GB); enable stays the class's opt-in OFF. */ public int PvsThresholdPercent { get; set; } = 40; public int PvsFloorGb { get; set; } = 1; + + /* #2349: OFF in the fakes so existing expectations are untouched. */ + public bool FileGrowthEnabled { get; set; } + public int FileGrowthRiseMb { get; set; } = 10240; + public int FileGrowthVolumePercent { get; set; } = 60; + public int FileGrowthLookbackMinutes { get; set; } = 60; public int LongRunningJobMultiplier { get; set; } = 3; public int FailedJobLookbackMinutes { get; set; } = 60; public int CooldownMinutes { get; set; } = 5; @@ -136,6 +142,12 @@ public Task> GetLongRunningQueriesAsync( public Task> GetVolumeFreeSpaceAsync(string serverKey, CancellationToken cancellationToken = default) => Task.FromResult(new List(Volumes)); + /* #2349: empty on purpose. These tests exercise other alerts, and a fabricated file would + make the file-growth gate fire inside an unrelated scenario. */ + public Task> GetDatabaseFileGrowthAsync( + string serverKey, int lookbackMinutes, CancellationToken cancellationToken = default) => + Task.FromResult(new List()); + public Task GetTempDbSpaceAsync(string serverKey, CancellationToken cancellationToken = default) => Task.FromResult(TempDb); @@ -1365,6 +1377,12 @@ public Task> GetLongRunningQueriesAsync(string server throw new InvalidOperationException("store down"); public Task> GetVolumeFreeSpaceAsync(string serverKey, CancellationToken cancellationToken = default) => throw new InvalidOperationException("store down"); + + /* #2349: empty on purpose. These tests exercise other alerts, and a fabricated file would + make the file-growth gate fire inside an unrelated scenario. */ + public Task> GetDatabaseFileGrowthAsync( + string serverKey, int lookbackMinutes, CancellationToken cancellationToken = default) => + Task.FromResult(new List()); public Task GetTempDbSpaceAsync(string serverKey, CancellationToken cancellationToken = default) => throw new InvalidOperationException("store down"); public Task> GetPvsPressureAsync(string serverKey, CancellationToken cancellationToken = default) => diff --git a/Darling/Darling.Tests/AlertStoredValueTests.cs b/Darling/Darling.Tests/AlertStoredValueTests.cs index fc876c649..4aaa39f8b 100644 --- a/Darling/Darling.Tests/AlertStoredValueTests.cs +++ b/Darling/Darling.Tests/AlertStoredValueTests.cs @@ -95,6 +95,12 @@ private sealed class Settings : IAlertEngineSettings public int CollectionFailureThreshold { get; set; } = 10; public int PvsThresholdPercent { get; set; } = 40; public int PvsFloorGb { get; set; } = 1; + + /* #2349: OFF in the fakes so existing expectations are untouched. */ + public bool FileGrowthEnabled { get; set; } + public int FileGrowthRiseMb { get; set; } = 10240; + public int FileGrowthVolumePercent { get; set; } = 60; + public int FileGrowthLookbackMinutes { get; set; } = 60; public int LongRunningJobMultiplier { get; set; } = 3; public int FailedJobLookbackMinutes { get; set; } = 60; public int CooldownMinutes { get; set; } = 5; diff --git a/Darling/Darling.Tests/AzureForeignStatePruneTests.cs b/Darling/Darling.Tests/AzureForeignStatePruneTests.cs index 345a3b439..5c0c33b46 100644 --- a/Darling/Darling.Tests/AzureForeignStatePruneTests.cs +++ b/Darling/Darling.Tests/AzureForeignStatePruneTests.cs @@ -125,17 +125,15 @@ public void ADatabaseNamedRegistrationIsPruned(string catalog) [Fact] public void BothArmsPruneEveryPerDatabasePrefix() { - Assert.Equal(5, QueryStorePerDatabaseState.PrunableKeys.Count); - Assert.Contains(QueryStorePerDatabaseState.PrunableKeys, - k => k.Prefix == QueryStorePlanXmlState.WatermarkKeyPrefix); + /* #2312 shrank this from five to three: the planwm:/textwm: watermark families retired with the + watermarks themselves (the fetches are activity-driven against the store now), and V77 deleted + their orphaned rows wholesale — a dropped-database prune has nothing left to own there. */ + Assert.Equal(3, QueryStorePerDatabaseState.PrunableKeys.Count); Assert.Contains(QueryStorePerDatabaseState.PrunableKeys, k => k.Prefix == QueryStoreBackfillState.DoneKeyPrefix); Assert.Contains(QueryStorePerDatabaseState.PrunableKeys, k => k.Prefix == QueryStoreBackfillState.HoleKeyPrefix); - /* #2150: the text watermark, keyed prefix + databaseName exactly like the plan watermark. */ - Assert.Contains(QueryStorePerDatabaseState.PrunableKeys, - k => k.Prefix == QueryStoreTextState.WatermarkKeyPrefix); - /* #2312: the open-interval refresh stamp, the fifth per-database prefix. */ + /* #2312: the open-interval refresh stamp. */ Assert.Contains(QueryStorePerDatabaseState.PrunableKeys, k => k.Prefix == QueryStoreOpenIntervalState.WatermarkKeyPrefix); diff --git a/Darling/Darling.Tests/ComposeStatementTimeoutStoreTests.cs b/Darling/Darling.Tests/ComposeStatementTimeoutStoreTests.cs new file mode 100644 index 000000000..1c888dc39 --- /dev/null +++ b/Darling/Darling.Tests/ComposeStatementTimeoutStoreTests.cs @@ -0,0 +1,174 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Linq; +using PerformanceMonitor.Darling.Service; +using PerformanceMonitor.Darling.Storage; +using PerformanceMonitor.Darling.Viewer; +using Xunit; + +namespace Darling.Tests; + +/// +/// The V78 rung (#2357) — the compose statement_timeout knob, and the top of the ladder. +/// +/// Why it could not just be a raised constant. The timeout is applied in role PROVISIONING DDL, +/// deliberately not a versioned migration: a role's statement_timeout has no probeable schema footprint, +/// so tying it to StorageVersion.SchemaVersion would break the viewer's connect-time version gate. An +/// existing install therefore already has the old value baked into its roles, and bumping a constant in a new +/// build would appear to do nothing. +/// +/// Why no new machinery was needed to deliver it. That same provisioning SQL is re-run on every +/// managed start — "idempotent + self-healing: re-run every managed start, converging role state" — so reading +/// the knob there means a changed value reaches a running install on its next restart. +/// +public class ComposeStatementTimeoutStoreTests +{ + [Fact] + public void TheRungIsRegisteredAndIsTheTopOfADenseLadder() + { + var versions = PgMigrations.Scripts.Select(s => s.Version).ToList(); + + /* #2349 added V79, so this rung is no longer the top -- the "I am the top" claim moves to the newest + rung's own test (FileGrowthAlertStoreTests). What stays true: it is PRESENT, the ladder is ordered + and dense, and the build's schema version tracks the maximum. */ + Assert.Equal("compose-statement-timeout", PgMigrations.Scripts.Single(s => s.Version == 78).Name); + Assert.Contains(78, versions); + Assert.Equal(StorageVersion.SchemaVersion, versions.Max()); + + /* Ordered, and dense above the one sanctioned historical hole at V45. */ + Assert.Equal(versions.Distinct().OrderBy(v => v), versions); + var above = versions.Where(v => v > 45).OrderBy(v => v).ToList(); + Assert.Equal(Enumerable.Range(above[0], above.Count), above); + } + + /// + /// The rung adds the column with the default that reproduces today's behaviour, and does it idempotently — + /// a rung that is not re-runnable turns a retried upgrade into a failed one. + /// + [Fact] + public void TheRungAddsTheColumn_Idempotently_WithTodaysValueAsTheDefault() + { + var sql = PgMigrations.Scripts.Single(s => s.Version == 78).Sql; + + Assert.Contains("ALTER TABLE config.config_service", sql, StringComparison.Ordinal); + Assert.Contains("ADD COLUMN IF NOT EXISTS compose_statement_timeout_seconds", sql, StringComparison.Ordinal); + Assert.Contains("DEFAULT 15", sql, StringComparison.Ordinal); + } + + /// + /// The probe, the reader ordinals and the map arity are three places that must agree; the reader hands over + /// exactly one argument per map parameter, so an added sentinel that forgot its ordinal fails here. + /// + [Fact] + public void TheProbeAsksForTheColumn_AndTheThreePlacesAgree() + { + Assert.Contains( + "table_name = 'config_service' AND column_name = 'compose_statement_timeout_seconds'", + ViewerDataService.StoreSchemaProbeSql, StringComparison.Ordinal); + + var mapParameters = typeof(ViewerDataService) + .GetMethod("MapProbedSchemaVersion", System.Reflection.BindingFlags.NonPublic | System.Reflection.BindingFlags.Static)! + .GetParameters().Length; + + var viewerSource = ReadViewerSource(); + + Assert.Contains($"reader.GetBoolean({mapParameters - 1})", viewerSource, StringComparison.Ordinal); + Assert.DoesNotContain($"reader.GetBoolean({mapParameters})", viewerSource, StringComparison.Ordinal); + } + + /// + /// A fully migrated store maps to exactly 78, and the previous arm still answers 77 rather than falling + /// through — the invariant that keeps the version banner from reporting a mismatch on a current store. + /// + [Fact] + public void TheProbeMapsAStoreAtExactly78To78() + { + Assert.Equal(StorageVersion.SchemaVersion, ViewerDataService.RequiredStoreSchemaVersion); + + var method = typeof(ViewerDataService) + .GetMethod("MapProbedSchemaVersion", System.Reflection.BindingFlags.NonPublic | System.Reflection.BindingFlags.Static)!; + var arity = method.GetParameters().Length; + + /* Every parameter appended by a LATER rung is padded FALSE, derived from arity so a future rung does + not have to edit this file -- the lesson from #2357, where four older rung tests broke at once. + + The LEADING count is fixed at this rung's own ordinal and must NOT be derived from arity: padding + that side instead would slide ownFlag one position right per new rung, so the assertion would drift + onto a newer arm while still passing. It read correctly with exactly one rung above it, which is + the only reason V79 did not catch it here too. */ + object[] Args(bool ownFlag) => Enumerable.Repeat(true, 53).Cast() + .Concat(new object[] { ownFlag }) + .Concat(Enumerable.Repeat((object)false, arity - 54)) + .ToArray(); + + Assert.Equal(78, (int)method.Invoke(null, Args(true))!); + Assert.Equal(77, (int)method.Invoke(null, Args(false))!); + } + + /// + /// The backstop must survive configuration. A LIMIT bounds OUTPUT; a group-by scans and sorts before + /// it. Something has to bound WORK, so the knob is clamped rather than trusted — zero or negative would + /// remove the ceiling entirely, which is the one outcome the whole design leans on not happening. + /// + [Theory] + [InlineData(0, 15)] + [InlineData(-1, 15)] + [InlineData(1, 5)] + [InlineData(5, 5)] + [InlineData(15, 15)] + [InlineData(120, 120)] + [InlineData(600, 600)] + [InlineData(9999, 600)] + public void TheKnobIsClamped(int stored, int effective) + { + Assert.Equal(effective, StoreConfigProvider.ClampComposeStatementTimeoutSeconds(stored)); + } + + /// + /// The provisioning DDL carries the configured value, and clamps independently — the method is public, and + /// a caller passing 0 must not be able to remove the ceiling. + /// + [Theory] + [InlineData(120, "120s")] + [InlineData(15, "15s")] + [InlineData(0, "15s")] + [InlineData(99999, "600s")] + public void TheProvisioningDdl_AppliesTheConfiguredTimeout(int seconds, string expected) + { + var sql = DarlingManagedRoles.BuildProvisioningSql( + "AdminPassword01", "ViewerPassword02", "McpPassword03", seconds); + + Assert.Contains($"SET statement_timeout = '{expected}'", sql, StringComparison.Ordinal); + + /* Both compose roles, not just one: the mcp role gets viewer's read surface and must get its ceiling. */ + Assert.Equal(2, System.Text.RegularExpressions.Regex.Matches(sql, @"SET statement_timeout = '").Count); + } + + /// Omitting it reproduces the constant it replaced, so an untouched install is unchanged. + [Fact] + public void TheDefaultReproducesTheOldConstant() + { + var sql = DarlingManagedRoles.BuildProvisioningSql("AdminPassword01", "ViewerPassword02", "McpPassword03"); + + Assert.Contains("SET statement_timeout = '15s'", sql, StringComparison.Ordinal); + } + + private static string ReadViewerSource([System.Runtime.CompilerServices.CallerFilePath] string thisFile = "") + { + var dir = System.IO.Path.GetDirectoryName(thisFile)!; + var relative = System.IO.Path.Combine("Darling", "PerformanceMonitor.Darling.Viewer", "ViewerDataService.cs"); + while (dir is not null && !System.IO.File.Exists(System.IO.Path.Combine(dir, relative))) + { + dir = System.IO.Path.GetDirectoryName(dir); + } + + return System.IO.File.ReadAllText(System.IO.Path.Combine(dir!, relative)); + } +} diff --git a/Darling/Darling.Tests/Darling.Tests.csproj b/Darling/Darling.Tests/Darling.Tests.csproj index 4395a579d..f134af585 100644 --- a/Darling/Darling.Tests/Darling.Tests.csproj +++ b/Darling/Darling.Tests/Darling.Tests.csproj @@ -6,6 +6,8 @@ enable false true + + Exe @@ -13,11 +15,6 @@ - - - all - runtime; build; native; contentfiles; analyzers; buildtransitive - diff --git a/Darling/Darling.Tests/DarlingAgReaderTests.cs b/Darling/Darling.Tests/DarlingAgReaderTests.cs index 2b11a1df8..9c1fa4b47 100644 --- a/Darling/Darling.Tests/DarlingAgReaderTests.cs +++ b/Darling/Darling.Tests/DarlingAgReaderTests.cs @@ -388,8 +388,8 @@ public void SerializedShape_CarriesTheFieldsBothConsumersRead() /* Severities serialize as NAMES (the JsonStringEnumConverter), not integers — the browser maps the name to a CSS class, so a numeric enum would silently break every color. */ - Assert.Contains("\"severity\": \"Critical\"", json, StringComparison.Ordinal); - Assert.DoesNotContain("\"severity\": 3", json, StringComparison.Ordinal); + JsonAssert.Contains("\"severity\": \"Critical\"", json); + JsonAssert.DoesNotContain("\"severity\": 3", json); } /* ─────────────────────────── SQL dialect pins ─────────────────────────── */ diff --git a/Darling/Darling.Tests/DarlingDimensionGcBoundTests.cs b/Darling/Darling.Tests/DarlingDimensionGcBoundTests.cs index c4c9b19da..61a296805 100644 --- a/Darling/Darling.Tests/DarlingDimensionGcBoundTests.cs +++ b/Darling/Darling.Tests/DarlingDimensionGcBoundTests.cs @@ -144,32 +144,9 @@ public void NeitherPruneOrder_CanLeaveAMapRowResolvingToAnAbsentDigest(int factR } } - /// - /// #2210: the re-verify cursor paces itself off RefreshAfter and NEVER touches the watermark. The - /// slice is a row count over an id range, which is the whole point — the old expiry walked BYTES and could - /// not finish inside a day on the catalogs that mattered (15.9 to 107.5 hours measured), so those restarted - /// forever. Redstone's 77k ids at a 5-minute cadence over a 1-day sweep is ~267 ids per pass. - /// - [Fact] - public void CursorSlice_PacesASweepWithinTheRefreshPeriod_AndNeverReturnsZeroForALiveCatalog() - { - var day = TimeSpan.FromDays(1); - var cadence = TimeSpan.FromMinutes(5); - - var redstone = QueryStorePlanMap.CursorSliceWidth(77_176, day, cadence); - Assert.InRange(redstone, 200, 350); - - /* A sweep must actually cover the range within the period: slice * passes >= watermark. */ - var passes = day.Ticks / cadence.Ticks; - Assert.True(redstone * passes >= 77_176, "the sweep must cover the id range inside one refresh period"); - - /* Never zero for a live catalog, and never wider than the range itself. */ - Assert.True(QueryStorePlanMap.CursorSliceWidth(10, day, cadence) > 0); - Assert.Equal(10, QueryStorePlanMap.CursorSliceWidth(10, day, cadence)); - - /* A fresh database has no watermark to re-verify, so there is nothing to slice. */ - Assert.Equal(0, QueryStorePlanMap.CursorSliceWidth(0, day, cadence)); - } + /* #2312: the CursorSlice facts that sat here are gone with the cursor itself — it was designed in + #2210 and never wired (Finding 4), and the in-place-rewrite job it existed for now runs per-cycle + through TouchAndProbeSql's hash comparison, pinned in QueryStorePlanFetchTests. */ /// /// #2210: the DIMENSION must outlive the MAP, expressed the way it actually matters — as cutoff DATES from diff --git a/Darling/Darling.Tests/DarlingFleetReaderTests.cs b/Darling/Darling.Tests/DarlingFleetReaderTests.cs index a7731f89f..7438fac7d 100644 --- a/Darling/Darling.Tests/DarlingFleetReaderTests.cs +++ b/Darling/Darling.Tests/DarlingFleetReaderTests.cs @@ -236,13 +236,13 @@ public void FleetServerCard_SerializesSnakeCase_WithStringBands() } /* Bands / severities serialize as strings, not ordinals — the frontend maps a name to a color. */ - Assert.Contains("\"band\": \"Critical\"", json, StringComparison.Ordinal); - Assert.Contains("\"cpu_severity\": \"Critical\"", json, StringComparison.Ordinal); - Assert.Contains("\"threads_severity\": \"Unknown\"", json, StringComparison.Ordinal); + JsonAssert.Contains("\"band\": \"Critical\"", json); + JsonAssert.Contains("\"cpu_severity\": \"Critical\"", json); + JsonAssert.Contains("\"threads_severity\": \"Unknown\"", json); /* naive-UTC instants carry no zone suffix (localized in the browser). */ - Assert.Contains("\"last_collection\": \"2026-07-18T03:30:00\"", json, StringComparison.Ordinal); - Assert.Contains("\"deadlock_last_seen\": \"2026-07-18T03:15:00\"", json, StringComparison.Ordinal); - Assert.DoesNotContain("\"cpu_severity\": 3", json, StringComparison.Ordinal); + JsonAssert.Contains("\"last_collection\": \"2026-07-18T03:30:00\"", json); + JsonAssert.Contains("\"deadlock_last_seen\": \"2026-07-18T03:15:00\"", json); + JsonAssert.DoesNotContain("\"cpu_severity\": 3", json); } [Fact] @@ -252,17 +252,17 @@ public void FleetServerCard_SerializesPerServerPlatform_ForComposerD4Greying() measure's appliesTo.azureSqlDb to auto-grey a measure that platform can't collect. */ var azure = new FleetServerCard { ServerId = 1, DisplayName = "az-db", ServerName = "az-db", EngineEdition = 5, IsAzureSqlDb = true }; var azureJson = JsonSerializer.Serialize(azure, DarlingFleetReader.JsonOptions); - Assert.Contains("\"engine_edition\": 5", azureJson, StringComparison.Ordinal); - Assert.Contains("\"is_azure_sql_db\": true", azureJson, StringComparison.Ordinal); - Assert.Contains("\"is_azure_mi\": false", azureJson, StringComparison.Ordinal); + JsonAssert.Contains("\"engine_edition\": 5", azureJson); + JsonAssert.Contains("\"is_azure_sql_db\": true", azureJson); + JsonAssert.Contains("\"is_azure_mi\": false", azureJson); /* A server that has not connected: null edition serializes as JSON null and both flags are false, so the frontend has no signal and keeps the measure badge rather than greying on a guess. */ var unknown = new FleetServerCard { ServerId = 2, DisplayName = "new", ServerName = "new" }; var unknownJson = JsonSerializer.Serialize(unknown, DarlingFleetReader.JsonOptions); - Assert.Contains("\"engine_edition\": null", unknownJson, StringComparison.Ordinal); - Assert.Contains("\"is_azure_sql_db\": false", unknownJson, StringComparison.Ordinal); - Assert.Contains("\"is_azure_mi\": false", unknownJson, StringComparison.Ordinal); + JsonAssert.Contains("\"engine_edition\": null", unknownJson); + JsonAssert.Contains("\"is_azure_sql_db\": false", unknownJson); + JsonAssert.Contains("\"is_azure_mi\": false", unknownJson); } [Fact] @@ -306,7 +306,7 @@ public void FleetOverviewResult_SerializesRollupShape() Assert.Contains(field, json, StringComparison.Ordinal); } - Assert.Contains("\"band_label\": \"Critical\"", json, StringComparison.Ordinal); + JsonAssert.Contains("\"band_label\": \"Critical\"", json); } } diff --git a/Darling/Darling.Tests/DarlingHardenFilesVerbTests.cs b/Darling/Darling.Tests/DarlingHardenFilesVerbTests.cs new file mode 100644 index 000000000..a49e5660b --- /dev/null +++ b/Darling/Darling.Tests/DarlingHardenFilesVerbTests.cs @@ -0,0 +1,229 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.IO; +using System.Runtime.CompilerServices; +using PerformanceMonitor.Darling.Service; +using Xunit; + +namespace Darling.Tests; + +/// +/// #2352: --harden-files — the actor that can actually apply the secret-file ACLs. +/// +/// Why the verb exists. The service already computes the correct DACL +/// (DarlingFileSecurity.HardenFile) and already detects when the real one is wrong +/// (IsReadableByOrdinaryUsers). What it lacks is authority: re-ACLing a file it does not own needs +/// WRITE_DAC, and taking ownership needs a privilege a virtual service account is not granted. So it logs the +/// remedy and continues — correctly, since a monitoring service must not refuse to monitor over a permissions +/// problem — and until now the only thing that ever APPLIED the rule to an existing install was +/// install-darling.ps1. A box registered by hand through the README's own sc create path left the +/// operator typing three icacls lines out of a log message. +/// +/// The ACL work itself is Windows-only and needs a real filesystem, so what is pinned here is everything +/// that decides whether the verb is REACHABLE and honest — the failure mode that shipped once already (#1912), +/// where a verb had a full dispatch block, appeared in the help text, and was bounced as "Unknown option" +/// because the allow-list never learned it. +/// +public class DarlingHardenFilesVerbTests +{ + [Theory] + [InlineData("--harden-files", true)] + [InlineData("--HARDEN-FILES", true)] + [InlineData("--Harden-Files", true)] + [InlineData("--harden", false)] + [InlineData("--harden-file", false)] + [InlineData("--configure-firewall", false)] + [InlineData("--nonsense", false)] + public void IsHardenFilesVerb_RecognizesTheVerb_CaseInsensitive(string arg, bool expected) + { + Assert.Equal(expected, DarlingCliCommands.IsHardenFilesVerb(arg)); + } + + /// + /// The allow-list must reach it or the Program.cs dispatch is dead code and the startup classifier answers + /// "Unknown option" instead. The generic reflection pin covers this too; naming it makes the intent local. + /// + [Fact] + public void IsKnownVerb_ReachesTheHardenVerb() + { + Assert.True(DarlingCliCommands.IsKnownVerb("--harden-files")); + } + + /// An operator who cannot find the verb does not have it. It is the remedy for a CRITICAL log line, + /// so it has to be listed where someone reading that line will look. + [Fact] + public void TheHelpText_ListsTheVerb_AndSaysItNeedsElevation() + { + var help = DarlingCliCommands.UsageText(); + + Assert.Contains("--harden-files", help, StringComparison.Ordinal); + Assert.Contains("elevated", help, StringComparison.OrdinalIgnoreCase); + } + + /// + /// Program.cs dispatches it, guarded on Windows, and calls it rather than something adjacent — pinned + /// structurally because the seam between the classifier and the dispatch is exactly where #1912's drift + /// lived, and no unit test that calls DarlingCliCommands directly can see it. + /// + [Fact] + public void ProgramDispatchesTheVerb_BehindAWindowsGuard() + { + var program = ReadRepoFile(Path.Combine( + "Darling", "PerformanceMonitor.Darling.Service", "Program.cs")); + + var at = program.IndexOf("DarlingCliCommands.IsHardenFilesVerb(args[0])", StringComparison.Ordinal); + Assert.True(at >= 0, "Program.cs no longer dispatches --harden-files (#2352)"); + + var block = program[at..Math.Min(program.Length, at + 900)]; + + Assert.Contains("OperatingSystem.IsWindows()", block, StringComparison.Ordinal); + Assert.Contains("DarlingCliCommands.HardenFiles(", block, StringComparison.Ordinal); + } + + /// + /// The verdict is the RE-READ, never the call that returned without throwing. "We tried" is not the same + /// statement as "the secret is not readable" — the distinction that let a permissions call which silently + /// did nothing hide in the field. Pinned on the source because the behaviour needs Windows ACLs to observe. + /// + [Fact] + public void TheVerb_VerifiesEachTarget_AndFailsWhenAnythingIsStillReadable() + { + var source = ReadRepoFile(Path.Combine( + "Darling", "PerformanceMonitor.Darling.Service", "DarlingCliCommands.cs")); + + var at = source.IndexOf("public static int HardenFiles(", StringComparison.Ordinal); + Assert.True(at >= 0, "HardenFiles is gone (#2352)"); + + var body = source[at..Math.Min(source.Length, at + 6000)]; + + /* It re-reads rather than trusting the call. */ + Assert.Contains("DarlingFileSecurity.IsReadableByOrdinaryUsers(", body, StringComparison.Ordinal); + + /* And a still-exposed target is a non-zero exit, so it is usable in a provisioning script. */ + Assert.Contains("STILL READABLE", body, StringComparison.Ordinal); + + /* The live config is the ONLY target the interactive operator keeps read on: the Viewer and the CLI + verbs run as that operator and must still read it. Nothing reads a backup (#1769). */ + Assert.Contains("AllowInteractive: true", body, StringComparison.Ordinal); + } + + /// + /// #2371: the verb hardens for the account the SERVICE is registered under, never for whoever is running it. + /// + /// The bug this pins. Every original caller of DarlingFileSecurity runs INSIDE the + /// service, so "the current identity" and "the account the service runs as" were the same value and the + /// distinction did not exist. This verb inverts that by construction: it exists because a virtual service + /// account cannot re-ACL a file it does not own, so it is ALWAYS run by somebody else. Resolving from the + /// caller therefore granted the operator and stripped the service — measured on a live box, where an + /// elevated run removed NT SERVICE\PerformanceMonitor Darling from all four targets and printed + /// All 4 item(s) secured while doing it. The install kept working until its next restart, which is + /// far enough away that nobody would connect the two. + /// + [Fact] + public void TheVerb_HardensForTheRegisteredServiceAccount_NotTheCaller() + { + var source = ReadRepoFile(Path.Combine( + "Darling", "PerformanceMonitor.Darling.Service", "DarlingCliCommands.cs")); + + var at = source.IndexOf("public static int HardenFiles(", StringComparison.Ordinal); + Assert.True(at >= 0, "HardenFiles is gone (#2352)"); + + var body = source[at..Math.Min(source.Length, at + 8000)]; + + /* It asks the SCM, and hardens for what it gets back. */ + Assert.Contains("RegisteredServiceAccount(ServiceName)", body, StringComparison.Ordinal); + Assert.Contains("HardenForAccount(", body, StringComparison.Ordinal); + + /* The resolution happens BEFORE the first target is hardened, or the early targets get the caller's + ACL and the later ones the service's -- a split-brain worse than either alone. */ + Assert.True( + body.IndexOf("HardenForAccount(", StringComparison.Ordinal) + < body.IndexOf("DarlingFileSecurity.HardenFile(", StringComparison.Ordinal), + "the service account must be resolved before anything is hardened (#2371)"); + } + + /// + /// #2371: private is only half of correct, so the verify pass checks BOTH directions. + /// + /// IsReadableByOrdinaryUsers asks whether anyone TOO MANY can read. It cannot see the opposite + /// failure — an ACL that excludes ordinary users AND the service is maximally private and completely broken, + /// and it passed the old check, which is why the live run reported success on four locked-out targets. + /// + [Fact] + public void TheVerb_AlsoVerifiesTheServiceCanStillReadEachTarget() + { + var source = ReadRepoFile(Path.Combine( + "Darling", "PerformanceMonitor.Darling.Service", "DarlingCliCommands.cs")); + + var at = source.IndexOf("public static int HardenFiles(", StringComparison.Ordinal); + var body = source[at..Math.Min(source.Length, at + 8000)]; + + Assert.Contains("GrantsHardenedAccount(", body, StringComparison.Ordinal); + Assert.Contains("LOCKED OUT", body, StringComparison.Ordinal); + + /* A lockout counts as exposure, so the verb exits non-zero and a provisioning script stops. Both + branches feed the same counter the STILL READABLE path does. */ + var lockedAt = body.IndexOf("LOCKED OUT", StringComparison.Ordinal); + Assert.Contains("exposed++", body[lockedAt..Math.Min(body.Length, lockedAt + 700)], StringComparison.Ordinal); + } + + /// + /// The resolver reads the SCM's own ObjectName, which is the account the service is logged on with — + /// and returns null rather than throwing when the service is not registered, so a console run or a + /// pre-install harden falls back to the caller instead of failing. + /// + [Fact] + public void TheResolver_ReturnsNull_WhenTheServiceIsNotRegistered() + { + if (!OperatingSystem.IsWindows()) + { + return; + } + + Assert.Null(DarlingFileSecurity.RegisteredServiceAccount( + "PerformanceMonitor Darling NoSuchService " + Guid.NewGuid().ToString("N"))); + } + + /// + /// LocalSystem is the one ObjectName with no + /// spelling to translate — the SCM stores it unqualified — so it is mapped to its well-known SID by hand. + /// An install re-homed to LocalSystem would otherwise fall back to the caller and reintroduce the bug on + /// exactly the configuration that looks most ordinary. + /// + [Fact] + public void TheResolver_HandlesTheUnqualifiedLocalSystemAlias() + { + var source = ReadRepoFile(Path.Combine( + "Darling", "PerformanceMonitor.Darling.Service", "DarlingFileSecurity.cs")); + + var at = source.IndexOf("RegisteredServiceAccount(", StringComparison.Ordinal); + Assert.True(at >= 0, "the SCM resolver is gone (#2371)"); + + var body = source[at..Math.Min(source.Length, at + 1800)]; + + Assert.Contains("ObjectName", body, StringComparison.Ordinal); + Assert.Contains("LocalSystem", body, StringComparison.Ordinal); + Assert.Contains("WellKnownSidType.LocalSystemSid", body, StringComparison.Ordinal); + } + + private static string ReadRepoFile(string relative, [CallerFilePath] string thisFile = "") + { + for (var dir = new DirectoryInfo(Path.GetDirectoryName(thisFile)!); dir is not null; dir = dir.Parent) + { + var candidate = Path.Combine(dir.FullName, relative); + if (File.Exists(candidate)) + { + return File.ReadAllText(candidate); + } + } + + throw new FileNotFoundException($"Could not locate {relative} walking up from {thisFile}"); + } +} diff --git a/Darling/Darling.Tests/DarlingInstallLocationTests.cs b/Darling/Darling.Tests/DarlingInstallLocationTests.cs index 7a988735f..deefcbcce 100644 --- a/Darling/Darling.Tests/DarlingInstallLocationTests.cs +++ b/Darling/Darling.Tests/DarlingInstallLocationTests.cs @@ -124,8 +124,10 @@ public void LocationGuard_ReadsTheProfileRootFromWindows_RatherThanAssumingCUser Assert.Contains("ProfilesDirectory", script, StringComparison.Ordinal); /* And the current user's own profile is checked as well: a profile redirected outside - ProfilesDirectory is still a profile, and it is the one whose owner is most likely running this. */ - Assert.Contains("Test-PathIsAtOrUnder $root $env:USERPROFILE", script, StringComparison.Ordinal); + ProfilesDirectory is still a profile, and it is the one whose owner is most likely running this. + Against $classifyRoot since #2348 — the normalized spelling, so the extended-length form of a + profile path is caught too. */ + Assert.Contains("Test-PathIsAtOrUnder $classifyRoot $env:USERPROFILE", script, StringComparison.Ordinal); } /// @@ -219,6 +221,12 @@ way out of the profile. */ /// /// The network half, also executed. \\?\C:\... is the long-path prefix on a LOCAL path, not a /// server name, and treating it as a share would refuse a perfectly ordinary install root. + /// + /// The probe composes Convert-ExtendedLengthPath ahead of Get-NetworkPathKind because + /// that is what the script itself does ($classifyRoot = Convert-ExtendedLengthPath $root). Calling + /// the kind function on a RAW path would be testing a contract the installer does not use: since #2348 it + /// classifies normalized paths, so it no longer carries a \\?\ carve-out of its own and would call a + /// bare \\?\C:\... a share. Keeping the composition here is the point — the pair is the unit. /// [Fact] public void GetNetworkPathKind_SeparatesAShareFromAnExtendedLengthLocalPath_AsShipped() @@ -230,13 +238,21 @@ public void GetNetworkPathKind_SeparatesAShareFromAnExtendedLengthLocalPath_AsSh (@"C:\PerformanceMonitorDarling", ""), (@"C:\Users\bob\Desktop\PerformanceMonitorDarling", ""), ("", ""), + + /* #2348: the extended-length spelling of a REAL share is a share. Previously the wholesale \\?\ + exclusion waved all three of these through as "". */ + (@"\\?\UNC\fileserver\share\PerformanceMonitorDarling", "UNC"), + (@"\\?\unc\fileserver\share\PerformanceMonitorDarling", "UNC"), + (@"\\?\UNC\fileserver\share", "UNC"), }; var probe = new StringBuilder(); + probe.AppendLine(ExtractFunction(InstallScript, "Convert-ExtendedLengthPath")); probe.AppendLine(ExtractFunction(InstallScript, "Get-NetworkPathKind")); foreach (var (path, _) in cases) { - probe.AppendLine($"$k = Get-NetworkPathKind '{path}'; if ($null -eq $k) {{ '' }} else {{ $k }}"); + probe.AppendLine( + $"$k = Get-NetworkPathKind (Convert-ExtendedLengthPath '{path}'); if ($null -eq $k) {{ '' }} else {{ $k }}"); } var answers = RunWindowsPowerShell(probe.ToString()); @@ -248,6 +264,48 @@ public void GetNetworkPathKind_SeparatesAShareFromAnExtendedLengthLocalPath_AsSh } } + /// + /// #2348, the normalization itself, executed as shipped. It is the one place in either implementation that + /// knows the \\?\ prefix exists, so every rule downstream can be written against real paths. + /// + /// The lowercase case is not padding: Windows accepts \\?\unc\, so an ordinal match on + /// UNC would leave a share spelled that way looking like the local path unc\server\share — + /// re-opening exactly the hole this closes, for the operator least likely to be checked on. + /// + [Fact] + public void ConvertExtendedLengthPath_RewritesBothSpellings_AsShipped() + { + var cases = new (string Path, string Expected)[] + { + (@"\\?\UNC\fileserver\share\dir", @"\\fileserver\share\dir"), + (@"\\?\unc\fileserver\share\dir", @"\\fileserver\share\dir"), + (@"\\?\C:\PerformanceMonitorDarling", @"C:\PerformanceMonitorDarling"), + (@"\\?\C:\Users\bob\dir", @"C:\Users\bob\dir"), + + /* Untouched: an ordinary local path, an ordinary share, and a path that merely CONTAINS the + characters without leading with them. */ + (@"C:\PerformanceMonitorDarling", @"C:\PerformanceMonitorDarling"), + (@"\\fileserver\share\dir", @"\\fileserver\share\dir"), + ("", ""), + }; + + var probe = new StringBuilder(); + probe.AppendLine(ExtractFunction(InstallScript, "Convert-ExtendedLengthPath")); + foreach (var (path, _) in cases) + { + probe.AppendLine($"$v = Convert-ExtendedLengthPath '{path}'; if ([string]::IsNullOrEmpty($v)) {{ '' }} else {{ $v }}"); + } + + var answers = RunWindowsPowerShell(probe.ToString()); + Assert.Equal(cases.Length, answers.Count); + + for (var i = 0; i < cases.Length; i++) + { + var expected = cases[i].Expected.Length == 0 ? "" : cases[i].Expected; + Assert.Equal(expected, answers[i]); + } + } + /// /// #2201: the mapped-drive probe must not fail OPEN when WMI is unavailable. /// @@ -334,6 +392,146 @@ function Get-PSDrive { } } + /// + /// The installer's verdict and the SERVICE's verdict must be the same verdict (#2185). + /// + /// Why this test exists. #2187 put the rule in PowerShell, where only installs that go + /// through the script can benefit; #2185 needed the service to reach the same conclusion by itself, for + /// the README's manual sc create path and for anyone who registers the exe by hand. That is two + /// implementations of one rule in two languages, and the one that drifted would be the one nobody was + /// reading. So both are run over ONE table — — and + /// disagreement is a failure regardless of which side is "right". + /// + /// What is really being executed. The PowerShell side is the shipped file: both helper + /// functions are extracted whole, and the two lines that COMPOSE them into a verdict are lifted verbatim + /// out of the script rather than retyped. Only the environment is injected — the profile root, the + /// operator's profile, and the drive type — which is exactly what the C# side takes as parameters. + /// + /// Windows-only in practice, like every test in this class: it shells out to + /// powershell.exe. + /// + [Fact] + public void TheInstallerAndTheService_ReachTheSameVerdict_OverOneTable() + { + const string ProfileRoot = @"C:\Users"; + const string UserProfile = @"C:\Users\installer"; + + var script = InstallScript; + var cases = DarlingServiceInstallLocationTests.Cases; + + var probe = new StringBuilder(); + + /* The environment, injected. Get-ProfilesDirectory reads HKLM and $env:USERPROFILE is the box's own, + and a parity test that took either from the machine it runs on would compare the two rules against + two different environments. */ + probe.AppendLine($"$env:USERPROFILE = '{UserProfile}'"); + probe.AppendLine($"function Get-ProfilesDirectory {{ '{ProfileRoot}' }}"); + + /* Z: is a mapped share and every other letter is a local fixed disk. A definite WMI answer is what the + shipped function trusts, so no Get-PSDrive shadow is needed - that fallback is #2201's territory and + is pinned on its own above. */ + probe.AppendLine(@" +function Get-CimInstance { + [CmdletBinding()] + param([Parameter(ValueFromRemainingArguments = $true)] $Rest) + if (($Rest -join ' ') -match ""DeviceID='[Zz]:'"") { return [pscustomobject]@{ DriveType = 4 } } + return [pscustomobject]@{ DriveType = 3 } +}"); + + probe.AppendLine(ExtractFunction(script, "Convert-ExtendedLengthPath")); + probe.AppendLine(ExtractFunction(script, "Test-PathIsAtOrUnder")); + probe.AppendLine(ExtractFunction(script, "Get-NetworkPathKind")); + + /* The composition, quoted out of the shipped script. The normalization line (#2348) is part of it: + both rules classify $classifyRoot, never the raw $root, and lifting only the two rule lines would + quietly test a composition the installer does not perform. */ + var normalizeLine = ExtractLine(script, "$classifyRoot = Convert-ExtendedLengthPath $root"); + var networkLine = ExtractLine(script, "$networkKind = Get-NetworkPathKind $classifyRoot"); + var profileLine = ExtractLine(script, "$underProfile = (Test-PathIsAtOrUnder $classifyRoot (Get-ProfilesDirectory))"); + + foreach (var (directory, _, _) in cases) + { + probe.AppendLine($"$root = '{directory}'"); + probe.AppendLine(normalizeLine); + probe.AppendLine(networkLine); + probe.AppendLine(profileLine); + /* The script's own precedence: the profile message is chosen first when a path is somehow both + (the 'if ($underProfile)' arm inside the guard block), and the guard fires on either. */ + probe.AppendLine("if ($underProfile) { 'UserProfile' } elseif ($networkKind -eq 'UNC') { 'UncPath' } elseif ($networkKind) { 'MappedDrive' } else { 'None' }"); + } + + var answers = RunWindowsPowerShell(probe.ToString()); + Assert.Equal(cases.Count, answers.Count); + + var disagreements = new List(); + for (var i = 0; i < cases.Count; i++) + { + var (directory, expected, because) = cases[i]; + + /* The C# side, with the same injected environment. Z: is the table's mapped drive. */ + var service = PerformanceMonitor.Darling.Service.DarlingInstallLocation.Classify( + directory, ProfileRoot, UserProfile, + static qualifier => string.Equals(qualifier, "Z:", StringComparison.OrdinalIgnoreCase)); + + /* Both sides are also checked against the table's own expectation, so a mutual mistake cannot pass + as agreement. */ + if (!string.Equals(answers[i], expected.ToString(), StringComparison.Ordinal) || service != expected) + { + disagreements.Add( + $"'{directory}': install-darling.ps1 said {answers[i]}, the service said {service}, the table says {expected} ({because})"); + } + } + + Assert.True(disagreements.Count == 0, + "the installer and the service disagree about which install locations cannot work (#2185):\n " + + string.Join("\n ", disagreements)); + } + + /// + /// The two rules must also agree about WHERE the profile root comes from, not just what they do with it + /// (review catch on #2185). + /// + /// The blind spot this closes. The table-parity test above injects the profile root into both + /// implementations, which is what makes the table comparable — and means neither side's own lookup is ever + /// exercised. The first C# version derived the root from %PUBLIC%'s parent on the belief that Windows + /// keeps PUBLIC in step with ProfilesDirectory; they are two INDEPENDENT values under one key + /// that merely default to the same tree. On a box where profiles were relocated without moving Public, the + /// installer would have refused an install the service waved through — a false negative on the exact case + /// #2185 exists to catch, invisible to every test that supplies the root itself. + /// + /// So this one supplies nothing: it runs the installer's Get-ProfilesDirectory as shipped and + /// requires to answer the same, on whatever box the + /// tests are running on. It passes trivially on an unrelocated box, which is fine — its job is to fail the + /// moment the two stop reading the same thing. + /// + [Fact] + public void TheInstallerAndTheService_ReadTheProfileRoot_FromTheSamePlace() + { + var probe = new StringBuilder(); + probe.AppendLine(ExtractFunction(InstallScript, "Get-ProfilesDirectory")); + probe.AppendLine("Get-ProfilesDirectory"); + + var answers = RunWindowsPowerShell(probe.ToString()); + Assert.Single(answers); + + var installer = answers[0].TrimEnd('\\'); + var service = PerformanceMonitor.Darling.Service.DarlingInstallLocation.MachineProfileRoot().TrimEnd('\\'); + + Assert.Equal(installer, service, ignoreCase: true); + } + + /// Returns the single line of containing , + /// verbatim — so a composition can be executed as shipped instead of retyped into a probe. + private static string ExtractLine(string script, string marker) + { + var at = script.IndexOf(marker, StringComparison.Ordinal); + Assert.True(at >= 0, $"install-darling.ps1 no longer contains '{marker}' (#2185)"); + + var start = script.LastIndexOf('\n', at) + 1; + var end = script.IndexOf('\n', at); + return (end < 0 ? script.Substring(start) : script.Substring(start, end - start)).Trim(); + } + /// Runs under Windows PowerShell 5.1 and returns its non-empty output /// lines. Written to a temp file rather than passed with -Command: the script under test is a whole /// function body, and quoting it through a command line is a source of failures that have nothing to do diff --git a/Darling/Darling.Tests/DarlingMcpConfigHistoryToolsTests.cs b/Darling/Darling.Tests/DarlingMcpConfigHistoryToolsTests.cs index 63ec2159a..d7011b338 100644 --- a/Darling/Darling.Tests/DarlingMcpConfigHistoryToolsTests.cs +++ b/Darling/Darling.Tests/DarlingMcpConfigHistoryToolsTests.cs @@ -395,9 +395,9 @@ await DarlingMcpTestData.ExecAsync(connection, ct, var qsh = await DarlingMcpConfigHistoryTools.GetQueryStoreHealth(postgres, ServerName); DarlingMcpTestData.AssertEnvelope(qsh, ServerName, "databases"); - Assert.Contains("\"state_matches_desired\": false", qsh, StringComparison.Ordinal); + JsonAssert.Contains("\"state_matches_desired\": false", qsh); Assert.Contains("storage cap reached", qsh, StringComparison.Ordinal); - Assert.Contains("\"pct_of_cap\": 100", qsh, StringComparison.Ordinal); + JsonAssert.Contains("\"pct_of_cap\": 100", qsh); bodySucceeded = true; } diff --git a/Darling/Darling.Tests/DarlingMcpDefaultTraceToolsTests.cs b/Darling/Darling.Tests/DarlingMcpDefaultTraceToolsTests.cs index b7be825d8..314b015aa 100644 --- a/Darling/Darling.Tests/DarlingMcpDefaultTraceToolsTests.cs +++ b/Darling/Darling.Tests/DarlingMcpDefaultTraceToolsTests.cs @@ -210,7 +210,7 @@ await DarlingMcpTestData.ExecAsync(connection, ct, Assert.Contains("AutoGrowShrink", json, StringComparison.Ordinal); Assert.Contains("SEVERE_MARKER", json, StringComparison.Ordinal); Assert.DoesNotContain("ROUTINE_MARKER", json, StringComparison.Ordinal); - Assert.Contains("\"total_events\": 2", json, StringComparison.Ordinal); + JsonAssert.Contains("\"total_events\": 2", json); /* Unknown server → the listing error; empty store → the miss. */ Assert.StartsWith("Could not resolve server.", await DarlingMcpDefaultTraceTools.GetDefaultTraceEvents(postgres, "darling-no-such-server"), StringComparison.Ordinal); diff --git a/Darling/Darling.Tests/DarlingPeerDisclosureTests.cs b/Darling/Darling.Tests/DarlingPeerDisclosureTests.cs new file mode 100644 index 000000000..432ef5722 --- /dev/null +++ b/Darling/Darling.Tests/DarlingPeerDisclosureTests.cs @@ -0,0 +1,585 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Linq; +using System.Text.Json; +using PerformanceMonitor.Darling.Service; +using PerformanceMonitor.Darling.Service.Mcp; +using Xunit; + +namespace Darling.Tests; + +/// +/// Pins the declared-peer disclosure (#2339, tier 1): the peers config block, the MCP-instructions +/// coverage section, list_servers' peer_fleets block, and the server-resolution miss message. +/// +/// The invariant that matters most is the negative one. With nothing declared — every existing +/// deployment, and the shipped sample — the PROSE surfaces must be byte-for-byte what they were, because this +/// feature is disclosure bolted onto paths that ~90 MCP tools and the whole web read dispatch already go +/// through. Three of the four are: the MCP instructions ( returns +/// the same reference), the server-resolution miss (pinned by exact string equality), and +/// list_servers' empty-registry sentence. Each gets a paired test: what it says with peers, and that +/// it says nothing extra without them. +/// +/// The one deliberate exception is list_servers' JSON envelope, which gains +/// this_store_covers / peer_fleets / peer_note on EVERY response, declared or not — so a +/// client doing an exact-shape comparison on that tool sees three new keys on upgrade even if it never +/// touches the peers config. That is the point rather than an oversight: an empty peer_fleets +/// means either "this is the only store" or "nobody declared the siblings", and an agent can only be told +/// the difference is unknowable if the note is there when the list is empty. A conditional block would say +/// nothing in exactly the case that produces the wrong conclusion. Called out here, and in the CHANGELOG, +/// rather than folded into the unchanged claim (raised in review on #2339). +/// +/// All of it is pure over an explicit except the one test +/// that exercises the ambient publish, which restores in a finally. +/// A concurrently-running test in another collection could observe a published snapshot for that instant; +/// the only effect would be APPENDED text on a resolution miss, which no assertion in the suite is sensitive +/// to (they all use StartsWith/Contains) — which is itself the reason the disclosure is additive. +/// +public sealed class DarlingPeerDisclosureTests +{ + private const string Use1Covers = "the 42 us-east-1 SQL Server primaries"; + + private static DarlingPeerDirectory.Snapshot TwoPeers() => + DarlingPeerDirectory.FromConfig(new PeersConfig + { + ThisStoreCovers = Use1Covers, + Stores = + { + new PeerStoreConfig + { + Name = "prod-pos-use2-monitor-01", + Covers = "the readable replicas of those same 42 primaries, from us-east-2", + Matches = { "use2" }, + }, + new PeerStoreConfig + { + Name = "prod-pos-pg-monitor-01", + Covers = "the Aurora PostgreSQL clusters", + /* No matches: a peer that declares none is still disclosed, just never singled out. */ + }, + }, + }); + + private static DarlingServerResolver.RegisteredServer Registered(string storageName, string? displayName = null) => + new(storageName.GetHashCode(StringComparison.Ordinal), storageName, displayName ?? storageName); + + /* ───────────────────────── the config block ───────────────────────── */ + + [Fact] + public void PeersBlock_ParsesFromDarlingJson() + { + var config = DarlingConfig.Parse(""" + { + "postgres": { "connectionString": "Host=localhost;Database=darling" }, + "servers": [ { "host": "SQL2022" } ], + "peers": { + "thisStoreCovers": "the 42 us-east-1 SQL Server primaries", + "stores": [ + { "name": "box2", "covers": "their readable replicas", "matches": ["use2", "-ro"] } + ] + } + } + """); + + Assert.Equal(Use1Covers, config.Peers.ThisStoreCovers); + var peer = Assert.Single(config.Peers.Stores); + Assert.Equal("box2", peer.Name); + Assert.Equal("their readable replicas", peer.Covers); + Assert.Equal(new[] { "use2", "-ro" }, peer.Matches); + Assert.DoesNotContain(config.Validate(), p => p.Contains("peer", StringComparison.OrdinalIgnoreCase)); + } + + [Fact] + public void PeersBlock_IsOptional_AndAbsenceDeclaresNothing() + { + /* No "peers" key at all — the shape every existing darling.json has. */ + var config = DarlingConfig.Parse(""" + { + "postgres": { "connectionString": "Host=localhost;Database=darling" }, + "servers": [ { "host": "SQL2022" } ] + } + """); + + Assert.Empty(config.Peers.Stores); + Assert.True(DarlingPeerDirectory.FromConfig(config.Peers).IsEmpty); + Assert.True(DarlingPeerDirectory.FromConfig(null).IsEmpty); + } + + [Fact] + public void SampleConfig_ShipsThePeersBlockDeclaringNothing() + { + /* The sample documents the block but must not declare fictional peers: a shipped declaration would + be a lie told to every agent that connects to a fresh install. */ + var samplePath = System.IO.Path.Combine(AppContext.BaseDirectory, "darling.sample.json"); + var config = DarlingConfig.Parse(System.IO.File.ReadAllText(samplePath)); + + Assert.NotNull(config.Peers); + Assert.Empty(config.Peers.Stores); + Assert.True(DarlingPeerDirectory.FromConfig(config.Peers).IsEmpty); + Assert.Empty(PeersConfig.Validate(config.Peers)); + } + + [Fact] + public void Validate_RequiresAPeerName() + { + /* A peer with only a description tells an agent "some other store has it" with nothing to point a + human at — no better off than the bare not-found this whole feature replaces. */ + var problems = PeersConfig.Validate(new PeersConfig + { + Stores = { new PeerStoreConfig { Covers = "the replicas" } }, + }); + + Assert.Contains(problems, p => p.Contains("name is required", StringComparison.Ordinal)); + } + + [Fact] + public void Validate_AllowsAPeerWithNoCoversSentence() + { + /* Half a disclosure still names an endpoint, so a name-only peer is legal (and renders as the name). */ + Assert.Empty(PeersConfig.Validate(new PeersConfig + { + Stores = { new PeerStoreConfig { Name = "box2" } }, + })); + } + + [Theory] + [InlineData("Host=box2;Password=hunter2")] + [InlineData("see its connectionString")] + [InlineData("Server=x;Integrated Security=true")] + public void Validate_RefusesCredentialShapedPeerText(string covers) + { + /* Everything in this block is sent verbatim to every connected MCP client, so failing OPEN here would + broadcast the secret — the one place where refusing to start is the proportionate response. */ + var inCovers = PeersConfig.Validate(new PeersConfig + { + Stores = { new PeerStoreConfig { Name = "box2", Covers = covers } }, + }); + Assert.Contains(inCovers, p => p.Contains("DISCLOSURE ONLY", StringComparison.Ordinal)); + + /* The same guard covers this store's own sentence and the match patterns — every disclosed string. */ + Assert.Contains( + PeersConfig.Validate(new PeersConfig { ThisStoreCovers = covers }), + p => p.Contains("DISCLOSURE ONLY", StringComparison.Ordinal)); + + Assert.Contains( + PeersConfig.Validate(new PeersConfig + { + Stores = { new PeerStoreConfig { Name = "box2", Matches = { covers } } }, + }), + p => p.Contains("DISCLOSURE ONLY", StringComparison.Ordinal)); + } + + [Fact] + public void Validate_ChecksThisStoreCovers_EvenWhenStoresIsExplicitJsonNull() + { + /* Review finding on #2339. System.Text.Json assigns null OVER the property initializer for an + explicit "stores": null (an omitted key leaves the default empty list), and the thisStoreCovers + guard used to sit after the per-peer loop behind an early `Stores is null` return — so this exact + config validated clean and then broadcast the credential through the instructions, + list_servers' this_store_covers, and every resolution miss. */ + var config = DarlingConfig.Parse(""" + { + "postgres": { "connectionString": "Host=localhost;Database=darling" }, + "servers": [ { "host": "SQL2022" } ], + "peers": { "thisStoreCovers": "internal db, see Password=hunter2", "stores": null } + } + """); + + Assert.Null(config.Peers.Stores); + Assert.Contains( + PeersConfig.Validate(config.Peers), + p => p.Contains("DISCLOSURE ONLY", StringComparison.Ordinal)); + Assert.Contains(config.Validate(), p => p.Contains("DISCLOSURE ONLY", StringComparison.Ordinal)); + + /* And nothing reaches the ambient snapshot, so no surface can disclose it. */ + try + { + var published = DarlingPeerDirectory.Publish(config.Peers); + Assert.True(published.Refused); + Assert.True(published.Snapshot.IsEmpty); + Assert.True(DarlingPeerDirectory.Current.IsEmpty); + } + finally + { + DarlingPeerDirectory.Reset(); + } + } + + [Fact] + public void Publish_RefusesTheWholeBlockOnAnyValidationProblem() + { + /* Review finding on #2339: the credential guard lived only in DarlingConfig.Validate, which the + WORKER runs — but the worker's abort is a return from its own hosted service, not a process exit, + and the MCP host loads its own config and deliberately never calls Validate. So the one path that + actually broadcasts peer text was the one path the guard never covered. Validating inside Publish + makes it structural: the ambient snapshot can only be written through here. */ + try + { + var leak = DarlingPeerDirectory.Publish(new PeersConfig + { + ThisStoreCovers = Use1Covers, + Stores = + { + new PeerStoreConfig { Name = "good", Covers = "the replicas", Matches = { "use2" } }, + new PeerStoreConfig { Name = "bad", Covers = "Host=x;Password=hunter2" }, + }, + }); + + Assert.True(leak.Refused); + Assert.Contains(leak.RefusedProblems, p => p.Contains("DISCLOSURE ONLY", StringComparison.Ordinal)); + + /* The WHOLE block, not the valid subset: a peers block that failed validation is one the + operator has not finished, and half a disclosure would state coverage that may be wrong while + the log says the config is broken. */ + Assert.True(leak.Snapshot.IsEmpty); + Assert.True(DarlingPeerDirectory.Current.IsEmpty); + + /* A nameless peer is not a secret, but it is still an unfinished block — same refusal. */ + Assert.True(DarlingPeerDirectory + .Publish(new PeersConfig { Stores = { new PeerStoreConfig { Covers = "the replicas" } } }) + .Refused); + Assert.True(DarlingPeerDirectory.Current.IsEmpty); + + /* A valid block publishes, and reports no problems. */ + var ok = DarlingPeerDirectory.Publish(new PeersConfig + { + ThisStoreCovers = Use1Covers, + Stores = { new PeerStoreConfig { Name = "box2", Covers = "the replicas" } }, + }); + + Assert.False(ok.Refused); + Assert.Empty(ok.RefusedProblems); + Assert.Same(ok.Snapshot, DarlingPeerDirectory.Current); + } + finally + { + DarlingPeerDirectory.Reset(); + } + } + + [Fact] + public void Validate_PeerProblemsSurfaceThroughTheWholeConfigValidate() + { + /* Reported even on a config with no servers — the peers check runs BEFORE that early return. */ + var config = new DarlingConfig + { + Postgres = new PostgresConfig { ConnectionString = "Host=localhost;Database=darling" }, + Peers = new PeersConfig { Stores = { new PeerStoreConfig { Covers = "the replicas" } } }, + }; + + var problems = config.Validate(); + Assert.Contains(problems, p => p.Contains("name is required", StringComparison.Ordinal)); + Assert.Contains(problems, p => p.Contains("servers must contain at least one entry", StringComparison.Ordinal)); + } + + /* ───────────────────────── normalization + matching ───────────────────────── */ + + [Fact] + public void FromConfig_TrimsAndDropsEmptyEntriesAndBlankPatterns() + { + var snapshot = DarlingPeerDirectory.FromConfig(new PeersConfig + { + ThisStoreCovers = " the primaries ", + Stores = + { + new PeerStoreConfig { Name = " box2 ", Covers = " the replicas ", Matches = { " use2 ", "", " " } }, + new PeerStoreConfig(), /* an empty object in the array is a typo, not a peer */ + }, + }); + + Assert.Equal("the primaries", snapshot.ThisStoreCovers); + var peer = Assert.Single(snapshot.Peers); + Assert.Equal("box2", peer.Name); + Assert.Equal("the replicas", peer.Covers); + + /* A blank pattern is a substring of EVERY name, so keeping one would make this peer claim the whole + fleet — the one normalization step that is a correctness fix rather than tidiness. */ + Assert.Equal(new[] { "use2" }, peer.Matches); + Assert.False(peer.CoversServerName("anything-at-all")); + Assert.True(peer.CoversServerName("prod-pos-USE2-apex-01")); + } + + [Fact] + public void CoversServerName_IsCaseInsensitiveSubstring_AndNeverTrueWithoutPatterns() + { + var snapshot = TwoPeers(); + var use2 = snapshot.Peers[0]; + var postgres = snapshot.Peers[1]; + + Assert.True(use2.CoversServerName("prod-pos-use2-ayr-01")); + Assert.True(use2.CoversServerName("PROD-POS-USE2-AYR-01")); + Assert.False(use2.CoversServerName("prod-pos-use1-ayr-01")); + Assert.False(use2.CoversServerName(null)); + Assert.False(use2.CoversServerName(" ")); + + /* No declared patterns means "cannot tell", which must never render as "yes". */ + Assert.False(postgres.CoversServerName("anything")); + + Assert.Equal(new[] { "prod-pos-use2-monitor-01" }, + snapshot.PeersCovering("prod-pos-use2-ayr-01").Select(p => p.Name)); + Assert.Empty(snapshot.PeersCovering("prod-pos-use1-ayr-01")); + } + + /* ───────────────────────── the MCP instructions ───────────────────────── */ + + [Fact] + public void Instructions_AreUnchangedWithNothingDeclared() + { + Assert.Equal("", DarlingPeerDirectory.InstructionsSection(DarlingPeerDirectory.Snapshot.Empty)); + Assert.Same(DarlingMcpInstructions.Text, DarlingMcpInstructions.Build(DarlingPeerDirectory.Snapshot.Empty)); + } + + [Fact] + public void Instructions_DiscloseThisStoreAndItsPeers_AboveTheToolCensus() + { + var text = DarlingMcpInstructions.Build(TwoPeers()); + + Assert.Contains(Use1Covers, text, StringComparison.Ordinal); + Assert.Contains("prod-pos-use2-monitor-01 — the readable replicas", text, StringComparison.Ordinal); + Assert.Contains("prod-pos-pg-monitor-01 — the Aurora PostgreSQL clusters", text, StringComparison.Ordinal); + + /* The point of the section is that a peer is NAMED, never contacted — say so where the agent will + read it, or it will try to route a query at the sibling. */ + Assert.Contains("NO cross-store connectivity", text, StringComparison.Ordinal); + + /* Placement is load-bearing: an agent must learn WHICH store it is talking to before it reads the + tool census and starts planning. */ + var readOnly = text.IndexOf("## CRITICAL: Read-Only Access", StringComparison.Ordinal); + var coverage = text.IndexOf("## Fleet Coverage", StringComparison.Ordinal); + var census = text.IndexOf("This server exposes", StringComparison.Ordinal); + Assert.True(readOnly >= 0 && coverage > readOnly && census > coverage, + $"the coverage section must sit between the read-only preamble and the tool census (read-only {readOnly}, coverage {coverage}, census {census})"); + + /* Inserting a section must not drop any of the body — the census sentence a cross-app test parses + (Lite.Tests/CrossAppMcpToolInventoryPinTests) lives in it, as does every tool table. */ + Assert.Contains("are unique to Darling", text, StringComparison.Ordinal); + Assert.EndsWith("mute a finding pattern the operator has accepted", text.TrimEnd(), StringComparison.Ordinal); + } + + [Fact] + public void Instructions_DiscloseCoverageEvenWithNoPeersDeclared() + { + /* "This store covers X" is worth saying on its own; the no-connectivity paragraph is not, because + there is nothing to warn about connecting to. */ + var text = DarlingPeerDirectory.InstructionsSection( + DarlingPeerDirectory.FromConfig(new PeersConfig { ThisStoreCovers = Use1Covers })); + + Assert.Contains(Use1Covers, text, StringComparison.Ordinal); + Assert.DoesNotContain("NO cross-store connectivity", text, StringComparison.Ordinal); + } + + /* ───────────────────────── list_servers ───────────────────────── */ + + private static JsonElement RenderedServerList(DarlingPeerDirectory.Snapshot peers) + { + var rows = new List + { + new(1, "prod-pos-use1-ayr-01", "ayr", 16, new DateTime(2026, 8, 19, 12, 0, 0, DateTimeKind.Utc)), + }; + + return JsonDocument + .Parse(DarlingMcpDataTools.RenderServerList(rows, new DateTime(2026, 8, 19, 12, 0, 30, DateTimeKind.Utc), peers)) + .RootElement; + } + + [Fact] + public void ListServers_CarriesThePeerFleetsSummary() + { + var root = RenderedServerList(TwoPeers()); + + Assert.Equal(1, root.GetProperty("server_count").GetInt32()); + Assert.Equal(Use1Covers, root.GetProperty("this_store_covers").GetString()); + + var fleets = root.GetProperty("peer_fleets").EnumerateArray().ToList(); + Assert.Equal(2, fleets.Count); + Assert.Equal("prod-pos-use2-monitor-01", fleets[0].GetProperty("name").GetString()); + Assert.Equal("the readable replicas of those same 42 primaries, from us-east-2", fleets[0].GetProperty("covers").GetString()); + Assert.Equal(new[] { "use2" }, fleets[0].GetProperty("matches").EnumerateArray().Select(m => m.GetString()).ToArray()); + Assert.Empty(fleets[1].GetProperty("matches").EnumerateArray()); + + Assert.Contains("cannot read a peer's data", root.GetProperty("peer_note").GetString(), StringComparison.Ordinal); + + /* The existing payload is untouched — the disclosure is additive here too. */ + var server = Assert.Single(root.GetProperty("servers").EnumerateArray()); + Assert.Equal("prod-pos-use1-ayr-01", server.GetProperty("server_name").GetString()); + Assert.Equal("ayr", server.GetProperty("display_name").GetString()); + } + + [Fact] + public void ListServers_EmptyPeerFleets_SaysWhatItDoesNotProve() + { + /* An empty peer list has two very different meanings — this is the only store, or the operator never + declared the siblings — and the service cannot tell them apart. Letting the empty array read as + "you are looking at the whole fleet" is exactly the mistake #2339 was filed about. */ + var root = RenderedServerList(DarlingPeerDirectory.Snapshot.Empty); + + Assert.Empty(root.GetProperty("peer_fleets").EnumerateArray()); + Assert.Equal(JsonValueKind.Null, root.GetProperty("this_store_covers").ValueKind); + + var note = root.GetProperty("peer_note").GetString(); + Assert.Contains("No peer stores are declared", note, StringComparison.Ordinal); + Assert.Contains("not proof that nobody monitors it", note, StringComparison.Ordinal); + } + + [Fact] + public void EmptyRegistry_StillDisclosesThePeers() + { + /* list_servers answers an empty registry with prose, not the JSON envelope — so that path carries the + disclosure explicitly, or it becomes the one place the declaration silently vanishes. And it is the + worst one to lose: an empty registry is a fresh or just-restarted box, where "no servers here" with + no mention of the siblings is the strongest version of the wrong conclusion. */ + var disclosure = DarlingPeerDirectory.EmptyRegistryDisclosure(TwoPeers()); + + Assert.Contains("one of SEVERAL monitoring this fleet", disclosure, StringComparison.Ordinal); + Assert.Contains("prod-pos-use2-monitor-01", disclosure, StringComparison.Ordinal); + Assert.Contains("prod-pos-pg-monitor-01", disclosure, StringComparison.Ordinal); + Assert.Contains($"This store covers: {Use1Covers}.", disclosure, StringComparison.Ordinal); + + /* Unchanged with nothing declared, like every other surface. */ + Assert.Equal("", DarlingPeerDirectory.EmptyRegistryDisclosure(DarlingPeerDirectory.Snapshot.Empty)); + } + + /* ───────────────────────── the resolution miss ───────────────────────── */ + + private const string MissWithoutPeers = + "Could not resolve server. Available servers:\nprod-pos-use1-ayr-01"; + + [Fact] + public void ResolutionMiss_IsByteForByteUnchangedWithNothingDeclared() + { + var (resolved, error) = DarlingServerResolver.ResolveOrError( + new[] { Registered("prod-pos-use1-ayr-01") }, + "prod-pos-use2-ayr-01", + DarlingPeerDirectory.Snapshot.Empty); + + Assert.Equal(default, resolved); + Assert.Equal(MissWithoutPeers, error); + } + + [Fact] + public void ResolutionMiss_NamesThePeerWhoseDeclaredCoverageMatches() + { + var (resolved, error) = DarlingServerResolver.ResolveOrError( + new[] { Registered("prod-pos-use1-ayr-01") }, + "prod-pos-use2-ayr-01", + TwoPeers()); + + Assert.Equal(default, resolved); + Assert.NotNull(error); + + /* The prefix and the local listing survive: 'Could not resolve server.' is what callers key off, and + the local list is still the right answer to the commonest miss (a typo). */ + Assert.StartsWith(MissWithoutPeers, error, StringComparison.Ordinal); + + Assert.Contains("'prod-pos-use2-ayr-01' is not monitored HERE", error, StringComparison.Ordinal); + Assert.Contains("matches the declared coverage of peer store prod-pos-use2-monitor-01", error, StringComparison.Ordinal); + Assert.Contains("That is a SEPARATE Darling store", error, StringComparison.Ordinal); + Assert.Contains("this server cannot read it", error, StringComparison.Ordinal); + Assert.Contains($"This store covers: {Use1Covers}.", error, StringComparison.Ordinal); + + /* The peer that declared no patterns must not be blamed for a name it never claimed. */ + Assert.DoesNotContain("prod-pos-pg-monitor-01", error, StringComparison.Ordinal); + } + + [Fact] + public void ResolutionMiss_WithNoMatchingPeer_ListsThemWithoutClaimingUnmonitored() + { + var (_, error) = DarlingServerResolver.ResolveOrError( + new[] { Registered("prod-pos-use1-ayr-01") }, + "some-other-box", + TwoPeers()); + + Assert.NotNull(error); + Assert.Contains("'some-other-box' is not monitored HERE", error, StringComparison.Ordinal); + Assert.Contains("matches no declared peer store's coverage either", error, StringComparison.Ordinal); + + /* Both peers are still disclosed: the declarations are prose plus optional patterns, not a live + registry, so "no pattern matched" is not evidence the server is unmonitored. */ + Assert.Contains("prod-pos-use2-monitor-01", error, StringComparison.Ordinal); + Assert.Contains("prod-pos-pg-monitor-01", error, StringComparison.Ordinal); + } + + [Fact] + public void ResolutionMiss_WithTwoPeersClaimingTheName_AgreesInNumber() + { + /* Two peers can legitimately both claim a name through overlapping `matches`, so the follow-on + sentence must not say "That is a SEPARATE store" about a list of two. */ + var overlapping = DarlingPeerDirectory.FromConfig(new PeersConfig + { + Stores = + { + new PeerStoreConfig { Name = "box2", Covers = "the replicas", Matches = { "use2" } }, + new PeerStoreConfig { Name = "box3", Covers = "the archive replicas", Matches = { "prod-pos" } }, + }, + }); + + var (_, error) = DarlingServerResolver.ResolveOrError( + new[] { Registered("prod-pos-use1-ayr-01") }, "prod-pos-use2-ayr-01", overlapping); + + Assert.NotNull(error); + Assert.Contains("these peer stores: box2 — the replicas; box3 — the archive replicas", error, StringComparison.Ordinal); + Assert.Contains("Those are SEPARATE Darling stores", error, StringComparison.Ordinal); + Assert.DoesNotContain("That is a SEPARATE Darling store", error, StringComparison.Ordinal); + } + + [Fact] + public void ResolutionMiss_WithNoNameGiven_DisclosesPeersWithoutAccusingOne() + { + /* The blank-name miss (several servers, no server_name passed) has no name to match, so the + disclosure must list rather than accuse. */ + var (_, error) = DarlingServerResolver.ResolveOrError( + new[] { Registered("prod-pos-use1-ayr-01"), Registered("prod-pos-use1-apex-01") }, + " ", + TwoPeers()); + + Assert.NotNull(error); + Assert.StartsWith("Could not resolve server.", error, StringComparison.Ordinal); + Assert.Contains("That server is not monitored HERE", error, StringComparison.Ordinal); + Assert.DoesNotContain("matches the declared coverage", error, StringComparison.Ordinal); + } + + [Fact] + public void ResolutionMiss_ReadsTheAmbientDeclaration_ThroughTheTwoArgOverload() + { + /* The ~90 tool methods all call the two-arg form, so the ambient publish is the seam that actually + delivers this to an MCP client. Reset in a finally: process-wide state must not leak between tests. */ + try + { + var published = DarlingPeerDirectory.Publish(new PeersConfig + { + ThisStoreCovers = Use1Covers, + Stores = { new PeerStoreConfig { Name = "box2", Covers = "the replicas", Matches = { "use2" } } }, + }); + + Assert.False(published.Refused); + Assert.False(published.Snapshot.IsEmpty); + Assert.Same(published.Snapshot, DarlingPeerDirectory.Current); + + var (_, error) = DarlingServerResolver.ResolveOrError( + new[] { Registered("prod-pos-use1-ayr-01") }, + "prod-pos-use2-ayr-01"); + + Assert.Contains("peer store box2 — the replicas", error, StringComparison.Ordinal); + } + finally + { + DarlingPeerDirectory.Reset(); + } + + Assert.True(DarlingPeerDirectory.Current.IsEmpty); + + var (_, afterReset) = DarlingServerResolver.ResolveOrError( + new[] { Registered("prod-pos-use1-ayr-01") }, + "prod-pos-use2-ayr-01"); + + Assert.Equal(MissWithoutPeers, afterReset); + } +} diff --git a/Darling/Darling.Tests/DarlingQueryTrendTieringTests.cs b/Darling/Darling.Tests/DarlingQueryTrendTieringTests.cs new file mode 100644 index 000000000..c0d35d25c --- /dev/null +++ b/Darling/Darling.Tests/DarlingQueryTrendTieringTests.cs @@ -0,0 +1,153 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Linq; +using PerformanceMonitor.Darling.Service.Mcp; +using PerformanceMonitor.Darling.Storage; +using Xunit; + +namespace Darling.Tests; + +/// +/// #2353: get_query_trend reads the tier that can actually serve the window, and says which one it read. +/// +/// The defect. The read went to raw query_stats only. The raw tier of a ROLLED table is +/// dropped at — four days — independently of the collector's +/// much longer advertised retention, so a request for 168 hours returned whatever had not aged out while +/// echoing hours_back: 168 back unchanged. Worse than the short array: the empty path asserted "No +/// history found ... within the last 168 hours", which for a query whose history had simply aged out is a +/// false statement rather than an incomplete one, and an agent acts on it by concluding the query never ran. +/// +public class DarlingQueryTrendTieringTests +{ + private static readonly DateTime Now = new(2026, 8, 19, 12, 0, 0, DateTimeKind.Utc); + + /// A window inside the raw horizon keeps per-collection resolution — the common case must not regress. + [Theory] + [InlineData(1)] + [InlineData(24)] + [InlineData(48)] + [InlineData(70)] + public void AWindowInsideTheRawHorizon_UsesRaw(int hoursBack) + { + Assert.True(DarlingTrendReader.ShouldUseRawTier(Now.AddHours(-hoursBack), Now)); + } + + /// + /// The reported case: 168 hours cannot come from a tier that keeps four days, so it must not be asked to. + /// + [Theory] + [InlineData(96)] + [InlineData(168)] + [InlineData(720)] + public void AWindowReachingPastTheRawHorizon_UsesTheAggregate(int hoursBack) + { + Assert.False(DarlingTrendReader.ShouldUseRawTier(Now.AddHours(-hoursBack), Now)); + } + + /// + /// The boundary itself, pinned rather than restated: the switch happens one margin INSIDE four days, so a + /// window landing exactly on the purge line takes the aggregate instead of depending on when the purge ran. + /// + [Fact] + public void TheBoundary_SitsOneMarginInsideTheRawHorizon() + { + var switchPoint = Now - TimescaleSupport.RawRetentionSpan + DarlingTrendReader.RawTierMargin; + + Assert.True(DarlingTrendReader.ShouldUseRawTier(switchPoint, Now)); + Assert.False(DarlingTrendReader.ShouldUseRawTier(switchPoint.AddSeconds(-1), Now)); + + /* And the margin really is inside the horizon, not outside it - a margin on the wrong side would send + windows to raw that raw has already dropped. */ + Assert.True(DarlingTrendReader.RawTierMargin > TimeSpan.Zero); + Assert.True(switchPoint > Now - TimescaleSupport.RawRetentionSpan); + } + + /// + /// Routing by the oldest point, never by width: a NARROW window sitting entirely in last week is exactly as + /// unservable from raw as a wide one. Routing by width would send it to a tier holding no rows for it, which + /// is the same silent-empty failure in a new place. + /// + [Fact] + public void ANarrowWindowInThePast_StillTakesTheAggregate() + { + var start = Now.AddDays(-10); + + /* Two hours wide and ten days old. Raw dropped these rows six days ago, so width must not rescue it. */ + Assert.False(DarlingTrendReader.ShouldUseRawTier(start, Now)); + + /* The regression this pins: measuring the start against the window's END instead of wall clock made + this window look "recent" - it IS recent relative to its own end - and routed it to a tier holding + nothing for it, which is the same silent-empty failure in a new place. Passing the window end here + would return true under that mistake; it must be false, because now is what retention answers to. */ + var windowEnd = start.AddHours(2); + Assert.True(windowEnd < Now - TimescaleSupport.RawRetentionSpan); + Assert.False(DarlingTrendReader.ShouldUseRawTier(start, Now)); + } + + /// + /// One mapper serves both queries, so the two projections must agree on ordinals. A column added to one and + /// not the other would mis-map silently — the reader would keep reading, just off by one. + /// + [Fact] + public void BothProjections_HaveTheSameShape() + { + static string[] Columns(string sql) => + sql[(sql.IndexOf("SELECT", StringComparison.Ordinal) + 6)..sql.IndexOf("FROM", StringComparison.Ordinal)] + .Split(',') + .Select(c => c.Trim()) + .Select(c => + { + var at = c.LastIndexOf(" AS ", StringComparison.OrdinalIgnoreCase); + return (at >= 0 ? c[(at + 4)..] : c).Trim(); + }) + .ToArray(); + + var raw = Columns(DarlingTrendReader.QueryHistorySql); + var hourly = Columns(DarlingTrendReader.QueryHistoryHourlySql); + + Assert.Equal(12, raw.Length); + Assert.Equal(raw.Length, hourly.Length); + Assert.Equal(raw, hourly); + } + + /// + /// The aggregate reads the rollup and buckets by it — asserted on the shipped SQL so a later edit that + /// quietly repoints it at the raw table (which would reintroduce the bug while every test still passed) + /// has to argue with this. + /// + [Fact] + public void TheAggregateQuery_ReadsTheRollup_AndBucketsByIt() + { + var sql = DarlingTrendReader.QueryHistoryHourlySql; + + Assert.Contains("FROM query_stats_hourly", sql, StringComparison.Ordinal); + Assert.Contains("bucket >=", sql, StringComparison.Ordinal); + Assert.Contains("ORDER BY bucket", sql, StringComparison.Ordinal); + + /* The columns the rollup does not carry are typed NULLs, never zeros: on an aggregate row a zero reads + as "none observed", which is a measurement nobody made. */ + Assert.Contains("CAST(NULL AS bigint) AS delta_logical_reads", sql, StringComparison.Ordinal); + Assert.Contains("CAST(NULL AS text) AS query_plan_hash", sql, StringComparison.Ordinal); + } + + /// + /// The raw query still reads the raw table, so the common path keeps per-collection resolution and every + /// column. Pinned alongside the above so the pair cannot drift into both reading the same place. + /// + [Fact] + public void TheRawQuery_StillReadsPerCollectionRows() + { + var sql = DarlingTrendReader.QueryHistorySql; + + Assert.Contains("FROM query_stats", sql, StringComparison.Ordinal); + Assert.DoesNotContain("query_stats_hourly", sql, StringComparison.Ordinal); + Assert.Contains("ORDER BY collection_time", sql, StringComparison.Ordinal); + } +} diff --git a/Darling/Darling.Tests/DarlingSelfAlertTests.cs b/Darling/Darling.Tests/DarlingSelfAlertTests.cs index 3c605a020..31ff2613e 100644 --- a/Darling/Darling.Tests/DarlingSelfAlertTests.cs +++ b/Darling/Darling.Tests/DarlingSelfAlertTests.cs @@ -79,6 +79,12 @@ private sealed class FakeSettings : IAlertEngineSettings public int CollectionFailureThreshold { get; set; } = 10; public int PvsThresholdPercent { get; set; } = 40; public int PvsFloorGb { get; set; } = 1; + + /* #2349: OFF in the fakes so existing expectations are untouched. */ + public bool FileGrowthEnabled { get; set; } + public int FileGrowthRiseMb { get; set; } = 10240; + public int FileGrowthVolumePercent { get; set; } = 60; + public int FileGrowthLookbackMinutes { get; set; } = 60; public int LongRunningJobMultiplier { get; set; } = 3; public int FailedJobLookbackMinutes { get; set; } = 60; public int CooldownMinutes { get; set; } = 5; @@ -2020,6 +2026,12 @@ public Task> GetLongRunningQueriesAsync( Task.FromResult(new List()); public Task> GetVolumeFreeSpaceAsync(string serverKey, CancellationToken cancellationToken = default) => Task.FromResult(new List()); + + /* #2349: empty on purpose. These tests exercise other alerts, and a fabricated file would + make the file-growth gate fire inside an unrelated scenario. */ + public Task> GetDatabaseFileGrowthAsync( + string serverKey, int lookbackMinutes, CancellationToken cancellationToken = default) => + Task.FromResult(new List()); public Task GetTempDbSpaceAsync(string serverKey, CancellationToken cancellationToken = default) => Task.FromResult(null); public Task> GetPvsPressureAsync(string serverKey, CancellationToken cancellationToken = default) => diff --git a/Darling/Darling.Tests/DarlingServiceInstallLocationTests.cs b/Darling/Darling.Tests/DarlingServiceInstallLocationTests.cs new file mode 100644 index 000000000..6072494d6 --- /dev/null +++ b/Darling/Darling.Tests/DarlingServiceInstallLocationTests.cs @@ -0,0 +1,501 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.IO; +using System.Text; +using Microsoft.Extensions.Logging; +using PerformanceMonitor.Darling.Service; +using Xunit; + +namespace Darling.Tests; + +/// +/// The SERVICE's own install-location diagnosis (#2185), as distinct from the installer's refusal, which +/// pins. +/// +/// Why the service needs its own. #2187 taught install-darling.ps1 to refuse a location +/// the service account cannot read, which covers every install that goes through the script — and none of the +/// ones that do not. The README's manual sc create path is documented, and registering the exe by hand +/// is exactly what someone does with a zip they just extracted. Those installs reached the reporter's +/// experience: an empty Output:, a bare loader status, then a missing pg-admin-credential.dpapi +/// and advice to start the service once, which they had. Nothing named the install directory. +/// +/// What is pinned here. The decision table both ways (a refused path must be named, and — the +/// half that decides whether an operator believes the first — a machine-scoped path must never be), the +/// message's three obligations, and the two silences: a console run, and a non-Windows host. The table is the +/// artifact: is shared with the cross-language parity test in +/// , so the two definitions of "a location that cannot work" cannot +/// drift apart without a red test. +/// +public sealed class DarlingServiceInstallLocationTests +{ + /// The machine profile root the table is written against — the ordinary one. + private const string ProfileRoot = @"C:\Users"; + + /// The installing operator's own profile, sitting inside the machine root as it normally does. + private const string UserProfile = @"C:\Users\installer"; + + /// + /// The decision table. Every row carries WHY, because a bare "expected UserProfile, got None" over + /// twenty-odd rows is not actionable, and half these rows exist to record a decision rather than a rule. + /// + internal static IReadOnlyList<(string InstallDirectory, InstallLocationVerdict Expected, string Because)> Cases => + [ + /* The reported shape (#2185), and the two ways the same folder gets written. */ + (@"C:\Users\username\Desktop\PerformanceMonitorDarling-3.2.0", InstallLocationVerdict.UserProfile, + "the reported install: a zip extracted to a Desktop"), + (@"C:\Users\username\Desktop\PerformanceMonitorDarling-3.2.0\", InstallLocationVerdict.UserProfile, + "a trailing separator is not a way out of the profile"), + (@"C:\Users\username\Downloads\PerformanceMonitorDarling", InstallLocationVerdict.UserProfile, + "Downloads is the other place a zip lands"), + + /* The profile root itself, and the ways a path can be spelled. An install root sitting AT the root is + exactly as unreadable as one below it. */ + (@"C:\Users", InstallLocationVerdict.UserProfile, "at the profile root, not under it"), + (@"C:\Users\", InstallLocationVerdict.UserProfile, "the profile root with a trailing separator"), + (@"c:\users\bob\x", InstallLocationVerdict.UserProfile, "Windows paths are case-insensitive"), + (@"C:\Users\bob\..\bob\x", InstallLocationVerdict.UserProfile, "a relative segment is not a way out"), + ("C:/Users/bob/x", InstallLocationVerdict.UserProfile, "a forward slash is not a way out"), + + /* Deliberate residual, shared with the installer and recorded rather than discovered later (#2187): + C:\Users\Public is under the profile root and is refused, even though an install there would + actually WORK - it grants NT AUTHORITY\SERVICE:(OI)(CI)(IO)(M,DC). Not carved out: nobody installs + a Windows service into the shared documents profile, a carve-out would amount to documenting it as + a reasonable place to install, and the refusal is not a dead end because it names the alternative. */ + (@"C:\Users\Public\PerformanceMonitorDarling", InstallLocationVerdict.UserProfile, + "C:\\Users\\Public is a deliberate refusal, not an oversight - one rule, no exceptions (#2187)"), + + /* The false-refusal cases. A bare StartsWith fails every one of these, and a false refusal is worse + than a missed one: it maligns an install that works. */ + (@"C:\UsersData\PerformanceMonitorDarling", InstallLocationVerdict.None, + "C:\\UsersData is not under C:\\Users - the boundary this table exists for"), + (@"C:\UsersData", InstallLocationVerdict.None, "the same boundary at the root itself"), + (@"C:\Users2\PerformanceMonitorDarling", InstallLocationVerdict.None, "nor is C:\\Users2"), + + /* The documented location and the other machine-scoped ones #2187 asked about explicitly. These are + the rows that decide whether the message is credible when it does appear. */ + (@"C:\PerformanceMonitorDarling", InstallLocationVerdict.None, "the documented install location"), + (@"C:\ProgramData\PerformanceMonitorDarling", InstallLocationVerdict.None, "machine-scoped"), + (@"C:\Program Files\PerformanceMonitorDarling", InstallLocationVerdict.None, "machine-scoped"), + (@"D:\PerformanceMonitorDarling", InstallLocationVerdict.None, "a second local volume is fine"), + + /* The network half. The reason differs from the profile case - a virtual account reaches the network + as the COMPUTER account, and a mapped letter belongs to one logon session - so the verdicts are + separate values rather than one "bad location". */ + (@"\\fileserver\share\PerformanceMonitorDarling", InstallLocationVerdict.UncPath, "a UNC share"), + (@"\\fileserver\share", InstallLocationVerdict.UncPath, "a UNC share root"), + (@"Z:\PerformanceMonitorDarling", InstallLocationVerdict.MappedDrive, "Z: is mapped to a share"), + (@"C:\Tools\PerformanceMonitorDarling", InstallLocationVerdict.None, "C: is a local volume, so not a mapped drive"), + + /* \\?\ is the long-path prefix on a LOCAL path, not a server name. Treating it as a share would + strand an ordinary install root written the extended-length way. This row is why the fix below + had to be a normalization rather than simply deleting the exclusion. */ + (@"\\?\C:\PerformanceMonitorDarling", InstallLocationVerdict.None, + "the extended-length prefix on a local path is not a UNC share"), + + /* #2348, fixed: both implementations now strip the extended-length prefix BEFORE classifying, so the + long spelling of a bad location gets the same verdict as the short one. These were pinned as known + residuals with a None verdict until the pair could move together; they now assert the fix, and they + still hold both sides to it - change one implementation and the cross-language parity test goes red. + + The lowercase and mixed-case UNC rows are not padding: Windows accepts \\?\unc\, so an + Ordinal match on 'UNC' would re-open exactly this hole for anyone who typed it that way. */ + (@"\\?\C:\Users\bob\PerformanceMonitorDarling", InstallLocationVerdict.UserProfile, + "the extended-length spelling of a profile path is still a profile path"), + (@"\\?\UNC\fileserver\share\PerformanceMonitorDarling", InstallLocationVerdict.UncPath, + "\\\\?\\UNC\\ is the extended-length spelling of a real share"), + (@"\\?\unc\fileserver\share\PerformanceMonitorDarling", InstallLocationVerdict.UncPath, + "Windows accepts the lowercase \\\\?\\unc\\ too, so the prefix match is case-insensitive"), + (@"\\?\UNC\fileserver\share", InstallLocationVerdict.UncPath, + "the extended-length spelling of a share ROOT is a share too"), + (@"\\?\Z:\PerformanceMonitorDarling", InstallLocationVerdict.MappedDrive, + "stripping the prefix exposes the drive letter, so a mapped drive is caught in either spelling"), + + /* Nothing is not somewhere. An empty path must never become a relative one resolved against whatever + the working directory happens to be. */ + ("", InstallLocationVerdict.None, "no path is not a bad path"), + (" ", InstallLocationVerdict.None, "nor is whitespace"), + ]; + + /// The drives the table's isNetworkDrive probe answers yes for. + private static bool IsMappedInTheTable(string qualifier) => + string.Equals(qualifier, "Z:", StringComparison.OrdinalIgnoreCase); + + /// + /// The whole decision, driven by the table. Runs every row and reports ALL disagreements at once: a + /// classifier is a table, and finding out about one wrong row per CI round is how a table gets fixed + /// wrong. + /// + [Fact] + public void Classify_DecidesEveryRowOfTheTable() + { + var wrong = new List(); + + foreach (var (directory, expected, because) in Cases) + { + var actual = DarlingInstallLocation.Classify(directory, ProfileRoot, UserProfile, IsMappedInTheTable); + if (actual != expected) + { + wrong.Add($"'{directory}' => {actual}, expected {expected} ({because})"); + } + } + + Assert.True(wrong.Count == 0, "the service's install-location table is wrong:\n " + string.Join("\n ", wrong)); + } + + /// + /// The profile root is whatever Windows says it is, not C:\Users. It is relocatable, and a literal + /// would quietly stop matching on precisely the box that moved it — the one box where a missed check costs + /// the most. + /// + [Fact] + public void Classify_HonoursARelocatedProfileRoot() + { + Assert.Equal( + InstallLocationVerdict.UserProfile, + DarlingInstallLocation.Classify(@"D:\Profiles\bob\PerformanceMonitorDarling", @"D:\Profiles", UserProfile, IsMappedInTheTable)); + + /* And the OLD location stops being special once the root has moved, which is the half that proves the + value is actually being read rather than checked alongside a hardcoded literal. */ + Assert.Equal( + InstallLocationVerdict.None, + DarlingInstallLocation.Classify(@"C:\Users\bob\PerformanceMonitorDarling", @"D:\Profiles", @"D:\Profiles\installer", IsMappedInTheTable)); + } + + /// + /// A profile redirected outside ProfilesDirectory is still a profile, so the current identity's own + /// profile is checked as well as the machine root — the same second arm the installer applies. + /// + [Fact] + public void Classify_AlsoChecksTheCurrentIdentitysOwnProfile() + { + Assert.Equal( + InstallLocationVerdict.UserProfile, + DarlingInstallLocation.Classify(@"E:\redirected\bob\PerformanceMonitorDarling", ProfileRoot, @"E:\redirected\bob", IsMappedInTheTable)); + } + + /// + /// The mapped-drive probe is only consulted for a path that HAS a drive qualifier, and an unknown drive is + /// not a refusal. A refusal needs evidence: the cost of missing one mapped drive is a message that does not + /// appear, while the cost of inventing one is telling an operator their working install is broken. + /// + [Fact] + public void Classify_NeverInventsAMappedDrive() + { + var asked = new List(); + bool Probe(string qualifier) + { + asked.Add(qualifier); + return false; + } + + Assert.Equal( + InstallLocationVerdict.None, + DarlingInstallLocation.Classify(@"Z:\PerformanceMonitorDarling", ProfileRoot, UserProfile, Probe)); + Assert.Equal(new[] { "Z:" }, asked); + + /* A UNC path is decided lexically and must not spend a drive probe at all - there is no letter to ask + about, and Split-Path -Qualifier's C# equivalent would have to invent one. */ + asked.Clear(); + Assert.Equal( + InstallLocationVerdict.UncPath, + DarlingInstallLocation.Classify(@"\\fileserver\share\x", ProfileRoot, UserProfile, Probe)); + Assert.Empty(asked); + } + + /// + /// The registry-unreadable fallback root must be a ROOTED path, not a drive-relative one. + /// + /// The bug this pins. The fallback was Path.Combine(systemDrive, "Users"), and + /// %SystemDrive% is documented to be a bare C: — which Path.Combine treats like a + /// trailing separator and concatenates, producing C:Users. + /// then resolves that against the process's current directory on the volume instead of the volume root, so + /// the profile check silently stopped matching on precisely the box the fallback exists for: one whose + /// ProfileList cannot be read. The registry read succeeds on every box CI runs on, which is exactly + /// why the branch had to be made reachable to be pinned at all (review catch on #2185). + /// + /// Asserted against the separator rather than a literal C:\Users, so the row that matters — + /// "the drive and the folder are not merely concatenated" — is the one being checked. + /// + [Fact] + public void ProfileRootForSystemDrive_IsRooted_NotDriveRelative() + { + var separator = Path.DirectorySeparatorChar; + + foreach (var (drive, expectedDrive, because) in new[] + { + ("C:", "C:", "the documented bare form, and the one that produced C:Users"), + (@"C:\", "C:", "a drive that already carries a separator must not double it"), + ("D:", "D:", "a box booted from another volume"), + ("", "C:", "an unset %SystemDrive% falls back to C:"), + (" ", "C:", "and so does a blank one"), + (null, "C:", "and so does a missing one"), + }) + { + var actual = DarlingInstallLocation.ProfileRootForSystemDrive(drive); + + Assert.Equal(expectedDrive + separator + "Users", actual); + Assert.DoesNotContain(expectedDrive + "Users", actual, StringComparison.Ordinal); + Assert.True(actual.Length > 0, because); + } + + /* And the shipped lookup's answer is rooted whichever branch it took - the registry read on a healthy + box, or the fallback on a locked-down one. A drive-relative answer here is the defect above. */ + Assert.True(Path.IsPathFullyQualified(DarlingInstallLocation.MachineProfileRoot()), + "MachineProfileRoot must return a fully-qualified path, or IsAtOrUnder resolves it against the " + + "process's current directory and the profile check stops matching (#2185)"); + } + + /// + /// The message's three obligations, one per verdict: name the offending PATH, say WHY this account cannot + /// read it, and say WHAT TO DO. It also has to name the downstream messages it displaces — those are what + /// the operator has already been chasing by the time they read this, and #2185 took four exchanges + /// precisely because nothing connected them. + /// + /* A Fact over the three verdicts rather than a Theory: xUnit needs a public signature, and the verdict + enum is internal to the service assembly. Every verdict is checked in one run either way. */ + [Fact] + public void Describe_NamesThePathTheReasonAndTheRemedy() + { + const string Directory = @"C:\Users\username\Desktop\PerformanceMonitorDarling-3.2.0"; + const string Account = @"NT SERVICE\PerformanceMonitor Darling"; + + foreach (var verdict in new[] + { + InstallLocationVerdict.UserProfile, + InstallLocationVerdict.UncPath, + InstallLocationVerdict.MappedDrive, + }) + { + var message = DarlingInstallLocation.Describe(verdict, Directory, Account); + + /* The path, exactly as the service knows it - an operator has to be able to match it against their + own log's content-root line without interpreting anything. */ + Assert.Contains(Directory, message, StringComparison.Ordinal); + + /* The account, resolved rather than assumed: an operator who re-homed the service to a domain + account or gMSA (#1802/#1823) must not be told to reason about a virtual account they do not + use. */ + Assert.Contains(Account, message, StringComparison.Ordinal); + + /* The remedy, with a destination. A refusal that does not name where to go instead is where + #2185's reporter already was. */ + Assert.Contains(DarlingInstallLocation.DocumentedInstallDirectory, message, StringComparison.Ordinal); + Assert.Contains("install-darling.ps1", message, StringComparison.Ordinal); + + /* And the connection to what they have already seen: the loader status (#2186) and the credential + message (#2197) are both downstream of this, and saying so is the whole point of saying it + early. */ + Assert.Contains("0xC0000135", message, StringComparison.Ordinal); + Assert.Contains("pg-admin-credential.dpapi", message, StringComparison.Ordinal); + Assert.Contains("#2185", message, StringComparison.Ordinal); + + /* BOTH store modes, because the message is built before darling.json is read and cannot know which + one is configured. The initdb narrative is the managed default's; an operator on + bring-your-own Postgres has no initdb to fail, and sending them to look for one would spend the + credibility this message exists to have (review catch on #2185). */ + Assert.Contains("postgres.managed = true", message, StringComparison.Ordinal); + Assert.Contains("postgres.managed = false", message, StringComparison.Ordinal); + Assert.Contains("Cannot load configuration", message, StringComparison.Ordinal); + } + } + + /// + /// The three reasons are actually different sentences. A single "bad location" message would be a + /// half-diagnosis: a profile fails on ACL inheritance, a share fails because a virtual account reaches + /// the network as the computer account, and a mapped letter fails because it belongs to a logon session a + /// service never joins. An operator who is told the wrong one goes and checks the wrong thing. + /// + [Fact] + public void Describe_GivesEachVerdictItsOwnReason() + { + var profile = DarlingInstallLocation.Describe(InstallLocationVerdict.UserProfile, @"C:\Users\bob\x", "ACCOUNT"); + var unc = DarlingInstallLocation.Describe(InstallLocationVerdict.UncPath, @"\\fs\share\x", "ACCOUNT"); + var mapped = DarlingInstallLocation.Describe(InstallLocationVerdict.MappedDrive, @"Z:\x", "ACCOUNT"); + + Assert.Contains("under a user profile", profile, StringComparison.Ordinal); + Assert.Contains("BUILTIN\\Users", profile, StringComparison.Ordinal); + + Assert.Contains("UNC network path", unc, StringComparison.Ordinal); + Assert.Contains("COMPUTER account", unc, StringComparison.Ordinal); + + Assert.Contains("mapped network drive", mapped, StringComparison.Ordinal); + Assert.Contains("logon session", mapped, StringComparison.Ordinal); + + Assert.NotEqual(profile, unc); + Assert.NotEqual(unc, mapped); + } + + /// + /// A refused location produces exactly ONE critical line. The issue's ask was one clear actionable + /// message, and a diagnosis split across four lines is how the useful one gets quoted without the others. + /// + [Fact] + public void Report_SaysItOnce_Critically() + { + var log = new CountingLogger(); + + DarlingInstallLocation.Report( + Path.Combine(DarlingInstallLocation.MachineProfileRoot(), "username", "Desktop", "PerformanceMonitorDarling-3.2.0"), + runningAsWindowsService: true, + log); + + Assert.Equal(1, log.Count(LogLevel.Critical)); + Assert.Equal(1, log.Total); + Assert.Contains("under a user profile", log.ToString(), StringComparison.Ordinal); + } + + /// + /// The documented install location says nothing at all. The most important property in the file: a + /// warning every healthy install prints is a warning nobody reads, and this one has to be believed on the + /// day it appears. + /// + [Fact] + public void Report_IsSilentForAMachineScopedInstall() + { + var log = new CountingLogger(); + + DarlingInstallLocation.Report(DarlingInstallLocation.DocumentedInstallDirectory, runningAsWindowsService: true, log); + + Assert.Equal(0, log.Total); + } + + /// + /// A console run says nothing either, even from the worst possible folder. Running the exe interactively + /// is something the README suggests, an interactive run IS the profile owner, and the tree it cannot read + /// as a service is perfectly readable as them — so this is the one false positive available to the check, + /// aimed at someone doing exactly what the docs told them to. + /// + [Fact] + public void Report_IsSilentOnAConsoleRun() + { + var log = new CountingLogger(); + + DarlingInstallLocation.Report( + Path.Combine(DarlingInstallLocation.MachineProfileRoot(), "username", "Desktop", "PerformanceMonitorDarling-3.2.0"), + runningAsWindowsService: false, + log); + + Assert.Equal(0, log.Total); + } + + /// + /// The worker's call site is platform-gated, so a Linux or container host (the compose deployment, which + /// runs bring-your-own Postgres and no virtual service account at all) can never reach this check. Pinned + /// on the SOURCE because the assertion is about the call site rather than the classifier: the test process + /// is Windows by construction, so nothing runnable here can observe the gate. + /// + [Fact] + public void TheWorker_CallsThisOnlyOnWindows_AndBeforeAnythingElse() + { + var worker = ReadRepoFile(Path.Combine( + "Darling", "PerformanceMonitor.Darling.Service", "DarlingWorker.cs")); + + var executeAsync = worker.IndexOf( + "protected override async Task ExecuteAsync(CancellationToken stoppingToken)", StringComparison.Ordinal); + Assert.True(executeAsync >= 0, "DarlingWorker.ExecuteAsync moved"); + + var report = worker.IndexOf("DarlingInstallLocation.Report(", executeAsync, StringComparison.Ordinal); + Assert.True(report >= 0, "DarlingWorker no longer diagnoses the install location (#2185)"); + + /* The platform gate, immediately above the call. */ + var gate = worker.LastIndexOf("if (OperatingSystem.IsWindows())", report, StringComparison.Ordinal); + Assert.True(gate > executeAsync, "the install-location report must sit inside a Windows guard (#2185)"); + + /* And it runs before anything that can fail in a way this explains. Config load first: an unreadable + tree takes darling.json out too, and "Cannot load configuration" is another message that never + names the install directory. */ + foreach (var (marker, what) in new[] + { + ("config = DarlingConfig.Load();", "loading darling.json"), + ("new DarlingManagedPostgres(config.Postgres, _logger)", "constructing the managed-Postgres bootstrap"), + ("await managedPostgres.EnsureRunningAsync(stoppingToken)", "the managed-Postgres bootstrap"), + }) + { + var at = worker.IndexOf(marker, executeAsync, StringComparison.Ordinal); + Assert.True(at >= 0, $"DarlingWorker.ExecuteAsync no longer contains {what} ('{marker}')"); + Assert.True(report < at, + $"the install-location diagnosis must run BEFORE {what}, or the operator reads the downstream failure first (#2185)"); + } + } + + /// Walks up from the test output directory to the repo root — the same idiom + /// uses. + private static string ReadRepoFile(string relativePath) + { + var directory = new DirectoryInfo(AppContext.BaseDirectory); + for (var i = 0; i < 10 && directory is not null; i++) + { + var candidate = Path.Combine(directory.FullName, relativePath); + if (File.Exists(Path.Combine(directory.FullName, "PerformanceMonitor.sln")) && File.Exists(candidate)) + { + return File.ReadAllText(candidate); + } + + directory = directory.Parent; + } + + Assert.Fail($"could not find {relativePath} above {AppContext.BaseDirectory}"); + return string.Empty; + } + + /// Counts lines per level as well as capturing them: "exactly one critical" is the assertion, and + /// a text-only capture cannot make it. + private sealed class CountingLogger : ILogger + { + private readonly List<(LogLevel Level, string Message)> _lines = new(); + + public int Total + { + get { lock (_lines) { return _lines.Count; } } + } + + public int Count(LogLevel level) + { + lock (_lines) + { + var n = 0; + foreach (var (l, _) in _lines) + { + if (l == level) { n++; } + } + + return n; + } + } + + public IDisposable? BeginScope(TState state) where TState : notnull => null; + + public bool IsEnabled(LogLevel logLevel) => true; + + public void Log( + LogLevel logLevel, EventId eventId, TState state, Exception? exception, Func formatter) + { + lock (_lines) + { + _lines.Add((logLevel, formatter(state, exception))); + } + } + + public override string ToString() + { + lock (_lines) + { + var builder = new StringBuilder(); + foreach (var (level, message) in _lines) + { + builder.Append('[').Append(level).Append("] ").AppendLine(message); + } + + return builder.ToString(); + } + } + } +} diff --git a/Darling/Darling.Tests/FileGrowthAlertStoreTests.cs b/Darling/Darling.Tests/FileGrowthAlertStoreTests.cs new file mode 100644 index 000000000..710358751 --- /dev/null +++ b/Darling/Darling.Tests/FileGrowthAlertStoreTests.cs @@ -0,0 +1,158 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Linq; +using PerformanceMonitor.Darling.Storage; +using PerformanceMonitor.Darling.Viewer; +using Xunit; + +namespace Darling.Tests; + +/// +/// The V79 rung (#2349) — the database file-growth alert's settings, and the top of the ladder. +/// +public class FileGrowthAlertStoreTests +{ + [Fact] + public void TheRungIsRegisteredAndIsTheTopOfADenseLadder() + { + var versions = PgMigrations.Scripts.Select(s => s.Version).ToList(); + + Assert.Equal("file-growth-alert", PgMigrations.Scripts.Single(s => s.Version == 79).Name); + Assert.Equal(79, versions.Max()); + Assert.Equal(79, StorageVersion.SchemaVersion); + Assert.Equal(StorageVersion.SchemaVersion, versions.Max()); + + Assert.Equal(versions.Distinct().OrderBy(v => v), versions); + var above = versions.Where(v => v > 45).OrderBy(v => v).ToList(); + Assert.Equal(Enumerable.Range(above[0], above.Count), above); + } + + /// + /// The alert ships OFF. A new alert that starts firing on upgrade is a bad citizen — the operator did not + /// ask for it, and the right thresholds are a property of their fleet rather than of the product. + /// + [Fact] + public void TheRungAddsTheKnobs_Idempotently_AndTheAlertShipsOff() + { + var sql = PgMigrations.Scripts.Single(s => s.Version == 79).Sql; + + Assert.Contains("ALTER TABLE config.config_alert_settings", sql, StringComparison.Ordinal); + Assert.Contains("ADD COLUMN IF NOT EXISTS file_growth_enabled boolean NOT NULL DEFAULT false", sql, StringComparison.Ordinal); + Assert.Contains("file_growth_rise_mb", sql, StringComparison.Ordinal); + Assert.Contains("file_growth_volume_percent", sql, StringComparison.Ordinal); + Assert.Contains("file_growth_lookback_minutes", sql, StringComparison.Ordinal); + } + + [Fact] + public void TheProbeAsksForTheColumn_AndTheThreePlacesAgree() + { + Assert.Contains( + "table_name = 'config_alert_settings' AND column_name = 'file_growth_enabled'", + ViewerDataService.StoreSchemaProbeSql, StringComparison.Ordinal); + + var mapParameters = typeof(ViewerDataService) + .GetMethod("MapProbedSchemaVersion", System.Reflection.BindingFlags.NonPublic | System.Reflection.BindingFlags.Static)! + .GetParameters().Length; + + var viewerSource = ReadViewerSource(); + + Assert.Contains($"reader.GetBoolean({mapParameters - 1})", viewerSource, StringComparison.Ordinal); + Assert.DoesNotContain($"reader.GetBoolean({mapParameters})", viewerSource, StringComparison.Ordinal); + } + + [Fact] + public void TheProbeMapsAFullyMigratedStoreTo79() + { + Assert.Equal(79, StorageVersion.SchemaVersion); + Assert.Equal(StorageVersion.SchemaVersion, ViewerDataService.RequiredStoreSchemaVersion); + + var method = typeof(ViewerDataService) + .GetMethod("MapProbedSchemaVersion", System.Reflection.BindingFlags.NonPublic | System.Reflection.BindingFlags.Static)!; + var arity = method.GetParameters().Length; + + /* 54 positional sentinels, then this rung's own, then FALSE for anything a later rung appends. The + leading count is FIXED at this rung's ordinal deliberately: deriving it from arity (`arity - 1`) + reads identically while this is the top rung, then slides the flag one place right per new rung — + the assertion keeps passing while quietly testing a newer arm. */ + var all = Enumerable.Repeat(true, 54).Cast().ToArray(); + object[] Args(bool ownFlag) => all + .Concat(new object[] { ownFlag }) + .Concat(Enumerable.Repeat((object)false, arity - 55)) + .ToArray(); + + Assert.Equal(79, (int)method.Invoke(null, Args(true))!); + Assert.Equal(78, (int)method.Invoke(null, Args(false))!); + } + + /// + /// The read must select and map every knob. ApplyToConfig replaces config.Alerts wholesale, + /// so a column selected but not mapped — or mapped but not selected — silently RESETS the operator's + /// setting on every worker start rather than failing, which is the failure mode the comments on every + /// appended knob above it warn about. + /// + [Fact] + public void TheStoreReadSelectsAndMapsEveryKnob() + { + var source = ReadRepoFile("Darling", "PerformanceMonitor.Darling.Service", "StoreConfigProvider.cs"); + + foreach (var column in new[] + { + "file_growth_enabled", "file_growth_rise_mb", + "file_growth_volume_percent", "file_growth_lookback_minutes", + }) + { + Assert.Contains(column, source, StringComparison.Ordinal); + } + + Assert.Contains("FileGrowthEnabled = reader.GetBoolean(54)", source, StringComparison.Ordinal); + Assert.Contains("FileGrowthRiseMb = reader.GetInt32(55)", source, StringComparison.Ordinal); + Assert.Contains("FileGrowthVolumePercent = reader.GetInt32(56)", source, StringComparison.Ordinal); + Assert.Contains("FileGrowthLookbackMinutes = reader.GetInt32(57)", source, StringComparison.Ordinal); + } + + /// + /// Both SKUs read the same shape. The Postgres side uses DISTINCT ON and DuckDB has none, so Lite + /// expresses the same selection with ROW_NUMBER() — different SQL, same rule, and the alert must not + /// behave differently depending on which product an operator bought. + /// + [Fact] + public void BothSkusReadTheSameShape() + { + var darling = ReadRepoFile("Darling", "PerformanceMonitor.Darling.Service", "DarlingAlertReadAdapter.cs"); + var lite = ReadRepoFile("Lite", "Services", "LocalDataService.FileGrowth.cs"); + + Assert.Contains("DISTINCT ON (database_name, file_name)", darling, StringComparison.Ordinal); + Assert.Contains("ROW_NUMBER() OVER (PARTITION BY database_name, file_name", lite, StringComparison.Ordinal); + + /* Both bound the window on collection_time, the partitioning column, so the read prunes rather than + scanning retention -- and both compute growth against a baseline from inside that window. */ + Assert.Contains("collection_time >= $2", darling, StringComparison.Ordinal); + Assert.Contains("collection_time >= $2", lite, StringComparison.Ordinal); + Assert.Contains("growth_window_minutes", darling, StringComparison.Ordinal); + Assert.Contains("growth_window_minutes", lite, StringComparison.Ordinal); + } + + private static string ReadViewerSource() => + ReadRepoFile("Darling", "PerformanceMonitor.Darling.Viewer", "ViewerDataService.cs"); + + private static string ReadRepoFile(params string[] parts) + { + var relative = System.IO.Path.Combine(parts); + var dir = System.IO.Path.GetDirectoryName(ThisFile())!; + while (dir is not null && !System.IO.File.Exists(System.IO.Path.Combine(dir, relative))) + { + dir = System.IO.Path.GetDirectoryName(dir); + } + + return System.IO.File.ReadAllText(System.IO.Path.Combine(dir!, relative)); + } + + private static string ThisFile([System.Runtime.CompilerServices.CallerFilePath] string path = "") => path; +} diff --git a/Darling/Darling.Tests/JsonAssert.cs b/Darling/Darling.Tests/JsonAssert.cs new file mode 100644 index 000000000..77edb8fd9 --- /dev/null +++ b/Darling/Darling.Tests/JsonAssert.cs @@ -0,0 +1,100 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Text; +using Xunit; + +namespace Darling.Tests; + +/// +/// Substring assertions over serialized JSON that ignore LAYOUT (#2350). +/// +/// A test written as Assert.Contains("\"severity\": \"Critical\"", json) reads as a claim about +/// content — this field serialized with this value — but is actually a claim about formatting, because the space +/// after the colon exists only under WriteIndented. When MCP tool results went compact, eighteen such +/// assertions failed across four files without a single one of the things they were testing having changed. +/// +/// These helpers normalize both sides by dropping whitespace that sits BETWEEN tokens while preserving +/// whitespace INSIDE strings, so "a": "b c" and "a":"b c" compare equal and the two-space value in +/// "b c" survives. The assertion then means what it always looked like it meant. +/// +/// Deliberately not a full JSON parse: these are substring assertions on purpose — they check a field +/// serialized a particular way (an enum as its string name rather than its ordinal, a null that stayed null) +/// without pinning the shape of the whole envelope around it. +/// +internal static class JsonAssert +{ + /// xUnit's argument order (expected first) so call sites read the same as the assertion they replace. + internal static void Contains(string expectedFragment, string json) + { + Assert.Contains(StripInsignificantWhitespace(expectedFragment), StripInsignificantWhitespace(json), StringComparison.Ordinal); + } + + /// + internal static void DoesNotContain(string unexpectedFragment, string json) + { + Assert.DoesNotContain(StripInsignificantWhitespace(unexpectedFragment), StripInsignificantWhitespace(json), StringComparison.Ordinal); + } + + /// + /// Removes whitespace outside string literals. Tracks escaping so a \" inside a string does not end it + /// and a \\ before a quote does not escape it — get that wrong and the parser falls out of the string, + /// starts stripping real spaces from values, and the assertion silently starts comparing something else. + /// + internal static string StripInsignificantWhitespace(string json) + { + if (string.IsNullOrEmpty(json)) + { + return json ?? string.Empty; + } + + var builder = new StringBuilder(json.Length); + var inString = false; + var escaped = false; + + foreach (var c in json) + { + if (inString) + { + builder.Append(c); + + if (escaped) + { + escaped = false; + } + else if (c == '\\') + { + escaped = true; + } + else if (c == '"') + { + inString = false; + } + + continue; + } + + if (c == '"') + { + inString = true; + builder.Append(c); + continue; + } + + if (c is ' ' or '\t' or '\r' or '\n') + { + continue; + } + + builder.Append(c); + } + + return builder.ToString(); + } +} diff --git a/Darling/Darling.Tests/PlanContentRetentionTests.cs b/Darling/Darling.Tests/PlanContentRetentionTests.cs index d30342feb..1def88be2 100644 --- a/Darling/Darling.Tests/PlanContentRetentionTests.cs +++ b/Darling/Darling.Tests/PlanContentRetentionTests.cs @@ -264,9 +264,16 @@ private static int InvokeMap(object[] leading, bool hasPlanContentRetentionKnob) var method = typeof(ViewerDataService) .GetMethod("MapProbedSchemaVersion", System.Reflection.BindingFlags.NonPublic | System.Reflection.BindingFlags.Static)!; - /* #2319 appended hasQueryStoreHealth after this rung's parameter — pass it FALSE so these - facts keep exercising the V75/V74 arms rather than the newer one. */ - var args = leading.Concat(new object[] { hasPlanContentRetentionKnob, false }).ToArray(); + /* #2319 appended hasQueryStoreHealth and #2312 appended hasQueryStoreTextHash after this rung's + parameter — pass both FALSE so these facts keep exercising the V75/V74 arms rather than the + newer ones. */ + /* Parameters appended by LATER rungs are padded FALSE, so this fact keeps exercising its own + arm rather than a newer one. Derived from the method's arity rather than listed by hand, so + a future rung does not have to edit this file -- #2357 (V78) was the fourth that would have. */ + var args = leading.Concat(new object[] { hasPlanContentRetentionKnob }).ToArray(); + args = args + .Concat(Enumerable.Repeat((object)false, method.GetParameters().Length - args.Length)) + .ToArray(); Assert.Equal(method.GetParameters().Length, args.Length); return (int)method.Invoke(null, args)!; diff --git a/Darling/Darling.Tests/PostgresTargetConfigTests.cs b/Darling/Darling.Tests/PostgresTargetConfigTests.cs index 3e25b03f6..c2fcf5e71 100644 --- a/Darling/Darling.Tests/PostgresTargetConfigTests.cs +++ b/Darling/Darling.Tests/PostgresTargetConfigTests.cs @@ -203,14 +203,37 @@ public void PostgresDetectionQueryUsesPortableSurfaces() Assert.Contains("server_version_num", sql, StringComparison.Ordinal); Assert.Contains("pg_is_in_recovery()", sql, StringComparison.Ordinal); - Assert.Contains("aurora_version", sql, StringComparison.Ordinal); - // Aurora detection must not hard-fail on stock PostgreSQL, so it is a pg_proc lookup. - Assert.Contains("pg_proc", sql, StringComparison.Ordinal); // No T-SQL leaked into the Postgres path. Assert.DoesNotContain("SERVERPROPERTY", sql, StringComparison.OrdinalIgnoreCase); Assert.DoesNotContain("@@VERSION", sql, StringComparison.OrdinalIgnoreCase); } + /// + /// #2340: Aurora is detected by CALLING aurora_version(), never by looking it up in a catalog. + /// + /// The catalog form this replaces — count(*) FROM pg_proc WHERE proname = 'aurora_version' + /// — was measured returning 0 on a live Aurora PostgreSQL 17.7 cluster as a pg_monitor + /// role whose SELECT aurora_version() returned 17.7.2. Because + /// PgWaitStatsCollector and PgStatementStatsCollector both gate on IsAurora, that + /// one wrong boolean silently disabled the two most valuable PostgreSQL reads on every Aurora target. + /// Pinned as an ABSENCE as well as a presence, because the tempting "just add pg_proc back as a + /// fallback" would restore a check that is wrong precisely where it matters. + /// + [Fact] + public void AuroraIsDetectedByCallingTheFunction_NotByACatalogLookup() + { + var probe = DarlingServerConnector.PostgresAuroraProbeQueryText; + + Assert.Contains("aurora_version()", probe, StringComparison.Ordinal); + Assert.DoesNotContain("pg_proc", probe, StringComparison.Ordinal); + Assert.DoesNotContain("count(", probe, StringComparison.OrdinalIgnoreCase); + + /* Its own statement: that is what lets a stock-PostgreSQL 42883 be caught and read as "not + Aurora" rather than failing the whole probe. Folded back into the detection query, the + failure would take version and recovery detection down with it. */ + Assert.DoesNotContain("aurora", DarlingServerConnector.PostgresDetectionQueryText, StringComparison.OrdinalIgnoreCase); + } + private static DarlingConfig ConfigWith(MonitoredServer server) { var config = new DarlingConfig(); diff --git a/Darling/Darling.Tests/QueryStoreFetchProbeLivePostgresTests.cs b/Darling/Darling.Tests/QueryStoreFetchProbeLivePostgresTests.cs new file mode 100644 index 000000000..65e78dae0 --- /dev/null +++ b/Darling/Darling.Tests/QueryStoreFetchProbeLivePostgresTests.cs @@ -0,0 +1,147 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading; +using System.Threading.Tasks; +using Npgsql; +using PerformanceMonitor.Common; +using PerformanceMonitor.Darling.Storage; +using Xunit; + +namespace Darling.Tests; + +/// +/// The #2312 touch-and-probe against a REAL store — the one statement the whole activity-driven fetch +/// runs on, whose risk lives entirely in how PostgreSQL evaluates the data-modifying CTEs, the hourly +/// guard, the hash comparisons and the LEFT JOIN together; no source pin can speak to any of it. Also the +/// writer's NULL-digest content-less marker, which V77's nullable column exists for: it must land, read as +/// RESOLVED, and never re-enter the fetch list. +/// +[Collection("live-postgres")] +public sealed class QueryStoreFetchProbeLivePostgresTests +{ + private const string ServerName = "darling-fetch-probe-e2e"; + private static readonly int ServerId = ServerIdHelper.GetDeterministicHashCode(ServerName); + private const string Db = "ProbeDb"; + private static string? ConnectionString => Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + + [Fact] + public async Task TouchAndProbe_AnswersMissingStaleAndMarker_AndRefreshesLiveness() + { + var cs = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(cs), "Set DARLING_TEST_PG to a Postgres connection string to run the live fetch-probe test."); + + var ct = TestContext.Current.CancellationToken; + using var connection = new NpgsqlConnection(cs); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + await DeleteRowsAsync(connection, ct); + + var bodySucceeded = false; + try + { + var landedAt = DateTime.UtcNow.AddHours(-3); + + /* Plan 1: real content with a hash. Plan 2: the engine had nothing to give — the writer must + land the NULL-digest marker rather than skipping the row. */ + var landed = await QueryStorePlanWriter.WriteAsync( + connection, ServerId, Db, + new[] + { + new FetchedPlan(1, "", "0xAAAA"), + new FetchedPlan(2, PlanXml: null, PlanHash: "0xBBBB"), + }, + landedAt, ct); + Assert.Equal(new long[] { 1, 2 }, landed); + + using (var marker = new NpgsqlCommand( + "SELECT digest IS NULL FROM collect.query_store_plan_map WHERE server_id = $1 AND database_name = $2 AND plan_id = 2", connection)) + { + marker.Parameters.AddWithValue(ServerId); + marker.Parameters.AddWithValue(Db); + Assert.Equal(true, await marker.ExecuteScalarAsync(ct)); + } + + /* The cycle references four plans: 1 current, 2 the marker, 3 never seen, and 1-with-a-new-hash + is exercised by a second batch below. The probe must return them all, in order. */ + var now = DateTime.UtcNow; + var verdicts = await QueryStoreFetchProbe.TouchAndProbePlansAsync( + connection, ServerId, Db, + new[] { (1L, (string?)"0xAAAA"), (2L, (string?)"0xBBBB"), (3L, (string?)"0xCCCC") }, + now, ct); + + Assert.Equal(3, verdicts.Count); + Assert.Equal(new FetchProbeVerdict(1, Resolved: true, HashStale: false), verdicts[0]); + /* The marker resolves — that is its entire job. */ + Assert.Equal(new FetchProbeVerdict(2, Resolved: true, HashStale: false), verdicts[1]); + Assert.Equal(new FetchProbeVerdict(3, Resolved: false, HashStale: false), verdicts[2]); + + /* Liveness: the touch advanced last_seen past the 3-hour-old landing stamp (the rows were + older than the hourly guard, so the update fired) — on the map AND on the dimension row the + real digest points at. */ + using (var freshness = new NpgsqlCommand(@" +SELECT + (SELECT COUNT(*) FROM collect.query_store_plan_map + WHERE server_id = $1 AND database_name = $2 AND last_seen > $3), + (SELECT COUNT(*) FROM collect.query_plan_dim d + JOIN collect.query_store_plan_map m ON m.digest = d.digest + WHERE m.server_id = $1 AND m.database_name = $2 AND d.last_seen > $3)", connection)) + { + freshness.Parameters.AddWithValue(ServerId); + freshness.Parameters.AddWithValue(Db); + freshness.Parameters.AddWithValue(QueryStorePlanMap.Naive(landedAt.AddMinutes(1))); + await using var reader = await freshness.ExecuteReaderAsync(ct); + Assert.True(await reader.ReadAsync(ct)); + Assert.Equal(2L, reader.GetInt64(0)); /* both referenced map rows touched */ + Assert.Equal(1L, reader.GetInt64(1)); /* the one real dim row touched */ + } + + /* An in-place rewrite: same plan_id, different live hash. Stale, and still resolved — the + caller refetches on the OR of the two. */ + var stale = await QueryStoreFetchProbe.TouchAndProbePlansAsync( + connection, ServerId, Db, new[] { (1L, (string?)"0xDEAD") }, now.AddHours(2), ct); + Assert.Equal(new FetchProbeVerdict(1, Resolved: true, HashStale: true), stale.Single()); + + /* Text side: land one row WITHOUT a hash (the legacy shape), then probe with a live hash — + NULL adopts rather than reading stale, and a second probe with a DIFFERENT hash is the + reset detector firing. */ + await QueryStoreTextWriter.WriteAsync( + connection, ServerId, Db, + new[] { new FetchedQueryText(10, "SELECT 1", QueryHash: null) }, landedAt, ct); + + var adopt = await QueryStoreFetchProbe.TouchAndProbeTextsAsync( + connection, ServerId, Db, new[] { (10L, (string?)"0x1111"), (11L, (string?)"0x2222") }, now, ct); + Assert.Equal(new FetchProbeVerdict(10, Resolved: true, HashStale: false), adopt[0]); + Assert.Equal(new FetchProbeVerdict(11, Resolved: false, HashStale: false), adopt[1]); + + var renumbered = await QueryStoreFetchProbe.TouchAndProbeTextsAsync( + connection, ServerId, Db, new[] { (10L, (string?)"0x9999") }, now.AddHours(2), ct); + Assert.Equal(new FetchProbeVerdict(10, Resolved: true, HashStale: true), renumbered.Single()); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, async (cleanup, cleanupCt) => + await DeleteRowsAsync(cleanup, cleanupCt)); + } + } + + private static async Task DeleteRowsAsync(NpgsqlConnection connection, CancellationToken ct) + { + var sql = + $"DELETE FROM collect.query_store_plan_map WHERE server_id = {ServerId};" + + $"DELETE FROM collect.query_store_text WHERE server_id = {ServerId};" + + $"DELETE FROM servers WHERE server_id = {ServerId};"; + using var cleanup = new NpgsqlCommand(sql, connection); + await cleanup.ExecuteNonQueryAsync(ct); + } +} diff --git a/Darling/Darling.Tests/QueryStoreHealthStoreTests.cs b/Darling/Darling.Tests/QueryStoreHealthStoreTests.cs index d6a50b0d4..68e815153 100644 --- a/Darling/Darling.Tests/QueryStoreHealthStoreTests.cs +++ b/Darling/Darling.Tests/QueryStoreHealthStoreTests.cs @@ -32,7 +32,11 @@ public void TheRungIsTheTopOfADenseLadder() { var versions = PgMigrations.Scripts.Select(s => s.Version).ToList(); - Assert.Equal(76, versions.Max()); + /* #2312 added V77, so this rung is no longer the top — the "I am the top" claim moves to the + newest rung's own test (ActivityDrivenPlanFetchStoreTests) and this one keeps the invariants + that stay true forever: the rung is PRESENT, the ladder is ordered and dense, and the build's + schema version tracks the maximum. */ + Assert.Contains(76, versions); Assert.Equal(StorageVersion.SchemaVersion, versions.Max()); Assert.Equal(versions.Distinct().OrderBy(v => v), versions); @@ -48,7 +52,8 @@ public void TheRungIsTheTopOfADenseLadder() [Fact] public void TheProbeMapsAFullyMigratedStoreTo76() { - Assert.Equal(76, StorageVersion.SchemaVersion); + /* #2312: no longer the top (that claim lives in ActivityDrivenPlanFetchStoreTests) — this fact + keeps pinning that a store at exactly 76 maps to 76 and one at 75 maps to 75, forever. */ Assert.Equal(StorageVersion.SchemaVersion, ViewerDataService.RequiredStoreSchemaVersion); /* 51 positional sentinels then the V76 one by name — the map takes 52 parameters. Present => 76, @@ -199,7 +204,15 @@ private static int InvokeMap(object[] leading, bool hasQueryStoreHealth) var method = typeof(ViewerDataService) .GetMethod("MapProbedSchemaVersion", System.Reflection.BindingFlags.NonPublic | System.Reflection.BindingFlags.Static)!; + /* #2312 appended hasQueryStoreTextHash after this rung's parameter — pass it FALSE so these + facts keep exercising the V76/V75 arms rather than the newer one. */ + /* Parameters appended by LATER rungs are padded FALSE, so this fact keeps exercising its own + arm rather than a newer one. Derived from the method's arity rather than listed by hand, so + a future rung does not have to edit this file -- #2357 (V78) was the fourth that would have. */ var args = leading.Concat(new object[] { hasQueryStoreHealth }).ToArray(); + args = args + .Concat(Enumerable.Repeat((object)false, method.GetParameters().Length - args.Length)) + .ToArray(); Assert.Equal(method.GetParameters().Length, args.Length); return (int)method.Invoke(null, args)!; diff --git a/Darling/Darling.Tests/QueryStorePlanFetchTests.cs b/Darling/Darling.Tests/QueryStorePlanFetchTests.cs new file mode 100644 index 000000000..0965eec5f --- /dev/null +++ b/Darling/Darling.Tests/QueryStorePlanFetchTests.cs @@ -0,0 +1,499 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Data; +using System.Linq; +using System.Threading; +using System.Threading.Tasks; +using PerformanceMonitor.Collectors; +using PerformanceMonitor.Darling.Storage; +using Xunit; + +namespace Darling.Tests; + +/// +/// #2312 Finding 2: the activity-driven plan/text fetch — the STORE is the watermark. The #2164/#2210 +/// watermark design this file's predecessor pinned (per-database plan-id resume points with a daily +/// refresh expiry) is retired, because Finding 4 measured what it actually did in production: the re-verify +/// cursor that was supposed to replace the expiry never got wired, so catalogs whose full walk needed more +/// than a day expired MID-walk, restarted from plan_id 0, and looped the full catalog fetch forever — +/// 40–110s per cycle, around the clock, to mostly rediscover held content (and 23s to discover "nothing +/// new" on a caught-up catalog). +/// +/// The replacement's contract, pinned here: the cycle's collected rows name their plans/texts; the +/// touch-and-probe answers which the store lacks (refreshing map/dim liveness in the same round trip — +/// Finding 3's unwired TouchSql); the fetch selects EXACTLY those by id under the same byte-budget +/// arithmetic; and NULL content lands as a stored marker instead of a per-cycle rediscovery. Live +/// round-trips for the probe SQL are in the gated Postgres suites; these are the shape and pure-function +/// pins. +/// +public class QueryStorePlanFetchTests +{ + private const string Db = "probedb"; + private static readonly DateTime Now = new(2026, 8, 11, 12, 0, 0, DateTimeKind.Utc); + private static readonly long[] SomeIds = { 900_001, 900_007, 900_042 }; + + /* ---------- the definition still owns no state ---------- */ + + [Fact] + public void TheDefinitionDeclaresNoStateKeys() + { + /* #2312 removed the plan/text watermark state entirely, and the definition must not gain state in + its place: a state-declaring definition is a two-host contract (CollectorStateContractTests pins + default_trace_events as the only one), and the fetch's only bookkeeping now lives in the store + itself plus in-memory carry-over. The one remaining query_store state family (qsowm:) is host + bookkeeping under its own owner, exactly like the backfill pair. */ + Assert.Empty(QueryStoreCollector.Instance.StateKeys); + } + + /* ---------- the runtime-stats query: no plan narrowing, no plan XML, flag-independent ---------- */ + + [Fact] + public void LiveQuery_NeverCarriesThePlanIdPredicateOrTheRowNumberGate() + { + var sql = LiveSql(Context(capturePlanXml: true)); + + Assert.DoesNotContain("qsp.plan_id > ", sql, StringComparison.Ordinal); + Assert.DoesNotContain("ROW_NUMBER()", sql, StringComparison.Ordinal); + Assert.DoesNotContain("query_plan_text = CASE", sql, StringComparison.Ordinal); + } + + [Fact] + public void LiveQuery_EmitsThePlaceholder_RegardlessOfCapturePlanXml() + { + /* #2210: the runtime query carries no plan XML in either mode, so CapturePlanXml no longer changes + this query's text at all. Darling (on) and Lite (off) get byte-identical SQL here; the only thing + CapturePlanXml still gates is the separate by-ids fetch. */ + var on = LiveSql(Context(capturePlanXml: true)); + var off = LiveSql(Context(capturePlanXml: false)); + + Assert.Contains("query_plan_text = CONVERT(nvarchar(1), NULL),", on, StringComparison.Ordinal); + Assert.True(string.Equals(on, off, StringComparison.Ordinal), "CapturePlanXml must not change this query's text"); + } + + /// #2312: the read loop stages NO state — the write-back that advanced planwm: from + /// inline-shipped plan XML retired with the watermark. The failure mode of it creeping back is a state + /// row nothing reads and a prune set that no longer owns its prefix. + [Fact] + public async Task ReadItemAsync_StagesNoPendingState() + { + var context = Context(capturePlanXml: true); + await Read(context, Plan(10, xml: true), Plan(30, xml: true)); + + Assert.Empty(context.PendingState); + } + + /* ---------- the by-ids plan fetch ---------- */ + + /// + /// The budget predicate admits a plan on the running total BEFORE it — running - own < budget + /// — so one oversized plan ships alone instead of stalling. Under store-as-watermark the naive + /// <= form is the same stall it always was, reached through the probe: the plan never lands, + /// stays missing, and rides every cycle's fetch list forever. + /// + /// A SHAPE pin: it asserts the predicate the SQL carries, not what SQL Server does with it. The + /// CONVERT sits inside the candidate CTE and the running total measures the converted text, so a + /// shipped plan is decompressed once (measured 133ms against 274ms for the join-back form on a + /// 73,163-plan catalog). + /// + [Fact] + public void PlanFetchByIds_SelectsExactlyTheNamedIds_UnderTheBudgetArithmetic() + { + var sql = QueryStoreCollector.Instance.BuildPlanFetchByIdsQuery( + Db, Context(capturePlanXml: true), SomeIds, budgetBytes: 12L * 1024 * 1024).Text; + + Assert.Contains("WHERE qsp.plan_id IN (900001, 900007, 900042)", sql, StringComparison.Ordinal); + Assert.DoesNotContain("qsp.plan_id > ", sql, StringComparison.Ordinal); + Assert.DoesNotContain("SELECT TOP", sql, StringComparison.Ordinal); + + Assert.Contains("b.running_bytes - b.plan_bytes < 12582912", sql, StringComparison.Ordinal); + Assert.DoesNotContain("b.running_bytes <= ", sql, StringComparison.Ordinal); + Assert.Contains("COALESCE(DATALENGTH(c.query_plan_text), 0)", sql, StringComparison.Ordinal); + Assert.Contains("ROWS UNBOUNDED PRECEDING", sql, StringComparison.Ordinal); + Assert.Contains("query_plan_text = CONVERT(nvarchar(max), qsp.query_plan)", sql, StringComparison.Ordinal); + Assert.DoesNotContain("JOIN sys.query_store_plan", sql, StringComparison.Ordinal); + + /* The hash rides along without decompressing anything, rendered the way the runtime payload + renders it — TouchAndProbeSql compares the two, so the formats must agree byte-for-byte. */ + Assert.Contains("query_plan_hash = CONVERT(varchar(64), qsp.query_plan_hash, 1)", sql, StringComparison.Ordinal); + + Assert.Contains("ORDER BY b.plan_id", sql, StringComparison.Ordinal); + Assert.Contains("OPTION(RECOMPILE)", sql, StringComparison.Ordinal); + } + + [Fact] + public void PlanFetchByIds_RunsInTheDatabasesOwnContext_WithBracketDoubling() + { + var sql = QueryStoreCollector.Instance.BuildPlanFetchByIdsQuery( + "we]ird", Context(capturePlanXml: true), SomeIds, 1024).Text; + + Assert.Contains("EXECUTE [we]]ird].sys.sp_executesql", sql, StringComparison.Ordinal); + } + + [Fact] + public void PlanFetchByIds_GuardsItsPreconditions() + { + var context = Context(capturePlanXml: true); + + /* Empty means "nothing missing" and the caller must not have called — a silent no-op query here + would hide the missing skip, and IN () is a syntax error anyway. */ + Assert.Throws(() => + QueryStoreCollector.Instance.BuildPlanFetchByIdsQuery(Db, context, Array.Empty(), 1024)); + + /* A non-positive budget ships nothing forever — the stall by another door. */ + Assert.Throws(() => + QueryStoreCollector.Instance.BuildPlanFetchByIdsQuery(Db, context, SomeIds, 0)); + + /* Plan capture off means no plan fetch exists at all. */ + Assert.Throws(() => + QueryStoreCollector.Instance.BuildPlanFetchByIdsQuery(Db, Context(capturePlanXml: false), SomeIds, 1024)); + } + + /* ---------- the by-ids text fetch ---------- */ + + [Fact] + public void TextFetchByIds_SelectsExactlyTheNamedIds_WithTheResetDetectorHash() + { + var context = Context(capturePlanXml: true, fetchTextSeparately: true); + + var sql = QueryStoreCollector.Instance.BuildTextFetchByIdsQuery( + Db, context, SomeIds, budgetBytes: 12L * 1024 * 1024).Text; + + Assert.Contains("WHERE qsq.query_id IN (900001, 900007, 900042)", sql, StringComparison.Ordinal); + Assert.DoesNotContain("qsq.query_id > ", sql, StringComparison.Ordinal); + Assert.DoesNotContain("SELECT TOP", sql, StringComparison.Ordinal); + + Assert.Contains("JOIN sys.query_store_query_text AS qst", sql, StringComparison.Ordinal); + Assert.Contains("b.running_bytes - b.text_bytes < 12582912", sql, StringComparison.Ordinal); + Assert.Contains("ROWS UNBOUNDED PRECEDING", sql, StringComparison.Ordinal); + + /* query_id is only unique until a Query Store reset renumbers it; the stored hash is how the probe + sees that id 5 now names a DIFFERENT statement. Same rendering as the runtime payload. */ + Assert.Contains("query_hash = CONVERT(varchar(64), qsq.query_hash, 1)", sql, StringComparison.Ordinal); + + Assert.Contains("ORDER BY b.query_id", sql, StringComparison.Ordinal); + Assert.Contains("OPTION(RECOMPILE)", sql, StringComparison.Ordinal); + } + + [Fact] + public void TextFetchByIds_GuardsItsPreconditions() + { + var context = Context(capturePlanXml: false, fetchTextSeparately: true); + + Assert.Throws(() => + QueryStoreCollector.Instance.BuildTextFetchByIdsQuery(Db, context, Array.Empty(), 1024)); + Assert.Throws(() => + QueryStoreCollector.Instance.BuildTextFetchByIdsQuery(Db, context, SomeIds, 0)); + + var inlineContext = Context(capturePlanXml: false); + Assert.Throws(() => + QueryStoreCollector.Instance.BuildTextFetchByIdsQuery(Db, inlineContext, SomeIds, 1024)); + } + + /* ---------- the touch-and-probe: liveness + missing set + rewrite/reset detection, one trip ---------- */ + + [Fact] + public void PlanTouchAndProbe_TouchesBothTables_ProbesResolvedAndStale() + { + var sql = QueryStorePlanMap.TouchAndProbeSql; + + /* The liveness half (Finding 3): map and dim last_seen stamped by the same pass, hourly-guarded, + and the dim update skips the NULL-digest content-less markers — there is no dim row to touch. */ + Assert.Contains("UPDATE collect.query_store_plan_map", sql, StringComparison.Ordinal); + Assert.Contains("UPDATE collect.query_plan_dim", sql, StringComparison.Ordinal); + Assert.Equal(2, CountOf(sql, "interval '1 hour'")); + Assert.Contains("FROM map_touch WHERE digest IS NOT NULL", sql, StringComparison.Ordinal); + + /* Hash adoption: legacy rows (stored hash NULL) take the batch's live hash on first touch — never + the other way around, so a stored baseline is never replaced by absence. */ + Assert.Contains("plan_hash = COALESCE(m.plan_hash, t.live_hash)", sql, StringComparison.Ordinal); + + /* The probe half (Finding 2): resolved is row EXISTENCE — a NULL-digest marker still resolves, or + unpersistable plans would ride every cycle's fetch list — and hash_stale fires only when BOTH + hashes exist and differ, so legacy-NULL rows and hash-less batches can never mass-refetch. */ + Assert.Contains("(m.plan_id IS NOT NULL) AS resolved", sql, StringComparison.Ordinal); + Assert.Contains("m.plan_hash IS NOT NULL AND batch.plan_hash IS NOT NULL", sql, StringComparison.Ordinal); + Assert.Contains("m.plan_hash <> batch.plan_hash", sql, StringComparison.Ordinal); + + /* Five parameters: ids + hashes + the stamp. */ + Assert.Contains("$5::timestamp", sql, StringComparison.Ordinal); + Assert.Contains("unnest($1::integer[], $2::text[], $3::bigint[], $4::text[])", sql, StringComparison.Ordinal); + } + + [Fact] + public void TextTouchAndProbe_MirrorsThePlanShape_SingleTable() + { + var sql = QueryStoreTextStore.TouchAndProbeSql; + + Assert.Contains("UPDATE collect.query_store_text", sql, StringComparison.Ordinal); + Assert.Contains("interval '1 hour'", sql, StringComparison.Ordinal); + Assert.Contains("query_hash = COALESCE(t.query_hash, x.live_hash)", sql, StringComparison.Ordinal); + Assert.Contains("(t.query_id IS NOT NULL) AS resolved", sql, StringComparison.Ordinal); + Assert.Contains("t.query_hash IS NOT NULL AND batch.query_hash IS NOT NULL", sql, StringComparison.Ordinal); + Assert.Contains("t.query_hash <> batch.query_hash", sql, StringComparison.Ordinal); + Assert.Contains("$5::timestamp", sql, StringComparison.Ordinal); + } + + /* ---------- the upserts never replace knowledge with absence ---------- */ + + [Fact] + public void PlanMapUpsert_CoalescesDigestAndHash_TowardKnowledge() + { + var sql = QueryStorePlanMap.UpsertSql; + + /* A NULL-XML refetch of a plan whose content the store already holds keeps the content; a fetch + that carried no hash keeps the stored baseline. Real values still advance both — that is how an + in-place rewrite's corrected content gets pointed at. */ + Assert.Contains("digest = COALESCE(EXCLUDED.digest, query_store_plan_map.digest)", sql, StringComparison.Ordinal); + Assert.Contains("plan_hash = COALESCE(EXCLUDED.plan_hash, query_store_plan_map.plan_hash)", sql, StringComparison.Ordinal); + Assert.Contains("WHERE EXCLUDED.last_seen >= query_store_plan_map.last_seen", sql, StringComparison.Ordinal); + Assert.Contains("ORDER BY server_id, database_name, plan_id", sql, StringComparison.Ordinal); + } + + [Fact] + public void TextUpsert_CarriesTheHash_AndOverwritesTextOnPurpose() + { + var sql = QueryStoreTextStore.UpsertSql; + + /* The TEXT is overwritten (a post-reset id names a different statement and the corrected text must + land); the HASH coalesces toward knowledge like the plan side. */ + Assert.Contains("query_sql_text = EXCLUDED.query_sql_text", sql, StringComparison.Ordinal); + Assert.Contains("query_hash = COALESCE(EXCLUDED.query_hash, query_store_text.query_hash)", sql, StringComparison.Ordinal); + Assert.Contains("unnest($1::integer[], $2::text[], $3::bigint[], $4::text[], $5::text[], $6::timestamp[])", sql, StringComparison.Ordinal); + } + + /* ---------- sizing: unchanged arithmetic, still load-bearing (it caps decompression) ---------- */ + + /// + /// The candidate cap sits just past what the budget can actually ship, at every plan size the fleet + /// ACTUALLY exhibits — per-quartile averages of 162 / 80 / 39 / 15 KB measured across 2,166 budget-cut + /// passes. The point of the pin is that none of these clamp: if a real fleet plan size hit a bound, the + /// bound would be doing the sizing instead of the measurement. + /// + [Theory] + [InlineData(162, 114)] + [InlineData(80, 231)] + [InlineData(39, 473)] + [InlineData(15, 1229)] + public void CandidatePlanCount_SitsJustPastTheBudget_AtEveryMeasuredFleetPlanSize(int avgKb, int expected) + { + var k = QueryStorePlanXmlState.CandidatePlanCount(avgKb * 1024L, 12L * 1024 * 1024, out var clamped); + + Assert.Equal(expected, k); + Assert.False(clamped, "a plan size the fleet actually shows must not hit a bound"); + + var actuallyFit = (12L * 1024 * 1024) / (avgKb * 1024L); + Assert.InRange(k / (double)actuallyFit, 1.4, 1.6); + } + + /// + /// First contact assumes LARGE plans on purpose. The estimate is a divisor, so over-stating plan size + /// yields a small cap — and small is the safe direction: it only spreads the catch-up out, where too + /// large decompresses a backlog to discover what fits, which is the trap the cap exists to prevent. + /// + [Fact] + public void CandidatePlanCount_WithNoPreviousPass_IsConservativelySmall() + { + var k = QueryStorePlanXmlState.CandidatePlanCount(null, 12L * 1024 * 1024, out var clamped); + + /* ceil(12MB / 160KB * 1.5) = 116 — the arithmetic, not the issue text's rounded "~118". */ + Assert.Equal(116, k); + Assert.False(clamped); + } + + [Theory] + [InlineData(10L * 1024 * 1024, 12L * 1024 * 1024, 32)] + [InlineData(1, 12L * 1024 * 1024, 2048)] + public void CandidatePlanCount_ClampsAndSaysSo(long avgBytes, long budget, int expected) + { + var k = QueryStorePlanXmlState.CandidatePlanCount(avgBytes, budget, out var clamped); + + Assert.Equal(expected, k); + Assert.True(clamped); + } + + [Fact] + public void CandidatePlanCount_LandingNaturallyOnABound_IsNotReportedAsClamped() + { + var exactlyTheFloor = (long)(QueryStorePlanXmlState.MinCandidatePlans / QueryStorePlanXmlState.CandidatePlanMargin); + var k = QueryStorePlanXmlState.CandidatePlanCount(1, exactlyTheFloor, out var clamped); + + Assert.Equal(QueryStorePlanXmlState.MinCandidatePlans, k); + Assert.False(clamped, "the measurement produced this value; no bound changed it"); + } + + /// + /// While catch-up is in progress the observed average is floored at the seed, because the sample is + /// biased then and measurably so: on one production catalog the plans the fetch shipped averaged 15 KB + /// while the newest 300 in the same catalog averaged 46 KB. Trusting the low figure inflates the cap + /// threefold and decompresses that much more than the budget can ship. + /// + [Fact] + public void CandidatePlanCount_DuringCatchUp_FloorsTheEstimateAtTheSeed() + { + const long budget = 12L * 1024 * 1024; + var biased = 15 * 1024L; + + var duringCatchUp = QueryStorePlanXmlState.CandidatePlanCount(biased, budget, catchUpInProgress: true, out _); + var converged = QueryStorePlanXmlState.CandidatePlanCount(biased, budget, catchUpInProgress: false, out _); + var seeded = QueryStorePlanXmlState.CandidatePlanCount(null, budget, out _); + + Assert.Equal(seeded, duringCatchUp); + Assert.True(converged > duringCatchUp, "the un-floored estimate must still be trusted once converged"); + + var large = 200 * 1024L; + Assert.Equal( + QueryStorePlanXmlState.CandidatePlanCount(large, budget, catchUpInProgress: false, out _), + QueryStorePlanXmlState.CandidatePlanCount(large, budget, catchUpInProgress: true, out _)); + } + + [Fact] + public void CandidatePlanCount_WithNonPositiveBudget_FloorsAndReportsClamped() + { + var k = QueryStorePlanXmlState.CandidatePlanCount(160 * 1024L, 0, out var clamped); + + Assert.Equal(QueryStorePlanXmlState.MinCandidatePlans, k); + Assert.True(clamped); + } + + /// + /// The estimator reproduces the measured fleet numbers from the same two inputs a pass already has. + /// + [Theory] + [InlineData(1.7, 11, 158)] + [InlineData(1.7, 22, 79)] + [InlineData(1.7, 44, 39)] + [InlineData(1.7, 116, 15)] + public void ObservedAvgPlanBytes_ReproducesTheMeasuredQuartiles(double shippedMb, int plans, int expectedKb) + { + var avg = QueryStorePlanXmlState.ObservedAvgPlanBytes((long)(shippedMb * 1024 * 1024), plans); + + Assert.NotNull(avg); + Assert.Equal(expectedKb, (int)(avg!.Value / 1024)); + } + + [Fact] + public void ObservedAvgPlanBytes_OnAQuietPass_IsNull() + { + Assert.Null(QueryStorePlanXmlState.ObservedAvgPlanBytes(0, 0)); + Assert.Null(QueryStorePlanXmlState.ObservedAvgPlanBytes(0, 5)); + Assert.Null(QueryStorePlanXmlState.ObservedAvgPlanBytes(1024, 0)); + } + + /* ---------- helpers ---------- */ + + private static int CountOf(string haystack, string needle) + { + var count = 0; + var index = 0; + while ((index = haystack.IndexOf(needle, index, StringComparison.Ordinal)) >= 0) + { + count++; + index += needle.Length; + } + + return count; + } + + private static string LiveSql(CollectorContext context) => + QueryStoreCollector.Instance.BuildPerItemQuery(Db, context).Text; + + private static CollectorContext Context(bool capturePlanXml, bool fetchTextSeparately = false) + { + var context = new CollectorContext + { + ServerId = 1, + ServerName = "probe", + CollectionTime = Now, + Deltas = new CollectorDeltaCalculator(), + CapturePlanXml = capturePlanXml, + FetchQueryTextSeparately = fetchTextSeparately, + State = new Dictionary(), + }; + context.CurrentDatabaseName = Db; + return context; + } + + private static (long PlanId, bool Xml) Plan(long planId, bool xml) => (planId, xml); + + private static async Task Read(CollectorContext context, params (long PlanId, bool Xml)[] plans) + { + using var reader = MakeReader(plans); + var rows = new List(); + await QueryStoreCollector.Instance.ReadItemAsync(Db, reader, rows, context, CancellationToken.None); + } + + /// + /// A real DbDataReader over the collector's OWN payload shape, generated from + /// PayloadColumns minus database_name (which the on-prem path takes from the enumerated + /// item, not the reader). Generated rather than hand-listed so a column added to the collector cannot + /// silently shift the ordinals the read loop depends on. + /// + private static DataTableReader MakeReader((long PlanId, bool Xml)[] plans) + { + var table = new DataTable("payload"); + var columns = QueryStoreCollector.Instance.PayloadColumns.Skip(1).ToList(); + + foreach (var column in columns) + { + table.Columns.Add(column.Name, ClrType(column.Name, column.Type)); + } + + for (var i = 0; i < plans.Length; i++) + { + var row = table.NewRow(); + + foreach (var column in columns) + { + row[column.Name] = column.Type switch + { + CollectorColumnType.BigInt => 0L, + CollectorColumnType.Integer => 160, + CollectorColumnType.Boolean => false, + /* Distinct per row, so a budget cut's boundary tie group ends on the very next row. */ + CollectorColumnType.Timestamp when ClrType(column.Name, column.Type) == typeof(DateTime) + => Now.AddMinutes(i), + CollectorColumnType.Timestamp => new DateTimeOffset(Now.AddMinutes(i), TimeSpan.Zero), + _ => "x", + }; + } + + row["query_id"] = plans[i].PlanId * 10; + row["plan_id"] = plans[i].PlanId; + row["execution_count"] = 1L; + /* Must not contain the self-query marker, or the read loop skips the row entirely. */ + row["query_text"] = "SELECT 1 FROM dbo.Whatever"; + row["query_plan_text"] = plans[i].Xml ? new string('p', 4096) : (object)DBNull.Value; + + table.Rows.Add(row); + } + + var dataSet = new DataSet(); + dataSet.Tables.Add(table); + return dataSet.CreateDataReader(); + } + + /// + /// The provider types the read loop actually expects, which are NOT uniform across the timestamp + /// columns: first_execution_time / last_execution_time come out of Query Store as + /// datetimeoffset and are read as DateTimeOffset, while interval_start_time_utc is + /// computed datetime2 and read with GetDateTime. A harness that types all three the same + /// way throws InvalidCastException inside the loop — which is how this was found. + /// + private static Type ClrType(string name, CollectorColumnType type) => type switch + { + CollectorColumnType.BigInt => typeof(long), + CollectorColumnType.Integer => typeof(int), + CollectorColumnType.Boolean => typeof(bool), + CollectorColumnType.Timestamp => + name.Equals("interval_start_time_utc", StringComparison.Ordinal) ? typeof(DateTime) : typeof(DateTimeOffset), + _ => typeof(string), + }; +} diff --git a/Darling/Darling.Tests/QueryStorePlanWatermarkTests.cs b/Darling/Darling.Tests/QueryStorePlanWatermarkTests.cs deleted file mode 100644 index 13c020c2e..000000000 --- a/Darling/Darling.Tests/QueryStorePlanWatermarkTests.cs +++ /dev/null @@ -1,640 +0,0 @@ -/* - * Copyright (c) 2026 Erik Darling, Darling Data LLC - * - * This file is part of the SQL Server Performance Monitor. - * - * Licensed under the MIT License. See LICENSE file in the project root for full license information. - */ - -using System; -using System.Collections.Generic; -using System.Data; -using System.Globalization; -using System.Linq; -using System.Threading; -using System.Threading.Tasks; -using PerformanceMonitor.Collectors; -using PerformanceMonitor.Darling.Storage; -using Xunit; - -namespace Darling.Tests; - -/// -/// #2164: the plan-XML watermark. 97% of the plan XML shipped in a three-hour fleet window was for plans the -/// store already held, and since drain is 94-97% of a pass and costs per-row LOB bytes, not fetching is worth -/// far more than fetching less. -/// -/// Driven entirely through the collector's PUBLIC surface — BuildPerItemQuery, -/// BuildBackfillPerItemQuery, ReadItemAsync — rather than reaching for the internal helpers, so -/// no production visibility is widened for the tests' benefit. It also makes the state format an explicit -/// pin: the stored string is written out literally here instead of being produced by the same formatter under -/// test, which would have agreed with itself no matter what it emitted. -/// -public class QueryStorePlanWatermarkTests -{ - private const string Db = "probedb"; - private static readonly DateTime Now = new(2026, 8, 11, 12, 0, 0, DateTimeKind.Utc); - - /* ---------- #2210: the SQL-side narrowing is gone; QueryStorePlanXmlState.Resolve is what remains ---------- - The ROW_NUMBER-gated CASE and its in-stream `AND qsp.plan_id > ` predicate are DELETED, not reworked — - BuildPlanFetchQuery is the only thing that reads plan XML now, and its `watermark` parameter is resolved - by the host calling Resolve directly rather than derived inline in this query. These tests used to drive - that predicate through the live SQL; they now drive Resolve directly, which is exactly what the host - still calls to get BuildPlanFetchQuery's watermark argument, so the state-format/malformed/expired/ - future-stamp/per-database coverage stays live even though the SQL it used to narrow does not exist. */ - - [Fact] - public void Resolve_Fresh_ReturnsTheStoredPlanId() - { - var resolved = QueryStorePlanXmlState.Resolve(Stored(900_000, Now), Db, Now); - - Assert.Equal(900_000, resolved); - } - - [Fact] - public void Resolve_Absent_ReturnsZero() - { - /* Absent is what a first run, a restarted host and a broken store all look like, and all three must - resolve to "fetch everything" rather than skip. */ - var resolved = QueryStorePlanXmlState.Resolve(new Dictionary(), Db, Now); - - Assert.Equal(0, resolved); - } - - [Theory] - [InlineData("")] - [InlineData(" ")] - [InlineData("900000")] /* no stamp */ - [InlineData("900000:")] /* empty stamp */ - [InlineData("notanumber:1786449600")] - [InlineData("900000:notanumber")] - [InlineData("900000:1786449600:extra")] - [InlineData("0:1786449600")] /* plan_id 0 is not a plan */ - [InlineData("-5:1786449600")] - public void Resolve_Malformed_ReturnsZero(string raw) - { - /* Anything unparseable degrades to a full fetch. Trusting a partially parsed value would suppress - plan XML based on a number nobody wrote. */ - var state = new Dictionary { [QueryStorePlanXmlState.WatermarkKeyPrefix + Db] = raw }; - - Assert.Equal(0, QueryStorePlanXmlState.Resolve(state, Db, Now)); - } - - [Fact] - public void Resolve_Expired_ReturnsZero_ButStillFreshReturnsTheStoredPlanId() - { - /* Bounded staleness is why the stamp is stored beside the id. Query Store can rewrite a plan's XML in - place (memory grant feedback and friends) without issuing a new plan_id, and a permanent watermark - would never look again. It also bounds the documented dormant-plan gap. */ - var stampedLongAgo = Now - QueryStorePlanXmlState.RefreshAfter - TimeSpan.FromMinutes(1); - - var expired = QueryStorePlanXmlState.Resolve(Stored(900_000, stampedLongAgo), Db, Now); - var stillFresh = QueryStorePlanXmlState.Resolve(Stored(900_000, Now - TimeSpan.FromMinutes(1)), Db, Now); - - Assert.Equal(0, expired); - Assert.Equal(900_000, stillFresh); - } - - [Fact] - public void Resolve_StampedInTheFuture_ReturnsZero() - { - /* A clock that moved backwards would otherwise pin the watermark for as long as the skew lasts. */ - var resolved = QueryStorePlanXmlState.Resolve(Stored(900_000, Now.AddDays(3)), Db, Now); - - Assert.Equal(0, resolved); - } - - [Fact] - public void Resolve_IsKeyedPerDatabase() - { - /* plan_id is monotonic WITHIN a database and means nothing across them, so one database's watermark - must never be read for another. */ - var state = new Dictionary - { - [QueryStorePlanXmlState.WatermarkKeyPrefix + "alpha"] = "900000:" + Unix(Now), - }; - - Assert.Equal(900_000, QueryStorePlanXmlState.Resolve(state, "alpha", Now)); - Assert.Equal(0, QueryStorePlanXmlState.Resolve(state, "beta", Now)); - } - - [Fact] - public void TheDefinitionDeclaresNoStateKeys_TheHostOwnsThisState() - { - /* The watermark keys are one per DATABASE and only known at runtime, so the definition could not - declare them even if it wanted to. More importantly it MUST NOT: a state-declaring definition is a - two-host contract (CollectorStateContractTests pins default_trace_events as the only one), while - this is host bookkeeping. The QueryStoreBackfillState seam — a separate state owner name — is what - lets the host persist per-database state without the definition claiming any. - - The failure mode if this ever flips is silent, which is why it is pinned from both ends: a row - written under the DEFINITION's name is never read back, so the watermark would resolve absent - forever and collection would quietly keep paying full price. */ - Assert.Empty(QueryStoreCollector.Instance.StateKeys); - Assert.NotEqual(QueryStorePlanXmlState.StateCollectorName, QueryStoreCollector.Instance.Name); - Assert.Equal("query_store_plan_xml", QueryStorePlanXmlState.StateCollectorName); - Assert.Equal("planwm:", QueryStorePlanXmlState.WatermarkKeyPrefix); - } - - /* ---------- #2210: the runtime-stats query itself, now flag-independent ---------- */ - - [Fact] - public void LiveQuery_NeverCarriesThePlanIdPredicateOrTheRowNumberGate_RegardlessOfWatermarkState() - { - /* The predicate and the CASE it used to narrow are gone from the query entirely, not just from the - conservative path — there is no live watermark state under which either can reappear. */ - var withState = LiveSql(Context(capturePlanXml: true, state: Stored(900_000, Now))); - var withoutState = LiveSql(Context(capturePlanXml: true, state: new Dictionary())); - - Assert.DoesNotContain("qsp.plan_id > ", withState, StringComparison.Ordinal); - Assert.DoesNotContain("qsp.plan_id > ", withoutState, StringComparison.Ordinal); - Assert.DoesNotContain("ROW_NUMBER()", withState, StringComparison.Ordinal); - Assert.DoesNotContain("query_plan_text = CASE", withState, StringComparison.Ordinal); - } - - [Fact] - public void LiveQuery_EmitsThePlaceholder_RegardlessOfCapturePlanXml() - { - /* #2210: both branches of the old ternary now emit the same nvarchar(1) NULL placeholder — the - runtime query carries no plan XML in either mode, so CapturePlanXml no longer changes this query's - text at all. Darling (on) and Lite (off) get byte-identical SQL here; the only thing CapturePlanXml - still gates is the separate BuildPlanFetchQuery fetch. */ - var on = LiveSql(Context(capturePlanXml: true, state: new Dictionary())); - var off = LiveSql(Context(capturePlanXml: false, state: new Dictionary())); - - Assert.Contains("query_plan_text = CONVERT(nvarchar(1), NULL),", on, StringComparison.Ordinal); - Assert.Contains("query_plan_text = CONVERT(nvarchar(1), NULL),", off, StringComparison.Ordinal); - Assert.True(string.Equals(on, off, StringComparison.Ordinal), "CapturePlanXml must no longer change this query's text"); - } - - /* ---------- write-back, driven through the real read loop ---------- */ - - [Fact] - public async Task WriteBack_NormalPass_AdvancesToTheHighestStoredPlanId() - { - var context = Context(capturePlanXml: true, state: new Dictionary()); - await Read(context, Plan(10, xml: true), Plan(20, xml: true), Plan(30, xml: true)); - - Assert.Equal("30:" + Unix(Now), Written(context)); - } - - [Fact] - public async Task WriteBack_CountsOnlyPlansWhoseXmlActuallyShipped() - { - /* The ROW_NUMBER gate NULLs the XML on all but one interval per plan, so "seen" and "stored" differ - on every real pass. Advancing on a plan whose XML was NULL would suppress that plan's XML from then - on without ever having sent it. */ - var context = Context(capturePlanXml: true, state: new Dictionary()); - await Read(context, Plan(10, xml: true), Plan(40, xml: false)); - - Assert.Equal("10:" + Unix(Now), Written(context)); - } - - /* WriteBack_BudgetCutPass_DoesNotAdvanceAtAll (#2164) deleted with the shape it pinned: its own comment - said it recorded known-broken behaviour under the last_execution_time-ordered fetch — the watermark - could never advance on a budget cut because a cut left an arbitrary SUBSET of plan_ids. #2210's - plan_id-ordered fetch removes that premise; the replacement is - AdvanceWatermark_OnABudgetCutPass_StillAdvances below, which pins the new behaviour directly against - QueryStorePlanXmlState.AdvanceWatermark. */ - - [Fact] - public async Task WriteBack_QuietWindow_NeverMovesTheWatermarkBackward() - { - /* This is the case that killed the first design. A window whose newest-EXECUTING plan is older than - the newest-COMPILED one is an ordinary quiet window, and on a steady workload it is most windows. - The first cut read it as a Query Store reset and dropped the watermark, which would have refetched - the whole catalog on nearly every pass — the exact cost being removed. */ - var context = Context(capturePlanXml: true, state: Stored(900_000, Now)); - await Read(context, Plan(800_000, xml: true), Plan(850_000, xml: true)); - - Assert.Null(Written(context)); - } - - [Fact] - public async Task WriteBack_Advance_CarriesTheOriginalStampForward_SoTheHorizonStillFires() - { - /* The stamp dates the last FULL fetch. If an advance re-stamped it to now, then any database that - keeps compiling new plans would push its refresh horizon out forever — and those are the busy - databases where a stale plan matters most. The bounded refresh would silently never happen. */ - var fetchedAt = Now - TimeSpan.FromHours(20); - var context = Context(capturePlanXml: true, state: Stored(900_000, fetchedAt)); - - await Read(context, Plan(950_000, xml: true)); - - Assert.Equal("950000:" + Unix(fetchedAt), Written(context)); - } - - [Fact] - public async Task WriteBack_AfterExpiry_StampsTheFullFetchAtNow() - { - /* The other half of the same rule: an expired watermark means THIS pass refetched everything, so it - is the one case that legitimately re-dates the horizon. Without this the stamp would never move and - every pass after the first expiry would be a full fetch. */ - var longAgo = Now - QueryStorePlanXmlState.RefreshAfter - TimeSpan.FromHours(1); - var context = Context(capturePlanXml: true, state: Stored(900_000, longAgo)); - - await Read(context, Plan(950_000, xml: true)); - - Assert.Equal("950000:" + Unix(Now), Written(context)); - } - - [Fact] - public async Task WriteBack_PlanCaptureOff_WritesNothing() - { - /* Lite reads the same rows with no XML. It must not leave a watermark behind that a plan-capturing - host would later honor, having never shipped a single plan. */ - var context = Context(capturePlanXml: false, state: new Dictionary()); - await Read(context, Plan(10, xml: true), Plan(30, xml: true)); - - Assert.Null(Written(context)); - } - - /* ---------- helpers ---------- */ - - private static Dictionary Stored(long planId, DateTime stampedAt) => - new() - { - [QueryStorePlanXmlState.WatermarkKeyPrefix + Db] = - planId.ToString(CultureInfo.InvariantCulture) + ":" + Unix(stampedAt), - }; - - private static string Unix(DateTime utc) => - new DateTimeOffset(DateTime.SpecifyKind(utc, DateTimeKind.Utc)).ToUnixTimeSeconds() - .ToString(CultureInfo.InvariantCulture); - - private static string? Written(CollectorContext context) => - context.PendingState.TryGetValue(QueryStorePlanXmlState.WatermarkKeyPrefix + Db, out var value) - ? value - : null; - - private static string LiveSql(CollectorContext context) => - QueryStoreCollector.Instance.BuildPerItemQuery(Db, context).Text; - - private static CollectorContext Context( - bool capturePlanXml, - IReadOnlyDictionary state, - int? budgetOverride = null) - { - var context = new CollectorContext - { - ServerId = 1, - ServerName = "probe", - CollectionTime = Now, - Deltas = new CollectorDeltaCalculator(), - CapturePlanXml = capturePlanXml, - State = state, - TextByteBudgetOverride = budgetOverride, - }; - context.CurrentDatabaseName = Db; - return context; - } - - private static (long PlanId, bool Xml) Plan(long planId, bool xml) => (planId, xml); - - private static async Task Read(CollectorContext context, params (long PlanId, bool Xml)[] plans) - { - using var reader = MakeReader(plans); - var rows = new List(); - await QueryStoreCollector.Instance.ReadItemAsync(Db, reader, rows, context, CancellationToken.None); - } - - /// - /// A real DbDataReader over the collector's OWN payload shape, generated from - /// PayloadColumns minus database_name (which the on-prem path takes from the enumerated - /// item, not the reader). Generated rather than hand-listed so a column added to the collector cannot - /// silently shift the ordinals the read loop depends on. - /// - private static DataTableReader MakeReader((long PlanId, bool Xml)[] plans) - { - var table = new DataTable("payload"); - var columns = QueryStoreCollector.Instance.PayloadColumns.Skip(1).ToList(); - - foreach (var column in columns) - { - table.Columns.Add(column.Name, ClrType(column.Name, column.Type)); - } - - for (var i = 0; i < plans.Length; i++) - { - var row = table.NewRow(); - - foreach (var column in columns) - { - row[column.Name] = column.Type switch - { - CollectorColumnType.BigInt => 0L, - CollectorColumnType.Integer => 160, - CollectorColumnType.Boolean => false, - /* Distinct per row, so a budget cut's boundary tie group ends on the very next row. */ - CollectorColumnType.Timestamp when ClrType(column.Name, column.Type) == typeof(DateTime) - => Now.AddMinutes(i), - CollectorColumnType.Timestamp => new DateTimeOffset(Now.AddMinutes(i), TimeSpan.Zero), - _ => "x", - }; - } - - row["query_id"] = plans[i].PlanId * 10; - row["plan_id"] = plans[i].PlanId; - row["execution_count"] = 1L; - /* Must not contain the self-query marker, or the read loop skips the row entirely. */ - row["query_text"] = "SELECT 1 FROM dbo.Whatever"; - row["query_plan_text"] = plans[i].Xml ? new string('p', 4096) : (object)DBNull.Value; - - table.Rows.Add(row); - } - - var dataSet = new DataSet(); - dataSet.Tables.Add(table); - return dataSet.CreateDataReader(); - } - - /// - /// The provider types the read loop actually expects, which are NOT uniform across the timestamp - /// columns: first_execution_time / last_execution_time come out of Query Store as - /// datetimeoffset and are read as DateTimeOffset, while interval_start_time_utc is - /// computed datetime2 and read with GetDateTime. A harness that types all three the same - /// way throws InvalidCastException inside the loop — which is how this was found. - /// - private static Type ClrType(string name, CollectorColumnType type) => type switch - { - CollectorColumnType.BigInt => typeof(long), - CollectorColumnType.Integer => typeof(int), - CollectorColumnType.Boolean => typeof(bool), - CollectorColumnType.Timestamp => - name.Equals("interval_start_time_utc", StringComparison.Ordinal) ? typeof(DateTime) : typeof(DateTimeOffset), - _ => typeof(string), - }; - - /* ---- #2210: the plan_id-ordered fetch policy. Pure functions, pinned like QueryStoreBackfillState - .AdaptiveSpan, because the candidate window and the watermark advance are the two places this - optimization can silently do nothing (attempt one) or silently lose plans (the ordering precondition). */ - - /// - /// The candidate window sits just past what the budget can actually ship, at every plan size the fleet - /// ACTUALLY exhibits — per-quartile averages of 162 / 80 / 39 / 15 KB measured across 2,166 budget-cut - /// passes. The point of the pin is that none of these clamp: if a real fleet plan size hit a bound, the - /// bound would be doing the sizing instead of the measurement. - /// - [Theory] - [InlineData(162, 114)] - [InlineData(80, 231)] - [InlineData(39, 473)] - [InlineData(15, 1229)] - public void CandidatePlanCount_SitsJustPastTheBudget_AtEveryMeasuredFleetPlanSize(int avgKb, int expected) - { - var k = QueryStorePlanXmlState.CandidatePlanCount(avgKb * 1024L, 12L * 1024 * 1024, out var clamped); - - Assert.Equal(expected, k); - Assert.False(clamped, "a plan size the fleet actually shows must not hit a bound"); - - /* Just past, not far past: the window is the coarse bound and the running byte total is the exact one, - and every plan IN the window is decompressed to compute that total. */ - var actuallyFit = (12L * 1024 * 1024) / (avgKb * 1024L); - Assert.InRange(k / (double)actuallyFit, 1.4, 1.6); - } - - /// - /// First contact assumes LARGE plans on purpose. The estimate is a divisor, so over-stating plan size - /// yields a small window — and small is the safe direction: it only slows the watermark down, where too - /// large decompresses a catalog to discover what fits, which is the trap the window exists to prevent. - /// - [Fact] - public void CandidatePlanCount_WithNoPreviousPass_IsConservativelySmall() - { - var seed = QueryStorePlanXmlState.CandidatePlanCount(null, 12L * 1024 * 1024, out var clamped); - var atLargestMeasured = QueryStorePlanXmlState.CandidatePlanCount(162 * 1024L, 12L * 1024 * 1024, out _); - - Assert.False(clamped); - Assert.InRange(seed, atLargestMeasured - 10, atLargestMeasured + 10); - } - - /// Bounds hold, and every clamp REPORTS itself — a window silently pinned at its ceiling reads - /// exactly like one that fit, which is how a cap becomes invisible. - [Theory] - [InlineData(1, 12L * 1024 * 1024, QueryStorePlanXmlState.MaxCandidatePlans)] - [InlineData(64 * 1024, 12L * 1024 * 1024, QueryStorePlanXmlState.MinCandidatePlans)] - public void CandidatePlanCount_ClampsAndSaysSo(long avgKb, long budget, int expected) - { - var k = QueryStorePlanXmlState.CandidatePlanCount(avgKb * 1024L, budget, out var clamped); - - Assert.Equal(expected, k); - Assert.True(clamped, "a clamped window must be reportable so the caller can log it"); - } - - /// - /// `clamped` means a bound CHANGED the answer, not that the answer equals one. A window whose measured size - /// lands naturally on a bound was sized by the measurement and needs no log line; reporting it as clamped is - /// a false positive, and a caller that logs on it trains its reader to ignore the message. - /// - [Fact] - public void CandidatePlanCount_LandingNaturallyOnABound_IsNotReportedAsClamped() - { - /* Budget chosen so budget/avg*margin is exactly MinCandidatePlans: 32 / 1.5 = 21.33 plans of 1 byte. */ - var exactlyTheFloor = (long)(QueryStorePlanXmlState.MinCandidatePlans / QueryStorePlanXmlState.CandidatePlanMargin); - var k = QueryStorePlanXmlState.CandidatePlanCount(1, exactlyTheFloor, out var clamped); - - Assert.Equal(QueryStorePlanXmlState.MinCandidatePlans, k); - Assert.False(clamped, "the measurement produced this value; no bound changed it"); - } - - /// - /// While catch-up is in progress the observed average is floored at the seed, because the sample is biased - /// then and measurably so: on one production catalog the plans the fetch shipped averaged 15 KB while the - /// newest 300 in the same catalog averaged 46 KB. Trusting the low figure inflates K threefold and - /// decompresses that much more than the budget can ship. After convergence the observed value is trusted. - /// - [Fact] - public void CandidatePlanCount_DuringCatchUp_FloorsTheEstimateAtTheSeed() - { - const long budget = 12L * 1024 * 1024; - var biased = 15 * 1024L; - - var duringCatchUp = QueryStorePlanXmlState.CandidatePlanCount(biased, budget, catchUpInProgress: true, out _); - var converged = QueryStorePlanXmlState.CandidatePlanCount(biased, budget, catchUpInProgress: false, out _); - var seeded = QueryStorePlanXmlState.CandidatePlanCount(null, budget, out _); - - Assert.Equal(seeded, duringCatchUp); - Assert.True(converged > duringCatchUp, "the un-floored estimate must still be trusted once converged"); - - /* A large observed average is NOT raised by the floor — over-estimating plan size is the safe direction - and the floor only ever makes the window smaller. */ - var large = 200 * 1024L; - Assert.Equal( - QueryStorePlanXmlState.CandidatePlanCount(large, budget, catchUpInProgress: false, out _), - QueryStorePlanXmlState.CandidatePlanCount(large, budget, catchUpInProgress: true, out _)); - } - - /// - /// #2210, ruling item 4: a FULL cursor sweep over a healthy catalog leaves the watermark byte-identical. - /// Only the reset arm may ever zero it. - /// - /// Simulated through the real functions rather than mocked, because the property is about them: a - /// healthy catalog means every slice the cursor walks finds its stored hash matching the live one, so - /// nothing is re-fetched and the pass lands no plan ids. The sweep is then ceil(range / slice) calls - /// to with nothing landed, and the persisted string - /// has to come out the same at the end — including its stamp, since re-stamping on a no-op sweep would push - /// the refresh period out forever on any database the cursor keeps visiting. - /// - [Fact] - public void AFullCursorSweepOverAHealthyCatalog_LeavesTheWatermarkByteIdentical() - { - const long watermark = 77_176; - var stamp = Now; - var before = QueryStorePlanXmlState.Format(watermark, stamp); - - var slice = QueryStorePlanMap.CursorSliceWidth(watermark, QueryStorePlanXmlState.RefreshAfter, TimeSpan.FromMinutes(5)); - Assert.True(slice > 0); - - var standing = watermark; - var passes = 0; - for (var floor = 0L; floor < watermark; floor += slice) - { - /* Healthy: hashes match across the slice, so nothing is re-fetched and nothing lands. */ - var advance = QueryStorePlanXmlState.AdvanceWatermark(standing, Array.Empty()); - - Assert.True(advance.ArrivedInPlanIdOrder); - Assert.Equal(standing, advance.Watermark); - standing = advance.Watermark; - passes++; - } - - Assert.Equal(watermark, standing); - Assert.Equal(before, QueryStorePlanXmlState.Format(standing, stamp)); - - /* And the sweep genuinely covered the range in one refresh period rather than needing a second. */ - Assert.True((long)passes * slice >= watermark); - Assert.True(passes <= QueryStorePlanXmlState.RefreshAfter.Ticks / TimeSpan.FromMinutes(5).Ticks); - } - - /// A misconfigured budget floors the window rather than producing zero or a negative one. - [Fact] - public void CandidatePlanCount_WithNonPositiveBudget_FloorsAndReportsClamped() - { - var k = QueryStorePlanXmlState.CandidatePlanCount(160 * 1024L, 0, out var clamped); - - Assert.Equal(QueryStorePlanXmlState.MinCandidatePlans, k); - Assert.True(clamped); - } - - /// - /// The estimator reproduces the measured fleet numbers from the same two inputs a pass already has, which - /// is the whole reason no probe is needed: 12.1 MB over 78 plans is the q1 average, 12.3 MB over 828 is q4. - /// - [Theory] - [InlineData(12.1, 78, 158)] - [InlineData(12.3, 828, 15)] - public void ObservedAvgPlanBytes_ReproducesTheMeasuredQuartiles(double shippedMb, int plans, int expectedKb) - { - var avg = QueryStorePlanXmlState.ObservedAvgPlanBytes((long)(shippedMb * 1024 * 1024), plans); - - Assert.NotNull(avg); - Assert.Equal(expectedKb, avg!.Value / 1024); - } - - /// A pass that shipped no plans teaches nothing about plan size and must leave the previous - /// estimate standing rather than replace it with a fallback. - [Fact] - public void ObservedAvgPlanBytes_OnAQuietPass_IsNull() - { - Assert.Null(QueryStorePlanXmlState.ObservedAvgPlanBytes(0, 0)); - Assert.Null(QueryStorePlanXmlState.ObservedAvgPlanBytes(5_000, 0)); - } - - /// - /// THE POINT OF THE WHOLE REDESIGN: a budget-cut pass still advances. Under plan_id-ordered shipping a cut - /// truncates a SUFFIX, so the highest landed id is safe. The previous design shipped in - /// last_execution_time order, where a cut left an arbitrary subset, no value was safe, and the guard that - /// followed meant the watermark could not advance on 97.8% of passes. - /// - [Fact] - public void AdvanceWatermark_OnABudgetCutPass_StillAdvances() - { - var cut = QueryStorePlanXmlState.AdvanceWatermark(100, new long[] { 101, 102 }); - - Assert.Equal(102, cut.Watermark); - Assert.True(cut.ArrivedInPlanIdOrder); - } - - /// Never backward, and a quiet pass earns nothing: lowering the watermark refetches the catalog, - /// and "no new plans this window" is an ordinary pass, not a reset. - [Theory] - [InlineData(new long[0], 100L)] - [InlineData(new[] { 98L, 99L }, 100L)] - [InlineData(new[] { 101L, 102L, 103L }, 103L)] - [InlineData(new[] { 101L, 101L, 102L }, 102L)] - public void AdvanceWatermark_NeverMovesBackward(long[] landed, long expected) - { - Assert.Equal(expected, QueryStorePlanXmlState.AdvanceWatermark(100, landed).Watermark); - } - - /// - /// A descent ABANDONS the advance rather than honouring the leading ascending run. Honouring it looks - /// safer and is not: given {105, 101} it would advance to 105, and with ordering broken there is no basis - /// for inferring that every SELECTED plan below 105 landed — so a plan whose XML never arrived would be - /// suppressed until the refresh horizon. One lost pass of progress is the cheap side of that trade. - /// - [Theory] - [InlineData(new[] { 101L, 102L, 99L, 105L })] - [InlineData(new[] { 105L, 101L })] - public void AdvanceWatermark_WhenOrderingIsViolated_RefusesToAdvance(long[] landed) - { - var refused = QueryStorePlanXmlState.AdvanceWatermark(100, landed); - - Assert.Equal(100, refused.Watermark); - Assert.False(refused.ArrivedInPlanIdOrder, - "the caller needs this to LOG the violation instead of just watching the watermark stop"); - } - - /// - /// #2210: the plan fetch admits a plan when the total BEFORE it was under budget, never on the cumulative - /// total alone. The naive running_bytes <= budget is a per-database STALL — a plan bigger than the - /// whole budget exceeds it on its own row, so it is excluded, every later row is excluded too, the pass ships - /// nothing, the watermark holds, and the next pass re-selects the same plan forever. One 13 MB plan against - /// the 12 MB default does it. - /// - /// A SHAPE pin, and worth being clear about its limit: it asserts the predicate the SQL carries, not - /// what SQL Server does with it. The two behavioural cases the reviewer asked for — an oversized plan ships - /// alone and advances the watermark, and an oversized plan mid-window cuts AFTER it rather than dropping it — - /// need a real Query Store to execute and belong to the measurement session before this leaves draft. What - /// this catches is the regression that reintroduces the naive form, which is the cheap half and the half a - /// future editor is most likely to trip. - /// - [Fact] - public void PlanFetch_AdmitsAPlanOnTheTotalBeforeIt_SoOneOversizedPlanCannotStall() - { - var sql = QueryStoreCollector.Instance.BuildPlanFetchQuery( - Db, Context(capturePlanXml: true, state: new Dictionary()), - watermark: 900_000, candidatePlans: 114, budgetBytes: 12L * 1024 * 1024).Text; - - Assert.Contains("b.running_bytes - b.plan_bytes < 12582912", sql, StringComparison.Ordinal); - Assert.DoesNotContain("b.running_bytes <= ", sql, StringComparison.Ordinal); - - /* The coarse bound sorts and filters on plan_id alone — no XML touched — which is what caps the - decompression the exact bound would otherwise pay across a whole catalog. */ - /* The CONVERT sits INSIDE the window and the running total measures the converted text, so a shipped - plan is decompressed once. Measured on a 73,163-plan production catalog: this shape 133ms against - 274ms for measuring DATALENGTH(qsp.query_plan) in the window and joining back for the text, same 114 - rows and 1.7MB out of both. Plan-id-only with no XML was 114ms. */ - Assert.Contains("query_plan_text = CONVERT(nvarchar(max), qsp.query_plan)", sql, StringComparison.Ordinal); - Assert.Contains("SELECT TOP (114)", sql, StringComparison.Ordinal); - Assert.DoesNotContain("JOIN sys.query_store_plan", sql, StringComparison.Ordinal); - - /* A NULL query_plan counts as zero bytes and still ships. Letting NULL propagate would make the budget - predicate NULL, filter the row out, and a window of all-NULL plans would then hold the watermark and - re-select forever — the oversized-plan stall by another route. */ - Assert.Contains("COALESCE(DATALENGTH(c.query_plan_text), 0)", sql, StringComparison.Ordinal); - Assert.Contains("WHERE qsp.plan_id > 900000", sql, StringComparison.Ordinal); - Assert.Contains("ROWS UNBOUNDED PRECEDING", sql, StringComparison.Ordinal); - } - - /// The ordering verdict rides along with the advance on the cases that are fine. - [Theory] - [InlineData(new[] { 101L, 102L, 103L })] - [InlineData(new[] { 101L, 101L, 102L })] - [InlineData(new[] { 7L })] - [InlineData(new long[0])] - public void AdvanceWatermark_AcceptsNonDescendingArrival(long[] landed) - { - Assert.True(QueryStorePlanXmlState.AdvanceWatermark(100, landed).ArrivedInPlanIdOrder); - } -} diff --git a/Darling/Darling.Tests/QueryStoreStatePruneLivePostgresTests.cs b/Darling/Darling.Tests/QueryStoreStatePruneLivePostgresTests.cs index 8a8793fb4..14aeb71f3 100644 --- a/Darling/Darling.Tests/QueryStoreStatePruneLivePostgresTests.cs +++ b/Darling/Darling.Tests/QueryStoreStatePruneLivePostgresTests.cs @@ -36,7 +36,7 @@ public sealed class QueryStoreStatePruneLivePostgresTests /// Distinctive fake ids — a real server_id is a storage-name hash, never these. private const int LiveServerId = -218800; private const int NeighborServerId = -218801; - private const string ServerName = "PLANWM-PRUNE-SRV"; + private const string ServerName = "QSOWM-PRUNE-SRV"; /// The snapshot's collection_time in every case below; state rows are dated relative to it. private static readonly DateTime Newest = new(2026, 8, 11, 9, 0, 0, DateTimeKind.Unspecified); @@ -44,7 +44,7 @@ public sealed class QueryStoreStatePruneLivePostgresTests /// Old enough to be judged by — the ordinary case for a real state row. private static readonly DateTime BeforeNewest = Newest.AddHours(-1); - private static string Planwm(string database) => QueryStorePlanXmlState.WatermarkKeyPrefix + database; + private static string Qsowm(string database) => QueryStoreOpenIntervalState.WatermarkKeyPrefix + database; private static string Done(string database) => QueryStoreBackfillState.DoneKeyPrefix + database; private static string Hole(string database) => QueryStoreBackfillState.HoleKeyPrefix + database; @@ -82,7 +82,7 @@ sys.databases keeps it. "App" is present and "AppArchive" is not, which is the name-shape trap: writing the anti-join as starts_with(state_key, prefix || ds.database_name) instead of an equality is a very - plausible variant, and it would spare planwm:AppArchive forever because "planwm:App" is a + plausible variant, and it would spare qsowm:AppArchive forever because "qsowm:App" is a prefix of it. For an issue whose subject is database name churn, that case has to be here. */ await SnapshotAsync(connection, ct, Newest, "Live", "Parked", "App"); @@ -90,11 +90,11 @@ as starts_with(state_key, prefix || ds.database_name) instead of an equality is newest, nothing would ever be retired. */ await SnapshotAsync(connection, ct, Newest.AddMinutes(-15), "Live", "Parked", "App", "Dropped", "AppArchive"); - await StateAsync(connection, ct, LiveServerId, QueryStorePlanXmlState.StateCollectorName, Planwm("Live"), "900000:1786449600"); - await StateAsync(connection, ct, LiveServerId, QueryStorePlanXmlState.StateCollectorName, Planwm("Parked"), "800000:1786449600"); - await StateAsync(connection, ct, LiveServerId, QueryStorePlanXmlState.StateCollectorName, Planwm("Dropped"), "700000:1786449600"); - await StateAsync(connection, ct, LiveServerId, QueryStorePlanXmlState.StateCollectorName, Planwm("App"), "500000:1786449600"); - await StateAsync(connection, ct, LiveServerId, QueryStorePlanXmlState.StateCollectorName, Planwm("AppArchive"), "400000:1786449600"); + await StateAsync(connection, ct, LiveServerId, QueryStoreOpenIntervalState.StateCollectorName, Qsowm("Live"), "900000:1786449600"); + await StateAsync(connection, ct, LiveServerId, QueryStoreOpenIntervalState.StateCollectorName, Qsowm("Parked"), "800000:1786449600"); + await StateAsync(connection, ct, LiveServerId, QueryStoreOpenIntervalState.StateCollectorName, Qsowm("Dropped"), "700000:1786449600"); + await StateAsync(connection, ct, LiveServerId, QueryStoreOpenIntervalState.StateCollectorName, Qsowm("App"), "500000:1786449600"); + await StateAsync(connection, ct, LiveServerId, QueryStoreOpenIntervalState.StateCollectorName, Qsowm("AppArchive"), "400000:1786449600"); /* The backfill worker's per-database keys, which orphan identically. Both prefixes get a survivor as well as a casualty: with only a delete case, a statement that deleted @@ -112,34 +112,34 @@ a prune written as "every key of this collector" would take it. */ here — server scoping is the difference between pruning one server and pruning the fleet. */ await StateAsync(connection, ct, LiveServerId, DefaultTraceEventsCollector.Instance.Name, DefaultTraceEventsCollector.LastTraceFilePathStateKey, @"S:\MSSQL\Log\log_766.trc"); - await StateAsync(connection, ct, NeighborServerId, QueryStorePlanXmlState.StateCollectorName, - Planwm("Dropped"), "600000:1786449600"); + await StateAsync(connection, ct, NeighborServerId, QueryStoreOpenIntervalState.StateCollectorName, + Qsowm("Dropped"), "600000:1786449600"); await runner.PruneOrphanedQueryStoreDatabaseStateAsync(LiveServerId, ct); /* Retired: gone from the newest snapshot, on every prefix it could have left behind. */ - Assert.Null(await ValueAsync(connection, ct, LiveServerId, QueryStorePlanXmlState.StateCollectorName, Planwm("Dropped"))); + Assert.Null(await ValueAsync(connection, ct, LiveServerId, QueryStoreOpenIntervalState.StateCollectorName, Qsowm("Dropped"))); Assert.Null(await ValueAsync(connection, ct, LiveServerId, QueryStoreBackfillState.StateCollectorName, Done("Dropped"))); Assert.Null(await ValueAsync(connection, ct, LiveServerId, QueryStoreBackfillState.StateCollectorName, Hole("Dropped"))); /* Retired even though a LIVE database's name is a prefix of it. */ - Assert.Null(await ValueAsync(connection, ct, LiveServerId, QueryStorePlanXmlState.StateCollectorName, Planwm("AppArchive"))); + Assert.Null(await ValueAsync(connection, ct, LiveServerId, QueryStoreOpenIntervalState.StateCollectorName, Qsowm("AppArchive"))); /* Kept: still collected. */ Assert.Equal("900000:1786449600", - await ValueAsync(connection, ct, LiveServerId, QueryStorePlanXmlState.StateCollectorName, Planwm("Live"))); + await ValueAsync(connection, ct, LiveServerId, QueryStoreOpenIntervalState.StateCollectorName, Qsowm("Live"))); Assert.Equal("500000:1786449600", - await ValueAsync(connection, ct, LiveServerId, QueryStorePlanXmlState.StateCollectorName, Planwm("App"))); + await ValueAsync(connection, ct, LiveServerId, QueryStoreOpenIntervalState.StateCollectorName, Qsowm("App"))); Assert.Equal("2026-08-11T09:00:00.0000000Z", await ValueAsync(connection, ct, LiveServerId, QueryStoreBackfillState.StateCollectorName, Done("Live"))); Assert.Equal(EncodedHole(), await ValueAsync(connection, ct, LiveServerId, QueryStoreBackfillState.StateCollectorName, Hole("Live"))); /* Kept, and this is the assertion the change exists for: present in sys.databases, absent from - every enumeration query_store runs. Pruning it costs a full plan-XML refetch of a database that - never went anywhere, on precisely the servers that keep databases parked. */ + every enumeration query_store runs. Pruning it costs re-including the open interval for a + database that never went anywhere, on precisely the servers that keep databases parked. */ Assert.Equal("800000:1786449600", - await ValueAsync(connection, ct, LiveServerId, QueryStorePlanXmlState.StateCollectorName, Planwm("Parked"))); + await ValueAsync(connection, ct, LiveServerId, QueryStoreOpenIntervalState.StateCollectorName, Qsowm("Parked"))); /* Kept: not database-keyed, not this collector, not this server. */ Assert.Equal("keep me", @@ -148,7 +148,7 @@ every enumeration query_store runs. Pruning it costs a full plan-XML refetch of await ValueAsync(connection, ct, LiveServerId, DefaultTraceEventsCollector.Instance.Name, DefaultTraceEventsCollector.LastTraceFilePathStateKey)); Assert.Equal("600000:1786449600", - await ValueAsync(connection, ct, NeighborServerId, QueryStorePlanXmlState.StateCollectorName, Planwm("Dropped"))); + await ValueAsync(connection, ct, NeighborServerId, QueryStoreOpenIntervalState.StateCollectorName, Qsowm("Dropped"))); /* The DIAGNOSTIC, which is the only thing that could ever make a wrong delete visible — the other symptom is a silent refetch. It comes from the statement's RETURNING clause, so if that ever @@ -159,7 +159,7 @@ await ValueAsync(connection, ct, LiveServerId, DefaultTraceEventsCollector.Insta Assert.DoesNotContain("Parked", logger.Joined, StringComparison.Ordinal); /* Idempotent — it runs on every query_store cycle of every server, so a second pass over a clean - store must touch nothing. Seven survivors: planwm for Live, Parked and App; done and hole for + store must touch nothing. Seven survivors: qsowm for Live, Parked and App; done and hole for Live; the non-database-keyed bookkeeping row; and the other collector's key. */ await runner.PruneOrphanedQueryStoreDatabaseStateAsync(LiveServerId, ct); Assert.Equal(7, await CountAsync(connection, ct, LiveServerId)); @@ -179,8 +179,8 @@ await LiveStoreCleanup.RunAsync(connectionString!, bodySucceeded, async (cleanup /// A snapshot that EXISTS is not a snapshot that is CURRENT. If database_states stops collecting /// for a server, its newest snapshot freezes, and every database created after that instant is missing /// from it while being perfectly alive. Pruning on presence alone would delete such a database's - /// watermark on every cycle forever — paying a full plan-XML refetch each time, which is the exact cost - /// #2164 exists to remove, while logging that a live database is gone. A snapshot cannot judge a row + /// state on every cycle forever — re-deriving what the stamp existed to skip, while logging that a + /// live database is gone. A snapshot cannot judge a row /// written after it was taken. /// [Fact] @@ -206,22 +206,22 @@ public async Task Prune_LeavesStateWrittenAfterTheSnapshot_AgainstDevPostgres() await SnapshotAsync(connection, ct, Newest, "OldDb"); /* Created after the snapshot froze — absent from it, and entirely alive. */ - await StateAsync(connection, ct, LiveServerId, QueryStorePlanXmlState.StateCollectorName, - Planwm("BornAfterTheSnapshot"), "10:1786449600", updatedAt: Newest.AddMinutes(30)); + await StateAsync(connection, ct, LiveServerId, QueryStoreOpenIntervalState.StateCollectorName, + Qsowm("BornAfterTheSnapshot"), "10:1786449600", updatedAt: Newest.AddMinutes(30)); /* Dropped before the snapshot froze: absent from it, and its last state write PRECEDES it, which is what still makes it prunable. Without this the test would pass for a prune that had simply stopped working. */ - await StateAsync(connection, ct, LiveServerId, QueryStorePlanXmlState.StateCollectorName, - Planwm("DroppedLongAgo"), "20:1786449600", updatedAt: BeforeNewest); + await StateAsync(connection, ct, LiveServerId, QueryStoreOpenIntervalState.StateCollectorName, + Qsowm("DroppedLongAgo"), "20:1786449600", updatedAt: BeforeNewest); await runner.PruneOrphanedQueryStoreDatabaseStateAsync(LiveServerId, ct); Assert.Equal("10:1786449600", - await ValueAsync(connection, ct, LiveServerId, QueryStorePlanXmlState.StateCollectorName, - Planwm("BornAfterTheSnapshot"))); - Assert.Null(await ValueAsync(connection, ct, LiveServerId, QueryStorePlanXmlState.StateCollectorName, - Planwm("DroppedLongAgo"))); + await ValueAsync(connection, ct, LiveServerId, QueryStoreOpenIntervalState.StateCollectorName, + Qsowm("BornAfterTheSnapshot"))); + Assert.Null(await ValueAsync(connection, ct, LiveServerId, QueryStoreOpenIntervalState.StateCollectorName, + Qsowm("DroppedLongAgo"))); bodySucceeded = true; } @@ -262,15 +262,15 @@ public async Task Prune_WithNoDatabaseSnapshot_RetiresNothing_AgainstDevPostgres so a prune that forgot to scope the snapshot read by server would wipe every row here. */ await SnapshotAsync(connection, ct, Newest, NeighborServerId, "SomeOtherServersDatabase"); - await StateAsync(connection, ct, LiveServerId, QueryStorePlanXmlState.StateCollectorName, Planwm("Alpha"), "1:1786449600"); - await StateAsync(connection, ct, LiveServerId, QueryStorePlanXmlState.StateCollectorName, Planwm("Beta"), "2:1786449600"); + await StateAsync(connection, ct, LiveServerId, QueryStoreOpenIntervalState.StateCollectorName, Qsowm("Alpha"), "1:1786449600"); + await StateAsync(connection, ct, LiveServerId, QueryStoreOpenIntervalState.StateCollectorName, Qsowm("Beta"), "2:1786449600"); await runner.PruneOrphanedQueryStoreDatabaseStateAsync(LiveServerId, ct); Assert.Equal("1:1786449600", - await ValueAsync(connection, ct, LiveServerId, QueryStorePlanXmlState.StateCollectorName, Planwm("Alpha"))); + await ValueAsync(connection, ct, LiveServerId, QueryStoreOpenIntervalState.StateCollectorName, Qsowm("Alpha"))); Assert.Equal("2:1786449600", - await ValueAsync(connection, ct, LiveServerId, QueryStorePlanXmlState.StateCollectorName, Planwm("Beta"))); + await ValueAsync(connection, ct, LiveServerId, QueryStoreOpenIntervalState.StateCollectorName, Qsowm("Beta"))); bodySucceeded = true; } @@ -314,32 +314,32 @@ public async Task Prune_RacingAnInFlightCycle_CannotLoseTheWatermark_AgainstDevP /* A snapshot that does NOT name Racer, and a state row old enough to be judged by it — the adversarial setup, since neither is true of a real live database. */ await SnapshotAsync(connection, ct, Newest, "Live"); - await StateAsync(connection, ct, LiveServerId, QueryStorePlanXmlState.StateCollectorName, - Planwm("Racer"), "900000:1786449600", updatedAt: BeforeNewest); + await StateAsync(connection, ct, LiveServerId, QueryStoreOpenIntervalState.StateCollectorName, + Qsowm("Racer"), "900000:1786449600", updatedAt: BeforeNewest); /* Cycle start: the collection pass reads its state. */ - var loaded = await runner.GetCollectorStateAsync(LiveServerId, QueryStorePlanXmlState.StateCollectorName, ct); - Assert.Equal("900000:1786449600", Assert.Contains(Planwm("Racer"), loaded)); + var loaded = await runner.GetCollectorStateAsync(LiveServerId, QueryStoreOpenIntervalState.StateCollectorName, ct); + Assert.Equal("900000:1786449600", Assert.Contains(Qsowm("Racer"), loaded)); /* Mid-flight: the prune fires and takes the row this cycle is still working from. */ await runner.PruneOrphanedQueryStoreDatabaseStateAsync(LiveServerId, ct); - Assert.Null(await ValueAsync(connection, ct, LiveServerId, QueryStorePlanXmlState.StateCollectorName, Planwm("Racer"))); + Assert.Null(await ValueAsync(connection, ct, LiveServerId, QueryStoreOpenIntervalState.StateCollectorName, Qsowm("Racer"))); /* Cycle end: the write-back is an INSERT ... ON CONFLICT, so it restores rather than failing on a row that is no longer there. The database keeps collecting; the delete cost nothing. */ await runner.SaveCollectorStateAsync( - LiveServerId, QueryStorePlanXmlState.StateCollectorName, - new Dictionary(StringComparer.Ordinal) { [Planwm("Racer")] = "950000:1786449600" }, + LiveServerId, QueryStoreOpenIntervalState.StateCollectorName, + new Dictionary(StringComparer.Ordinal) { [Qsowm("Racer")] = "950000:1786449600" }, ct); Assert.Equal("950000:1786449600", - await ValueAsync(connection, ct, LiveServerId, QueryStorePlanXmlState.StateCollectorName, Planwm("Racer"))); + await ValueAsync(connection, ct, LiveServerId, QueryStoreOpenIntervalState.StateCollectorName, Qsowm("Racer"))); /* And it stays: the restored row is stamped NOW, which is after the snapshot, so the freshness guard keeps the next cycle's prune off it too. Without that the two would fight forever. */ await runner.PruneOrphanedQueryStoreDatabaseStateAsync(LiveServerId, ct); Assert.Equal("950000:1786449600", - await ValueAsync(connection, ct, LiveServerId, QueryStorePlanXmlState.StateCollectorName, Planwm("Racer"))); + await ValueAsync(connection, ct, LiveServerId, QueryStoreOpenIntervalState.StateCollectorName, Qsowm("Racer"))); bodySucceeded = true; } diff --git a/Darling/Darling.Tests/QueryStoreStatePruneTests.cs b/Darling/Darling.Tests/QueryStoreStatePruneTests.cs index 2c0c3e0ae..a8f6d4214 100644 --- a/Darling/Darling.Tests/QueryStoreStatePruneTests.cs +++ b/Darling/Darling.Tests/QueryStoreStatePruneTests.cs @@ -19,9 +19,10 @@ namespace Darling.Tests; /// /// #2188: retiring the per-database collector_state rows query_store leaves behind for databases the -/// server no longer has. The #2164 plan-XML watermark writes one planwm: row per database and the -/// #2022 backfill worker writes done: and hole: rows the same way, and until this nothing -/// deleted any of them for a dropped or renamed database. +/// server no longer has. The #2022 backfill worker writes done: and hole: rows per database +/// and the #2312 open-interval stamp writes qsowm: the same way, and until this nothing deleted any +/// of them for a dropped or renamed database. (The #2164/#2150 watermark families this prune originally +/// existed for retired with the watermarks themselves in #2312; V77 deleted their rows wholesale.) /// /// What actually needs pinning is the input, not the delete. A delete keyed on the wrong list is /// the failure mode: query_store's own enumeration is filtered by ONLINE state, AG primary-ness, the @@ -32,17 +33,16 @@ namespace Darling.Tests; /// is where that is proven against a real store; this /// class holds the policy and the cross-host wiring, which no store can see. /// -/// Both SKUs. Lite writes no planwm: (it never sets -/// CollectorContext.CapturePlanXml) but it DOES write done: and hole: through its own -/// backfill worker, and it only ever deletes a hole it services or expires — so the orphan class is real on -/// both sides and the prune is ported, not declared Darling-only. +/// Both SKUs. Lite writes done: and hole: through its own backfill worker, and +/// it only ever deletes a hole it services or expires — so the orphan class is real on both sides and the +/// prune is ported, not declared Darling-only. /// pins that in both directions, and the key set /// itself lives in the shared so a prefix cannot end up pruned on /// one SKU and orphaning on the other. /// public sealed class QueryStoreStatePruneTests { - private static string Planwm(string database) => QueryStorePlanXmlState.WatermarkKeyPrefix + database; + private static string Qsowm(string database) => QueryStoreOpenIntervalState.WatermarkKeyPrefix + database; /* ---------------- the design's premise, pinned without a store ---------------- */ @@ -121,13 +121,12 @@ It demands a DECISION rather than an addition. A prefix must appear in exactly o && type.Name.EndsWith("State", StringComparison.Ordinal)) .ToArray(); - Assert.Contains(typeof(QueryStorePlanXmlState), stateClasses); Assert.Contains(typeof(QueryStoreBackfillState), stateClasses); - /* #2150 added a third, and the discovery above found it without being told — which is the property + /* #2312: the open-interval stamp, found by the discovery without being told — which is the property this guard exists for. Named here anyway so a rename that quietly drops it out of the pattern - fails rather than silently shrinking the set under test. */ - Assert.Contains(typeof(QueryStoreTextState), stateClasses); - /* #2312 added a fourth, same treatment. */ + fails rather than silently shrinking the set under test. (QueryStorePlanXmlState still matches the + name pattern but declares no key prefixes any more — its watermark retired in #2312 — so it + contributes nothing to `declared`, which is exactly right.) */ Assert.Contains(typeof(QueryStoreOpenIntervalState), stateClasses); var declared = stateClasses @@ -159,19 +158,10 @@ fails rather than silently shrinking the set under test. */ /* Owner and prefix must travel together: a prefix pruned under the wrong collector_name silently deletes nothing, which looks exactly like "there was nothing to prune". */ - Assert.Contains( - (QueryStorePlanXmlState.StateCollectorName, QueryStorePlanXmlState.WatermarkKeyPrefix), - QueryStorePerDatabaseState.PrunableKeys); Assert.Contains( (QueryStoreBackfillState.StateCollectorName, QueryStoreBackfillState.HoleKeyPrefix), QueryStorePerDatabaseState.PrunableKeys); - /* #2150: paired with its OWN collector name, not the plan fetch's. The two watermarks are stored - separately on purpose (they walk different catalogs at different rates), so borrowing the plan - fetch's owner here would prune nothing and look exactly like having nothing to prune. */ - Assert.Contains( - (QueryStoreTextState.StateCollectorName, QueryStoreTextState.WatermarkKeyPrefix), - QueryStorePerDatabaseState.PrunableKeys); - /* #2312: the open-interval stamp, per database like the three above, under its own owner. */ + /* #2312: the open-interval stamp, per database like the backfill pair, under its own owner. */ Assert.Contains( (QueryStoreOpenIntervalState.StateCollectorName, QueryStoreOpenIntervalState.WatermarkKeyPrefix), QueryStorePerDatabaseState.PrunableKeys); @@ -225,15 +215,12 @@ change nobody chose. */ [Fact] public void LiteWritesTheBackfillKeysButNeverTheWatermark() { - /* The parity FACT, which the first cut of this change got wrong: Lite writes no planwm: (it never - sets CapturePlanXml) but it DOES write done: and hole: through its own backfill worker, and it - only ever deletes a hole it services or expires. So the orphan class is real on both SKUs and the - prune had to be ported, not declared Darling-only. - - Pinned at source in both directions so neither half can rot: if Lite ever starts capturing plans - it inherits a planwm: prune that is already there (the shared PrunableKeys carries the watermark - on both hosts precisely so that day needs no code change), and if Lite ever stops writing the - backfill keys this test says so rather than leaving a prune nobody needs. */ + /* The parity FACT: Lite writes done: and hole: through its own backfill worker, and it only ever + deletes a hole it services or expires. So the orphan class is real on both SKUs and the prune had + to be ported, not declared Darling-only. Pinned at source so if Lite ever stops writing the + backfill keys this test says so rather than leaving a prune nobody needs; the CapturePlanXml pin + below survives the watermark's retirement because the flag still gates the DARLING-only + activity-driven plan fetch, and Lite growing one would be a real design event. */ var root = FindRepoRoot(); Assert.True(root is not null, "repo root not found -- the source pin cannot run"); @@ -244,10 +231,9 @@ backfill keys this test says so rather than leaving a prune nobody needs. */ Assert.False( liteRunner.Contains("CapturePlanXml", StringComparison.Ordinal), - "Lite's definition runner now sets CapturePlanXml, so Lite writes planwm: rows too. The shared " - + "PrunableKeys already covers that prefix on both hosts, so the prune needs no change — but " - + "QueryStorePlanWatermarkTests.WriteBack_PlanCaptureOff_WritesNothing and this file's prose " - + "both describe Lite as never writing them, and that is now wrong."); + "Lite's definition runner now sets CapturePlanXml — the flag that gates the Darling-only " + + "activity-driven plan fetch (#2312). That is a real design event: Lite has no plan dimension " + + "or map to fetch into, so decide what the flag means there before shipping it."); foreach (var prefix in new[] { "DoneKeyPrefix", "HoleKeyPrefix" }) { @@ -265,43 +251,13 @@ still passed. */ } } - /// - /// The recreate-with-the-same-name case, which is the only shape here that could cost data rather than a - /// refetch: a dropped and recreated database restarts Query Store's plan_id numbering at 1, so every plan - /// in the NEW database sorts below the OLD database's watermark and has its XML suppressed. - /// - /// #2183 ships no reset detection — it was written, found unsound, and removed, because the - /// tempting test ("the highest plan_id seen this pass is below the standing watermark") is TRUE in any - /// ordinary window where nothing new compiled. What actually bounds this is - /// : the stamp dates the last FULL fetch, so a stale - /// watermark stops applying within a day no matter what. This test states that mechanism explicitly, so - /// the claim is a checked fact rather than a PR-description assertion. - /// - /// The prune strictly improves on that bound without replacing it — it removes the row outright - /// when the drop is observed between cycles — but it cannot be the guarantee, because a drop and recreate - /// entirely within one cycle is never observed as an absence at all. - /// - [Fact] - public void RecreatedDatabase_IsBoundedByTheRefreshHorizon_NotByResetDetection() - { - var now = new DateTime(2026, 8, 11, 12, 0, 0, DateTimeKind.Utc); - var state = new Dictionary(StringComparer.Ordinal) - { - [Planwm("Recreated")] = QueryStorePlanXmlState.Format(900_000, now - QueryStorePlanXmlState.RefreshAfter), - }; - - /* At the horizon the watermark stops applying, so the recreated database's plan_ids (which start at 1 - and would all fail a > 900000 predicate) are fetched again. */ - Assert.Equal(0, QueryStorePlanXmlState.Resolve(state, "Recreated", now)); - - /* And one second inside it, the stale watermark DOES still apply — which is the exposure this bounds, - and the reason the prune is worth having even though it is not the guarantee. */ - Assert.Equal(900_000, QueryStorePlanXmlState.Resolve(state, "Recreated", now - TimeSpan.FromSeconds(1))); - - /* A pruned row is simply absent, and absent is the conservative full-fetch path — so a recreate that - happens after an observed drop inherits nothing at all. */ - Assert.Equal(0, QueryStorePlanXmlState.Resolve(new Dictionary(StringComparer.Ordinal), "Recreated", now)); - } + /* #2312: the RecreatedDatabase_IsBoundedByTheRefreshHorizon fact that sat here retired with the + watermark. The recreate-with-the-same-name exposure it bounded (a recreated database restarts plan_id + numbering, so old state suppressed the new database's XML for up to a day) no longer exists in that + shape: the store-as-watermark probe sees a recreated database's plan_ids as unresolved-or-hash-stale + and refetches them within ONE cycle — pinned as SQL shape in QueryStorePlanFetchTests and as a live + round-trip in the gated Postgres suite. The prune keeps its own job either way: retiring rows for + databases that are gone for good. */ /// /// Walks up from the test output directory to the repo root — the same walk-up idiom diff --git a/Darling/Darling.Tests/QueryStoreTextStoreTests.cs b/Darling/Darling.Tests/QueryStoreTextStoreTests.cs index 1eea31cbc..bd56b8779 100644 --- a/Darling/Darling.Tests/QueryStoreTextStoreTests.cs +++ b/Darling/Darling.Tests/QueryStoreTextStoreTests.cs @@ -208,21 +208,19 @@ public void TheFetchIsOnForTheSweepAndOffForTheOnDemandRead() } /// - /// The text watermark is saved under its OWN state owner. The load merges both owners into one - /// dictionary, so writing it under the plan fetch's owner would still READ back — and then never be - /// pruned, because the shared prune set pairs textwm: with query_store_text and a prefix - /// pruned under the wrong owner deletes nothing. + /// #2312: the text watermark retired with the plan one — the fetch is activity-driven against this + /// store's own rows now, so there is no textwm: family to save, and the shared prune set must not + /// claim it (a prefix listed there without a writer is a standing invitation to delete nothing and + /// call it hygiene). The V77 migration deleted the orphaned rows wholesale. /// [Fact] - public void TheTextWatermarkIsSavedUnderItsOwnOwner() + public void TheTextWatermarkFamilyIsRetired() { var source = ReadRunnerSource(); - Assert.Contains("QueryStoreTextState.StateCollectorName, textKeys", source, StringComparison.Ordinal); - Assert.Contains("QueryStoreTextState.WatermarkKeyPrefix", source, StringComparison.Ordinal); - Assert.Contains( - (QueryStoreTextState.StateCollectorName, QueryStoreTextState.WatermarkKeyPrefix), - QueryStorePerDatabaseState.PrunableKeys); + Assert.DoesNotContain("textwm:", source, StringComparison.Ordinal); + Assert.DoesNotContain(QueryStorePerDatabaseState.PrunableKeys, + k => k.Prefix == "textwm:"); } /// @@ -237,7 +235,13 @@ private static int InvokeMap(object[] leading, bool hasQueryStoreText) /* #2316 and #2319 appended parameters after this rung's — pass them FALSE so these facts keep exercising the V74/V73 arms rather than the newer ones. */ - var args = leading.Concat(new object[] { hasQueryStoreText, false, false }).ToArray(); + /* Parameters appended by LATER rungs are padded FALSE, so this fact keeps exercising its own + arm rather than a newer one. Derived from the method's arity rather than listed by hand, so + a future rung does not have to edit this file -- #2357 (V78) was the fourth that would have. */ + var args = leading.Concat(new object[] { hasQueryStoreText }).ToArray(); + args = args + .Concat(Enumerable.Repeat((object)false, method.GetParameters().Length - args.Length)) + .ToArray(); Assert.Equal(method.GetParameters().Length, args.Length); return (int)method.Invoke(null, args)!; diff --git a/Darling/Darling.Tests/QueryStoreTextWatermarkTests.cs b/Darling/Darling.Tests/QueryStoreTextWatermarkTests.cs deleted file mode 100644 index 536f5bc95..000000000 --- a/Darling/Darling.Tests/QueryStoreTextWatermarkTests.cs +++ /dev/null @@ -1,165 +0,0 @@ -/* - * Copyright (c) 2026 Erik Darling, Darling Data LLC - * - * This file is part of the SQL Server Performance Monitor. - * - * Licensed under the MIT License. See LICENSE file in the project root for full license information. - */ - -using System; -using System.Collections.Generic; -using PerformanceMonitor.Collectors; -using Xunit; - -namespace Darling.Tests; - -/// -/// #2150: the per-database watermark for the query-text fetch — the sibling of -/// , pinning the same conservative-zero rules on a second -/// catalog. -/// -/// Why the text fetch exists. The runtime payload selected query_sql_text -/// (nvarchar(max)) inside a TOP ... WITH TIES ... ORDER BY last_execution_time. A Top-N Sort -/// carries every output column through the sort and reads ALL of its input before emitting a row, so -/// choosing the rows to ship materialized text for the entire qualifying set. With #2210's plan XML -/// already gone and that column as the only difference, time-to-first-row measured 4.67s against 0.45s at -/// 1,505 rows and 5.02s against 0.57s at 4,037. Neither the row cap nor the client byte budget bounds it: -/// TOP (500) measured the same as TOP (50000), and wall time was flat from a 4 MB to a -/// 256 MB budget, because the server is finished before the client sees a byte. -/// -/// Every zero below is the same deliberate choice: an absent, malformed, expired or future-stamped -/// watermark means "fetch everything", because a first run, a restarted host and a broken store are -/// indistinguishable from here and all three must refetch rather than skip. -/// -public sealed class QueryStoreTextWatermarkTests -{ - private const string Db = "SO"; - - private static DateTime Now => new(2026, 8, 16, 12, 0, 0, DateTimeKind.Utc); - - private static Dictionary StateWith(long queryId, DateTime stampedAt) => - new() { [QueryStoreTextState.KeyFor(Db)] = QueryStoreTextState.Format(queryId, stampedAt) }; - - [Fact] - public void AFreshWatermarkRoundTrips() - { - Assert.Equal(900, QueryStoreTextState.Resolve(StateWith(900, Now), Db, Now)); - Assert.Equal(Now, QueryStoreTextState.ResolveStamp(StateWith(900, Now), Db)); - } - - /// - /// Past the refresh horizon the watermark expires to 0 — a full re-walk. - /// - /// Not decoration: query_id is monotonic in FIRST-SEEN order, not in "we have stored it", - /// so a Query Store reset renumbers ids from the start and every text would arrive below a standing - /// watermark. Without a bounded horizon that suppresses text forever. - /// - [Fact] - public void PastTheRefreshHorizonItRefetchesEverything() - { - var state = StateWith(900, Now); - - Assert.Equal(900, QueryStoreTextState.Resolve(state, Db, Now + QueryStoreTextState.RefreshAfter - TimeSpan.FromMinutes(1))); - Assert.Equal(0, QueryStoreTextState.Resolve(state, Db, Now + QueryStoreTextState.RefreshAfter)); - } - - /// - /// A future stamp is refused, or a backwards clock would pin the watermark for as long as the skew - /// lasts. - /// - [Fact] - public void AFutureStampIsRefused() - => Assert.Equal(0, QueryStoreTextState.Resolve(StateWith(900, Now), Db, Now.AddHours(-1))); - - [Theory] - [InlineData("")] - [InlineData(" ")] - [InlineData("garbage")] - [InlineData("900")] - [InlineData(":123")] - [InlineData("900:")] - [InlineData("-5:123")] - [InlineData("900:notanumber")] - public void AMalformedWatermarkRefetchesEverything(string raw) - { - var state = new Dictionary { [QueryStoreTextState.KeyFor(Db)] = raw }; - - Assert.Equal(0, QueryStoreTextState.Resolve(state, Db, Now)); - Assert.Null(QueryStoreTextState.ResolveStamp(state, Db)); - } - - [Fact] - public void AnAbsentDatabaseRefetchesEverything() - { - Assert.Equal(0, QueryStoreTextState.Resolve(StateWith(900, Now), "somewhere-else", Now)); - Assert.Equal(0, QueryStoreTextState.Resolve(new Dictionary(), Db, Now)); - Assert.Equal(0, QueryStoreTextState.Resolve(null!, Db, Now)); - } - - [Fact] - public void TheWatermarkAdvancesToTheHighestLandedId() - { - var advance = QueryStoreTextState.AdvanceWatermark(100, new long[] { 101, 102, 103 }); - - Assert.Equal(103, advance.Watermark); - Assert.True(advance.ArrivedInQueryIdOrder); - } - - /// - /// Out-of-order arrival HOLDS the watermark, because the ordering is the whole safety argument: a - /// budget cut is only a suffix if the ids arrived sorted, and advancing past a gap would strand - /// unstored text behind a strict comparison permanently. - /// - [Fact] - public void OutOfOrderArrivalHoldsTheWatermark() - { - var advance = QueryStoreTextState.AdvanceWatermark(100, new long[] { 101, 99, 102 }); - - Assert.Equal(100, advance.Watermark); - Assert.False(advance.ArrivedInQueryIdOrder); - } - - /// - /// A quiet pass is a quiet pass, not a reset. Lowering the watermark because nothing new arrived would - /// refetch the catalog on every idle cycle. - /// - [Fact] - public void AQuietPassNeverLowersTheWatermark() - { - Assert.Equal(100, QueryStoreTextState.AdvanceWatermark(100, Array.Empty()).Watermark); - Assert.Equal(100, QueryStoreTextState.AdvanceWatermark(100, null!).Watermark); - Assert.Equal(100, QueryStoreTextState.AdvanceWatermark(100, new long[] { 5, 6 }).Watermark); - Assert.True(QueryStoreTextState.AdvanceWatermark(100, Array.Empty()).ArrivedInQueryIdOrder); - } - - /// - /// The stamp survives an advance, which is what makes the refresh horizon reachable at all: re-stamping - /// on every advance would push it out forever on any database that keeps seeing new statements — which - /// is exactly where a Query Store reset would hurt most. - /// - [Fact] - public void AnAdvanceCanCarryTheOriginalStampForward() - { - var originalStamp = Now.AddHours(-6); - var carried = QueryStoreTextState.Format(950, originalStamp); - var state = new Dictionary { [QueryStoreTextState.KeyFor(Db)] = carried }; - - Assert.Equal(950, QueryStoreTextState.Resolve(state, Db, Now)); - Assert.Equal(originalStamp, QueryStoreTextState.ResolveStamp(state, Db)); - /* Six hours in, the horizon is still six hours closer than a re-stamp would have left it. */ - Assert.Equal(0, QueryStoreTextState.Resolve(state, Db, originalStamp + QueryStoreTextState.RefreshAfter)); - } - - /// - /// Text and plan watermarks live under DIFFERENT collector names. They walk different catalogs at - /// different rates, and sharing state would let one side's reset drop the other's watermark for no - /// reason. - /// - [Fact] - public void TheTextWatermarkIsStoredSeparatelyFromThePlanWatermark() - { - Assert.NotEqual(QueryStorePlanXmlState.StateCollectorName, QueryStoreTextState.StateCollectorName); - Assert.NotEqual(QueryStorePlanXmlState.WatermarkKeyPrefix, QueryStoreTextState.WatermarkKeyPrefix); - Assert.NotEqual(QueryStorePlanXmlState.KeyFor(Db), QueryStoreTextState.KeyFor(Db)); - } -} diff --git a/Darling/Darling.Tests/QueryStoreTopWindowTests.cs b/Darling/Darling.Tests/QueryStoreTopWindowTests.cs new file mode 100644 index 000000000..58dd0eab9 --- /dev/null +++ b/Darling/Darling.Tests/QueryStoreTopWindowTests.cs @@ -0,0 +1,135 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.IO; +using System.Runtime.CompilerServices; +using PerformanceMonitor.Darling.Service.Mcp; +using Xunit; + +namespace Darling.Tests; + +/// +/// #2364: get_query_store_top reports the window it actually served. +/// +/// The defect. It reads raw query_store_stats, and on a store with the rollups armed that +/// table is dropped at four days — measured: a 4d 13h span over 17,162,516 rows under a +/// policy_retention {"drop_after": "4 days"}. A caller asking for 30 days therefore got at most four, +/// with hours_back: 720 echoed back unchanged and nothing marking the difference. +/// +/// Why this could not be fixed the way #2353 was. That one routed to the hourly rollup. Here there +/// is no route: QueryStoreTopSql groups by database_name, query_id, plan_id, query_hash, +/// replica_role and the corrected CAGG groups by database_name, module_name, query_hash — no +/// query_id, no plan_id. Plan identity is the entire purpose of this tool, and a rollup grained to +/// it would approach the size of the raw data. So the fix is honesty, not routing. +/// +public class QueryStoreTopWindowTests +{ + private static string ReaderSource => ReadRepoFile(Path.Combine( + "Darling", "PerformanceMonitor.Darling.Service", "Mcp", "DarlingDataReader.cs")); + + private static string ToolSource => ReadRepoFile(Path.Combine( + "Darling", "PerformanceMonitor.Darling.Service", "Mcp", "DarlingMcpDataTools.cs")); + + /// + /// The floor probe is bounded on BOTH sides of the window. Bounding the partitioning column is what lets + /// TimescaleDB exclude chunks; an unbounded MIN would read the retention window to answer a question + /// about it. + /// + [Fact] + public void TheWindowFloorProbe_IsBoundedOnBothSides() + { + var sql = DarlingDataReader.QueryStoreWindowFloorSql; + + Assert.Contains("MIN(collection_time)", sql, StringComparison.Ordinal); + Assert.Contains("FROM query_store_stats", sql, StringComparison.Ordinal); + Assert.Contains("collection_time >=", sql, StringComparison.Ordinal); + Assert.Contains("collection_time <=", sql, StringComparison.Ordinal); + } + + /// + /// The floor must come from a probe, never from the returned rows. The result set is the top N by + /// COST, so its timestamps say nothing about how far back the read reached — the most expensive query in a + /// month may have run this morning. Deriving the window from the rows would produce a confident, wrong + /// answer, which is worse than the silence it replaces. + /// + [Fact] + public void TheEffectiveWindow_ComesFromTheProbe_NotTheRows() + { + Assert.Contains("GetQueryStoreWindowFloorAsync", ToolSource, StringComparison.Ordinal); + + var at = ToolSource.IndexOf("var effectiveStart =", StringComparison.Ordinal); + Assert.True(at >= 0, "the tool no longer computes an effective window (#2364)"); + + var line = ToolSource[at..ToolSource.IndexOf(';', at)]; + Assert.Contains("floor", line, StringComparison.Ordinal); + + /* Not from the projection: rows are ordered by cost, and LastExecutionTime is a per-query fact. */ + Assert.DoesNotContain("rows.Min(", ToolSource, StringComparison.Ordinal); + Assert.DoesNotContain("rows.Max(", ToolSource, StringComparison.Ordinal); + } + + /// The payload describes the data, not just the request. + [Fact] + public void ThePayload_CarriesTheServedWindow() + { + Assert.Contains("effective_start", ToolSource, StringComparison.Ordinal); + Assert.Contains("effective_hours_back", ToolSource, StringComparison.Ordinal); + Assert.Contains("truncated", ToolSource, StringComparison.Ordinal); + + /* hours_back is still echoed — the caller needs to see what it asked for beside what it got. */ + Assert.Contains("hours_back,", ToolSource, StringComparison.Ordinal); + } + + /// + /// The empty path must not assert absence over a span it never read. The old message said Query Store "may + /// not be enabled", which for a window reaching past raw retention is a confident wrong diagnosis — an agent + /// acts on it by going to look at a Query Store configuration that is fine. + /// + [Fact] + public void TheEmptyPath_NamesTheWindowSearched_AndDoesNotOnlyBlameConfiguration() + { + var at = ToolSource.IndexOf("No Query Store rows for this server", StringComparison.Ordinal); + Assert.True(at >= 0, "the empty message no longer names the window searched (#2364)"); + + var message = ToolSource[at..Math.Min(ToolSource.Length, at + 700)]; + + Assert.Contains("hours_back", message, StringComparison.Ordinal); + Assert.Contains("4 days", message, StringComparison.Ordinal); + Assert.Contains("shorter window", message, StringComparison.Ordinal); + } + + /// + /// The tool still reads RAW, and that is deliberate rather than an oversight — pinned so a well-meaning + /// "route it to the rollup like #2353" edit has to confront why it cannot. + /// + [Fact] + public void TheTool_StillReadsRaw_BecauseTheRollupCannotCarryPlanIdentity() + { + Assert.Contains("FROM query_store_stats", DarlingDataReader.QueryStoreTopSql, StringComparison.Ordinal); + Assert.DoesNotContain("query_store_stats_corrected", DarlingDataReader.QueryStoreTopSql, StringComparison.Ordinal); + Assert.DoesNotContain("query_store_stats_hourly", DarlingDataReader.QueryStoreTopSql, StringComparison.Ordinal); + + /* plan_id is the reason: it is a grouping key here and absent from every rollup. */ + Assert.Contains("plan_id", DarlingDataReader.QueryStoreTopSql, StringComparison.Ordinal); + } + + private static string ReadRepoFile(string relative, [CallerFilePath] string thisFile = "") + { + for (var dir = new DirectoryInfo(Path.GetDirectoryName(thisFile)!); dir is not null; dir = dir.Parent) + { + var candidate = Path.Combine(dir.FullName, relative); + if (File.Exists(candidate)) + { + return File.ReadAllText(candidate); + } + } + + throw new FileNotFoundException($"Could not locate {relative} walking up from {thisFile}"); + } +} diff --git a/Darling/Darling.Tests/StoreSelfMetricsTests.cs b/Darling/Darling.Tests/StoreSelfMetricsTests.cs index f8e24231d..618e78835 100644 --- a/Darling/Darling.Tests/StoreSelfMetricsTests.cs +++ b/Darling/Darling.Tests/StoreSelfMetricsTests.cs @@ -305,32 +305,50 @@ public async Task Sweep_EndToEnd_WritesEveryObjectKind_IncludingBackgroundJobs_A /* ---------------- #2136 synthetic scale test ---------------- */ /// - /// The #2136 capacity claim, proven end to end rather than asserted from one production observation: - /// job runtimes scale with raw volume, the V56 telemetry RECORDS that growth, and the #2141 alert - /// FIRES from real store readings. One throwaway hypertable with a compression policy that is PARKED - /// except when a measurement deliberately arms it (created parked in one transaction — the #1888 - /// discipline — so no background tick ever races a measurement, the #2143 class), driven at 1x and - /// then 10x row volume: + /// The #2136 capacity claim, proven end to end rather than asserted from one production observation. + /// One throwaway hypertable with a compression policy that is PARKED except when a measurement + /// deliberately arms it (created parked in one transaction — the #1888 discipline — so no background + /// tick ever races a measurement, the #2143 class), driven at 1x and then 10x row volume: /// - /// a scheduler-driven run at each scale (arm, poll total_runs, park — foreground run_job does - /// NOT update this accounting, CI-proved); job_stats.last_run_duration must be measurable (the - /// premise the whole telemetry stands on) and must GROW with volume; + /// a scheduler-driven run at each scale (arm, poll last_successful_finish, park — foreground + /// run_job does NOT update this accounting, CI-proved). Each run must COMPRESS THE CHUNK ITS SEED + /// CREATED, and that chunk must hold exactly the seeded row count, so the escalation is real work at + /// two genuinely different scales rather than two no-ops; and job_stats.last_run_duration must be + /// measurable at BOTH scales — the premise the whole telemetry stands on; /// a self-metrics sweep after each run; the store_metrics series must carry both readings, in - /// order, growing — this is the series an operator (and the cadence alert's detail text) trends; + /// order — this is the series an operator (and the cadence alert's detail text) trends; /// alter_job shrinks the schedule interval to half the measured 10x duration, and the REAL /// evaluator, fed by the REAL against this /// store, must fire the Critical tier under the storejob: key. /// - /// Volumes (50k vs 500k rows in one closed chunk each, after a discarded warm-up run) are chosen so - /// the big run does strictly more compression work than the 1x run by a margin no runner jitter - /// plausibly inverts; the assertion is monotonicity, not a ratio, for exactly that reason. The - /// margin is a full order of magnitude because 4x was NOT enough (#2160): a fast runner's fixed - /// per-run cost plus cache warmth accumulating across the two measured runs inverted 50k-vs-200k - /// in the field (d1=279ms, d4=217ms). - /// Seeds are midday-anchored (#1972) so a run near midnight cannot split a chunk. + /// Seeds are midday-anchored (#1972) so a run near midnight cannot split a chunk. Ordering the + /// per-day counts ascending is safe across a midnight rollover too: the anchors are re-evaluated per + /// seed, so warm-up stays the oldest day and 10x the newest whichever side of midnight each lands on. + /// + /// #2266: there is deliberately NO assertion that the 10x run took LONGER than the 1x run, + /// and one must not be reintroduced. That assertion was the flake, and it is unfixable by tuning + /// because it is a benchmark of TimescaleDB's compression throughput on shared CI hardware, not a + /// claim about this product. Measured on a rig (TimescaleDB 2.29 / PG17, 15 consecutive runs of this + /// exact sequence): d1 lands at 24–39 ms and d10 at 109–167 ms, so a 10x volume increase buys only + /// ~3.2x the duration — about 85 ms of absolute signal, because compression cost is largely + /// fixed per run. CI's observed baseline for the same pair is 690–970 ms, i.e. roughly twenty times + /// that fixed cost, so the volume-dependent component there is ~10% of the measurement's own + /// magnitude and sits comfortably inside the run-to-run variance of launching a background worker on + /// Windows. Both reported failures are exactly that: d1=970/d10=863 and then d1=689/d10=689. The + /// earlier reading of the byte-identical pair as proof of a mechanism (both runs compressing nothing) + /// is refuted by the rig — chunk counts go 1, 2, 3 and the per-day counts are exactly + /// 2000/50000/500000 on all 15 runs — and it was never as improbable as it looked, because the pair + /// is only ever read when the test FAILS, which selects for differences already near zero. + /// Raising the volumes cannot rescue it either: at ~0.19 ms per thousand rows it would take millions + /// of rows per chunk to clear a variance nobody has measured on the platform that actually fails. + /// What the product owns is that a real duration is measured, recorded in order, and drives the + /// cadence alert — all three asserted below, deterministically. What TimescaleDB owns is how long + /// compressing a chunk takes, and this suite is not the place to police it. #2160's 4x-was-not-enough + /// finding (d1=279ms, d4=217ms) was the same signal being read as a volume problem; 10x did not fix + /// it and no multiple would have. /// [Fact] - public async Task ScaleTest_JobDurationGrowsWithVolume_TelemetryRecordsIt_AndTheAlertFires_AgainstDevPostgres() + public async Task ScaleTest_EachRunCompressesItsOwnChunk_TelemetryRecordsBothRuns_AndTheAlertFires_AgainstDevPostgres() { var baseConnectionString = Environment.GetEnvironmentVariable("DARLING_TEST_PG"); Assert.SkipWhen(string.IsNullOrEmpty(baseConnectionString), @@ -377,9 +395,11 @@ FROM timescaledb_information.jobs AND (proc_name LIKE '%compression%' OR proc_name LIKE '%columnstore%')", connection).ExecuteScalarAsync(ct))!); /* Warm-up: the first run of a policy pays one-time costs (worker spin-up, catalog warm-up) that - would inflate d1 and could invert the monotonicity assertion. Run once on a token chunk and - discard the measurement. Doubles as the canary that this scratch database HAS a scheduler: - if it never runs, the arm-and-poll below fails with its own diagnosis rather than a mystery. */ + would swamp d1. Run once on a token chunk and discard the measurement. Doubles as the canary + that this scratch database HAS a scheduler: if it never runs, the arm-and-poll below fails with + its own diagnosis rather than a mystery. It also establishes the compressed-chunk baseline the + two measured runs are counted against, so a warm-up that silently compressed nothing shows up + as a wrong count after 1x rather than as a mystery duration. */ await SeedTickRowsAsync(connection, Table, daysBack: 12, rows: 2_000, ct); await RunJobViaSchedulerAsync(connection, jobId, ct); @@ -387,42 +407,69 @@ would inflate d1 and could invert the monotonicity assertion. Run once on a toke await SeedTickRowsAsync(connection, Table, daysBack: 10, rows: 50_000, ct); await RunJobViaSchedulerAsync(connection, jobId, ct); long d1 = await ReadJobDurationMsAsync(connection, jobId, ct); - var work1 = await DescribeJobWorkAsync(connection, Table, jobId, ct); /* #2266 */ - Assert.True(d1 > 0, - "a scheduler-driven run left job_stats.last_run_duration unmeasurable — the premise the " + - "V56 telemetry and the #2141 alert both stand on. (Foreground run_job is already known " + - "not to update this accounting — CI proved that on this test's first version — which is " + - "why the runs go through the real scheduler.)"); + var work1 = await ReadCompressionWorkAsync(connection, Table, ct); + /* Branch rather than pass the describe call into Assert.True's message: that argument is a plain + string, so it is evaluated eagerly and would spend two live catalog queries on every PASSING run + to build a message nobody reads (review catch). The helper exists to explain a failure, so it + should only run when there is one. */ + if (d1 <= 0) + { + Assert.Fail( + "a scheduler-driven run left job_stats.last_run_duration unmeasurable — the premise the " + + "V56 telemetry and the #2141 alert both stand on. (Foreground run_job is already known " + + "not to update this accounting — CI proved that on this test's first version — which is " + + "why the runs go through the real scheduler.)" + + $"\n what the job did: {await DescribeJobWorkAsync(connection, Table, jobId, ct)}"); + } + await StoreSelfMetrics.SweepAsync(connection, timescaleAvailable: true, DateTime.UtcNow, null, ct); /* 10x: one closed chunk, 500k rows. */ await SeedTickRowsAsync(connection, Table, daysBack: 8, rows: 500_000, ct); await RunJobViaSchedulerAsync(connection, jobId, ct); long d10 = await ReadJobDurationMsAsync(connection, jobId, ct); - var work10 = await DescribeJobWorkAsync(connection, Table, jobId, ct); /* #2266 */ + var work10 = await ReadCompressionWorkAsync(connection, Table, ct); + if (d10 <= 0) + { + /* Same eager-evaluation reason as the d1 branch above. */ + Assert.Fail( + "the 10x run left job_stats.last_run_duration unmeasurable. Asserted separately from d1 " + + "(#2266): ReadJobDurationMsAsync maps a NULL duration to 0, and the telemetry check below " + + "compares the series against these same variables, so an unmeasurable 10x run used to " + + "satisfy 0 == 0 and pass." + + $"\n what the job did: {await DescribeJobWorkAsync(connection, Table, jobId, ct)}"); + } + await StoreSelfMetrics.SweepAsync(connection, timescaleAvailable: true, DateTime.UtcNow.AddSeconds(2), null, ct); - /* 1. The capacity claim itself: more volume, longer run. Monotonicity, not a ratio — runner - jitter owns the constant factor, the direction is ours. - - #2266: the failure message now reports what the job DID, not only how long it took. This test - has failed intermittently on diffs that cannot reach it, and the reading that mattered was - d1=689ms / d10=689ms — BYTE-IDENTICAL. Two independent sub-second timings of different - workloads do not land on the same millisecond by chance, so the earlier "runner jitter" - explanation cannot be right; something is making both runs do the same work. The scheduler - helper already rules out a stale read (it waits for last_successful_finish to ADVANCE), which - leaves "both runs compressed the same amount, plausibly none" — and that is invisible from a - duration alone. Chunk counts make it visible the first time it recurs, without a rig. */ - Assert.True(d10 > d1, - $"10x volume did not run longer than 1x (d1={d1}ms, d10={d10}ms) — job runtime is not " + - "scaling with volume, which invalidates the #2136 capacity model." + - $"\n after 1x ({50_000} rows seeded): {work1}" + - $"\n after 10x ({500_000} rows seeded): {work10}" + - "\n If the compressed-chunk counts are EQUAL, the two runs did the same work and this " + - "assertion was never measuring the capacity model — the volumes are not producing " + - "compressible chunks, which is a fixture defect rather than a timing tolerance one (#2266)."); - - /* 2. The telemetry recorded the growth: two series points for this job, in order, growing. */ + /* 1. The escalation is REAL WORK at two different scales — asserted on rows and chunks, which are + exact, instead of on the two durations, which are a benchmark of somebody else's compression + engine (see the #2266 block in the summary for the measurements that settle that). Each measured + run must have compressed the chunk its own seed created, and that chunk must hold exactly the + seeded row count. Per-day counts double as the "one chunk per seed" check: 1-day chunks make day + groups and chunks the same thing, so a seed that straddled midnight would show up as an extra + group rather than as a quietly halved workload. + + This is the assertion the durations were standing in for, and it is strictly stronger: the + hypothesis the intermittent failures raised — that both runs compressed nothing and the whole + cost was fixed overhead — is a hard failure here, at the step where it happens, instead of + being invisible behind a timing comparison that fails for two unrelated reasons. */ + Assert.Equal(new long[] { 2_000, 50_000 }, work1.RowsPerDay); + Assert.Equal(2, work1.ChunksTotal); + Assert.True(work1.ChunksCompressed == 2, + $"the 1x run did not leave both chunks compressed ({work1}) — the 50k seed did not become " + + "compressible work, so this test would be measuring fixed overhead twice rather than the " + + $"#2136 capacity model (#2266). d1={d1}ms."); + + Assert.Equal(new long[] { 2_000, 50_000, 500_000 }, work10.RowsPerDay); + Assert.Equal(3, work10.ChunksTotal); + Assert.True(work10.ChunksCompressed == 3, + $"the 10x run did not leave all three chunks compressed ({work10}) — the 500k seed did not " + + $"become compressible work (#2266). d1={d1}ms, d10={d10}ms."); + + /* 2. The telemetry recorded both runs: two series points for this job, in order, carrying the + durations the job actually reported. This is the product's half of #2136 — whatever duration + TimescaleDB took, the V56 series has it, in order, ready for the cadence comparison in step 3. */ await using (var series = new NpgsqlCommand(@" SELECT last_run_duration_ms FROM collect.store_metrics @@ -515,6 +562,62 @@ a read of a finished run's accounting. */ await ExecAsync(connection, $"SELECT alter_job({jobId}::integer, scheduled => false)", ct); } + /// + /// What the compression job measurably ACHIEVED, as exact counts the scale test asserts on (#2266) — + /// as opposed to , which is a best-effort string for explaining a + /// failure and deliberately swallows its own faults. This one THROWS, because a Timescale view that + /// stopped answering is a real failure of the thing being asserted rather than a cosmetic gap in a + /// message. + /// + /// RowsPerDay is ordered by day ascending, which — with the 1-day chunk interval this + /// hypertable is created at — makes it both the per-chunk row census and the "each seed produced + /// exactly one chunk" check. Counting rows through the hypertable rather than reading a compression + /// stats view is deliberate: compressed chunks stay transparently queryable, so a plain + /// count(*) is exact and needs none of the pre/post-2.18 columnstore-vs-compression view + /// vocabulary the rest of this file has to hedge on. + /// + private sealed record CompressionWork(long ChunksTotal, long ChunksCompressed, long[] RowsPerDay) + { + public override string ToString() => + $"chunks={ChunksTotal} compressed={ChunksCompressed} rowsPerDay=[{string.Join(", ", RowsPerDay)}]"; + } + + private static async Task ReadCompressionWorkAsync( + NpgsqlConnection connection, string table, CancellationToken ct) + { + long total; + long compressed; + await using (var chunks = new NpgsqlCommand(@" +SELECT + count(*) AS chunks_total, + count(*) FILTER (WHERE is_compressed) AS chunks_compressed +FROM timescaledb_information.chunks +WHERE hypertable_schema = 'collect' AND hypertable_name = $1", connection)) + { + chunks.Parameters.AddWithValue(table); + await using var reader = await chunks.ExecuteReaderAsync(ct); + Assert.True(await reader.ReadAsync(ct), "timescaledb_information.chunks returned no row"); + total = reader.GetInt64(0); + compressed = reader.GetInt64(1); + } + + var rowsPerDay = new List(); + await using (var perDay = new NpgsqlCommand($@" +SELECT count(*) AS row_count +FROM collect.{table} +GROUP BY date_trunc('day', collection_time) +ORDER BY date_trunc('day', collection_time)", connection)) + { + await using var reader = await perDay.ExecuteReaderAsync(ct); + while (await reader.ReadAsync(ct)) + { + rowsPerDay.Add(reader.GetInt64(0)); + } + } + + return new CompressionWork(total, compressed, rowsPerDay.ToArray()); + } + private static async Task ReadLastSuccessfulFinishAsync(NpgsqlConnection connection, long jobId, CancellationToken ct) { /* -infinity (never finished) maps to DateTime.MinValue via Npgsql, which orders below every real @@ -528,12 +631,14 @@ private static async Task ReadLastSuccessfulFinishAsync(NpgsqlConnecti } /// - /// What the compression job actually DID, as one line for a failure message (#2266). + /// What the compression job actually DID, as one line for a failure message (#2266) — the job-side + /// context (total_runs, last_run_status, last_successful_finish) that says whether a + /// run happened at all and whether it succeeded. Attached to the two duration-measurability assertions, + /// which are the ones where "did the run even complete" is the question a reader has next. /// - /// Added because a duration alone cannot distinguish "this run compressed ten times as much and the - /// machine was noisy" from "both runs compressed nothing and the cost is all fixed overhead" — and the - /// intermittent failures of this test have produced BYTE-IDENTICAL durations, which only the second story - /// explains. Reporting chunk counts turns the next recurrence into a diagnosis instead of another re-run. + /// Distinct from , which returns exact counts the test + /// ASSERTS on. The split is the point: this one is prose for a human reading a failure, so it must never + /// throw, and that same property makes it unfit to assert against. /// /// Deliberately best-effort and never throwing: it exists to explain a failure, so a fault here must /// not replace the assertion's own message with its own — that is the #1902 mistake in miniature. A missing @@ -650,6 +755,12 @@ private sealed class CadenceFakeSettings : IAlertEngineSettings public int CollectionFailureThreshold { get; set; } = 10; public int PvsThresholdPercent { get; set; } = 40; public int PvsFloorGb { get; set; } = 1; + + /* #2349: OFF in the fakes so existing expectations are untouched. */ + public bool FileGrowthEnabled { get; set; } + public int FileGrowthRiseMb { get; set; } = 10240; + public int FileGrowthVolumePercent { get; set; } = 60; + public int FileGrowthLookbackMinutes { get; set; } = 60; public int LongRunningJobMultiplier { get; set; } = 3; public int FailedJobLookbackMinutes { get; set; } = 60; public int CooldownMinutes { get; set; } = 5; diff --git a/Darling/Darling.Tests/ViewerServerInventoryEnabledTests.cs b/Darling/Darling.Tests/ViewerServerInventoryEnabledTests.cs new file mode 100644 index 000000000..0e05c0c92 --- /dev/null +++ b/Darling/Darling.Tests/ViewerServerInventoryEnabledTests.cs @@ -0,0 +1,168 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.IO; +using System.Linq; +using System.Runtime.CompilerServices; +using Xunit; + +namespace Darling.Tests; + +/// +/// #2359: Server Inventory says whether a server is still monitored, so an old Last Updated is legible. +/// +/// The bug. The grid lists every REGISTERED server and joined servers without ever reading +/// is_enabled. A decommissioned server therefore kept the collection_time it had when monitoring +/// stopped — accurate, and read by every operator as a broken freshness column. Measured on a 61-server fleet the +/// split was total: 19 disabled servers all last collected within five minutes of each other on the day they were +/// removed, and 42 enabled servers all fresh within a minute of each other. Nothing was stale; nineteen things +/// were finished. +/// +/// The rows are kept rather than filtered out. This is the FinOps tab, and a decommissioned server's cost +/// history is exactly what someone opens it to look at — dropping them would trade a confusing grid for a lying +/// one. +/// +public class ViewerServerInventoryEnabledTests +{ + private static string InventorySource => ReadRepoFile(Path.Combine( + "Darling", "PerformanceMonitor.Darling.Viewer", "ViewerDataService.FinOps.Inventory.cs")); + + /// + /// The projection carries is_enabled, and it comes from the REGISTRY rather than from the properties + /// snapshot — server_properties has no such column, and a disabled server's newest snapshot is + /// precisely the row that cannot tell you it is disabled. + /// + [Fact] + public void TheInventoryQuery_SelectsIsEnabledFromTheRegistry() + { + Assert.Contains("s.is_enabled", InventorySource, StringComparison.Ordinal); + } + + /// + /// The whole fleet a person still runs sorts first. A grid that interleaves nineteen finished servers with + /// forty-two live ones is the same confusion in a different arrangement. + /// + [Fact] + public void TheInventoryQuery_SortsLiveServersFirst() + { + Assert.Contains("ORDER BY s.is_enabled DESC, server_name", InventorySource, StringComparison.Ordinal); + } + + /// + /// The ordinal shift, pinned. Inserting a column mid-projection moves every ordinal after it, and a + /// positional reader does not fail when that happens — it keeps reading, one column off, and turns a cost + /// into a boolean. That exact mistake shipped once already in the Aurora detection query, so the + /// neighbours are asserted by number rather than trusted. + /// + /// #2359's last_collection was appended LAST for the same reason: at ordinal 19 it moves + /// nothing that came before it. + /// + [Fact] + public void TheReader_ReadsIsEnabledAt17_MonthlyCostAt18_AndLastCollectedAt19() + { + Assert.Contains("IsEnabled = reader.IsDBNull(17)", InventorySource, StringComparison.Ordinal); + Assert.Contains("MonthlyCost = reader.IsDBNull(18)", InventorySource, StringComparison.Ordinal); + Assert.Contains("LastCollected = reader.IsDBNull(19)", InventorySource, StringComparison.Ordinal); + + /* And nothing still reads the pre-shift position for the cost. */ + Assert.DoesNotContain("MonthlyCost = reader.IsDBNull(17)", InventorySource, StringComparison.Ordinal); + } + + /// + /// #2359, the actual reported bug: server_properties ships with FrequencyMinutes 0 — "collect + /// once on server load only" — so its timestamp is the last service start, not a heartbeat. Calling it + /// Last Updated made every actively-monitored server on a long-running install look stale. + /// + /// The field is named for what it is, and the value people were actually asking for is carried + /// alongside it rather than instead of it — a decommissioned server still needs its snapshot time. + /// + [Fact] + public void TheGrid_SeparatesTheConfigSnapshotFromRealFreshness() + { + var xaml = ReadRepoFile(Path.Combine("Darling", "PerformanceMonitor.Darling.Viewer", "FinOpsTab.xaml")); + + Assert.Contains("{Binding InventoryAsOf", xaml, StringComparison.Ordinal); + Assert.Contains("Text=\"Inventory As Of\"", xaml, StringComparison.Ordinal); + Assert.Contains("{Binding LastCollected", xaml, StringComparison.Ordinal); + Assert.Contains("Text=\"Last Collected\"", xaml, StringComparison.Ordinal); + + /* The misleading label must not come back. */ + Assert.DoesNotContain("Text=\"Last Updated\"", xaml, StringComparison.Ordinal); + Assert.DoesNotContain("{Binding LastUpdated", xaml, StringComparison.Ordinal); + } + + /// + /// Freshness comes from the collection log, not from the properties snapshot — reading it off + /// server_properties would just reproduce the bug under a better column name. + /// + [Fact] + public void FreshnessComesFromTheCollectionLog() + { + Assert.Contains("FROM v_collection_log", InventorySource, StringComparison.Ordinal); + Assert.Contains("AS last_collection", InventorySource, StringComparison.Ordinal); + } + + /// + /// SELECT order and reader order have to agree, so the count is derived from the SQL rather than restated: + /// is_enabled must sit immediately before monthly_cost_usd, which is what makes 17/18 correct. + /// + [Fact] + public void TheProjection_PutsIsEnabledImmediatelyBeforeMonthlyCost() + { + var enabled = InventorySource.IndexOf("s.is_enabled", StringComparison.Ordinal); + var cost = InventorySource.IndexOf("COALESCE(s.monthly_cost_usd, 0) AS monthly_cost_usd", StringComparison.Ordinal); + + Assert.True(enabled > 0 && cost > 0, "both columns must be present"); + Assert.True(enabled < cost, "is_enabled must be selected before monthly_cost_usd"); + + var between = InventorySource[enabled..cost]; + Assert.DoesNotContain(",", between[(between.IndexOf(',', StringComparison.Ordinal) + 1)..], StringComparison.Ordinal); + } + + /// + /// Disabled rows are KEPT. A future edit that "fixes" the confusing dates by filtering them away has to + /// argue with this: it would silently drop cost history from the tab that exists to show cost history. + /// + [Fact] + public void TheInventoryQuery_DoesNotFilterOutDisabledServers() + { + Assert.DoesNotContain("WHERE s.is_enabled", InventorySource, StringComparison.Ordinal); + Assert.DoesNotContain("AND s.is_enabled", InventorySource, StringComparison.Ordinal); + Assert.DoesNotContain("is_enabled = true", InventorySource, StringComparison.Ordinal); + } + + /// The grid actually shows it — a flag nobody can see fixes nothing. + [Fact] + public void TheGrid_ShowsTheMonitoringColumn() + { + var xaml = ReadRepoFile(Path.Combine("Darling", "PerformanceMonitor.Darling.Viewer", "FinOpsTab.xaml")); + + Assert.Contains("{Binding MonitoringStatus}", xaml, StringComparison.Ordinal); + + /* #2181/#2331: a column carrying an explicit Style whose key lives in MainWindow.xaml's window + resources resolves at parse time and throws when the grid is realized. This one carries none. */ + var at = xaml.IndexOf("{Binding MonitoringStatus}", StringComparison.Ordinal); + var column = xaml[(xaml.LastIndexOf('<', at))..(xaml.IndexOf("/>", at, StringComparison.Ordinal) + 2)]; + Assert.DoesNotContain("StaticResource", column, StringComparison.Ordinal); + } + + private static string ReadRepoFile(string relative, [CallerFilePath] string thisFile = "") + { + for (var dir = new DirectoryInfo(Path.GetDirectoryName(thisFile)!); dir is not null; dir = dir.Parent) + { + var candidate = Path.Combine(dir.FullName, relative); + if (File.Exists(candidate)) + { + return File.ReadAllText(candidate); + } + } + + throw new FileNotFoundException($"Could not locate {relative} walking up from {thisFile}"); + } +} diff --git a/Darling/Darling.Tests/packages.lock.json b/Darling/Darling.Tests/packages.lock.json index d870e712d..ca736f2fb 100644 --- a/Darling/Darling.Tests/packages.lock.json +++ b/Darling/Darling.Tests/packages.lock.json @@ -2,22 +2,6 @@ "version": 2, "dependencies": { "net10.0-windows7.0": { - "Microsoft.NET.Test.Sdk": { - "type": "Direct", - "requested": "[18.8.1, )", - "resolved": "18.8.1", - "contentHash": "dknJL3/9Y3t4XuCBqnc0PevPxgLsUMmVhjwup/b1HNovA8zWcj3XsfIf7c6p05363DWcqL7X/YhDL9B+Zymv1w==", - "dependencies": { - "Microsoft.CodeCoverage": "18.8.1", - "Microsoft.TestPlatform.TestHost": "18.8.1" - } - }, - "xunit.runner.visualstudio": { - "type": "Direct", - "requested": "[3.1.5, )", - "resolved": "3.1.5", - "contentHash": "tKi7dSTwP4m5m9eXPM2Ime4Kn7xNf4x4zT9sdLO/G4hZVnQCRiMTWoSZqI/pYTVeI27oPPqHBKYI/DjJ9GsYgA==" - }, "xunit.v3": { "type": "Direct", "requested": "[3.2.2, )", @@ -66,11 +50,6 @@ "resolved": "9.0.13", "contentHash": "5T+bH3Lb1nEe8Hf/ixMxLmhlrx5wRi53wv7OhVwG2F1ZviW1ejFRS1NHur3uqPpJRGtkQwUchtY6zhVK2R+v+w==" }, - "Microsoft.CodeCoverage": { - "type": "Transitive", - "resolved": "18.8.1", - "contentHash": "Eclse/ZZjr4lmWzZFNN9h/OluhKL+SK/QbUyKUewgX139aGeyMEO/DkMPwuFs2MixvanTnz6891rF8UHDg+W4Q==" - }, "Microsoft.Data.SqlClient.Extensions.Abstractions": { "type": "Transitive", "resolved": "7.0.2", @@ -116,215 +95,215 @@ }, "Microsoft.Extensions.Configuration.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "5Vnd2I75DmZCVEjSynIdJ/0EGafgnLQwgR3t2C2/fkjx/nRG+cLwxLLdInoHeCEpkD5K4Ov/g9ZCRYrl4TRsaA==", + "resolved": "10.0.11", + "contentHash": "fVi053xdpda9Em7vSkmgVxO/PtgC2m78ekReKWsgcyskqY0U82Bz/MONwxpGzI0hElYKJfw+fupqMVeKW3fSaA==", "dependencies": { - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Configuration.Binder": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "GqmN2o1CkJvk7uWp+p4CwBYW0w/zfoEbvsiFDbO2G8l1Uz+mrDAbAcZiXhU2lufKPby1cjAUdd5GTWpebYOkOA==", + "resolved": "10.0.11", + "contentHash": "rFn8RuszZn3qquPVkDytMUlPc2+rXl9MCoygwc1XmAgC5vg5/oXJ8hkOosOrLoBLsqdTy4lFwP6iQdPS9uSYOA==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Configuration.CommandLine": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "33cBeR2HRbzHUTtmcmLdNOApneNGcymwwL4arHuotgVK9Frba8kcDTrvVTj7cSCmF1R9OiSbZH0KxNOwab3HUg==", + "resolved": "10.0.11", + "contentHash": "1KHr/1L56llwQ/yI0tAisEA31UpPsn8aasjASIwELOaN4JIUcbjuQBMdFOIzfNBBeULoUa0XfBe5QDtRRUY+fg==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Configuration.EnvironmentVariables": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "KRfFSSCV58vEdU7mPED/YMzeovIWF5P0g8s9K8n9HEfy0/WzMq37SrPdXdFN5/dFT/rPMHpF7AvpoXHckbcBFg==", + "resolved": "10.0.11", + "contentHash": "KICyU3eVi5jvloKm01EXV69L97H/zkhISVtV98cIuzuFOxNx3xTUVcXqvWTz3aq7OvUuDB/MFlPFjmxRaKF7/A==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Configuration.FileExtensions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "ZOhZYwvbXGTgGVRwswIirofEMVHuWdxjdh0JeUZXwaF9cgcjXdz/t0ELtgaevw7ezTyv47yPNCgGreWtLkn3IQ==", + "resolved": "10.0.11", + "contentHash": "mDW7KVFB05M6jiRUyaZiOMWhS31n5HlSZwoYctHAZAucD4sMDJ70IxOmkGDt6RpstchD+keWBjhdzcMpSkWvWQ==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10", - "Microsoft.Extensions.FileProviders.Physical": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11", + "Microsoft.Extensions.FileProviders.Physical": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Configuration.UserSecrets": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "1s1sKFTk/Foam64JY6+m/diH8drL3Wx6V3gtSd5v1IEZtszZYyc1pW8uRnMblzpNiR0l0t8gGk7tXj3xHzFgdg==", + "resolved": "10.0.11", + "contentHash": "BRliLdUowglV8GS+J1G/QsSofCJYYFg3U8QZx0ACRn+a91az/Qnpy+h6PyHS94WgV2TSazX5D/cuuk6wnCJatw==", "dependencies": { - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Configuration.Json": "10.0.10", - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10", - "Microsoft.Extensions.FileProviders.Physical": "10.0.10" + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Configuration.Json": "10.0.11", + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11", + "Microsoft.Extensions.FileProviders.Physical": "10.0.11" } }, "Microsoft.Extensions.DependencyInjection": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "ANyvsgkNBRvcJh2XLgn8veGmajf+8m0AbKK+HPWdRL1yraSNVVSmQhFntLtdz/C795jxqqup+k05cs/3jZQPOA==", + "resolved": "10.0.11", + "contentHash": "PSmotV19c7E3lKed++uYo1kSiXFI+uTl37CBSrhq+CfLC3FCHjG7R91+xPnNehQfHS1b0Tzo/CCLPWH3qaEheg==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11" } }, "Microsoft.Extensions.DependencyInjection.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "z/2xXlFw2aLGjHyEm6E0tQ+In6VfzQzTrtArbQ2c0TQE16ZbyDCMGPvaUT9I0s8rgy9sRWlU2P9waW37qV04qA==" + "resolved": "10.0.11", + "contentHash": "/a1aJz4m7ylhEDf25ugQChLQoN5XwoGjWw/BoR/ZWWKsO1v4DdJElS1uyngahz4B/eOzjFk1KNTkarRLE5wsIg==" }, "Microsoft.Extensions.Diagnostics": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "Kr/e7lUf4+N8tacbqJ2Ctwe/HarKdAc9ZkgKVVqvtJDBKbez+T/KnUwu82KSlnBp/SrpBcxc7u7xkE2oUZT/5Q==", + "resolved": "10.0.11", + "contentHash": "HT70uGPxMLqqnOzKMcnQtDmeV4r0KHr4qVCLhP7SXil9jMEm8sQXwcybxVVFGXZJ1V44xV0mLqQ54aZbcR2OiQ==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Diagnostics.Abstractions": "10.0.10", - "Microsoft.Extensions.Options.ConfigurationExtensions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Diagnostics.Abstractions": "10.0.11", + "Microsoft.Extensions.Options.ConfigurationExtensions": "10.0.11" } }, "Microsoft.Extensions.Diagnostics.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "9uWiKpeOVac355STyChWR/pliFX/5CeLqChW9kKsaxyDH4EUTZxMkT4Jwp/J/peLm0GBFmSX5c0WCse3yCnq1Q==", + "resolved": "10.0.11", + "contentHash": "se7Kx8QpJEt+nf26L4qIVAofGTDr1wbexxsh/Fm3Xc04xUkqUXK06KUS7FLwSQYSjqb7q9n+T7MEcXYBhI1Y5g==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11" } }, "Microsoft.Extensions.FileProviders.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "c5zqFCY9DiIpMovLd7/d/CTiEtrMOuQ639dhv3PABtKQIKNQikSHwQt8+N679uii9q+B55lgK28Uv64FOwEu8w==", + "resolved": "10.0.11", + "contentHash": "JOjac6SQQgZmdmB8WGEw61/7siqMZoWJMkmq2p1goJGxqI59lO6oB4bOl0jNsbaPBdYy5Mlkb+6U7T4+CjnD8Q==", "dependencies": { - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.FileProviders.Physical": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "jhJAyo38kSrH3ARvWUk0h8itogVnQu2DCZuPo+s0Z+tXes0ugTxMPaHYzap85785eHQmPFqD9TYERqBbtGxn/w==", + "resolved": "10.0.11", + "contentHash": "Tq/UqMaczePv9yWwSsJZRgKtgA46djVR5xHj/lZBCueQ3ag8f9v5mu0EdhrNx7tXxNk+Y9OurG2oKuSKINjr0A==", "dependencies": { - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10", - "Microsoft.Extensions.FileSystemGlobbing": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11", + "Microsoft.Extensions.FileSystemGlobbing": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.FileSystemGlobbing": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "jSOCVxEwCd4Aq925kJVz1kSO1EpX2OHYKL04qVREXkDU7Ce3pVDdHPYm+fEy8y/th2kJf/DAstRHpJAqoNWP8w==" + "resolved": "10.0.11", + "contentHash": "2i6rtW/B5rCnWCnhdmWWEmaM9O0HD0zsPY9eRqa++y4tclI3Uw8zvGbBvhY/LjAdtf8gUHhUPcAWj3DRlWMXmQ==" }, "Microsoft.Extensions.Hosting.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "5LugpYGHk+mkn0a8IZgcyfBca8PCTAU9RQFoMrTdtOOidq88M2SI5f3px6ugnzgxC+eTkvYYJi8pzlUnG5xdAQ==", + "resolved": "10.0.11", + "contentHash": "pwtpF7iF/NNaOBcX+pvMZ7y2+JAVbH5KkNrH9uMZtuVxVJsFTDiWiCR7Tk3HVptsAijaetipUZVRVK2LLq+nvA==", "dependencies": { - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Diagnostics.Abstractions": "10.0.10", - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10" + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Diagnostics.Abstractions": "10.0.11", + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Logging.Configuration": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "cLrqxkuEfcilZ8SjK+9KAnpLk9lOoMPaOokF+wRUYie+iUEcdX4/p/+gJkt0BYgWLthjpBUCkVTBI6Kxg0nsOw==", + "resolved": "10.0.11", + "contentHash": "S7LvLeVHKNPaY2NMyxW7c2TBGsLgxoSUBCV5Ev5iN8kgC7EPR2UB7eW7vHsElGMcIUDwRmoxLfvGDynCn3q6EA==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Configuration.Binder": "10.0.10", - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10", - "Microsoft.Extensions.Options.ConfigurationExtensions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Configuration.Binder": "10.0.11", + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11", + "Microsoft.Extensions.Options.ConfigurationExtensions": "10.0.11" } }, "Microsoft.Extensions.Logging.Console": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "VIlNzPwPS0GeQVSmCqqo36ugryX3LpE9ul6gEkks5VLET3weH/XMLeWmclwfoGn4Nxi2mwVibB+OZBVJ9tDqvg==", + "resolved": "10.0.11", + "contentHash": "dFc0yDudyD1iIg6z9XT7ofsT3hVO7Y4ylrxGHIVRR0GaZ4CUk4ujOrMoy7wWEdNHZhvJooySg6hZpOxyS8zEVA==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging.Configuration": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging.Configuration": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11" } }, "Microsoft.Extensions.Logging.Debug": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "8+TZBnV5fgBXoVNJ5ROSErUwYogk4hOgV7c2HWK1u5cqKGmiUTUn7+KqZ35iQu8e/B7Ykccyz5OTjdXcidNZ9g==", + "resolved": "10.0.11", + "contentHash": "wr+j1bjdFXhc8lKTLoq+RbwFM8M+orcMS9xrcqLmDGOxJcXpKizEeE5h6v/GKwCZV02FmhaA7OlNjoq072jZpQ==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Logging.EventLog": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "0RE4951AzQ+YD4gVrvbq0BhdsiBgSDo44yM7+QBZ2mrmMJeNjY+teCIYfUjqDPVYnKs0HR6SkkhgrX1YgXZq3Q==", + "resolved": "10.0.11", + "contentHash": "Eck9GpCCpvZ3f6L7IUlN+mPtRVefnf7PsiIG5vi61QawPtLNCEAv2TPD/M3SojcU0PFaef+BxiVGPOShFHtDog==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10", - "System.Diagnostics.EventLog": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11", + "System.Diagnostics.EventLog": "10.0.11" } }, "Microsoft.Extensions.Logging.EventSource": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "85SAPwXhJtdBInzN2k7SChiFiBGh3KOWay5AfoY+GREF6P7oZA98+ST2p7Z9384iLKYjkZSKIZ/FqIO5aojtNw==", + "resolved": "10.0.11", + "contentHash": "hs6QWECLLohi2VKqUvSGRUvrg7eXR1DqKL95Jrtz3cdD2g2nBA+yJPdRQLZ7SLmnTZWycxfMDK2s0ho+rfst5w==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Options": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "srnhnk7nE8krBiIXp71LvBmKBtraBONWSRzdjJgRv1Ko9Mp8IVNqv4vIS9hGeVteBig8aQkva9ZG+sC+o5sVcA==", + "resolved": "10.0.11", + "contentHash": "eY1GAKcTfD2maP27J84X9IovT3yjHJ2dVDzPmDg6/XqYvt3jMzJhtfQCLjG9pVsZGAd+8DQ2QrjaDcs2+VQLGw==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Options.ConfigurationExtensions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "tnBmu/LwF25ZQK+HBNCu2xrwnkKoB/XEbJyooGGoYxHrhvxbSKi7eOFiJ4AXBy/QU4vtCvCJfoi8k9Ej72qzOQ==", + "resolved": "10.0.11", + "contentHash": "syEhXQ/sEaSBFaqzlp9gDGHX/nk6gkQkh1sIUpBO1mlBj3Phu1rmb4ML1uCiyPW9N6Kxfxv3y5FGObC+bV01Qw==", "dependencies": { - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Configuration.Binder": "10.0.10", - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Configuration.Binder": "10.0.11", + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Primitives": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "5wu/GrYVd8mG2DVUw3vFJzF+O336TyTGg/Kmcgw9bfwYhCoFiV5lR5QeEmKecJyrW4W54nMfD3p3589E8a7czQ==" + "resolved": "10.0.11", + "contentHash": "SXcz+kF+4Oo9b1+55zntpJFYfwb1jw66ioxptyNOOTDc8g2FHnBFWjZpsWfCvZIhzr0x+4e2trVTs4OKwQfBtw==" }, "Microsoft.IdentityModel.Abstractions": { "type": "Transitive", @@ -408,19 +387,6 @@ "Microsoft.Testing.Platform": "1.9.1" } }, - "Microsoft.TestPlatform.ObjectModel": { - "type": "Transitive", - "resolved": "18.8.1", - "contentHash": "qLbktNB1+b1XZLNJBTzaWVVJAd6PEzD7cgD406geMb6PcFZhp3EDNa1tctWx1+mtMU6MP/6ozVvFPC9vs2a9rw==" - }, - "Microsoft.TestPlatform.TestHost": { - "type": "Transitive", - "resolved": "18.8.1", - "contentHash": "FaQHPDTUOcE+SFTjssNPfrub2lT9Zyon4J2W/KLHt/efLJACb1TCeWXyOgh0D/4Q1e4n+S3E6mOKud+9nLZlEA==", - "dependencies": { - "Microsoft.TestPlatform.ObjectModel": "18.8.1" - } - }, "Microsoft.Win32.Registry": { "type": "Transitive", "resolved": "5.0.0", @@ -428,8 +394,8 @@ }, "ModelContextProtocol.Core": { "type": "Transitive", - "resolved": "2.1.0", - "contentHash": "cU/urrhRxE4/iSyBIJI7QOaFqSP1FOEnwEHsct9n6t6/XluCAFD9iqnrPkBAsEYr+f/G4tVQ21U+6wN/6fQvOg==", + "resolved": "2.2.0", + "contentHash": "FeBfXU6T8k+jw4afg4sfxdEX2rL/e5oKOk9ROOGztu9k47+7Bz08sdaToYt2XvMY1opNbwxYQOFMj6wH9TInhA==", "dependencies": { "Microsoft.Extensions.AI.Abstractions": "10.8.3", "Microsoft.Extensions.Logging.Abstractions": "10.0.10" @@ -596,8 +562,8 @@ }, "System.Diagnostics.EventLog": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "OvGz3PrzuAI/Sj7LTcXcCe3FClRI1IyRMZjNONcZtFh+Ww7nAtSh4kh08r8KVe/xxkXJPjR0Y1jF7H+N42d4xQ==" + "resolved": "10.0.11", + "contentHash": "QTXEoQBzz00SFWbo7nAg1Ogd4f99lwqcO9uAJ7MYSLEUR28f6As32QktrqG2Fr9cfAfd1GjLyGYspE7Ipj7P6w==" }, "System.IdentityModel.Tokens.Jwt": { "type": "Transitive", @@ -610,10 +576,10 @@ }, "System.ServiceProcess.ServiceController": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "9tI/EtlMimwJn9EEtbqFrVqGCD98VfVdnW6UekC7/YHO5Ky/LMUfXibBv9+AV8g9u7uim20oHgiOqzaeEWAhqg==", + "resolved": "10.0.11", + "contentHash": "khROyIQ3GAZy5MbCabfmlbvxf9nuvdveFP2XbRCteZ9Lq8OFEvImmrjV6acl1LegRu/9T6VMxi9Lr+KZu29VHg==", "dependencies": { - "System.Diagnostics.EventLog": "10.0.10" + "System.Diagnostics.EventLog": "10.0.11" } }, "xunit.analyzers": { @@ -699,29 +665,31 @@ "type": "Project", "dependencies": { "CredentialManagement": "[1.0.2, )", - "Microsoft.Extensions.Logging.Abstractions": "[10.0.10, )", - "ModelContextProtocol": "[2.1.0, )" + "Microsoft.Extensions.Logging.Abstractions": "[10.0.11, )", + "ModelContextProtocol": "[2.2.0, )" } }, "performancemonitor.darling.analysis": { "type": "Project", "dependencies": { "Microsoft.Data.SqlClient": "[7.0.2, )", + "ModelContextProtocol": "[2.2.0, )", "PerformanceMonitor.Analysis": "[1.0.0, )", "PerformanceMonitor.Collectors": "[1.0.0, )", "PerformanceMonitor.Darling.Storage": "[1.0.0, )", "PerformanceMonitor.Notifications": "[1.0.0, )", - "PerformanceMonitor.PlanAnalysis": "[1.0.0, )" + "PerformanceMonitor.PlanAnalysis": "[1.0.0, )", + "System.Security.Cryptography.ProtectedData": "[10.0.11, )" } }, "performancemonitor.darling.service": { "type": "Project", "dependencies": { "Microsoft.Data.SqlClient": "[7.0.2, )", - "Microsoft.Extensions.Hosting": "[10.0.10, )", - "Microsoft.Extensions.Hosting.WindowsServices": "[10.0.10, )", - "ModelContextProtocol": "[2.1.0, )", - "ModelContextProtocol.AspNetCore": "[2.1.0, )", + "Microsoft.Extensions.Hosting": "[10.0.11, )", + "Microsoft.Extensions.Hosting.WindowsServices": "[10.0.11, )", + "ModelContextProtocol": "[2.2.0, )", + "ModelContextProtocol.AspNetCore": "[2.2.0, )", "PerformanceMonitor.Alerting": "[1.0.0, )", "PerformanceMonitor.Collectors": "[1.0.0, )", "PerformanceMonitor.Common": "[1.0.0, )", @@ -729,7 +697,7 @@ "PerformanceMonitor.Darling.Storage": "[1.0.0, )", "PerformanceMonitor.PlanAnalysis": "[1.0.0, )", "System.IO.FileSystem.AccessControl": "[5.0.0, )", - "System.Security.Cryptography.ProtectedData": "[10.0.10, )" + "System.Security.Cryptography.ProtectedData": "[10.0.11, )" } }, "performancemonitor.darling.storage": { @@ -758,7 +726,7 @@ "performancemonitor.notifications": { "type": "Project", "dependencies": { - "Microsoft.Extensions.Logging.Abstractions": "[10.0.10, )", + "Microsoft.Extensions.Logging.Abstractions": "[10.0.11, )", "PerformanceMonitor.Analysis": "[1.0.0, )" } }, @@ -766,7 +734,8 @@ "type": "Project", "dependencies": { "Microsoft.Data.SqlClient": "[7.0.2, )", - "PerformanceMonitor.Common": "[1.0.0, )" + "PerformanceMonitor.Common": "[1.0.0, )", + "System.Security.Cryptography.ProtectedData": "[10.0.11, )" } }, "performancemonitor.ui": { @@ -807,105 +776,105 @@ }, "Microsoft.Extensions.Configuration": { "type": "CentralTransitive", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "plJWK2zpWuuyxI8F8s2scx6Je7N1Ajjs6HvYUGKwRnDMWIVIz9FHwAkiT7ASgrvAOd10T0FPVlh9BzAJJME+jg==", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "wlhRqZW8LcJPa+vk2oLAc/REXDItHtkFQdf/QcXYGZbZOO13izcsKY1pCvuFQYwUiZD+hwSZwsKASjqT+BNaVg==", "dependencies": { - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Configuration.Json": { "type": "CentralTransitive", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "uvJ6sHwjgrkMEJOgiC76G0mcZGXerwyyWkwX34EOjCbxKG6TCtfAoqDKAMsCvEBf9HxjlGQEgqsSMOGCmGBf+A==", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "nSPrT8U/cNoB4coqkmnanAMK9PsL7lsjG+LLUKEwHRFwS6E78b8S1wdv/y88EOxBhasWov1rLd7RTHmmsYPOLg==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Configuration.FileExtensions": "10.0.10", - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Configuration.FileExtensions": "10.0.11", + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Hosting": { "type": "CentralTransitive", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "tL9FkfV64GPUDSPvwrgyw42LVzsnVAnyrqJEuZVJbODgrQ3eL63zmzEcVWoCHzfgqUhWggzbgAyUCnz/zfI3Pg==", - "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Configuration.Binder": "10.0.10", - "Microsoft.Extensions.Configuration.CommandLine": "10.0.10", - "Microsoft.Extensions.Configuration.EnvironmentVariables": "10.0.10", - "Microsoft.Extensions.Configuration.FileExtensions": "10.0.10", - "Microsoft.Extensions.Configuration.Json": "10.0.10", - "Microsoft.Extensions.Configuration.UserSecrets": "10.0.10", - "Microsoft.Extensions.DependencyInjection": "10.0.10", - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Diagnostics": "10.0.10", - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10", - "Microsoft.Extensions.FileProviders.Physical": "10.0.10", - "Microsoft.Extensions.Hosting.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging.Configuration": "10.0.10", - "Microsoft.Extensions.Logging.Console": "10.0.10", - "Microsoft.Extensions.Logging.Debug": "10.0.10", - "Microsoft.Extensions.Logging.EventLog": "10.0.10", - "Microsoft.Extensions.Logging.EventSource": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10" + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "eIDa/Rl+93aj17gMlFsJJx+LhBvb3CP0Mu1PeVYkDp2Y3S4Jock8UynfGQEcx7lrlq+gKW+ECQJHbro/LTPDEQ==", + "dependencies": { + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Configuration.Binder": "10.0.11", + "Microsoft.Extensions.Configuration.CommandLine": "10.0.11", + "Microsoft.Extensions.Configuration.EnvironmentVariables": "10.0.11", + "Microsoft.Extensions.Configuration.FileExtensions": "10.0.11", + "Microsoft.Extensions.Configuration.Json": "10.0.11", + "Microsoft.Extensions.Configuration.UserSecrets": "10.0.11", + "Microsoft.Extensions.DependencyInjection": "10.0.11", + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Diagnostics": "10.0.11", + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11", + "Microsoft.Extensions.FileProviders.Physical": "10.0.11", + "Microsoft.Extensions.Hosting.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging.Configuration": "10.0.11", + "Microsoft.Extensions.Logging.Console": "10.0.11", + "Microsoft.Extensions.Logging.Debug": "10.0.11", + "Microsoft.Extensions.Logging.EventLog": "10.0.11", + "Microsoft.Extensions.Logging.EventSource": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11" } }, "Microsoft.Extensions.Hosting.WindowsServices": { "type": "CentralTransitive", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "qY7XhE2ljtqCwDKexJf3uG4E7r0teE/DU0eEUDbFqGN2HAIz0WK7P3u4RnnnqOy/3Jz5vdTfTuDMFsrwxrcmeg==", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "OTlOHRxmt4IyYR5tBejF5ffdVMbo9n5NbPJRzCe0bnJmzTQt0kHNiDiEU96t4KkB3wE3vZqPETtXJVLujrRJnA==", "dependencies": { - "Microsoft.Extensions.Hosting": "10.0.10", - "Microsoft.Extensions.Logging.EventLog": "10.0.10", - "System.ServiceProcess.ServiceController": "10.0.10" + "Microsoft.Extensions.Hosting": "10.0.11", + "Microsoft.Extensions.Logging.EventLog": "10.0.11", + "System.ServiceProcess.ServiceController": "10.0.11" } }, "Microsoft.Extensions.Logging": { "type": "CentralTransitive", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "Tf6z5HsL0VDYRTfvsoNrTGHGheCwkTsZBA2FFh5ATJUbkAwug+FFNISJK2gjpUNemlAOoWllAK52HOWCjto3EQ==", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "nUOJwgFkSiLHiVGFpU22pIJtuWYewuSYQ3JVuP/gdK8ASMT807Px+TYQiRWs6uSsOmoyFTaVCwKXTasczV6BpA==", "dependencies": { - "Microsoft.Extensions.DependencyInjection": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10" + "Microsoft.Extensions.DependencyInjection": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11" } }, "Microsoft.Extensions.Logging.Abstractions": { "type": "CentralTransitive", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "zkFxGYUvdxAvIKTyXHrmW+Sux53D4SezD9dMyZ6hrwwzPQJNuwCRy1f5W7AvYTqacEGhWF2XderRQG1OvbV8og==", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "Ljd0Uxoq5XpScD2Bg0nM/r3mwx7Ao5Uq24eo2ARxbGvqJ7Zht6rt2cJtwVRH4Cv+1ZVMdXz6TB43KbpmsxRrvQ==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11" } }, "ModelContextProtocol": { "type": "CentralTransitive", - "requested": "[2.1.0, )", - "resolved": "2.1.0", - "contentHash": "Oa4rU7EL9C2qyFjQj1dx+ysGMzfWDRpM8RRaUMmLGs5vPvfJ9xyz4ZtyF4ychY+Nx1b/auGCqIQLqSz/IpPkKA==", + "requested": "[2.2.0, )", + "resolved": "2.2.0", + "contentHash": "4Pb9u02Nwsp0poueDsqNdyGRojFxOYpljB7zDBsq+aHL+Afou3OgxlBc3GWFVnsRMRJrUtWqDh3s6k2JgPzmrQ==", "dependencies": { "Microsoft.Extensions.Caching.Abstractions": "10.0.10", "Microsoft.Extensions.Hosting.Abstractions": "10.0.10", - "ModelContextProtocol.Core": "[2.1.0]" + "ModelContextProtocol.Core": "[2.2.0]" } }, "ModelContextProtocol.AspNetCore": { "type": "CentralTransitive", - "requested": "[2.1.0, )", - "resolved": "2.1.0", - "contentHash": "yhJ8bBXIgrX0mAgRYRgzcbH3bLdv3MDSkG52utRW9EAAtQrPw/g7Q/T6EurxKV+L+Zefv8VVUYcNbXEOd9GgfA==", + "requested": "[2.2.0, )", + "resolved": "2.2.0", + "contentHash": "3JelDMuFIwFzXybsh6K30G6wXu5gmqKnuRnieqgBMuR0FkqqXRv4B+NYZuPgthSyNq5UwRyvv2U2VJBE3RD3PQ==", "dependencies": { - "ModelContextProtocol": "[2.1.0]" + "ModelContextProtocol": "[2.2.0]" } }, "Npgsql": { @@ -937,9 +906,9 @@ }, "System.Security.Cryptography.ProtectedData": { "type": "CentralTransitive", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "BKt0SQgq2lq3ESE68jkeLwv95ypANrPDtkTOIFGcnhg2aRUeUUBDrQlkVDZctrg1WenVcvn5P5XZnuYI7q6rFQ==" + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "PNoxCTPb+Tlux+GJyq4c89ddYdpioVSqfGx8pqOF6shCSKwUNNctXhQtRkCICjbJmGrJJsW8NY52kVYO/b8mlQ==" }, "Velopack": { "type": "CentralTransitive", diff --git a/Darling/PerformanceMonitor.Darling.Analysis/PerformanceMonitor.Darling.Analysis.csproj b/Darling/PerformanceMonitor.Darling.Analysis/PerformanceMonitor.Darling.Analysis.csproj index ddc4f7476..4a5efcf10 100644 --- a/Darling/PerformanceMonitor.Darling.Analysis/PerformanceMonitor.Darling.Analysis.csproj +++ b/Darling/PerformanceMonitor.Darling.Analysis/PerformanceMonitor.Darling.Analysis.csproj @@ -1,44 +1,46 @@ - - - net10.0 - enable - disable - PerformanceMonitor.Darling.Analysis - PerformanceMonitor.Darling.Analysis - Darling Data, LLC - Copyright © 2026 Darling Data, LLC - true - latest-recommended - CA1849;CA2007;CA1508;CA1822;CA1805;CA1510;CA1816;CA1861;CA1845;CA2201;CA1848;CA1852;CA1305;CA1860;CA1707;CA1507;CA1806;CA2254 - CS4014 - - - - - - - - - - - - - - - - - - - - - - - - - + + + net10.0 + enable + disable + PerformanceMonitor.Darling.Analysis + PerformanceMonitor.Darling.Analysis + Darling Data, LLC + Copyright © 2026 Darling Data, LLC + true + latest-recommended + CA1849;CA2007;CA1508;CA1822;CA1805;CA1510;CA1816;CA1861;CA1845;CA2201;CA1848;CA1852;CA1305;CA1860;CA1707;CA1507;CA1806;CA2254 + CS4014 + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingAlertReadAdapter.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingAlertReadAdapter.cs index d48c2f166..16fed51e7 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingAlertReadAdapter.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingAlertReadAdapter.cs @@ -443,6 +443,102 @@ public async Task> GetLongRunningQueriesAsync( return items; } + /* ---------------- database file growth (#2349) ---------------- */ + + /// + /// Per-file current size, growth over the lookback window, and the file's volume. + /// + /// Newest per file, and a baseline from the window's far edge. DISTINCT ON takes the + /// current row per (database, file); the baseline join takes the OLDEST sample inside the window for the + /// same key. Growth is the difference, and is 0 when the window holds a single sample — which reads as "no + /// rise observed" rather than as a rise of the whole file, the wrong answer for a server that just started + /// collecting. + /// + /// Both sides are bounded on collection_time, the partitioning column, so the window prunes + /// chunks rather than scanning retention. The reported window width is measured rather than assumed, so a + /// gap in collection cannot make a slow rise look fast. + /// + /// $1 server_id, $2 window start (naive UTC). + /// + public const string DatabaseFileGrowthSql = @" +WITH current_files AS ( + SELECT DISTINCT ON (database_name, file_name) + database_name, file_name, physical_name, file_type_desc, collection_time, + total_size_mb, auto_growth_mb, is_percent_growth, growth_pct, max_size_mb, + volume_mount_point, volume_total_mb, volume_free_mb + FROM database_size_stats + WHERE server_id = $1 + AND collection_time >= $2 + ORDER BY database_name, file_name, collection_time DESC +), +baseline AS ( + SELECT DISTINCT ON (database_name, file_name) + database_name, file_name, collection_time, total_size_mb + FROM database_size_stats + WHERE server_id = $1 + AND collection_time >= $2 + ORDER BY database_name, file_name, collection_time ASC +) +SELECT + c.database_name, + c.file_name, + COALESCE(c.physical_name, '') AS physical_name, + COALESCE(c.file_type_desc, '') AS file_type_desc, + COALESCE(c.total_size_mb, 0) AS total_size_mb, + COALESCE(c.total_size_mb, 0) - COALESCE(b.total_size_mb, c.total_size_mb, 0) AS growth_mb, + COALESCE(EXTRACT(EPOCH FROM (c.collection_time - b.collection_time)) / 60.0, 0) AS growth_window_minutes, + COALESCE(c.volume_mount_point, '') AS volume_mount_point, + COALESCE(c.volume_total_mb, 0) AS volume_total_mb, + COALESCE(c.volume_free_mb, 0) AS volume_free_mb, + c.auto_growth_mb, + COALESCE(c.is_percent_growth, false) AS is_percent_growth, + c.growth_pct, + c.max_size_mb +FROM current_files c +LEFT JOIN baseline b + ON b.database_name = c.database_name + AND b.file_name = c.file_name +WHERE c.total_size_mb IS NOT NULL +ORDER BY c.database_name, c.file_name"; + + public async Task> GetDatabaseFileGrowthAsync( + string serverKey, int lookbackMinutes, CancellationToken cancellationToken = default) + { + var serverId = ParseServerKey(serverKey); + var windowStart = DateTime.SpecifyKind( + DateTime.UtcNow.AddMinutes(-Math.Max(1, lookbackMinutes)), DateTimeKind.Unspecified); + + var items = new List(); + await using var connection = await _postgres.OpenConnectionAsync(cancellationToken); + using var command = new NpgsqlCommand(DatabaseFileGrowthSql, connection); + command.Parameters.AddWithValue(serverId); + command.Parameters.AddWithValue(windowStart); + + await using var reader = await command.ExecuteReaderAsync(cancellationToken); + while (await reader.ReadAsync(cancellationToken)) + { + items.Add(new DatabaseFileGrowthInfo + { + DatabaseName = reader.IsDBNull(0) ? "" : reader.GetString(0), + FileName = reader.IsDBNull(1) ? "" : reader.GetString(1), + PhysicalName = reader.GetString(2), + FileTypeDesc = reader.GetString(3), + TotalSizeMb = Convert.ToDouble(reader.GetValue(4)), + GrowthMb = Convert.ToDouble(reader.GetValue(5)), + GrowthWindowMinutes = Convert.ToDouble(reader.GetValue(6)), + VolumeMountPoint = reader.GetString(7), + VolumeTotalMb = Convert.ToDouble(reader.GetValue(8)), + VolumeFreeMb = Convert.ToDouble(reader.GetValue(9)), + AutoGrowthMb = reader.IsDBNull(10) ? null : Convert.ToDouble(reader.GetValue(10)), + IsPercentGrowth = !reader.IsDBNull(11) && reader.GetBoolean(11), + GrowthPct = reader.IsDBNull(12) ? null : Convert.ToDouble(reader.GetValue(12)), + MaxSizeMb = reader.IsDBNull(13) ? null : Convert.ToDouble(reader.GetValue(13)), + }); + } + + return items; + } + /* ---------------- volume free space ---------------- */ /// Lite's per-volume free-space read verbatim (database_size_stats table). diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingAlertSettings.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingAlertSettings.cs index 81df67e99..8d050829e 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingAlertSettings.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingAlertSettings.cs @@ -85,6 +85,15 @@ public DarlingAlertSettings(DarlingConfig config) the percent it has no meaningful upper bound. */ public int PvsThresholdPercent => Math.Clamp(_config.Alerts.PvsThresholdPercent, 0, 100); public int PvsFloorGb => Math.Max(0, _config.Alerts.PvsFloorGb); + + /* #2349: the file-growth gates. Clamped the same way the neighbours are -- a negative threshold would + make the comparison always true, which for a gate whose whole job is to be quiet until something moves + is the worst possible default. A ZERO is meaningful here rather than nonsense: it disables that one + gate, so an operator can run rise-only or level-only without a second switch. */ + public bool FileGrowthEnabled => _config.Alerts.FileGrowthEnabled; + public int FileGrowthRiseMb => Math.Max(0, _config.Alerts.FileGrowthRiseMb); + public int FileGrowthVolumePercent => Math.Clamp(_config.Alerts.FileGrowthVolumePercent, 0, 100); + public int FileGrowthLookbackMinutes => Math.Clamp(_config.Alerts.FileGrowthLookbackMinutes, 5, 1440); public int LongRunningJobMultiplier => _config.Alerts.LongRunningJobMultiplier; public int FailedJobLookbackMinutes => Math.Clamp(_config.Alerts.FailedJobLookbackMinutes, 1, 1440); public int CooldownMinutes => Math.Clamp(_config.Alerts.CooldownMinutes, 1, 120); diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingCliCommands.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingCliCommands.cs index 64f6ee93c..ae595e4fe 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingCliCommands.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingCliCommands.cs @@ -91,6 +91,13 @@ public static bool IsConfigureNetworkVerb(string arg) => public static bool IsConfigureFirewallVerb(string arg) => string.Equals(arg, "--configure-firewall", StringComparison.OrdinalIgnoreCase); + /// The verb handles — re-apply the secret-file ACLs from an elevated + /// prompt (#2352). The running service knows the correct ACL and cannot apply it: re-ACLing a file it does + /// not own needs WRITE_DAC, and taking ownership needs a privilege a virtual service account is not + /// granted, so it can only log the remedy. This is the actor that carries it out. + public static bool IsHardenFilesVerb(string arg) => + string.Equals(arg, "--harden-files", StringComparison.OrdinalIgnoreCase); + /// The verb handles — enable the MCP endpoint in the store (+ firewall). public static bool IsEnableMcpVerb(string arg) => string.Equals(arg, "--enable-mcp", StringComparison.OrdinalIgnoreCase); @@ -154,6 +161,7 @@ public static bool IsKnownVerb(string arg) => || IsExportViewerConfigVerb(arg) || IsConfigureNetworkVerb(arg) || IsConfigureFirewallVerb(arg) + || IsHardenFilesVerb(arg) || IsEnableMcpVerb(arg) || IsDisableMcpVerb(arg) || IsEnableWebVerb(arg) @@ -227,6 +235,7 @@ public static string UsageText() => " PerformanceMonitor.Darling.Service.exe --export-viewer-config [dir] [--config ] Write a ready-to-copy viewer folder (darling.json + server.crt + README.txt)." + Environment.NewLine + " PerformanceMonitor.Darling.Service.exe --configure-network Interactive LAN-exposure wizard." + Environment.NewLine + " PerformanceMonitor.Darling.Service.exe --configure-firewall Create/remove the scoped firewall rules to match darling.json (run elevated)." + Environment.NewLine + + " PerformanceMonitor.Darling.Service.exe --harden-files Re-apply the ACLs on darling.json and the store credentials (run elevated)." + Environment.NewLine + " PerformanceMonitor.Darling.Service.exe --enable-mcp Enable the MCP endpoint in the store and open its firewall (run elevated)." + Environment.NewLine + " PerformanceMonitor.Darling.Service.exe --disable-mcp Disable the MCP endpoint in the store and remove its firewall rule (run elevated)." + Environment.NewLine + " PerformanceMonitor.Darling.Service.exe --enable-web Enable the web dashboard in the store and open its firewall (run elevated)." + Environment.NewLine + @@ -2740,6 +2749,206 @@ public static IReadOnlyList PlanFirewallRules(DarlingConfig co _ => null, }; + /// + /// Creates or removes every scoped Darling firewall rule so the live firewall matches darling.json. + /// Requires elevation (that is the entire point of the verb) and is idempotent — safe to re-run on every + /// upgrade, which is exactly how install-darling.ps1 uses it. Returns 0 when the firewall ends up matching + /// the config, 1 when it could not be made to. + /// + [SupportedOSPlatform("windows")] + /// + /// One target of : a path, whether the interactive operator legitimately reads it, + /// and what it is called in the report. Kept as data so the list is readable as a policy rather than as + /// control flow — which file gets INTERACTIVE read is the only judgement in this verb, and it should be + /// visible at a glance. + /// + private readonly record struct HardenTarget(string Path, bool AllowInteractive, bool IsDirectory, string What); + + /// + /// Re-applies the secret-file ACLs, elevated (#2352). + /// + /// Why this exists as a verb. The service already computes the correct DACL + /// () and already detects when the real one is wrong + /// (). What it lacks is authority: re-ACLing a + /// file it does not own needs WRITE_DAC, and taking ownership needs a privilege a virtual service account is + /// not granted, so it can only log the remedy and continue. Until now the only thing that ever APPLIED the + /// rule to an existing install was install-darling.ps1, which leaves anyone who registered the exe by + /// hand — the README's own sc create path — typing three icacls lines out of a log message. + /// + /// It verifies rather than claims. Every target is re-read after the attempt and reported as + /// SECURED or STILL READABLE, and the exit code is 1 if anything is still exposed. "We tried" is not the + /// same statement as "the secret is not readable" — a distinction 's own + /// contract already makes, and the reason a permissions call that silently did nothing was able to hide. + /// + /// Idempotent, so it is safe on a healthy box and can simply live in a runbook. Missing targets are + /// skipped quietly: a BYO-Postgres install has no managed credential files, and their absence is not a + /// fault. + /// + [SupportedOSPlatform("windows")] + public static int HardenFiles(string? configPath, TextWriter output, TextWriter error) + { + var resolvedConfig = DarlingConfig.ResolveConfigPath(configPath); + + /* The managed store's directory is derived from config when it loads, but this verb has to work when + darling.json is exactly what is unreadable — that is the failure it exists to repair. So a config that + will not load is a warning, not a stop: the config file itself is still hardened, and the store + targets fall back to the documented default location. */ + string? dataDirectory = null; + try + { + var config = DarlingConfig.Load(configPath); + dataDirectory = DarlingManagedPostgres.ResolveDataDirectory(config.Postgres); + } + catch (Exception ex) + { + output.WriteLine($"NOTE: could not load {resolvedConfig} ({ex.Message})."); + output.WriteLine(" Continuing with the documented default store location — hardening darling.json"); + output.WriteLine(" is very often what makes it loadable again."); + dataDirectory = Path.Combine( + Environment.GetFolderPath(Environment.SpecialFolder.CommonApplicationData), + "PerformanceMonitorDarling", "pg"); + } + + var storeRoot = Path.GetDirectoryName(Path.GetFullPath(dataDirectory)); + var targets = new List + { + /* INTERACTIVE read, alone in this list: the Viewer (ViewerSettings.ResolveConfigPath) and the CLI + verbs run as the operator and must still read the live config. Nothing reads a backup (#1769). */ + new(resolvedConfig, AllowInteractive: true, IsDirectory: false, "the live config"), + }; + + foreach (var backup in SafeEnumerate(Path.GetDirectoryName(resolvedConfig), "darling.json.bak-*")) + { + targets.Add(new(backup, AllowInteractive: false, IsDirectory: false, "a config backup")); + } + + if (!string.IsNullOrEmpty(storeRoot)) + { + targets.Add(new(storeRoot, AllowInteractive: false, IsDirectory: true, "the store directory")); + targets.Add(new(Path.Combine(storeRoot, "pg-credential.dpapi"), false, false, "the store credential")); + targets.Add(new(Path.Combine(storeRoot, "pg-admin-credential.dpapi"), false, false, "the admin credential")); + } + + /* #2371: harden for the account the SERVICE runs as, not for whoever is running THIS. The verb is + documented to be run elevated and exists because the service cannot re-ACL a file it does not own, + so the caller is never the service — and resolving from the caller granted the operator and stripped + the service, leaving an install that worked until its next restart. Falls back to the caller when + the service is not registered (a console run, or hardening a tree before install), which is the only + case where those two are legitimately the same account. */ + var registered = DarlingFileSecurity.RegisteredServiceAccount(ServiceName); + if (registered is not null) + { + DarlingFileSecurity.HardenForAccount(registered); + } + + output.WriteLine($"Hardening for service account: {DarlingFileSecurity.ServiceAccountDisplayName}"); + if (registered is null) + { + output.WriteLine( + $" (the '{ServiceName}' service is not registered on this machine, so this is the account " + + "you are running as - install the service first if you expected its own account here)"); + } + + output.WriteLine(); + + var exposed = 0; + var touched = 0; + + foreach (var target in targets) + { + var exists = target.IsDirectory ? Directory.Exists(target.Path) : File.Exists(target.Path); + if (!exists) + { + continue; + } + + touched++; + + try + { + if (target.IsDirectory) + { + /* Traverse, not read: the operator's Viewer needs to walk to the config, never to read the + credential blobs sitting in here. Mirrors DarlingManagedPostgres' own call. */ + DarlingFileSecurity.HardenDirectory(target.Path, allowInteractiveTraverse: true); + } + else + { + DarlingFileSecurity.HardenFile(target.Path, target.AllowInteractive); + } + } + catch (Exception ex) + { + error.WriteLine($" FAILED {target.Path} ({target.What}): {ex.Message}"); + error.WriteLine($" {DarlingFileSecurity.DescribeOwnerAndExposure(target.Path)}"); + exposed++; + continue; + } + + /* The claim is the re-read, not the call that returned without throwing. */ + if (DarlingFileSecurity.IsReadableByOrdinaryUsers(target.Path)) + { + error.WriteLine($" STILL READABLE {target.Path} ({target.What})"); + error.WriteLine($" {DarlingFileSecurity.DescribeOwnerAndExposure(target.Path)}"); + exposed++; + } + else if (!DarlingFileSecurity.GrantsHardenedAccount(target.Path)) + { + /* #2371: private is only half of correct. An ACL that excludes ordinary users but also excludes + the service is a locked-out install, and it reports as SECURED under the readability check + alone — then fails on the next start, far enough away that nobody connects the two. */ + error.WriteLine($" LOCKED OUT {target.Path} ({target.What})"); + error.WriteLine( + $" secured, but {DarlingFileSecurity.ServiceAccountDisplayName} cannot read it - " + + "the service would fail on its next start. Grant that account and re-run."); + exposed++; + } + else + { + output.WriteLine($" SECURED {target.Path} ({target.What})"); + } + } + + output.WriteLine(); + + if (touched == 0) + { + error.WriteLine("Nothing to harden: no config and no store files were found at the expected paths."); + error.WriteLine($"Looked for the config at {resolvedConfig}. Pass an explicit path as the second argument."); + return 1; + } + + if (exposed > 0) + { + error.WriteLine($"{exposed} of {touched} item(s) are STILL readable by ordinary users."); + error.WriteLine("Run this from an ELEVATED prompt. If it was already elevated, the owner is the"); + error.WriteLine("problem: ownership carries WRITE_DAC, so take ownership first, then re-run."); + return 1; + } + + output.WriteLine($"All {touched} item(s) secured. The service re-asserts these ACLs at every start."); + return 0; + } + + /// Directory enumeration that treats an unreadable or missing folder as empty — this verb runs + /// precisely when permissions are broken, so a throw here would defeat its purpose. + private static IEnumerable SafeEnumerate(string? directory, string pattern) + { + if (string.IsNullOrEmpty(directory) || !Directory.Exists(directory)) + { + return Array.Empty(); + } + + try + { + return Directory.EnumerateFiles(directory, pattern).ToList(); + } + catch (Exception) + { + return Array.Empty(); + } + } + /// /// Creates or removes every scoped Darling firewall rule so the live firewall matches darling.json. /// Requires elevation (that is the entire point of the verb) and is idempotent — safe to re-run on every diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingCollectorRunner.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingCollectorRunner.cs index 17ab1fa73..d0dd78424 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingCollectorRunner.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingCollectorRunner.cs @@ -126,6 +126,32 @@ private void OnQueryStoreItemSucceeded(int serverId, string database) /// private readonly ConcurrentDictionary<(int ServerId, string Database), QueryStorePlanXmlState.PlanSizeEstimate> _observedPlanSize = new(); + /// + /// Per-database ids the activity-driven fetch (#2312 Finding 2) still owes the store: probed missing in + /// an earlier cycle but deferred by the candidate cap or the byte budget. Carried IN MEMORY because the + /// probe's input is each cycle's batch references, and a plan referenced once — its delta rows shipped, + /// never executed again — would otherwise never re-enter the probe and never get its XML. The honest + /// costs of in-memory: a restart forgets the debt, and the ids re-enter only if their plans execute + /// again — for the literal-churn plans that dominate deferrals, XML nobody can reach from a fact is the + /// cheap thing to lose. Bounded: ids are 8 bytes and a first-contact backlog is one catalog's worth. + /// + private readonly ConcurrentDictionary<(int ServerId, string Database), long[]> _planFetchCarryover = new(); + + /// Text twin of — same deferral contract, keyed by query_id. + private readonly ConcurrentDictionary<(int ServerId, string Database), long[]> _textFetchCarryover = new(); + + /// + /// Ids per IN-list statement for the plan fetch. Small on purpose: each id in the list is a plan the + /// server will DECOMPRESS to run the budget's running total, so the statement size is never the real + /// bound — the candidate cap from is — and 400 + /// keeps the SQL text itself a few KB. + /// + private const int PlanFetchIdsPerStatement = 400; + + /// Ids per IN-list statement for the text fetch. Larger than the plan side because + /// DATALENGTH(query_sql_text) is cheap — no decompression — so the only cost is statement size. + private const int TextFetchIdsPerStatement = 1000; + private static readonly TimeSpan AzureMasterRecheckInterval = TimeSpan.FromMinutes(15); public const int CommandTimeoutSeconds = 60; @@ -229,56 +255,14 @@ delete every live watermark it legitimately has. */ } } - /* #2164: the per-database plan-XML watermarks, owned by the HOST under its own state collector name - rather than declared by the definition — the QueryStoreBackfillState seam. The definition cannot - declare these: the keys are one per DATABASE and only known at runtime, and declaring a prefix - would make query_store a second state-declaring collector, which is a two-host contract change - (CollectorStateContractTests) rather than the local one this is. Loaded only when plan capture is - on, because that is the only case where anything reads or writes them. */ - if (collectorState is null - && string.Equals(definition.Name, "query_store", StringComparison.Ordinal) - && _capturePlans()) - { - collectorState = await GetCollectorStateAsync( - server.ServerId, QueryStorePlanXmlState.StateCollectorName, cancellationToken); - } - - /* #2150: the text watermark lives under its OWN state owner, so it is a second read merged into the - same dictionary — the two prefixes (planwm: / textwm:) cannot collide, and the definition still - sees one flat State. Read unconditionally for query_store rather than behind _capturePlans(), - because the text fetch is not gated on plan capture: a host that turned plans off still needs its - statement text. Merged rather than replacing, so a store that has plan state but no text state - yet (every store before this rung) keeps working. */ - if (string.Equals(definition.Name, "query_store", StringComparison.Ordinal)) - { - var textState = await GetCollectorStateAsync( - server.ServerId, QueryStoreTextState.StateCollectorName, cancellationToken); - - if (textState is { Count: > 0 }) - { - var merged = new Dictionary(StringComparer.Ordinal); - if (collectorState is not null) - { - foreach (var entry in collectorState) - { - merged[entry.Key] = entry.Value; - } - } + /* #2312: the plan and text watermark reads that used to merge in here (the #2164/#2150 host-owned + state families) are GONE — the fetches are activity-driven against the store's own map/text + tables now, so there is no persisted resume point to load. V77 deleted the orphaned rows. */ - foreach (var entry in textState) - { - merged[entry.Key] = entry.Value; - } - - collectorState = merged; - } - } - - /* #2312: the open-interval refresh stamps, the third owner merged into the same flat State — - qsowm: cannot collide with planwm:/textwm:. Read unconditionally for query_store like the - text watermark (the skip applies regardless of plan capture), and merged the same way so a + /* #2312: the open-interval refresh stamps, merged into the flat State. Read unconditionally for + query_store (the skip applies regardless of plan capture), and merged rather than replacing so a store predating this state keeps working: absent keys read as "include the open interval", - which is today's behavior exactly. */ + which is the conservative behavior. */ if (string.Equals(definition.Name, "query_store", StringComparison.Ordinal)) { var openIntervalState = await GetCollectorStateAsync( @@ -446,9 +430,16 @@ store round-trips on cancellationToken made THIS loop — the one the field repo specifically: a budget expiry abandons the whole pass, so the watermark does not advance, the clamp is re-derived next cycle, and the hole is re-recorded (merged wider with any already pending) rather than lost. */ + /* #2344: same bound as the enumerated arm. Safe here for the same reason and + by a different route — this branch does not clamp itself, but query_store's own + BuildCutoffParameters does (the #1836 double-clamp the policy documents), so the + value this read returns is clamped before anything uses it. */ + var azureReadFloor = string.Equals(definition.Name, QueryStoreCollector.Instance.Name, StringComparison.Ordinal) + ? WatermarkPolicy.ReadFloor(collectionTime) + : null; context.Watermark = await GetLastCollectedTimeForDatabaseAsync( server.ServerId, definition.TargetTable, definition.WatermarkColumn!, - definition.PerDatabaseWatermarkColumn!, databaseName, dbToken); + definition.PerDatabaseWatermarkColumn!, databaseName, dbToken, azureReadFloor); /* #2111 adaptive shrink, Azure arm — tighten BEFORE BuildQuery: the definition's own clamp only floors OLDER watermarks, so a tighter one @@ -761,9 +752,17 @@ would otherwise be silently counted as row-streaming time. Measured here so DrainMsFrom can subtract it; the whole point of the split is that each number names one real phase. */ var watermarkWatch = Stopwatch.StartNew(); + /* #2344: bound the read for the ONE collector whose value is clamped right + below. Name-guarded rather than applied to every enumerating definition, + for the reason WatermarkPolicy's remarks give: a ring-buffer source whose + legitimate catch-up spans days must keep reading its whole history, and the + floor would silently truncate it. The clamp and the bound travel together. */ + var readFloor = string.Equals(definition.Name, QueryStoreCollector.Instance.Name, StringComparison.Ordinal) + ? WatermarkPolicy.ReadFloor(collectionTime) + : null; var raw = await GetLastCollectedTimeForDatabaseAsync( server.ServerId, definition.TargetTable, definition.WatermarkColumn!, - definition.PerDatabaseWatermarkColumn!, item, ct); + definition.PerDatabaseWatermarkColumn!, item, ct, readFloor); var clamped = WatermarkPolicy.ClampCatchup(raw, collectionTime); if (raw.HasValue && clamped != raw) { @@ -882,7 +881,8 @@ type the signature needs AND gates the engine in one expression that cannot drif per-cycle cost lives HERE rather than in the payload — a 0-row cycle's blended sql: could not distinguish them. */ var planFetchWatch = Stopwatch.StartNew(); - await FetchAndStorePlansAsync(planFetchConnection, server, item, context, itemTimeout, ct); + await FetchAndStorePlansAsync(planFetchConnection, + server, item, context, itemTimeout, ExtractPlanReferences(batch), ct); context.PerItemPlanFetchMs = planFetchWatch.ElapsedMilliseconds; } @@ -899,7 +899,8 @@ advance from a cut pass. { /* #2312 investigation: same split as the plan fetch above. */ var textFetchWatch = Stopwatch.StartNew(); - await FetchAndStoreQueryTextAsync(textFetchConnection, server, item, context, itemTimeout, ct); + await FetchAndStoreQueryTextAsync(textFetchConnection, + server, item, context, itemTimeout, ExtractTextReferences(batch), ct); context.PerItemTextFetchMs = textFetchWatch.ElapsedMilliseconds; } @@ -1053,52 +1054,32 @@ exactly what they were. */ path. Outside the storage-phase timer: this is host bookkeeping, not collected data. */ if (context.PendingState.Count > 0) { - /* #2164: query_store's pending state is the plan-XML watermark set, which belongs to the host's - own state owner, NOT to the definition's name — the definition declares no state keys, so a row - written under "query_store" would never be read back and the watermark would silently never - apply. Everything else keeps writing under its definition. */ - var stateOwner = string.Equals(definition.Name, "query_store", StringComparison.Ordinal) - ? QueryStorePlanXmlState.StateCollectorName - : definition.Name; - - /* #2150/#2312: query_store's pending state now carries THREE watermark families with three - owners, so it is split by prefix on the way out. Writing one under another's owner would - still read back (the load above merges all three), but it would then never be pruned: the - shared prune set pairs each prefix with its owner, and a prefix pruned under the wrong - owner deletes nothing — which is indistinguishable from having nothing to prune. */ - var textKeys = context.PendingState - .Where(entry => entry.Key.StartsWith(QueryStoreTextState.WatermarkKeyPrefix, StringComparison.Ordinal)) - .ToDictionary(entry => entry.Key, entry => entry.Value, StringComparer.Ordinal); + /* #2312: query_store's pending state is down to ONE family — the open-interval refresh stamps + (qsowm:), which belong to the host's own state owner rather than the definition's name (the + definition declares no state keys, so a row written under "query_store" would never be read + back). The plan/text watermark families that used to be split out here retired with the + watermarks themselves; the split-by-prefix survives only as the qsowm: extraction, so a + future fourth family cannot silently land under the wrong owner and become unprunable. */ var openIntervalKeys = context.PendingState .Where(entry => entry.Key.StartsWith(QueryStoreOpenIntervalState.WatermarkKeyPrefix, StringComparison.Ordinal)) .ToDictionary(entry => entry.Key, entry => entry.Value, StringComparer.Ordinal); - if (textKeys.Count > 0 || openIntervalKeys.Count > 0) + if (openIntervalKeys.Count > 0) { + await SaveCollectorStateAsync( + server.ServerId, QueryStoreOpenIntervalState.StateCollectorName, openIntervalKeys, cancellationToken); + var others = context.PendingState - .Where(entry => !textKeys.ContainsKey(entry.Key) && !openIntervalKeys.ContainsKey(entry.Key)) + .Where(entry => !openIntervalKeys.ContainsKey(entry.Key)) .ToDictionary(entry => entry.Key, entry => entry.Value, StringComparer.Ordinal); - - if (textKeys.Count > 0) - { - await SaveCollectorStateAsync( - server.ServerId, QueryStoreTextState.StateCollectorName, textKeys, cancellationToken); - } - - if (openIntervalKeys.Count > 0) - { - await SaveCollectorStateAsync( - server.ServerId, QueryStoreOpenIntervalState.StateCollectorName, openIntervalKeys, cancellationToken); - } - if (others.Count > 0) { - await SaveCollectorStateAsync(server.ServerId, stateOwner, others, cancellationToken); + await SaveCollectorStateAsync(server.ServerId, definition.Name, others, cancellationToken); } } else { - await SaveCollectorStateAsync(server.ServerId, stateOwner, context.PendingState, cancellationToken); + await SaveCollectorStateAsync(server.ServerId, definition.Name, context.PendingState, cancellationToken); } } @@ -1359,23 +1340,94 @@ public async Task> GetCollectorStateAsync( } /// - /// Fetches one database's un-stored plan XML in plan_id order, lands it into the shared plan dimension - /// plus the map, and advances that database's watermark to what actually LANDED (#2210). + /// The cycle's distinct referenced plans with their live hashes — the probe's whole input (#2312). + /// Generic because the dispatch loop is; any batch that is not query_store rows extracts nothing, and + /// the caller's CapturePlanXml gate means that never actually happens. When one plan appears in + /// several rows (several intervals), a non-null hash wins over a null one — the probe compares against + /// whatever the engine reported, and null only means the payload row predated the hash column. + /// + private static IReadOnlyList<(long PlanId, string? PlanHash)> ExtractPlanReferences(List batch) + { + if (batch is not List rows || rows.Count == 0) + { + return Array.Empty<(long, string?)>(); + } + + var seen = new Dictionary(); + foreach (var row in rows) + { + if (row.PlanId <= 0) + { + continue; + } + + if (!seen.TryGetValue(row.PlanId, out var hash) || (hash is null && row.QueryPlanHash is not null)) + { + seen[row.PlanId] = row.QueryPlanHash; + } + } + + var references = new List<(long PlanId, string? PlanHash)>(seen.Count); + foreach (var entry in seen) + { + references.Add((entry.Key, entry.Value)); + } + + references.Sort((a, b) => a.PlanId.CompareTo(b.PlanId)); + return references; + } + + /// Text twin of , keyed by query_id with query_hash. + private static IReadOnlyList<(long QueryId, string? QueryHash)> ExtractTextReferences(List batch) + { + if (batch is not List rows || rows.Count == 0) + { + return Array.Empty<(long, string?)>(); + } + + var seen = new Dictionary(); + foreach (var row in rows) + { + if (row.QueryId <= 0) + { + continue; + } + + if (!seen.TryGetValue(row.QueryId, out var hash) || (hash is null && row.QueryHash is not null)) + { + seen[row.QueryId] = row.QueryHash; + } + } + + var references = new List<(long QueryId, string? QueryHash)>(seen.Count); + foreach (var entry in seen) + { + references.Add((entry.Key, entry.Value)); + } + + references.Sort((a, b) => a.QueryId.CompareTo(b.QueryId)); + return references; + } + + /// + /// The activity-driven plan-XML fetch for one database (#2312 Finding 2): touch-and-probe the store for + /// the cycle's referenced plans — which refreshes map/dim liveness (Finding 3's unwired TouchSql, now + /// the same round trip) and answers which plans are missing or hash-stale — then fetch exactly those by + /// id, budget-bounded, and land them into the shared dimension plus the map. The store is the + /// watermark: a caught-up database's missing set is EMPTY and no target query runs at all, which is the + /// property the retired catalog walk lacked (measured 23s per cycle to discover "nothing new"). /// - /// Failure-isolated, and that is load-bearing rather than defensive: plan XML is an enrichment on top - /// of runtime statistics, so a fetch that throws must not cost the database its runtime stats. It logs and - /// returns with the watermark untouched, which is safe by construction — the watermark only ever advances to - /// content already written, so the next pass simply re-selects the same plans. + /// Failure-isolated, and that is load-bearing rather than defensive: plan XML is an enrichment on + /// top of runtime statistics, so a fetch that throws must not cost the database its runtime stats. It + /// logs and returns; whatever did not land is still missing from the store, so the next cycle that + /// references it re-selects it by construction. /// - /// The candidate window is seeded conservatively rather than adapted, DELIBERATELY, and this is the one - /// piece of the ratified design not yet wired: the adaptive input is the previous pass's own - /// bytes-per-plan, and there is nowhere to keep it. CollectorContext is shared with Lite, so adding a - /// field is a two-host contract change — the same reasoning that put the watermark under its own state owner - /// rather than on the definition — and the state VALUE is a parsed planId:stamp pair that cannot carry - /// a third field without a format change and a migration for readers. Passing null means K comes from - /// FirstContactAvgPlanBytes, which over-estimates plan size and therefore under-sizes the window: it - /// fetches fewer plans per pass than it could, and never more than it should. Slower convergence, never - /// unsafe. + /// Budget-deferred and capped ids go to , because the probe's + /// input is each cycle's batch references: a plan referenced ONCE whose fetch was deferred would + /// otherwise never re-enter the probe. Ids the target no longer has (Query Store cleanup took the plan + /// between reference and fetch) are dropped from the debt — but only on a pass that provably completed + /// uncut, because inside a cut pass "absent from the result" and "excluded by the budget predicate" are + /// indistinguishable from the client. /// private async Task FetchAndStorePlansAsync( SqlConnection sqlConnection, @@ -1383,122 +1435,193 @@ private async Task FetchAndStorePlansAsync( string databaseName, CollectorContext context, int itemTimeout, + IReadOnlyList<(long PlanId, string? PlanHash)> references, CancellationToken cancellationToken) { try { - var watermark = QueryStorePlanXmlState.Resolve(context.State, databaseName, context.CollectionTime); + var carryKey = (server.ServerId, databaseName); + var hasCarryover = _planFetchCarryover.TryGetValue(carryKey, out var carriedIds); + if (references.Count == 0 && !hasCarryover) + { + /* The steady quiet cycle: nothing referenced, nothing owed. Zero store reads, zero target + queries — the whole point of the reshape. */ + return; + } + + await using var pgConnection = await _postgres.OpenConnectionAsync(cancellationToken); + + var missing = new SortedSet(); + if (hasCarryover) + { + foreach (var id in carriedIds!) + { + missing.Add(id); + } + } + + if (references.Count > 0) + { + var verdicts = await QueryStoreFetchProbe.TouchAndProbePlansAsync( + pgConnection, server.ServerId, databaseName, references, context.CollectionTime, cancellationToken); + foreach (var verdict in verdicts) + { + if (!verdict.Resolved || verdict.HashStale) + { + missing.Add(verdict.Id); + } + else + { + /* Resolved and current: if it was carried debt, it is paid. */ + missing.Remove(verdict.Id); + } + } + } + + if (missing.Count == 0) + { + _planFetchCarryover.TryRemove(carryKey, out _); + return; + } + var budget = context.TextByteBudgetOverride ?? 12 * 1024 * 1024; - /* #2312 Finding 1: size the window from THIS database's learned average instead of the - 160KB seed every pass — zero AvgBytes means never learned, which is the seed's job. */ - var estimate = _observedPlanSize.TryGetValue((server.ServerId, databaseName), out var carried) - ? carried + /* #2312 Finding 1 (#2322): cap the attempt from THIS database's learned average instead of the + 160KB seed every pass — zero AvgBytes means never learned, which is the seed's job. The cap + bounds server-side DECOMPRESSION (the running total materializes every plan it measures), so + it stays load-bearing even though the walk it originally sized is gone. */ + var estimate = _observedPlanSize.TryGetValue(carryKey, out var carriedEstimate) + ? carriedEstimate : default; - var candidates = QueryStorePlanXmlState.CandidatePlanCount( + var cap = QueryStorePlanXmlState.CandidatePlanCount( estimate.AvgBytes > 0 ? estimate.AvgBytes : null, budget, estimate.CatchUpInProgress, out var clamped); - if (clamped) { _logger?.LogInformation( - "query_store plan fetch on '{Server}' database [{Database}]: candidate window clamped to {K} — a bound sized this pass, not a measurement.", - server.Config.DisplayName, databaseName, candidates); + "query_store plan fetch on '{Server}' database [{Database}]: candidate cap clamped to {K} — a bound sized this pass, not a measurement.", + server.Config.DisplayName, databaseName, cap); } - var query = QueryStoreCollector.Instance.BuildPlanFetchQuery( - databaseName, context, watermark, candidates, budget); - + /* Ascending ids (SortedSet order) so the budget's in-SQL cut and the cross-chunk break are + deterministic — the same debt is retried in the same order until paid. */ + var attempt = missing.Take(cap).ToList(); + var attempted = new List(attempt.Count); var fetched = new List(); - using (var command = CreateCollectorCommand(query, sqlConnection, itemTimeout)) - await using (var reader = await command.ExecuteReaderAsync(cancellationToken)) + var shippedBytes = 0L; + var brokeOnBudget = false; + + foreach (var chunk in attempt.Chunk(PlanFetchIdsPerStatement)) { + if (shippedBytes >= budget) + { + brokeOnBudget = true; + break; + } + + var query = QueryStoreCollector.Instance.BuildPlanFetchByIdsQuery( + databaseName, context, chunk, budget - shippedBytes); + + using var command = CreateCollectorCommand(query, sqlConnection, itemTimeout); + await using var reader = await command.ExecuteReaderAsync(cancellationToken); + attempted.AddRange(chunk); while (await reader.ReadAsync(cancellationToken)) { + var planXml = reader.IsDBNull(2) ? null : reader.GetString(2); fetched.Add(new FetchedPlan( reader.GetInt64(0), - reader.IsDBNull(1) ? null : reader.GetString(1), - PlanHash: null)); + planXml, + reader.IsDBNull(1) ? null : reader.GetString(1))); + if (planXml is not null) + { + /* nvarchar length * 2 is DATALENGTH exactly — no server round-trip needed. */ + shippedBytes += (long)planXml.Length * 2; + } } } - /* Learn from what this pass actually decompressed and shipped — BEFORE the empty-pass - early return, because an empty pass is the one that proves the walk caught up (nvarchar - length * 2 is DATALENGTH exactly, no server round-trip needed). NULL-XML rows count for - the window (they shipped, the watermark passes them) but not for the average's divisor - (they carried no bytes to average — the review catch). */ - var shippedBytes = 0L; + /* NULL-XML rows count for the cap/catch-up comparison (they shipped, and the writer records + their content-less marker) but not for the average's divisor (they carried no bytes). */ var plansMeasured = 0; foreach (var plan in fetched) { if (plan.PlanXml is not null) { - shippedBytes += (long)plan.PlanXml.Length * 2; plansMeasured++; } } - _observedPlanSize[(server.ServerId, databaseName)] = - QueryStorePlanXmlState.Learn(estimate, shippedBytes, fetched.Count, plansMeasured, candidates, budget); + _observedPlanSize[carryKey] = + QueryStorePlanXmlState.Learn(estimate, shippedBytes, fetched.Count, plansMeasured, cap, budget); - if (fetched.Count == 0) + var returned = new HashSet(fetched.Count); + if (fetched.Count > 0) { - return; - } + var landed = await QueryStorePlanWriter.WriteAsync( + pgConnection, server.ServerId, databaseName, fetched, context.CollectionTime, cancellationToken); + foreach (var id in landed) + { + missing.Remove(id); + } - await using var pgConnection = await _postgres.OpenConnectionAsync(cancellationToken); - var landed = await QueryStorePlanWriter.WriteAsync( - pgConnection, server.ServerId, databaseName, fetched, context.CollectionTime, cancellationToken); + foreach (var plan in fetched) + { + returned.Add(plan.PlanId); + } + } - var advance = QueryStorePlanXmlState.AdvanceWatermark(watermark, landed); - if (!advance.ArrivedInPlanIdOrder) + /* Target-side-gone cleanup, only when the pass provably completed UNCUT: every chunk issued + and the in-SQL predicate never fired (a fired cut leaves shipped at or past the remaining + budget by the oversized-admission arithmetic). On such a pass an attempted id with no + returned row does not exist in sys.query_store_plan any more — Query Store cleanup took it + between reference and fetch — and carrying it forever would be the content-less stall + wearing a new hat. */ + if (!brokeOnBudget && shippedBytes < budget && attempted.Count == attempt.Count) { - /* Loud rather than swallowed: the fetch's ORDER BY is what makes a budget cut a suffix, so - out-of-order arrival means that safety argument no longer holds and the pass earns nothing. */ - _logger?.LogWarning( - "query_store plan fetch on '{Server}' database [{Database}]: plans arrived OUT OF plan_id order — watermark held at {Watermark}. The ORDER BY is what makes a cut safe, so this pass earned no advance.", - server.Config.DisplayName, databaseName, watermark); - return; + foreach (var id in attempted) + { + if (!returned.Contains(id)) + { + missing.Remove(id); + } + } } - if (advance.Watermark > watermark) + if (missing.Count > 0) + { + var owed = new long[missing.Count]; + missing.CopyTo(owed); + _planFetchCarryover[carryKey] = owed; + } + else { - /* Same stamp discipline as the runtime write-back: carried FORWARD across an advance, stamped - fresh only when the standing watermark was 0 (this pass WAS the full fetch). Re-stamping on - every advance would push the sweep period out forever on any database that keeps compiling. */ - var stamp = watermark > 0 - ? QueryStorePlanXmlState.ResolveStamp(context.State, databaseName) ?? context.CollectionTime - : context.CollectionTime; - - context.PendingState[QueryStorePlanXmlState.KeyFor(databaseName)] = - QueryStorePlanXmlState.Format(advance.Watermark, stamp); + _planFetchCarryover.TryRemove(carryKey, out _); } } catch (Exception ex) when (ex is not OperationCanceledException) { _logger?.LogWarning(ex, - "query_store plan fetch failed on '{Server}' database [{Database}] — runtime statistics are unaffected and the watermark is unchanged, so the next pass re-selects the same plans.", + "query_store plan fetch failed on '{Server}' database [{Database}] — runtime statistics are unaffected, and whatever did not land is still missing from the store, so the next cycle that references it re-selects it.", server.Config.DisplayName, databaseName); } } /// - /// One database's statement-text fetch (#2150), the sibling of . - /// - /// Exists because the runtime-stats payload stopped carrying query_sql_text: selecting it - /// inside the shipping TOP ... ORDER BY made a Top-N Sort materialize nvarchar(max) text - /// for the entire qualifying set before emitting row one (measured 4.67s against 0.45s - /// time-to-first-row, and neither the row cap nor the byte budget could bound it). + /// One database's statement-text fetch, the sibling of — the + /// #2150 split (text out of the runtime stream) driven the #2312 way (activity, not a watermark walk): + /// touch-and-probe query_store_text for the cycle's referenced query_ids, fetch exactly the + /// missing or hash-stale ones by id, land them. The hash-stale arm is the Query Store RESET detector — + /// ids renumber on a reset, so a stored hash differing from the live one means the id names a + /// different statement now, and its text is refetched within one cycle instead of waiting on the + /// retired daily re-walk. /// /// A failure here is text-only. Runtime statistics are already written by the time this - /// runs, and the watermark only advances on what LANDED — so a throw leaves the rows in place with their - /// text unresolved and the next pass re-selects the same statements. That is why this is a warning - /// rather than a failure of the collector. + /// runs, and whatever did not land is still missing from the store — so a throw leaves the rows in + /// place with their text unresolved and the next cycle that references them re-selects them. That is + /// why this is a warning rather than a failure of the collector. /// - /// Known property of a first fill, stated rather than discovered. The walk is ASCENDING by - /// query_id, because that is what makes a byte-budget cut a resumable suffix. On a store whose - /// watermark is still 0 that means the OLDEST statements resolve first, while the rows being collected - /// right now reference the newest ids — so a fresh store shows missing text for recent statements until - /// the walk catches up. Steady state is the opposite and is the case that matters: the watermark sits - /// near the top, so a newly-seen statement is fetched on the next pass. + /// Simpler than the plan fetch on purpose, in the same two ways the builders differ: no + /// candidate-cap estimator (DATALENGTH on text is cheap — no decompression to bound) and larger id + /// chunks. The budget, the carry-over debt, and the uncut-pass target-side-gone cleanup all work + /// exactly as the plan side documents. /// private async Task FetchAndStoreQueryTextAsync( SqlConnection sqlConnection, @@ -1506,67 +1629,136 @@ private async Task FetchAndStoreQueryTextAsync( string databaseName, CollectorContext context, int itemTimeout, + IReadOnlyList<(long QueryId, string? QueryHash)> references, CancellationToken cancellationToken) { try { - var watermark = QueryStoreTextState.Resolve(context.State, databaseName, context.CollectionTime); - var budget = context.TextByteBudgetOverride ?? 12 * 1024 * 1024; + var carryKey = (server.ServerId, databaseName); + var hasCarryover = _textFetchCarryover.TryGetValue(carryKey, out var carriedIds); + if (references.Count == 0 && !hasCarryover) + { + return; + } + + await using var pgConnection = await _postgres.OpenConnectionAsync(cancellationToken); + + var missing = new SortedSet(); + if (hasCarryover) + { + foreach (var id in carriedIds!) + { + missing.Add(id); + } + } - var query = QueryStoreCollector.Instance.BuildTextFetchQuery( - databaseName, context, watermark, QueryStoreTextState.CandidateTexts, budget); + if (references.Count > 0) + { + var verdicts = await QueryStoreFetchProbe.TouchAndProbeTextsAsync( + pgConnection, server.ServerId, databaseName, references, context.CollectionTime, cancellationToken); + foreach (var verdict in verdicts) + { + if (!verdict.Resolved || verdict.HashStale) + { + missing.Add(verdict.Id); + } + else + { + missing.Remove(verdict.Id); + } + } + } + if (missing.Count == 0) + { + _textFetchCarryover.TryRemove(carryKey, out _); + return; + } + + var budget = context.TextByteBudgetOverride ?? 12 * 1024 * 1024; + var attempt = new List(missing.Count); + foreach (var id in missing) + { + attempt.Add(id); + } + + var attempted = new List(attempt.Count); var fetched = new List(); - using (var command = CreateCollectorCommand(query, sqlConnection, itemTimeout)) - await using (var reader = await command.ExecuteReaderAsync(cancellationToken)) + var shippedBytes = 0L; + var brokeOnBudget = false; + + foreach (var chunk in attempt.Chunk(TextFetchIdsPerStatement)) { + if (shippedBytes >= budget) + { + brokeOnBudget = true; + break; + } + + var query = QueryStoreCollector.Instance.BuildTextFetchByIdsQuery( + databaseName, context, chunk, budget - shippedBytes); + + using var command = CreateCollectorCommand(query, sqlConnection, itemTimeout); + await using var reader = await command.ExecuteReaderAsync(cancellationToken); + attempted.AddRange(chunk); while (await reader.ReadAsync(cancellationToken)) { + var text = reader.IsDBNull(2) ? null : reader.GetString(2); fetched.Add(new FetchedQueryText( reader.GetInt64(0), + text, reader.IsDBNull(1) ? null : reader.GetString(1))); + if (text is not null) + { + shippedBytes += (long)text.Length * 2; + } } } - if (fetched.Count == 0) + var returned = new HashSet(fetched.Count); + if (fetched.Count > 0) { - return; - } + var landed = await QueryStoreTextWriter.WriteAsync( + pgConnection, server.ServerId, databaseName, fetched, context.CollectionTime, cancellationToken); + foreach (var id in landed) + { + missing.Remove(id); + } - await using var pgConnection = await _postgres.OpenConnectionAsync(cancellationToken); - var landed = await QueryStoreTextWriter.WriteAsync( - pgConnection, server.ServerId, databaseName, fetched, context.CollectionTime, cancellationToken); + foreach (var text in fetched) + { + returned.Add(text.QueryId); + } + } - var advance = QueryStoreTextState.AdvanceWatermark(watermark, landed); - if (!advance.ArrivedInQueryIdOrder) + /* Same uncut-pass cleanup as the plan side: an id the target no longer serves must not become + permanent debt. */ + if (!brokeOnBudget && shippedBytes < budget && attempted.Count == attempt.Count) { - /* Loud rather than swallowed, same as the plan fetch: the ORDER BY is what makes a budget cut - a suffix, so out-of-order arrival means that safety argument no longer holds and the pass - earns no advance. */ - _logger?.LogWarning( - "query_store text fetch on '{Server}' database [{Database}]: statements arrived OUT OF query_id order — watermark held at {Watermark}. The ORDER BY is what makes a cut safe, so this pass earned no advance.", - server.Config.DisplayName, databaseName, watermark); - return; + foreach (var id in attempted) + { + if (!returned.Contains(id)) + { + missing.Remove(id); + } + } } - if (advance.Watermark > watermark) + if (missing.Count > 0) + { + var owed = new long[missing.Count]; + missing.CopyTo(owed); + _textFetchCarryover[carryKey] = owed; + } + else { - /* Stamp carried FORWARD across an advance and stamped fresh only when the standing watermark - was 0 (this pass WAS the full walk). Re-stamping on every advance would push the refresh - horizon out forever on any database that keeps seeing new statements — which is exactly - where a Query Store reset, the thing the horizon exists to recover from, would hurt most. */ - var stamp = watermark > 0 - ? QueryStoreTextState.ResolveStamp(context.State, databaseName) ?? context.CollectionTime - : context.CollectionTime; - - context.PendingState[QueryStoreTextState.KeyFor(databaseName)] = - QueryStoreTextState.Format(advance.Watermark, stamp); + _textFetchCarryover.TryRemove(carryKey, out _); } } catch (Exception ex) when (ex is not OperationCanceledException) { _logger?.LogWarning(ex, - "query_store text fetch failed on '{Server}' database [{Database}] — runtime statistics are already written and the watermark is unchanged, so those rows keep unresolved text and the next pass re-selects the same statements.", + "query_store text fetch failed on '{Server}' database [{Database}] — runtime statistics are already written, and whatever did not land is still missing from the store, so the next cycle that references those statements re-selects them.", server.Config.DisplayName, databaseName); } } @@ -1899,17 +2091,35 @@ public async Task WriteBackfillBatchAsync( /// value for ONE database, for definitions with a PerDatabaseWatermarkColumn (Azure SQL DB /// per-database XE capture, #1535). Null on first run for that database or on failure — the /// caller falls back to the definition's documented window. + /// + /// bounds the read on collection_time — the + /// PARTITIONING column, so the bound actually prunes chunks (#2344). Null keeps the unbounded + /// behaviour, which is correct for any reader whose watermark is NOT clamped; pass + /// only from a caller whose value is, and read that method's + /// remarks for why the bound provably changes no answer. Unbounded, this is a MAX over a + /// non-partitioning column with no time predicate — every chunk in retention, per database, per + /// cycle, at a cost that grows with the store rather than the workload. /// public async Task GetLastCollectedTimeForDatabaseAsync( - int serverId, string tableName, string columnName, string databaseColumnName, string databaseName, CancellationToken cancellationToken) + int serverId, string tableName, string columnName, string databaseColumnName, string databaseName, + CancellationToken cancellationToken, DateTime? collectedSince = null) { try { await using var connection = await _postgres.OpenConnectionAsync(cancellationToken); - using var command = new NpgsqlCommand( - $"SELECT MAX({columnName}) FROM {tableName} WHERE server_id = $1 AND {databaseColumnName} = $2", connection); + var sql = collectedSince is null + ? $"SELECT MAX({columnName}) FROM {tableName} WHERE server_id = $1 AND {databaseColumnName} = $2" + : $"SELECT MAX({columnName}) FROM {tableName} WHERE server_id = $1 AND {databaseColumnName} = $2 AND collection_time > $3"; + using var command = new NpgsqlCommand(sql, connection); command.Parameters.AddWithValue(serverId); command.Parameters.AddWithValue(databaseName); + if (collectedSince is DateTime floor) + { + /* Naive like every other timestamp bound in this store (#1969): a Utc Kind infers + timestamptz and Postgres would convert it into the session zone on the way in. */ + command.Parameters.AddWithValue(DateTime.SpecifyKind(floor, DateTimeKind.Unspecified)); + } + var result = await command.ExecuteScalarAsync(cancellationToken); if (result is DateTime dt) { diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingConfig.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingConfig.cs index 5fa83dd29..2c153efe5 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingConfig.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingConfig.cs @@ -9,6 +9,7 @@ using System; using System.Collections.Generic; using System.IO; +using System.Linq; using System.Net; using System.Net.Sockets; using System.Text.Json; @@ -84,6 +85,17 @@ public sealed class DarlingConfig [JsonPropertyName("planContentRetentionDays")] public int PlanContentRetentionDays { get; set; } = 21; + /// + /// The per-session statement_timeout applied to the viewer and mcp roles — the hard backstop a + /// composed query can never exceed (#2357). Seeds config_service.compose_statement_timeout_seconds; + /// the store is authoritative afterwards, like every other value here. + /// + /// 15 preserves the constant it replaces. It is a judgement about store size and disk speed, which + /// this product cannot make for someone else's deployment — a fleet-wide aggregate over a wide window on a + /// large store can exceed 15s with nothing wrong. + /// + public int ComposeStatementTimeoutSeconds { get; set; } = 15; + /// /// The plan-XML storage codec (#2171). Store-backed (config_service, V62), normalized to 'gzip' or /// 'none' on read. 'gzip' (default, unchanged): plans live as gzip bytes in query_plan_gz - 14.0x @@ -178,6 +190,14 @@ public sealed class DarlingConfig [JsonPropertyName("web")] public WebConfig Web { get; set; } = new(); + /// + /// Declared PEER STORES (#2339): what this store covers, and which sibling Darling stores cover the + /// rest of the fleet. Pure disclosure — see . Optional; omit it entirely on a + /// single-store deployment and every surface behaves exactly as it did before. + /// + [JsonPropertyName("peers")] + public PeersConfig Peers { get; set; } = new(); + public static string ResolveConfigPath(string? explicitPath = null) { if (!string.IsNullOrWhiteSpace(explicitPath)) @@ -262,6 +282,10 @@ public IReadOnlyList Validate() problems.Add("postgres.connectionString is required (or set postgres.managed = true to run the bundled server)."); } + /* Peer-disclosure problems are checked BEFORE the servers early-return so a broken peers block is + reported even on a config that has no servers yet. */ + problems.AddRange(PeersConfig.Validate(Peers)); + if (Servers is null || Servers.Count == 0) { problems.Add("servers must contain at least one entry."); @@ -504,6 +528,13 @@ public sealed class AlertsConfig [JsonPropertyName("pvsFloorGb")] public int PvsFloorGb { get; set; } = 1; + /* #2349: the database file-growth alert. Ships OFF -- a new alert that starts firing on upgrade is a bad + citizen, and the right thresholds are a property of the fleet rather than of the product. */ + public bool FileGrowthEnabled { get; set; } + public int FileGrowthRiseMb { get; set; } = 10240; + public int FileGrowthVolumePercent { get; set; } = 60; + public int FileGrowthLookbackMinutes { get; set; } = 60; + /// #2107: the store volume's self-alert warning percent (was a compile-time 10.0; /// 0 disables the check — percent is its only trigger). [JsonPropertyName("selfDiskFreeWarnPercent")] @@ -1232,3 +1263,164 @@ public sealed class MonitoredServer public int ServerId => StoredServerId ?? PerformanceMonitor.Common.ServerIdHelper.GetDeterministicHashCode(StorageName); } + +/// +/// Declared peer stores (#2339, tier 1) — the peers block. A fleet split across several boxes gives +/// every box a Darling store that knows only its own slice, so an agent asking the wrong endpoint about a +/// server gets a not-found that is indistinguishable from "nobody monitors it". Declaring the siblings here +/// makes the split legible in the MCP instructions, in list_servers, and in the server-resolution +/// miss message. +/// +/// Disclosure only — there are no credentials in this block and there is no connectivity behind +/// it. A peer is a NAME and a sentence; this service never contacts one, cannot read one's data, and +/// cannot tell whether one is even running. Deliberately so: cross-store reads (auth between stores, +/// latency, partial failures) are a much larger surface and may never be worth building if disclosure alone +/// makes the split legible. therefore REFUSES text that looks like a connection +/// string or credential, because everything here is sent verbatim to every connected MCP client. +/// +/// A file-only block (not seeded into the control-plane store): it describes the DEPLOYMENT TOPOLOGY of +/// the box this config sits on, which is exactly the kind of thing that must not be editable from a peer's +/// Viewer. An edit takes effect on the next service restart. +/// +public sealed class PeersConfig +{ + /// + /// One sentence naming what THIS store monitors — the anchor everything else is relative to + /// ("the 42 us-east-1 SQL Server primaries"). Optional; omit it and the peer list is still disclosed. + /// + [JsonPropertyName("thisStoreCovers")] + public string ThisStoreCovers { get; set; } = ""; + + /// The sibling Darling stores. Empty (the default) = nothing declared, and every surface behaves as before. + [JsonPropertyName("stores")] + public List Stores { get; set; } = new(); + + /* Substrings that mean the operator pasted a credential or a whole connection string into a field whose + entire purpose is to be broadcast. Matched case-insensitively against every peer string. Deliberately + a short, high-signal list rather than a secret detector: it catches the realistic mistake (copying a + peer's connectionString in as its description) without pretending to be a scanner. */ + private static readonly string[] CredentialShapedTokens = + { + "password=", "pwd=", "connectionstring", "integrated security=", "accountkey=", "secretaccesskey", + }; + + /// + /// Validates a peers block; returns human-readable problems (empty = valid). Fatal, like the rest of + /// , for exactly two shapes: a peer that cannot be NAMED (an agent + /// told "some other store has it" with no name to point a human at is no better off than before), and + /// any peer text that looks like it carries a secret (failing open there would broadcast it). + /// A peer with a name but no covers sentence is allowed — half a disclosure still names an + /// endpoint — and renders as just the name. + /// + public static IReadOnlyList Validate(PeersConfig? peers) + { + var problems = new List(); + if (peers is null) + { + return problems; + } + + /* thisStoreCovers is checked FIRST, before anything can short-circuit. It used to live after the + per-peer loop behind an early `peers?.Stores is null` return, which meant an explicit + `"stores": null` in the JSON (System.Text.Json assigns null over the property initializer, unlike + an omitted key) skipped the credential guard on the one field that is still disclosed in that + config — instructions, list_servers' this_store_covers, and every resolution miss. Caught in + review on #2339. Ordering, not an extra check, is the fix: an unconditional guard cannot be + bypassed by a shape nobody thought to enumerate. */ + var selfText = peers.ThisStoreCovers ?? ""; + var selfOffending = CredentialShapedTokens.FirstOrDefault( + t => selfText.Contains(t, StringComparison.OrdinalIgnoreCase)); + + if (selfOffending is not null) + { + problems.Add( + $"peers.thisStoreCovers contains '{selfOffending}'. The peers block is DISCLOSURE ONLY — its text " + + "is sent verbatim to every connected MCP client — so it must carry no connection string and no " + + "credential."); + } + + if (peers.Stores is null) + { + return problems; + } + + for (int i = 0; i < peers.Stores.Count; i++) + { + var peer = peers.Stores[i]; + if (peer is null) + { + continue; + } + + var label = string.IsNullOrWhiteSpace(peer.Name) ? $"peers.stores[{i}]" : $"peer '{peer.Name.Trim()}'"; + + if (string.IsNullOrWhiteSpace(peer.Name)) + { + problems.Add( + $"{label}: name is required — it is what an agent tells its human to point at, so a peer " + + "with only a description cannot be acted on."); + } + + foreach (var value in PeerStrings(peer)) + { + var offending = CredentialShapedTokens.FirstOrDefault( + t => value.Contains(t, StringComparison.OrdinalIgnoreCase)); + + if (offending is not null) + { + problems.Add( + $"{label}: peer text contains '{offending}'. The peers block is DISCLOSURE ONLY — its " + + "text is sent verbatim to every connected MCP client — so it must carry no connection " + + "string and no credential. Describe what the peer monitors, not how to reach it."); + break; + } + } + } + + return problems; + } + + private static IEnumerable PeerStrings(PeerStoreConfig peer) + { + yield return peer.Name ?? ""; + yield return peer.Covers ?? ""; + + foreach (var match in peer.Matches ?? new List()) + { + yield return match ?? ""; + } + } +} + +/// +/// One declared peer store: what to call it, what it covers in prose, and optionally which server names it +/// owns. No address and no credential by design — see . +/// +public sealed class PeerStoreConfig +{ + /// + /// What to call this store — whatever an operator would recognize (the box name, "the use1 store"). + /// Required: it is the actionable half of the disclosure. + /// + [JsonPropertyName("name")] + public string Name { get; set; } = ""; + + /// + /// A short sentence naming what that store monitors ("the readable replicas of the use1 primaries, from + /// us-east-2"). Human prose — never parsed, only shown. + /// + [JsonPropertyName("covers")] + public string Covers { get; set; } = ""; + + /// + /// Optional server-name substrings this peer declares it monitors ("use1", "-replica"), + /// matched case-insensitively. This is the ONLY machine-checked field: it is what lets a resolution miss + /// name the specific peer that owns the server instead of listing them all. Blank entries are dropped — + /// an empty substring matches every name, which would make one peer claim the whole fleet. No globbing + /// and no regex on purpose: a pattern language is a config surface with its own failure modes, and a + /// substring answers the question actually being asked (which region/role does this name belong to?). + /// A peer that declares none is still disclosed everywhere; it just cannot be singled out on a miss. + /// + [JsonPropertyName("matches")] + public List Matches { get; set; } = new(); +} diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingFileSecurity.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingFileSecurity.cs index 25eb87cf7..880cd11e4 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingFileSecurity.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingFileSecurity.cs @@ -60,9 +60,97 @@ public static class DarlingFileSecurity new(WellKnownSidType.WorldSid, null), ]; - /// The account this process runs as — the service account when hosted as a service. + /* #2371: set by --harden-files to the account the SERVICE is registered under, because that verb is the + one caller that is deliberately NOT the service. Null everywhere else, so the in-service callers keep + resolving themselves exactly as before. */ + private static SecurityIdentifier? _serviceAccountOverride; + + /// + /// Harden FOR rather than for whoever is running this process (#2371). + /// + /// Every original caller of this class runs INSIDE the service, so "the current identity" and "the + /// account the service runs as" were the same value and the distinction did not exist. --harden-files + /// breaks that: it exists precisely because a virtual service account cannot re-ACL a file it does not own, + /// so it is always run by somebody else — and resolving from the caller there grants the OPERATOR and drops + /// the service, which is a working install turned into one that fails on its next start. + /// + public static void HardenForAccount(SecurityIdentifier account) => _serviceAccountOverride = account; + + /// + /// The account the service is REGISTERED under, read from its SCM entry rather than from this process — + /// `ObjectName` is what the SCM logs the service on with. Returns null when the service is not registered + /// (a console run, or hardening a tree before install), leaving the caller to fall back. + /// + /// Handles the well-known aliases the SCM stores unqualified: LocalSystem has no + /// spelling to translate, while a virtual account (NT SERVICE\…), a domain + /// account and a gMSA all translate directly. + /// + public static SecurityIdentifier? RegisteredServiceAccount(string serviceName) + { + try + { + using var key = Microsoft.Win32.Registry.LocalMachine.OpenSubKey( + $@"SYSTEM\CurrentControlSet\Services\{serviceName}"); + + if (key?.GetValue("ObjectName") is not string account || string.IsNullOrWhiteSpace(account)) + { + return null; + } + + account = account.Trim(); + + return account.Equals("LocalSystem", StringComparison.OrdinalIgnoreCase) + ? new SecurityIdentifier(WellKnownSidType.LocalSystemSid, null) + : (SecurityIdentifier)new NTAccount(account).Translate(typeof(SecurityIdentifier)); + } + catch (Exception) + { + /* Same reasoning as the display name below: a resolution failure must degrade to the old + behaviour, never take the harden down. */ + return null; + } + } + + /// + /// Whether still grants the account being hardened for. The harden's own check is + /// , which asks whether anyone TOO MANY can read — this is the + /// other half, and the one #2371 needed: an ACL can be perfectly private and still lock the service out + /// of its own credentials. + /// + public static bool GrantsHardenedAccount(string path) + { + try + { + var target = ServiceAccount; + var rules = (Directory.Exists(path) + ? new DirectoryInfo(path).GetAccessControl().GetAccessRules(true, true, typeof(SecurityIdentifier)) + : new FileInfo(path).GetAccessControl().GetAccessRules(true, true, typeof(SecurityIdentifier))); + + foreach (FileSystemAccessRule rule in rules) + { + if (rule.AccessControlType == AccessControlType.Allow + && rule.IdentityReference is SecurityIdentifier sid + && sid.Equals(target) + && (rule.FileSystemRights & FileSystemRights.Read) != 0) + { + return true; + } + } + + return false; + } + catch (Exception) + { + /* Unreadable ACL is not proof of absence, so do not report a lockout we cannot see. */ + return true; + } + } + + /// The account to harden FOR: the registered service account when + /// named one, otherwise the account this process runs as. private static SecurityIdentifier ServiceAccount => - WindowsIdentity.GetCurrent().User + _serviceAccountOverride + ?? WindowsIdentity.GetCurrent().User ?? throw new InvalidOperationException("Cannot resolve the current Windows identity for ACL hardening."); /// @@ -79,7 +167,11 @@ public static string ServiceAccountDisplayName { try { - return WindowsIdentity.GetCurrent().Name; + /* #2371: when hardening for the REGISTERED account, name that one — printing the caller here + would describe an ACL the harden is not writing. */ + return _serviceAccountOverride is not null + ? ((NTAccount)_serviceAccountOverride.Translate(typeof(NTAccount))).Value + : WindowsIdentity.GetCurrent().Name; } catch (Exception) { diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingInstallLocation.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingInstallLocation.cs new file mode 100644 index 000000000..e900b5c3a --- /dev/null +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingInstallLocation.cs @@ -0,0 +1,421 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.IO; +using System.Runtime.Versioning; +using System.Security; +using System.Text; +using Microsoft.Extensions.Logging; +using Microsoft.Win32; + +namespace PerformanceMonitor.Darling.Service; + +/// Why an install directory is one the service's own account cannot read. is the +/// only value that means "this location is fine". +internal enum InstallLocationVerdict +{ + /// A machine-scoped local path. Nothing to say. + None, + + /// At or under a user profile — the reported shape (#2185): a zip extracted to a Desktop. + UserProfile, + + /// A UNC path (\\server\share\...). + UncPath, + + /// A drive letter mapped to a network share. + MappedDrive, +} + +/// +/// Says out loud, at service start, that the install directory is one the service account can never read +/// (#2185) — before the bundled PostgreSQL bootstrap turns that into an exit code nobody can read. +/// +/// The reported experience. A zip extracted to +/// C:\Users\username\Desktop\PerformanceMonitorDarling-3.2.0\ — a completely reasonable thing to do +/// with a download — produced initdb failed (exit code -1073741515) with an empty Output:, and +/// then a message about a missing pg-admin-credential.dpapi telling the operator to "start the service +/// once", which they had. Every visible message was downstream of the install location, and none of them +/// named it. #2186 decoded the exit code, #2197 fixed the credential message's advice, and #2187 taught +/// install-darling.ps1 to refuse the location — but the installer can only guard the installs that go +/// through it. The README's manual sc create path, and anyone who registers the exe by hand, bypass it +/// entirely, which is why the SERVICE has to be able to reach this conclusion by itself. +/// +/// The mechanism, measured rather than assumed (#2187, on a clean Windows 11 box). The service +/// runs as the virtual account NT SERVICE\PerformanceMonitor Darling and never as LocalSystem, because +/// the bundled PostgreSQL refuses to run with administrative privileges. A directory created under a profile +/// inherits exactly SYSTEM / Administrators / the profile owner — no BUILTIN\Users, no +/// Authenticated Users, no CREATOR OWNER — so that account cannot read the program files it was pointed at, +/// and cannot read back what it writes there itself. A directory created under C:\ inherits +/// BUILTIN\Users:(RX) from the volume root instead, which every service account is a member of, which +/// is why the documented location works and a profile does not. +/// +/// Why this diagnoses and does not refuse to start. The installer's asymmetry, for the same +/// reason (#2187): a fresh install is refused outright, but an existing service already living there is asked +/// rather than blocked, because an operator may have granted the tree read + execute by hand — #2187's +/// rejected option 2 — and stranding a deployment that runs today would be worse than the disease. A service +/// has nobody to ask, so it takes the same side: it states the cause, unmissably and BEFORE the failure it +/// predicts, and then gets on with the start it was asked for. An operator who hand-granted the tree keeps a +/// working service and a standing note that it is one ACL reset away from breaking, which is true. +/// +/// The decision table is deliberately the same one install-darling.ps1 applies — profile +/// root from ProfileList\ProfilesDirectory plus %USERPROFILE%, UNC, and a drive letter whose type +/// is network, each applied to the path with any \\?\ extended-length prefix already stripped (#2348). Two definitions of "a location that cannot +/// work" would drift, and the one that drifted would be the one nobody was reading. DarlingInstallLocationTests +/// runs BOTH over one shared table and fails if they ever disagree. Making the C# the single source and having +/// the script call it is the better end state and is NOT done here: it would change the installer's behavior, +/// which is not this change's to make. +/// +[SupportedOSPlatform("windows")] +internal static class DarlingInstallLocation +{ + /// The documented machine-scoped location, named in every remedy this class prints so the + /// operator is never left with a refusal and no destination. + internal const string DocumentedInstallDirectory = @"C:\PerformanceMonitorDarling"; + + /// + /// The one message, logged critical, when is somewhere the service + /// account cannot read. Never throws: a diagnostic that costs the service its start is worse than the + /// diagnostic being absent. + /// + /// Normally AppContext.BaseDirectory. + /// + /// silences this entirely, and that is the point of the parameter rather than a + /// convenience. A console run from a Desktop folder is a SUPPORTED thing to do — the README suggests + /// test-driving the service interactively, and an interactive run is the profile owner, so the tree it + /// cannot read as a service is perfectly readable as them. Warning there would be the one false positive + /// available to this check, aimed at someone doing exactly what the docs told them to. + /// + internal static void Report(string installDirectory, bool runningAsWindowsService, ILogger logger) + { + if (!runningAsWindowsService) + { + return; + } + + try + { + var verdict = Classify( + installDirectory, + MachineProfileRoot(), + Environment.GetEnvironmentVariable("USERPROFILE"), + IsNetworkDrive); + + if (verdict == InstallLocationVerdict.None) + { + return; + } + + logger.LogCritical("{Diagnosis}", Describe(verdict, installDirectory, DarlingFileSecurity.ServiceAccountDisplayName)); + } + catch (Exception ex) + { + logger.LogWarning( + "Could not check whether the install directory {Root} is one this service's account can read ({Message}).", + installDirectory, ex.Message); + } + } + + /// + /// The decision, with every environment lookup passed in so the table can be driven by a test rather than + /// by the box the test runs on. + /// + /// The profile answer WINS over the network answer when a path is somehow both, matching the + /// installer: the profile case is the one that gets reported, so it is the one whose explanation an + /// operator most needs to read. + /// + /// The machine's profile root, e.g. C:\Users. + /// + /// The current identity's own profile. Checked as well as the machine root because a profile redirected + /// outside ProfilesDirectory is still a profile. Under a service this is the service account's + /// profile rather than an operator's, so it is the arm that almost never fires here — kept anyway, + /// because a definition that is the same as the installer's except for one arm is exactly the drift this + /// class exists to avoid. + /// + /// Given a qualifier such as Z:, whether it maps to a share. + internal static InstallLocationVerdict Classify( + string installDirectory, + string? profilesDirectory, + string? userProfileDirectory, + Func isNetworkDrive) + { + if (string.IsNullOrWhiteSpace(installDirectory)) + { + /* Nothing is not somewhere. An empty path must never become a relative one resolved against + whatever the working directory happens to be. */ + return InstallLocationVerdict.None; + } + + /* #2348: strip the extended-length prefix FIRST, so every test below sees the ordinary spelling of + the same location. The prefix is an instruction to the path parser, never part of the identity of + where the install lives, so classifying its stripped form is classifying the same directory. + + This replaces a wholesale `\\?\` exclusion in the UNC test below, which was correct about + \\?\C:\PerformanceMonitorDarling (an ordinary local root written the long way, which must not be + refused) and wrong about the two locations this class exists to catch: \\?\UNC\server\share is a + REAL share, and \\?\C:\Users\bob is a REAL profile that `C:\Users` prefix-matching misses. Skipping + the check is not the same as passing it, and the old exclusion conflated them. + + Normalizing at the entry rather than carving out each test is what keeps this honest: there is one + place that knows about the prefix, and every rule downstream is written against real paths. */ + installDirectory = StripExtendedLengthPrefix(installDirectory); + + if (IsAtOrUnder(installDirectory, profilesDirectory) || IsAtOrUnder(installDirectory, userProfileDirectory)) + { + return InstallLocationVerdict.UserProfile; + } + + /* No `\\?\` carve-out any more: the prefix is gone by here, so an extended-length LOCAL root has + already become C:\... and cannot reach this test, while an extended-length SHARE has become + \\server\share and correctly does. */ + if (installDirectory.StartsWith(@"\\", StringComparison.Ordinal)) + { + return InstallLocationVerdict.UncPath; + } + + var qualifier = DriveQualifier(installDirectory); + if (qualifier is not null && isNetworkDrive(qualifier)) + { + return InstallLocationVerdict.MappedDrive; + } + + return InstallLocationVerdict.None; + } + + /// + /// Rewrites an extended-length path to its ordinary spelling (#2348), leaving anything else alone: + /// \\?\UNC\server\share becomes \\server\share and \\?\C:\dir becomes C:\dir. + /// + /// The UNC form is tested FIRST because it is the longer, more specific prefix — checking + /// \\?\ first would strip four characters off a share and leave the nonsense UNC\server\share, + /// which is neither a share nor a local path and would classify as neither. + /// + /// The UNC segment is matched case-insensitively because Windows accepts \\?\unc\ too, + /// and a miss there would silently re-open exactly the hole this closes. The \\?\ prefix itself has + /// no letters, so it is matched ordinally. + /// + /// This normalizes for CLASSIFICATION only. The caller keeps its own path for everything else — the + /// service still starts from, and every message still names, the directory as it actually is. + /// + internal static string StripExtendedLengthPrefix(string path) + { + const string extendedUncPrefix = @"\\?\UNC\"; + const string extendedPrefix = @"\\?\"; + + if (string.IsNullOrEmpty(path)) + { + return path; + } + + if (path.StartsWith(extendedUncPrefix, StringComparison.OrdinalIgnoreCase)) + { + return @"\\" + path[extendedUncPrefix.Length..]; + } + + if (path.StartsWith(extendedPrefix, StringComparison.Ordinal)) + { + return path[extendedPrefix.Length..]; + } + + return path; + } + + /// + /// True when IS or sits underneath it. + /// + /// The separator is appended before the prefix test on purpose, and it is the one thing here worth + /// being exact about: a bare StartsWith reads C:\UsersData as living under C:\Users, + /// and a false refusal is a worse failure than the one this check exists to catch — it maligns an install + /// that would have worked. Equality counts as under: a root sitting AT the profile root is exactly as + /// unreadable as one below it. + /// + internal static bool IsAtOrUnder(string? candidate, string? parent) + { + if (string.IsNullOrWhiteSpace(candidate) || string.IsNullOrWhiteSpace(parent)) + { + return false; + } + + string child; + string root; + try + { + /* Normalizes separators, resolves . and .., and makes the answer independent of how the path was + typed. Anything GetFullPath rejects is not a path we can reason about, so it is not matched. */ + child = Path.TrimEndingDirectorySeparator(Path.GetFullPath(candidate)); + root = Path.TrimEndingDirectorySeparator(Path.GetFullPath(parent)); + } + catch (Exception ex) when (ex is ArgumentException or NotSupportedException or PathTooLongException or IOException) + { + return false; + } + + return child.Equals(root, StringComparison.OrdinalIgnoreCase) + || child.StartsWith(root + Path.DirectorySeparatorChar, StringComparison.OrdinalIgnoreCase); + } + + /// + /// Z: for Z:\Darling, null for anything without a drive letter — the C# equivalent of the + /// installer's Split-Path -Qualifier. + /// + /// Read off the string rather than through so it answers the + /// same way on the machine running the tests as on Windows: a drive qualifier is a Windows concept, and + /// GetPathRoot returns nothing for one on any other platform, which would make the mapped-drive + /// arm of the table unrunnable anywhere but Windows. + /// + internal static string? DriveQualifier(string path) + { + if (path.Length < 2 || path[1] != ':') + { + return null; + } + + return char.IsLetter(path[0]) ? path.Substring(0, 2) : null; + } + + /// + /// The message. One log line, three jobs: name the path, say why the account cannot read it, and say what + /// to do about it — with the downstream messages named explicitly, because those are the ones the operator + /// has already been chasing by the time they read this (#2185). + /// + internal static string Describe(InstallLocationVerdict verdict, string installDirectory, string serviceAccount) + { + var message = new StringBuilder(); + + message.Append("This service is installed somewhere its own account cannot read: ") + .Append(installDirectory).Append('.'); + + switch (verdict) + { + case InstallLocationVerdict.UserProfile: + message.Append(" That path is under a user profile, and this service is running as ") + .Append(serviceAccount) + .Append(" — which is not you, not SYSTEM, and not Administrators. A directory created under a profile grants access to about those three and nobody else (no BUILTIN\\Users, no Authenticated Users, no CREATOR OWNER), so the service cannot read its own program files there, and cannot even read back what it writes there itself."); + break; + + case InstallLocationVerdict.UncPath: + message.Append(" That path is a UNC network path, and this service is running as ") + .Append(serviceAccount) + .Append(" — a virtual service account reaches the network as the COMPUTER account rather than as the operator who installed it, so a share that opens for you is not open for it."); + break; + + case InstallLocationVerdict.MappedDrive: + message.Append(" That path is on a mapped network drive, and this service is running as ") + .Append(serviceAccount) + .Append(" — a drive letter belongs to the logon session that mapped it, which a service does not share and cannot see at all, and a virtual service account reaches the network as the COMPUTER account rather than as you."); + break; + + default: + /* Describe is only reached with a real verdict; Report returns on None. Kept explicit rather + than throwing, because a diagnostic must never be the thing that fails. */ + break; + } + + /* WHAT fails next depends on postgres.managed, and this message is built BEFORE darling.json is read + (deliberately - an unreadable tree takes the config out too), so it cannot know which. Naming both + costs two sentences and is the difference between a diagnosis and a confident guess: an operator on + bring-your-own Postgres has no initdb to fail, and telling them to go looking for one would spend + the credibility this message exists to have (review catch on #2185). */ + message.Append(" What fails next depends on how this service is configured, and the location is the cause either way.") + .Append(" With the shipped managed store (postgres.managed = true, the default) it is the bundled PostgreSQL bootstrap: initdb.exe dies at exit code -1073741515 (0xC0000135, STATUS_DLL_NOT_FOUND) in the Windows loader, before it can write a word of output — and the messages after it are downstream of THIS rather than faults of their own, including a missing pg-admin-credential.dpapi and advice to start the service once, which starting the service again will not satisfy (#2185).") + .Append(" Pointed at your own PostgreSQL (postgres.managed = false) there is no initdb to fail, and the unreadability surfaces wherever the service next reads this folder instead — \"Cannot load configuration\", from a darling.json sitting right here that it cannot open, is the usual one.") + .Append(" FIX: stop the service, move the install to a machine-scoped local path — ") + .Append(DocumentedInstallDirectory) + .Append(" is the documented one, and a folder created there inherits read + execute for BUILTIN\\Users, which the service account is a member of — then re-run install-darling.ps1 from the new location. It updates the service in place; darling.json can move with the folder, and the store under %ProgramData%\\PerformanceMonitorDarling and its credentials stay exactly where they are."); + + return message.ToString(); + } + + /// + /// The machine's profile root, read from where Windows actually keeps it rather than assumed to be + /// C:\Users: ProfilesDirectory is relocatable, and a hardcoded literal would quietly stop + /// matching on precisely the box that moved it — the one box where a missed check costs the most. + /// + /// Read from the same registry value the installer reads, and this is the second attempt: the + /// first derived it from %PUBLIC%'s parent, on the belief that Windows keeps PUBLIC in step + /// with ProfilesDirectory. It does not. ProfilesDirectory and Public are two + /// INDEPENDENT REG_EXPAND_SZ values under the same key that merely default to the same tree, so an + /// administrator who relocates profiles to D:\Profiles — a documented, supported move — without also + /// moving Public leaves %PUBLIC% at C:\Users\Public. The service would then have answered + /// C:\Users while the installer answered D:\Profiles, and an install under the box's real + /// profile root would have passed silently: a false negative on exactly the case this exists to catch, on + /// exactly the box where a missed check costs the most. Reading the value itself removes the divergence + /// instead of documenting it (review catch on #2185). + /// + /// Falls back to %SystemDrive%\Users, which is what the installer falls back to. An + /// unreadable ProfileList is not a reason to skip the check. + /// + internal static string MachineProfileRoot() + { + try + { + using var profileList = Registry.LocalMachine.OpenSubKey( + @"SOFTWARE\Microsoft\Windows NT\CurrentVersion\ProfileList"); + + /* GetValue expands a REG_EXPAND_SZ by default; expanding again is harmless and keeps this + correct if the value is ever stored as a plain string containing %SystemDrive%. */ + if (profileList?.GetValue("ProfilesDirectory") is string configured + && !string.IsNullOrWhiteSpace(configured)) + { + var expanded = Environment.ExpandEnvironmentVariables(configured); + if (!string.IsNullOrWhiteSpace(expanded)) + { + return expanded; + } + } + } + catch (Exception ex) when (ex is SecurityException or UnauthorizedAccessException or IOException or ObjectDisposedException or PlatformNotSupportedException) + { + /* An unreadable ProfileList is not a reason to skip the check — fall through to the default. */ + } + + return ProfileRootForSystemDrive(Environment.GetEnvironmentVariable("SystemDrive")); + } + + /// + /// The fallback root, as a pure function of the system drive, so the branch is reachable from a test — the + /// registry read succeeds on any box CI runs on, which is precisely why the bug below shipped. + /// + /// Not , and this is the whole reason the method + /// exists. %SystemDrive% is documented to be a bare drive and colon with no trailing separator + /// (C:), and Path.Combine treats a trailing volume separator the way it treats a trailing + /// directory separator: it inserts nothing. So Path.Combine("C:", "Users") is C:Users, a + /// DRIVE-RELATIVE path that then resolves against the process's + /// current directory on that volume rather than against the volume root. The profile check would have + /// silently stopped matching real profile installs on every box whose ProfileList is unreadable — + /// the one box the fallback exists for. PowerShell's Join-Path 'C:' 'Users' does NOT have this + /// behavior, so it was also a silent divergence from the installer (review catch on #2185). + /// + internal static string ProfileRootForSystemDrive(string? systemDrive) + { + var drive = string.IsNullOrWhiteSpace(systemDrive) ? "C:" : systemDrive.Trim(); + + /* Tolerate a drive that already carries a separator (C:\) as well as the documented bare form. */ + drive = drive.TrimEnd(Path.DirectorySeparatorChar, Path.AltDirectorySeparatorChar); + + return drive + Path.DirectorySeparatorChar + "Users"; + } + + /// + /// Whether a drive qualifier maps to a share. Unknown answers "no": a refusal needs evidence, and the + /// cost of missing one mapped drive is a message that does not appear, while the cost of inventing one is + /// telling an operator their working install is broken. + /// + private static bool IsNetworkDrive(string qualifier) + { + try + { + return new DriveInfo(qualifier).DriveType == DriveType.Network; + } + catch (Exception ex) when (ex is ArgumentException or IOException or UnauthorizedAccessException) + { + return false; + } + } +} diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingManagedRoles.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingManagedRoles.cs index f3557b8c4..9ed028db4 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingManagedRoles.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingManagedRoles.cs @@ -200,7 +200,14 @@ public static async Task EnsureProvisionedAsync( allowInteractiveRead: false, logger); await using var connection = await dataSource.OpenConnectionAsync(cancellationToken); - await using var command = new NpgsqlCommand(BuildProvisioningSql(adminPassword, viewerPassword, mcpPassword), connection); + /* #2357: read the live knob rather than a constant. Ordering is what makes this safe -- migrations + run before provisioning at startup, so the column exists by now -- and because this DDL is re-run + on every managed start, a changed value reaches an existing install on its next restart without + any new machinery. A store whose config row is not seeded yet answers with the default. */ + var composeTimeoutSeconds = await ReadComposeStatementTimeoutAsync(connection, cancellationToken); + + await using var command = new NpgsqlCommand( + BuildProvisioningSql(adminPassword, viewerPassword, mcpPassword, composeTimeoutSeconds), connection); await command.ExecuteNonQueryAsync(cancellationToken); logger.LogInformation( @@ -287,6 +294,30 @@ private static void TryDelete(string path, ILogger logger) } } + /// + /// The store's compose statement_timeout in seconds (#2357), or 15 when it cannot be read. + /// + /// Defensive on purpose. This runs during startup provisioning, before the config row is + /// necessarily seeded and on stores that may predate the column, and a role-provisioning step that threw + /// over a tuning knob would stop the service from starting over something that has a perfectly good + /// default. + /// + private static async Task ReadComposeStatementTimeoutAsync( + NpgsqlConnection connection, CancellationToken cancellationToken) + { + try + { + await using var command = new NpgsqlCommand( + "SELECT compose_statement_timeout_seconds FROM config.config_service WHERE id = 1", connection); + var value = await command.ExecuteScalarAsync(cancellationToken); + return value is int seconds ? seconds : 15; + } + catch (Exception) + { + return 15; + } + } + /// /// The idempotent, self-healing provisioning DDL with the role passwords injected. Passwords are /// alnum-only (), verified here before the @@ -297,7 +328,12 @@ private static void TryDelete(string path, ILogger logger) /// batch requires those tables to already exist — safe because provisioning runs AFTER migration (see /// ); a dropped/recreated table re-grants on the next start. /// - public static string BuildProvisioningSql(string adminPassword, string viewerPassword, string mcpPassword) + /// + /// The per-session statement_timeout for the viewer and mcp roles (#2357). Defaults to the 15 the + /// constant used to hard-code, so a caller that does not care gets today's behaviour exactly. + /// + public static string BuildProvisioningSql( + string adminPassword, string viewerPassword, string mcpPassword, int composeStatementTimeoutSeconds = 15) { RequireAlphanumeric(adminPassword, nameof(adminPassword)); RequireAlphanumeric(viewerPassword, nameof(viewerPassword)); @@ -311,7 +347,12 @@ public static string BuildProvisioningSql(string adminPassword, string viewerPas const string collect = PgSchemaGenerator.CollectSchema; const string config = PgSchemaGenerator.ConfigSchema; const string marker = RoleMarker; - const string statementTimeout = ComposeLimits.StatementTimeout; + /* #2357: was ComposeLimits.StatementTimeout, a bare "15s". Clamped here as well as on the config + read, because this method is public and a caller passing 0 would remove the backstop entirely -- + and the backstop is the whole reason the value exists: a LIMIT bounds output, a group-by scans and + sorts before it, so something has to bound WORK. */ + var statementTimeout = + $"{Math.Clamp(composeStatementTimeoutSeconds <= 0 ? 15 : composeStatementTimeoutSeconds, 5, 600)}s"; /* The fail-closed viewer column-ACL carve for the secret-bearing config tables (see ViewerRestrictedConfigTables). Runs AFTER the blanket config GRANT below, so it strips diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingRetention.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingRetention.cs index 8b75c5535..fa6fcf25f 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingRetention.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingRetention.cs @@ -721,7 +721,7 @@ internal static DateTime ComputeDimTableCutoff(string dimTable, DateTime coupled /// renders "not collected" and self-corrects; content pruned while a map row survives is a live fact /// resolving to absent XML, silently. The coupled pair keeps that gap at ChunkIntervalDays; the /// dedicated pair keeps it at one day (map at knob, dim at knob + 1 — the same one-day stamp-skew - /// margin as everywhere else, because TouchSql refreshes the map's stamp eagerly while the + /// margin as everywhere else, because TouchAndProbeSql refreshes the map's stamp eagerly while the /// dim's refresh is hourly-guarded, so the dim's stamp can trail). Both components are strictly /// ordered, so the max-of-newer composition preserves the ordering under every knob value — /// pinned in PlanContentRetentionTests across the full age sweep. diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingServerConnector.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingServerConnector.cs index 8ecd55bad..9bbe764fa 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingServerConnector.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingServerConnector.cs @@ -188,9 +188,9 @@ public static string ResolveConnectionString(MonitoredServer config, ILogger? lo than string parsing (version() text formatting has changed across releases). pg_is_in_recovery() -> reader vs writer. On Aurora every reader endpoint is its own instance with its own statistics, so this is identity, not a routing hint. - aurora_version() -> present only on Aurora. Wrapped: on stock PostgreSQL - the function does not exist, and a missing function must read as "not Aurora" rather than - failing the whole probe. + + The Aurora marker is NOT here. It used to be a pg_proc lookup on this query and was wrong on real + Aurora (#2340) — it now lives in PostgresAuroraProbeQueryText, which CALLS the function instead. No timezone offset column: unlike SQL Server's DATEDIFF-on-GETDATE idiom, Postgres timestamps here are read as-is and the store's convention is naive UTC either way. */ @@ -199,11 +199,61 @@ here are read as-is and the store's convention is naive UTC either way. */ version() AS server_version_text, current_setting('server_version_num')::int / 10000 AS major_version, pg_is_in_recovery() AS is_in_recovery, - (SELECT count(*) FROM pg_proc WHERE proname = 'aurora_version') > 0 AS has_aurora_marker, current_setting('server_version_num')::int AS server_version_num, -- #2228: which database this connection actually landed in. Appended; see the comment above. current_database() AS connected_database"; + /// + /// The Aurora probe (#2340), a SEPARATE statement because it decides by CALLING the marker function + /// rather than looking it up in a catalog — and that distinction is the whole bug it fixes. + /// + /// This used to be a column on the detection query above: + /// (SELECT count(*) FROM pg_proc WHERE proname = 'aurora_version') > 0. Measured against a live + /// Aurora PostgreSQL 17.7 cluster as a pg_monitor-only role: that lookup returns 0 while + /// SELECT aurora_version() returns 17.7.2. So a genuine Aurora target read as stock + /// PostgreSQL, and because both and + /// gate on + /// IsAurora, ONE wrong boolean silently dropped the two most valuable PostgreSQL reads — with a + /// healthy-looking log line and a pre-flight that just printed a smaller collector count. + /// + /// Existence-by-catalog-lookup and callability are different questions, and the collectors care + /// about the second one. Its own statement because that is what lets a stock-PostgreSQL + /// 42883 undefined_function be caught and read as "not Aurora" instead of failing the whole + /// probe — the wrapping the old column comment claimed but a catalog subquery never actually needed. + /// + public const string PostgresAuroraProbeQueryText = @"SELECT aurora_version()"; + + /// + /// Whether this target is Aurora, decided by CALLING aurora_version() (#2340). True when the call + /// succeeds; false when it raises — 42883 undefined_function is stock PostgreSQL's answer and is + /// the expected negative, so it is caught rather than propagated. + /// + /// Any other error is also caught and read as "not Aurora", deliberately: this probe decides which + /// OPTIONAL collectors apply, and a target that answers the version/recovery questions but trips over + /// this one must still be monitored for everything else rather than failing to connect. The direction + /// matters and is the pre-#2340 behaviour anyway — the difference is that a real Aurora cluster now + /// answers true. + /// + /// Logged at debug on failure rather than swallowed silently, so "why is this Aurora cluster + /// reading as stock PostgreSQL" is answerable from the service log instead of requiring a live psql + /// session against the target, which is what diagnosing #2340 actually took. + /// + private static async Task ProbeAuroraAsync( + NpgsqlConnection connection, CancellationToken cancellationToken, ILogger? logger = null) + { + try + { + using var command = new NpgsqlCommand(PostgresAuroraProbeQueryText, connection) { CommandTimeout = 15 }; + var version = await command.ExecuteScalarAsync(cancellationToken); + return version is not null and not DBNull; + } + catch (Exception ex) when (ex is not OperationCanceledException) + { + logger?.LogDebug(ex, "aurora_version() probe did not succeed; treating the target as stock PostgreSQL"); + return false; + } + } + /// Connects, probes, and returns the runtime state for one configured server. public static async Task ConnectAsync(MonitoredServer config, ILogger? logger, CancellationToken cancellationToken) { @@ -285,7 +335,7 @@ private static async Task ConnectPostgresAsync( using var reader = await command.ExecuteReaderAsync(cancellationToken); int majorVersion = 0, versionNum = 0; - bool isInRecovery = false, isAurora = false; + bool isInRecovery = false; string versionText = ""; string? connectedDatabase = null; if (await reader.ReadAsync(cancellationToken)) @@ -293,11 +343,15 @@ private static async Task ConnectPostgresAsync( versionText = reader.IsDBNull(0) ? "" : reader.GetString(0); majorVersion = reader.IsDBNull(1) ? 0 : reader.GetInt32(1); isInRecovery = !reader.IsDBNull(2) && reader.GetBoolean(2); - isAurora = !reader.IsDBNull(3) && reader.GetBoolean(3); - versionNum = reader.IsDBNull(4) ? 0 : reader.GetInt32(4); - connectedDatabase = reader.IsDBNull(5) ? null : reader.GetString(5); /* #2228 */ + versionNum = reader.IsDBNull(3) ? 0 : reader.GetInt32(3); + connectedDatabase = reader.IsDBNull(4) ? null : reader.GetString(4); /* #2228 */ } + /* The reader must be closed before another command runs on this connection. */ + await reader.CloseAsync(); + + var isAurora = await ProbeAuroraAsync(connection, cancellationToken, logger); + logger?.LogInformation( "Connected to PostgreSQL target '{Server}': major {Major} (server_version_num {Num}), {Role}, Aurora: {Aurora} — {VersionText}", config.DisplayName, majorVersion, versionNum, isInRecovery ? "reader (in recovery)" : "writer", isAurora, diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs index 2dc65b021..b33fea452 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingWorker.cs @@ -19,6 +19,7 @@ using System.Threading.Tasks; using Microsoft.Data.SqlClient; using Microsoft.Extensions.Hosting; +using Microsoft.Extensions.Hosting.WindowsServices; using Microsoft.Extensions.Logging; using Npgsql; using PerformanceMonitor.Darling.Service.Targets; @@ -543,6 +544,26 @@ sweep body and the connect path (both run on the pool thread, one at a time per protected override async Task ExecuteAsync(CancellationToken stoppingToken) { + /* #2185: an install directory the service account cannot read is diagnosed HERE — first, ahead of + reading darling.json, and a long way ahead of the managed-Postgres bootstrap. Order is the whole + point. Every message the reporter saw was downstream of this one: an unreadable tree takes out + darling.json ("Cannot load configuration") and the bundled PostgreSQL's initdb (an empty Output: + and a bare loader status) before anything says WHERE the install is. Stated first, it is the line + above the failure in the log an operator is reading bottom-up. + + Diagnose and continue, deliberately, rather than refuse to start — see DarlingInstallLocation for + why (the installer asks rather than refuses on an upgrade for the same reason, and a service has + nobody to ask). Silent unless this process really is running as a Windows service: a console + test-drive from a Desktop folder runs as the profile owner, reads the tree fine, and is something + the README suggests doing. */ + if (OperatingSystem.IsWindows()) + { + DarlingInstallLocation.Report( + AppContext.BaseDirectory, + WindowsServiceHelpers.IsWindowsService(), + _logger); + } + DarlingConfig config; string configPath; try @@ -573,6 +594,21 @@ protected override async Task ExecuteAsync(CancellationToken stoppingToken) return; } + /* #2339: publish the declared peer stores as soon as a VALIDATED config is in hand, so the web + dashboard's read dispatch (which reuses the MCP tool methods) discloses the same peers even when + the MCP endpoint is disabled. The MCP host publishes the identical snapshot from its own load; + whichever runs first wins and they cannot disagree, both reading darling.json. + + Publish re-validates and refuses rather than trusting the Validate() above: it cannot fire here + (we already returned on any problem), but the check belongs to the publish, not to this call site, + because the MCP host reaches Publish WITHOUT ever calling Validate. Logged if it ever does, rather + than discarded, so an impossible state cannot become a silent one. */ + var peerPublish = DarlingPeerDirectory.Publish(config.Peers); + foreach (var problem in peerPublish.RefusedProblems) + { + _logger.LogCritical("Peer disclosure refused (nothing published): {Problem}", problem); + } + /* Network-endpoint caller warnings (darling-network-endpoints, D-BYO / D7) — emitted AFTER Validate() passes and NEVER inside it (Validate is all-fatal; an optional-endpoint note must not abort startup). Covers BYO-mode network.* being ignored and the network.role=admin pivot risk. */ diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingAgReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingAgReader.cs index db5d483a0..b34ac08c6 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingAgReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingAgReader.cs @@ -57,12 +57,12 @@ namespace PerformanceMonitor.Darling.Service.Mcp; internal static class DarlingAgReader { /// Shared serializer options — snake_case field names come from the DTOs' [JsonPropertyName] - /// attributes, severities serialize as their string names, and the output is indented (the MCP tool - /// convention). ONE options object so /api/ag and get_ag_health serialize the identical - /// shape. + /// attributes, severities serialize as their string names, and the output is COMPACT (#2350 - the MCP tool + /// convention, since the reader on both ends is a parser rather than a person). ONE options object so + /// /api/ag and get_ag_health serialize the identical shape. public static readonly JsonSerializerOptions JsonOptions = new() { - WriteIndented = true, + WriteIndented = false, Converters = { new JsonStringEnumConverter() }, }; diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingDataReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingDataReader.cs index fd265bee7..432d4f53c 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingDataReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingDataReader.cs @@ -879,6 +879,41 @@ public static async Task> GetTopProceduresByCpuAsync( /* ─────────────────────────── query store ─────────────────────────── */ + /// + /// The oldest collection_time this server actually has inside the requested window (#2364). + /// + /// Why a separate probe rather than reading the returned rows. + /// returns the top N by COST, not by time, so the timestamps on those rows say nothing about how far back + /// the window reaches — the most expensive query in a month might have run this morning. The window floor is + /// a property of the tier, not of the result set, and has to be asked for separately. + /// + /// Bounded on both sides, so it prunes chunks and answers from an ordered scan that stops at the first + /// row rather than reading the window. $1 server_id, $2/$3 window (naive UTC). + /// + public const string QueryStoreWindowFloorSql = """ + SELECT MIN(collection_time) + FROM query_store_stats + WHERE server_id = $1 + AND collection_time >= $2 + AND collection_time <= $3 + """; + + /// + /// Reads . Null when the window holds nothing at all, which the caller + /// reports as "nothing was read" rather than as an absence of activity. + /// + public static async Task GetQueryStoreWindowFloorAsync( + NpgsqlDataSource postgres, int serverId, DateTime startUtc, DateTime endUtc, + CancellationToken cancellationToken = default) + { + await using var command = postgres.CreateCommand(QueryStoreWindowFloorSql); + DarlingMcpReadParameters.AddInt(command, serverId); + DarlingMcpReadParameters.AddTimestamp(command, startUtc); + DarlingMcpReadParameters.AddTimestamp(command, endUtc); + var value = await command.ExecuteScalarAsync(cancellationToken); + return value is DateTime dt ? dt : null; + } + /// /// Top Query Store groups over the window — a focused projection of the viewer's /// QueryStoreTopSql (the columns Lite's get_query_store_top returns): group by diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingFleetReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingFleetReader.cs index 804421a86..85ddfe5b3 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingFleetReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingFleetReader.cs @@ -43,11 +43,11 @@ internal static class DarlingFleetReader { /// Shared serializer options for the fleet DTOs — snake_case field names come from the DTOs' /// [JsonPropertyName] attributes, enum bands serialize as their string names, and the output is - /// indented (matching the MCP tool convention). ONE options object so the web endpoint and the MCP tool - /// serialize the identical shape. + /// COMPACT (#2350, matching the MCP tool convention). ONE options object so the web endpoint and the MCP + /// tool serialize the identical shape. public static readonly JsonSerializerOptions JsonOptions = new() { - WriteIndented = true, + WriteIndented = false, Converters = { new JsonStringEnumConverter() }, }; diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpDataTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpDataTools.cs index 8cb1fdcff..202840f3d 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpDataTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpDataTools.cs @@ -663,9 +663,27 @@ public static async Task GetQueryStoreTop( try { var now = DateTime.UtcNow; - var rows = await DarlingDataReader.GetQueryStoreTopAsync(postgres, resolved.ServerId, now.AddHours(-hours_back), now, top, database_name); + var requestedStart = now.AddHours(-hours_back); + var rows = await DarlingDataReader.GetQueryStoreTopAsync(postgres, resolved.ServerId, requestedStart, now, top, database_name); + + /* #2364: what the window ACTUALLY holds. The rows above are the top N by COST, so their timestamps + say nothing about how far back the read reached -- the most expensive query in a month may have + run this morning. raw query_store_stats is dropped at 4 days on a store with the rollups armed, + and this tool has no rollup to fall back to (the corrected CAGGs carry no query_id or plan_id, + and plan identity is the whole point of this tool). So the honest move is to report the window + that was served rather than echo the one that was asked for. */ + var floor = await DarlingDataReader.GetQueryStoreWindowFloorAsync(postgres, resolved.ServerId, requestedStart, now); + var effectiveStart = floor ?? requestedStart; + var truncated = floor is DateTime f && f > requestedStart.AddMinutes(90); + if (rows.Count == 0) - return McpHelpers.Status("unavailable", "No Query Store data available. Query Store may not be enabled on target databases."); + return McpHelpers.Status( + "unavailable", + $"No Query Store rows for this server in the {hours_back}-hour window searched. Query Store " + + "may not be enabled on the target databases -- or the window reaches past what the raw tier " + + "retains (query_store_stats is dropped at 4 days when the rollups are armed), in which case " + + "nothing was read for the older part of it. Try a shorter window before concluding the " + + "queries did not run."); var result = rows.Select(r => new { @@ -693,6 +711,16 @@ to tell them apart. NULL when the server did not attribute the row. */ { server = resolved.ServerName, hours_back, + /* #2364: what was served, beside what was asked for. hours_back alone was a request echoed + back as though it described the data. */ + effective_start = effectiveStart.ToString("o"), + effective_hours_back = Math.Round((now - effectiveStart).TotalHours, 1), + truncated, + truncation_note = truncated + ? "The window reaches further back than this server's raw query_store_stats retains, so the " + + "older part of it was not read. This tool reads the raw tier only: the corrected rollups " + + "carry no query_id or plan_id, and plan identity is what it exists to return." + : null, queries = result }, McpHelpers.JsonOptions); } @@ -704,7 +732,7 @@ to tell them apart. NULL when the server did not attribute the row. */ /* ═══════════════════════════ discovery / health ═══════════════════════════ */ - [McpServerTool(Name = "list_servers"), Description("Lists all monitored SQL Server instances with their collection freshness status and last collection time. Use this first to see available servers before calling other tools. The service has no live connection to the monitored servers, so status is derived from how recently each server was collected (Online = fresh, Warning = stale, Offline = no recent collection).")] + [McpServerTool(Name = "list_servers"), Description("Lists all monitored SQL Server instances with their collection freshness status and last collection time. Use this first to see available servers before calling other tools. The service has no live connection to the monitored servers, so status is derived from how recently each server was collected (Online = fresh, Warning = stale, Offline = no recent collection). The peer_fleets block names the SIBLING Darling stores that monitor the rest of a split fleet, with what each one covers — this server can only NAME them (no cross-store reads), and peer_note says what an empty peer_fleets does and does not prove.")] public static async Task ListServers( NpgsqlDataSource postgres) { @@ -720,25 +748,16 @@ public static async Task ListServers( return $"Could not read the servers registry from the Postgres store: {ex.Message}"; } + /* #2339: the empty-registry answer is prose, not the JSON envelope, so it carries the peer + disclosure explicitly — otherwise it is the one path where the declaration silently vanishes, + and it is the worst one to lose it on: a store with nothing registered is a fresh or + just-restarted box, where "no servers here" with no mention of the siblings is the strongest + version of the wrong conclusion. */ if (servers.Count == 0) - return "No servers are registered yet. The service registers each monitored server on its first successful connection."; - - var now = DateTime.UtcNow; - var result = servers.Select(s => new - { - server_name = s.ServerName, - display_name = string.IsNullOrEmpty(s.DisplayName) ? s.ServerName : s.DisplayName, - sql_version = SqlVersionLabel(s.SqlMajorVersion), - status = FreshnessStatus(s.LastCollection, now), - read_only = s.ServerName.EndsWith(":RO", StringComparison.Ordinal), - last_collection = s.LastCollection?.ToString("o") - }); + return "No servers are registered yet. The service registers each monitored server on its first successful connection." + + DarlingPeerDirectory.EmptyRegistryDisclosure(DarlingPeerDirectory.Current); - return JsonSerializer.Serialize(new - { - server_count = servers.Count, - servers = result - }, McpHelpers.JsonOptions); + return RenderServerList(servers, DateTime.UtcNow, DarlingPeerDirectory.Current); } catch (Exception ex) { @@ -746,6 +765,49 @@ public static async Task ListServers( } } + /// + /// The list_servers envelope, pure over (registry rows, now, declared peers) — separated from the + /// store read so the response SHAPE, including the #2339 peer disclosure, pins without a live store. + /// + /// Why the peer block lives on THIS tool. list_servers is the discovery read: it is + /// where an agent forms its model of "who is monitored", so it is where the fact that SIBLING stores hold + /// the rest of a split fleet has to appear. peer_fleets is therefore always present and + /// peer_note is always populated — an EMPTY peer list has two very different meanings (this really + /// is the only store, or the operator never declared its siblings) and this server cannot tell them + /// apart, so it says so rather than letting an empty array read as "this is the whole fleet". + /// + internal static string RenderServerList( + IReadOnlyList servers, + DateTime nowUtc, + DarlingPeerDirectory.Snapshot peers) + { + var result = servers.Select(s => new + { + server_name = s.ServerName, + display_name = string.IsNullOrEmpty(s.DisplayName) ? s.ServerName : s.DisplayName, + sql_version = SqlVersionLabel(s.SqlMajorVersion), + status = FreshnessStatus(s.LastCollection, nowUtc), + read_only = s.ServerName.EndsWith(":RO", StringComparison.Ordinal), + last_collection = s.LastCollection?.ToString("o") + }); + + return JsonSerializer.Serialize(new + { + server_count = servers.Count, + this_store_covers = peers.ThisStoreCovers.Length == 0 ? null : peers.ThisStoreCovers, + peer_fleets = peers.Peers.Select(p => new + { + name = p.Name, + covers = p.Covers, + matches = p.Matches + }), + peer_note = peers.Peers.Count == 0 + ? DarlingPeerDirectory.NoPeersDeclaredNote + : DarlingPeerDirectory.PeersDeclaredNote, + servers = result + }, McpHelpers.JsonOptions); + } + [McpServerTool(Name = "get_collection_health"), Description("Shows the health status of all data collectors for a server — whether they're running successfully, failing, or stale. Check this before investigating data to ensure collectors are working properly. Each row also carries last_note/note_count: what a NON-failing run reported, e.g. an enumeration that came back with 0 items. note_count equal to total_runs means the collector has been collecting nothing all window — not a fault (the target may be legitimately empty), but the reason a HEALTHY collector can still have no data. target_has_user_databases tells those two apart: true means the target DID have user databases in the same window, so an all-window empty enumeration is worth investigating (a login that cannot enter them, an exclusion filter that matched everything); false means either no user databases or no inventory to go on. The sweep_pressure block is the server-level roll-up: it compares the collectors' combined execution demand (average duration amortized by cadence) against the minute the fastest cadence holds. SATURATED means the collection body cannot fit inside its cadence, so relaunches are skipped and the server collects at a multiple of its configured interval while every collector still reads healthy — heaviest_collectors names where that budget goes.")] public static async Task GetCollectionHealth( NpgsqlDataSource postgres, diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpHostService.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpHostService.cs index 5a49a5e9e..00c24d73b 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpHostService.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpHostService.cs @@ -445,6 +445,39 @@ listen value is itself loopback or a wildcard (0.0.0.0/::), which would collide builder.Services.AddSingleton(postgres); builder.Services.AddSingleton(new DarlingAnalysisService(postgres, planFetcher, _logger)); + /* #2339: publish the declared peer stores before the instructions are rendered, so the same + snapshot feeds the instructions section, list_servers' peer_fleets block, and the + server-resolution miss message. Publishing here as well as in the worker is deliberate: either + may reach its config first, the value is identical (both read darling.json), and the disclosure + should not depend on which one won. An empty declaration is Snapshot.Empty, which leaves every + one of those three surfaces exactly as it was. + + THIS host never calls DarlingConfig.Validate (see the class doc: its fail-closed checks are + host-local, because the worker's abort is a return from the worker and would not stop this + server). Publish therefore validates the peers block itself and refuses the whole thing on any + problem — the same host-local fail-closed posture as ResolveMcpBind, and the reason a + credential pasted into a peer description cannot reach a client from here. Reported at CRITICAL + because it is a configuration defect that silently costs the operator their disclosure. */ + var peerPublish = DarlingPeerDirectory.Publish(config.Peers); + var declaredPeers = peerPublish.Snapshot; + + if (peerPublish.Refused) + { + foreach (var problem in peerPublish.RefusedProblems) + { + _logger.LogCritical( + "MCP peer disclosure REFUSED (nothing published; peers are not disclosed until this is fixed): {Problem}", + problem); + } + } + else if (!declaredPeers.IsEmpty) + { + _logger.LogInformation( + "MCP peer disclosure active: {PeerCount} declared peer store(s){Coverage}. Disclosure only — this service never contacts a peer.", + declaredPeers.Peers.Count, + declaredPeers.ThisStoreCovers.Length > 0 ? $"; this store covers {declaredPeers.ThisStoreCovers}" : ""); + } + /* Register MCP server with the analysis tool class. */ builder.Services .AddMcpServer(options => @@ -454,7 +487,7 @@ listen value is itself loopback or a wildcard (0.0.0.0/::), which would collide Name = "PerformanceMonitorDarling", Version = "1.0.0" }; - options.ServerInstructions = DarlingMcpInstructions.Text; + options.ServerInstructions = DarlingMcpInstructions.Build(declaredPeers); }) /* Stateless mode: each request is self-contained (no Mcp-Session-Id round-trip). Required for clients like Google Antigravity that don't echo the session id, diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs index 085d4f4a8..e11a2082b 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpInstructions.cs @@ -16,7 +16,26 @@ namespace PerformanceMonitor.Darling.Service.Mcp; /// internal static class DarlingMcpInstructions { - public const string Text = """ + /// + /// The instructions with NO peer disclosure — a single-store deployment's text, byte-for-byte what this + /// server sent before #2339. Composed from + ; the split exists + /// only so can put the fleet-coverage section between them, high enough that an + /// agent reads which store it is talking to before it reads the tool census. + /// + public static readonly string Text = Preamble + "\n\n" + Body; + + /// + /// Renders the instructions for THIS store, inserting the declared fleet-coverage section (#2339) when + /// the operator declared any. Returns unchanged for an empty declaration, so nothing + /// about a single-store deployment moves. + /// + public static string Build(DarlingPeerDirectory.Snapshot peers) + { + var section = DarlingPeerDirectory.InstructionsSection(peers); + return section.Length == 0 ? Text : Preamble + "\n\n" + section + "\n\n" + Body; + } + + private const string Preamble = """ You are connected to a SQL Server performance monitoring tool via PerformanceMonitor Darling, the headless collector service. ## CRITICAL: Read-Only Access @@ -29,7 +48,9 @@ internal static class DarlingMcpInstructions - Run any ad-hoc diagnostics beyond what the collectors have already captured The writes this server performs are all to the MONITORING store, never a monitored SQL Server: mute_analysis_finding records a mute rule, analyze_server persists its findings, the custom-view management tools (create_custom_view / update_custom_view / delete_custom_view) save user-authored dashboard/notebook definitions to config.custom_views, and the alert-tuning tools (update_alert_settings / create_mute_rule / delete_mute_rule) change the shared alert configuration the service delivers on (all the same store the web viewer / Settings window writes). None of these touches a monitored SQL Server or the collected performance data itself. + """; + private const string Body = """ ## How Data Is Collected The Darling service collects from monitored SQL Server instances 24/7 and stores the data in a Postgres/TimescaleDB store. Data is collected in snapshots at regular intervals (typically every 1-15 minutes depending on the collector). This means: @@ -84,7 +105,7 @@ internal static class DarlingMcpInstructions | `get_top_queries_by_cpu` | Expensive queries from query stats (plan cache) with query_hash / sql_handle; `cpu_attribution.attributed_cpu_ratio` says how much of the box's measured CPU the returned rows explain | `server_name`, `hours_back` (default 24), `top` (default 20), `database_name`, `parallel_only`, `min_dop` | | `get_top_procedures_by_cpu` | Most expensive stored procedures by total CPU, with the same `cpu_attribution` disclosure | `server_name`, `hours_back` (default 24), `top` (default 20), `database_name` | | `get_query_store_top` | Expensive queries from Query Store with query_id / plan_id (survives restarts) | `server_name`, `hours_back` (default 24), `top` (default 20), `database_name` | - | `list_servers` | All monitored servers with collection-freshness status and last collection time | none | + | `list_servers` | All monitored servers with collection-freshness status and last collection time, plus `peer_fleets` — the declared SIBLING Darling stores and what each covers (disclosure only; this server cannot read them) and `peer_note`, which says what an EMPTY `peer_fleets` does and does not prove | none | | `get_collection_health` | Per-collector health (running / failing / stale) over the last 7 days, plus the server's sweep_pressure verdict (a SATURATED body collects at a multiple of its configured cadence with every collector healthy) | `server_name` | | `get_server_properties` | Instance properties: edition, version, CPU count, memory, socket/core topology, HADR | `server_name` | diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTrendTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTrendTools.cs index 07e04a765..57a7a9fa0 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTrendTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpTrendTools.cs @@ -222,9 +222,26 @@ public static async Task GetQueryTrend( try { var now = DateTime.UtcNow; - var rows = await DarlingTrendReader.GetQueryHistoryAsync(postgres, resolved.ServerId, database_name, query_hash, now.AddHours(-hours_back), now); + var history = await DarlingTrendReader.GetQueryHistoryAsync(postgres, resolved.ServerId, database_name, query_hash, now.AddHours(-hours_back), now); + var rows = history.Points; if (rows.Count == 0) - return McpHelpers.Status("empty", $"No history found for query_hash '{query_hash}' in database '{database_name}' within the last {hours_back} hours."); + { + /* #2353: say what was READ, never what happened. This message used to assert "no history in the + last N hours" over a span the read never covered — for a query whose history had aged out of + the raw tier that is a false statement, not an incomplete one, and an agent acts on it by + concluding the query did not run. */ + return McpHelpers.Status( + "empty", + $"No history found for query_hash '{query_hash}' in database '{database_name}' in the " + + $"{history.Source} tier over the last {hours_back} hours. This means nothing was recorded " + + "for that query_hash in that window in the tier searched — confirm the hash and database " + + "with get_top_queries_by_cpu before concluding the query did not run."); + } + + /* The hourly rollup keeps executions, CPU and elapsed and nothing else. Those columns arrive as + NULL and the mapper floors them to 0, so they are emitted as null HERE rather than as zero: on an + aggregate row a zero would read as "none observed", which is a measurement we did not make. */ + var aggregated = history.Source != "raw"; var result = rows.Select(r => new { @@ -234,14 +251,14 @@ public static async Task GetQueryTrend( elapsed_ms = Math.Round(r.DeltaElapsedUs / 1000.0, 2), avg_cpu_ms = Math.Round(r.DeltaExecutions > 0 ? r.DeltaCpuUs / 1000.0 / r.DeltaExecutions : 0, 2), avg_elapsed_ms = Math.Round(r.DeltaExecutions > 0 ? r.DeltaElapsedUs / 1000.0 / r.DeltaExecutions : 0, 2), - logical_reads = r.DeltaLogicalReads, - logical_writes = r.DeltaLogicalWrites, - physical_reads = r.DeltaPhysicalReads, - rows = r.DeltaRows, - spills = r.DeltaSpills, - min_dop = r.MinDop, - max_dop = r.MaxDop, - query_plan_hash = r.QueryPlanHash + logical_reads = aggregated ? (long?)null : r.DeltaLogicalReads, + logical_writes = aggregated ? (long?)null : r.DeltaLogicalWrites, + physical_reads = aggregated ? (long?)null : r.DeltaPhysicalReads, + rows = aggregated ? (long?)null : r.DeltaRows, + spills = aggregated ? (long?)null : r.DeltaSpills, + min_dop = aggregated ? (int?)null : r.MinDop, + max_dop = aggregated ? (int?)null : r.MaxDop, + query_plan_hash = aggregated ? null : r.QueryPlanHash }); return JsonSerializer.Serialize(new @@ -250,6 +267,19 @@ public static async Task GetQueryTrend( database_name, query_hash, hours_back, + /* #2353: what was actually served, alongside what was asked for. hours_back on its own was a + request echoed back as if it were a description of the data. */ + source = history.Source, + effective_start = history.EffectiveStartUtc.ToString("o"), + effective_hours_back = Math.Round((now - history.EffectiveStartUtc).TotalHours, 1), + truncated = history.Truncated, + bucket = history.Source == "raw" ? "per-collection" : "1 hour", + aggregate_note = aggregated + ? "Served from the hourly rollup because the requested window reaches past the raw tier's " + + "4-day retention. Executions, CPU and elapsed time are summed per hour; logical_reads, " + + "logical_writes, physical_reads, rows, spills, min_dop, max_dop and query_plan_hash are " + + "null because the rollup does not carry them - null means not measured, not zero." + : null, data_points = rows.Count, trend = result }, McpHelpers.JsonOptions); diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingPeerDirectory.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingPeerDirectory.cs new file mode 100644 index 000000000..8c7fa48fa --- /dev/null +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingPeerDirectory.cs @@ -0,0 +1,330 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Linq; +using System.Text; + +namespace PerformanceMonitor.Darling.Service.Mcp; + +/// +/// The declared-peer directory (#2339, tier 1): what THIS Darling store covers, and which SIBLING stores +/// cover the rest of the fleet. Pure DISCLOSURE — a name and a human coverage sentence per peer, nothing +/// more. There is deliberately no address, no credential and no connectivity of any kind here; a peer is +/// something this server can NAME, never something it can read. +/// +/// The problem it solves. A fleet split across several boxes (one store per box — SQL Server +/// primaries on one, their readable replicas on another, PostgreSQL on a third) gives every box an MCP +/// server that answers over its own store only. A server monitored by a sibling resolves as not-found, +/// which is indistinguishable from a server NOBODY monitors — so an agent asking the wrong endpoint gets +/// "unknown server" when the true answer is "the other box has that one." Declaring the peers makes the +/// split legible at the three places an agent forms its model of the fleet: the MCP instructions, the +/// list_servers discovery read, and the server-resolution miss. +/// +/// Ambient, not injected, and why. The snapshot is published once from config at startup and +/// read from a process-wide static. Roughly ninety MCP tool methods resolve a server name through +/// and take only an NpgsqlDataSource; threading a peer +/// parameter through all of them (and through the web dashboard's read dispatch, which reuses the same +/// tool methods) would touch every one of them to deliver a constant. The stored value is an IMMUTABLE +/// snapshot behind a volatile field, so a publish is a single reference swap and every reader sees a +/// coherent list. is the default, which is byte-for-byte today's behavior — a +/// single-store deployment that declares nothing is unaffected everywhere. +/// +/// Matching is opt-in and deliberately dumb. A peer's covers text is prose for a human +/// (and for an LLM reading the instructions); it is NOT parsed. Naming the peer that owns a missed server +/// needs a machine-checkable rule, so a peer may also declare matches — plain case-insensitive +/// substrings of the server names it monitors ("use1", "-replica"). No globbing, no regex: a +/// pattern language here would be a config surface with its own bugs, and substrings answer the real +/// question (which region/role prefix is this?). A peer with no matches is still disclosed +/// everywhere — it just cannot be singled out on a miss, which the miss message says rather than +/// implying the server is unmonitored. +/// +internal static class DarlingPeerDirectory +{ + /// One declared sibling store: its name, what it covers in prose, and the optional + /// server-name substrings that let a miss point at it. + internal sealed record Peer(string Name, string Covers, IReadOnlyList Matches) + { + /// + /// True when falls inside this peer's DECLARED coverage — a + /// case-insensitive substring hit on any matches entry. False for a peer that declared no + /// patterns, which is "cannot tell", never "not this peer" (the callers distinguish the two). + /// + public bool CoversServerName(string? serverName) => + !string.IsNullOrWhiteSpace(serverName) + && Matches.Any(m => serverName!.Contains(m, StringComparison.OrdinalIgnoreCase)); + } + + /// + /// The immutable published state: this store's own coverage sentence plus the declared peers. + /// Handed around as one value so a caller cannot read a coverage line that disagrees with the peer + /// list it was published beside. + /// + internal sealed record Snapshot(string ThisStoreCovers, IReadOnlyList Peers) + { + /// Nothing declared — the shipped default, and byte-for-byte today's behavior. + public static readonly Snapshot Empty = new("", Array.Empty()); + + /// True when the operator declared neither a coverage sentence nor any peer. + public bool IsEmpty => string.IsNullOrWhiteSpace(ThisStoreCovers) && Peers.Count == 0; + + /// The peers whose declared matches cover a name, in declaration order. + public IReadOnlyList PeersCovering(string? serverName) => + Peers.Where(p => p.CoversServerName(serverName)).ToList(); + } + + private static volatile Snapshot s_current = Snapshot.Empty; + + /// The live snapshot. until something publishes. + internal static Snapshot Current => s_current; + + /// + /// What a installed, and why it installed nothing if it refused. One value rather + /// than a snapshot plus an out-parameter, so a caller cannot take the snapshot and drop the reason. + /// + internal readonly record struct PublishResult(Snapshot Snapshot, IReadOnlyList RefusedProblems) + { + /// True when the config failed validation and NOTHING was published. + public bool Refused => RefusedProblems.Count > 0; + } + + /// + /// Publishes the declared coverage from config, returning what was installed so the caller can render + /// the MCP instructions from the same value. Idempotent by design: the worker and the MCP host both load + /// darling.json and both publish, because either may reach its config first and neither should have to + /// wait on the other to make the disclosure available. + /// + /// Fail-closed HERE, not at the callers (review finding on #2339). The credential guard + /// originally lived only in , which the worker runs — but the + /// worker's abort is a return from its own hosted service, not a process exit, and the MCP host + /// loads its own copy of the config and deliberately never calls Validate (its network-exposure + /// checks are host-local for exactly that reason). So the one path that actually broadcasts peer text to + /// clients was the one path the guard never covered. Validating inside Publish makes it + /// structural: the ambient snapshot can only be written through here, so a future third publish site + /// cannot reintroduce the hole. + /// + /// A refusal publishes — the WHOLE block, not the valid subset. A + /// peers block that failed validation is one the operator has not finished, and half a disclosure is + /// worse than none: it would state coverage that may be wrong while the log says the config is broken. + /// The cost is that the fleet split goes undisclosed until it is fixed, which the peer_note + /// already reports honestly ("this server cannot tell those apart"), and the caller logs the problems at + /// CRITICAL. Leaking a credential is not recoverable; losing disclosure for one restart is. + /// + internal static PublishResult Publish(PeersConfig? config) + { + var problems = PeersConfig.Validate(config); + if (problems.Count > 0) + { + s_current = Snapshot.Empty; + return new PublishResult(Snapshot.Empty, problems); + } + + s_current = FromConfig(config); + return new PublishResult(s_current, Array.Empty()); + } + + /// Resets the ambient snapshot — for tests, which must not leak declared peers into each other. + internal static void Reset() => s_current = Snapshot.Empty; + + /// + /// Normalizes a config block into a snapshot: trims everything, drops entries with no name and no + /// coverage text (an empty JSON object in the array is a typo, not a peer), and drops blank match + /// patterns — an empty pattern is a substring of EVERY name, so keeping one would make a peer claim + /// the whole fleet. + /// + internal static Snapshot FromConfig(PeersConfig? config) + { + if (config is null) + { + return Snapshot.Empty; + } + + var peers = (config.Stores ?? new List()) + .Where(p => p is not null) + .Select(p => new Peer( + (p.Name ?? "").Trim(), + (p.Covers ?? "").Trim(), + (p.Matches ?? new List()) + .Where(m => !string.IsNullOrWhiteSpace(m)) + .Select(m => m.Trim()) + .ToList())) + .Where(p => p.Name.Length > 0 || p.Covers.Length > 0) + .ToList(); + + return new Snapshot((config.ThisStoreCovers ?? "").Trim(), peers); + } + + /// A peer's one-line disclosure — "name — what it covers", or just whichever half exists. + private static string Describe(Peer peer) => + peer.Covers.Length == 0 ? peer.Name + : peer.Name.Length == 0 ? peer.Covers + : $"{peer.Name} — {peer.Covers}"; + + /// + /// The MCP-instructions section (#2339): what this store covers, what the siblings cover, and the + /// flat statement that there is no path from here to there. Returns "" for an empty snapshot, so the + /// single-store instructions text is unchanged. + /// + internal static string InstructionsSection(Snapshot snapshot) + { + if (snapshot.IsEmpty) + { + return ""; + } + + var text = new StringBuilder(); + text.Append("## Fleet Coverage: This Store and Its Peers\n\n"); + text.Append( + "This is ONE Darling store among several, each monitoring a different slice of the fleet from its own " + + "MCP endpoint and its own server registry. Every tool here answers over THIS store only."); + + if (snapshot.ThisStoreCovers.Length > 0) + { + text.Append(" This store covers: ").Append(snapshot.ThisStoreCovers).Append('.'); + } + + if (snapshot.Peers.Count > 0) + { + text.Append(" The declared peer stores and what they cover:\n\n"); + foreach (var peer in snapshot.Peers) + { + text.Append("- ").Append(Describe(peer)).Append('\n'); + } + + text.Append( + "\nThere is NO cross-store connectivity: this server cannot read a peer's data, forward a query to " + + "it, or confirm that a peer is up. The peers are DECLARED here, not contacted. So a server that " + + "belongs to a peer's coverage must be asked of THAT endpoint — a not-found here is not evidence " + + "the server is unmonitored, and `list_servers` repeats this list as `peer_fleets` for a client " + + "that reads tool output rather than these instructions."); + } + + return text.ToString(); + } + + /// + /// The peer half of a server-resolution miss (#2339) — appended AFTER the existing + /// "Could not resolve server. Available servers:" listing, never replacing it, because that prefix is + /// what a caller (and several tests) keys off and the local listing is still the useful answer to a typo. + /// + /// Returns "" when nothing is declared, so a single-store deployment's miss message is + /// byte-for-byte unchanged. The honest-empty disclosure for "no peers declared, so absence here is not + /// proof nobody monitors it" belongs on list_servers instead: that is where an agent builds its + /// model of the fleet ONCE, whereas a resolution miss is usually a typo and would carry the same + /// paragraph on every one of them. + /// + internal static string ResolutionMissDisclosure(Snapshot snapshot, string? requestedName) + { + if (snapshot.IsEmpty) + { + return ""; + } + + var text = new StringBuilder(); + var named = !string.IsNullOrWhiteSpace(requestedName); + var subject = named ? $"'{requestedName!.Trim()}'" : "That server"; + + var matching = named ? snapshot.PeersCovering(requestedName) : Array.Empty(); + if (matching.Count > 0) + { + /* Two peers can legitimately both claim a name (overlapping `matches`, e.g. "use1" and + "prod-pos"), so the follow-on sentence agrees in number rather than saying "That is a SEPARATE + store" about a list of two. */ + var single = matching.Count == 1; + text.Append(subject) + .Append(" is not monitored HERE, and it matches the declared coverage of ") + .Append(single ? "peer store " : "these peer stores: ") + .Append(string.Join("; ", matching.Select(Describe))) + .Append(single + ? ". That is a SEPARATE Darling store with its own MCP endpoint; this server cannot read it, " + + "so point the client at that endpoint (or tell your operator which store answers for this " + + "server) rather than concluding the server is unmonitored." + : ". Those are SEPARATE Darling stores, each with its own MCP endpoint; this server cannot read " + + "them, so point the client at whichever one owns this server (or ask your operator) rather " + + "than concluding the server is unmonitored."); + } + else + { + text.Append(subject) + .Append(" is not monitored HERE."); + + if (snapshot.Peers.Count > 0) + { + text.Append(" It matches no declared peer store's coverage either, so it may genuinely be " + + "unmonitored — but the peer declarations are prose plus optional name patterns, not a " + + "live registry, so check the coverage list before concluding that. Declared peer stores " + + "(separate Darling stores with their own MCP endpoints; this server cannot read them): ") + .Append(string.Join("; ", snapshot.Peers.Select(Describe))) + .Append('.'); + } + } + + if (snapshot.ThisStoreCovers.Length > 0) + { + text.Append(" This store covers: ").Append(snapshot.ThisStoreCovers).Append('.'); + } + + return text.ToString(); + } + + /// + /// The disclosure appended when the registry itself is EMPTY (#2339) — list_servers answers that + /// case with prose rather than the JSON envelope, so it would otherwise be the one place the peer block + /// silently disappears. + /// + /// That is the worst place to lose it: a store with nothing registered is a fresh or just-restarted + /// box, and "no servers here" plus no mention of the siblings is the strongest possible version of the + /// wrong conclusion this whole feature exists to prevent. Returns "" when nothing is declared, so the + /// single-store message is unchanged. + /// + internal static string EmptyRegistryDisclosure(Snapshot snapshot) + { + if (snapshot.IsEmpty) + { + return ""; + } + + var text = new StringBuilder(); + + if (snapshot.Peers.Count > 0) + { + text.Append(" This store is one of SEVERAL monitoring this fleet, so an empty registry here says " + + "nothing about what the others hold. Declared peer stores (separate Darling stores with " + + "their own MCP endpoints; this server cannot read them): ") + .Append(string.Join("; ", snapshot.Peers.Select(Describe))) + .Append('.'); + } + + if (snapshot.ThisStoreCovers.Length > 0) + { + text.Append(" This store covers: ").Append(snapshot.ThisStoreCovers).Append('.'); + } + + return text.ToString(); + } + + /// + /// The peer_fleets note list_servers carries when nothing is declared. An empty peer + /// list has two very different meanings — this really is the only store, or the operator never + /// declared the siblings — and this server cannot tell them apart, so it says so instead of letting + /// an empty array read as "you are looking at the whole fleet" (the house rule that an empty result + /// must never be mistakable for "nothing happened"). + /// + internal const string NoPeersDeclaredNote = + "No peer stores are declared (darling.json's \"peers\" block is absent or empty). That means EITHER this is " + + "the only Darling store monitoring this fleet, OR the operator has not declared its siblings — this server " + + "cannot tell those apart, so a server missing from the list above is not proof that nobody monitors it."; + + /// The peer_fleets note when peers ARE declared: what they are, and what they are not. + internal const string PeersDeclaredNote = + "Peer fleets are SEPARATE Darling stores, each with its own MCP endpoint and its own server registry. They " + + "are DECLARED here for disclosure only: this server cannot read a peer's data, forward a query to it, or " + + "confirm it is up. A server inside a peer's coverage must be asked of that peer's endpoint. 'matches' lists " + + "the server-name substrings that peer declares it monitors, and is empty when the peer declared none."; +} diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingServerResolver.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingServerResolver.cs index 601983091..46cebf6d8 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingServerResolver.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingServerResolver.cs @@ -26,6 +26,11 @@ namespace PerformanceMonitor.Darling.Service.Mcp; /// case-insensitive) beats partial (Contains) match; a miss returns a ready-to-return error /// listing the available servers, with Lite's [Read-Only] tag derived from the /// storage-name :RO suffix (the registry's encoding of ReadOnlyIntent). +/// +/// One headless-only addition (#2339): the miss message also discloses the DECLARED PEER STORES, so a +/// fleet split across several Darling boxes does not answer "unknown server" where the true answer is "the +/// other box has that one." Purely additive — see the +/// overload. /// internal static class DarlingServerResolver { @@ -72,16 +77,41 @@ WHERE is_enabled /// /// The pure matching half — Lite's semantics over materialized registry rows, separated - /// from the Postgres read so the resolution rules unit-test without a live store. + /// from the Postgres read so the resolution rules unit-test without a live store. Reads the ambient + /// peer declaration (#2339) for the miss message; the overload below takes it explicitly. /// internal static ((int ServerId, string ServerName) resolved, string? error) ResolveOrError( IReadOnlyList servers, - string? serverName) + string? serverName) => + ResolveOrError(servers, serverName, DarlingPeerDirectory.Current); + + /// + /// The resolution rules over an EXPLICIT peer declaration — the pure form, so the miss message's peer + /// disclosure is testable without publishing process-wide state. + /// + /// The miss message is additive on purpose (#2339). A fleet split across several Darling + /// stores makes "not monitored here" the normal case rather than an edge, and the bare + /// "Could not resolve server" it produced is indistinguishable from "nobody monitors this server" — so + /// the peer disclosure is appended, naming the sibling store whose declared coverage matches. It is + /// APPENDED rather than substituted because the local server listing is still the right answer to the + /// commonest miss (a typo), and because the leading "Could not resolve server." is what callers key off. + /// With nothing declared the message is byte-for-byte what it was. + /// + internal static ((int ServerId, string ServerName) resolved, string? error) ResolveOrError( + IReadOnlyList servers, + string? serverName, + DarlingPeerDirectory.Snapshot peers) { var resolved = Resolve(servers, serverName); - return resolved is null - ? (default, $"Could not resolve server. Available servers:\n{ListAvailableServers(servers)}") - : (resolved.Value, null); + if (resolved is not null) + { + return (resolved.Value, null); + } + + var message = $"Could not resolve server. Available servers:\n{ListAvailableServers(servers)}"; + var disclosure = DarlingPeerDirectory.ResolutionMissDisclosure(peers, serverName); + + return (default, disclosure.Length == 0 ? message : $"{message}\n\n{disclosure}"); } /// diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingTrendReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingTrendReader.cs index 2780bebe6..7c613c070 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingTrendReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingTrendReader.cs @@ -11,6 +11,7 @@ using System.Threading; using System.Threading.Tasks; using Npgsql; +using PerformanceMonitor.Darling.Storage; namespace PerformanceMonitor.Darling.Service.Mcp; @@ -75,6 +76,56 @@ public sealed record QueryHistoryPoint( long DeltaLogicalReads, long DeltaLogicalWrites, long DeltaPhysicalReads, long DeltaRows, long DeltaSpills, int MinDop, int MaxDop, string QueryPlanHash); + /// + /// Which physical tier answered a query-history read, and over what window (#2353). + /// + /// raw is the per-collection query_stats table and carries every column. + /// hourly is the query_stats_hourly continuous aggregate: hour buckets, and only the measures + /// the rollup keeps — executions, CPU and elapsed. The columns it does not keep are reported as NULL rather + /// than zero, because zero is a claim and NULL is the absence of one. + /// + /// is what was actually read, which can be LATER than the start the + /// caller asked for when even the aggregate does not reach that far back. It exists so the tool can say so + /// instead of returning a short array under the requested window's label. + /// + public sealed record QueryHistoryResult( + List Points, string Source, DateTime EffectiveStartUtc, bool Truncated); + + /// + /// The aggregate twin of , reading the hourly continuous aggregate (#2353). + /// + /// Shaped to the SAME reader ordinals as the raw query so one mapper serves both. The eight columns + /// the rollup does not carry — reads, writes, physical reads, rows, spills, the DOP pair and the plan hash — + /// are selected as typed NULLs and surface as NULL in the payload. That is the honest answer: an hour bucket + /// has no single plan hash and no single DOP, and inventing a zero would read as "none observed". + /// + /// What survives the rollup is exactly what a trend is usually asked for: executions, CPU and elapsed + /// time. bucket is projected as collection_time so the series column name does not change + /// underneath a caller that got a raw answer last time. + /// + public const string QueryHistoryHourlySql = """ + SELECT + bucket AS collection_time, + execution_count_sum AS delta_execution_count, + worker_time_sum AS delta_worker_time, + elapsed_time_sum AS delta_elapsed_time, + CAST(NULL AS bigint) AS delta_logical_reads, + CAST(NULL AS bigint) AS delta_logical_writes, + CAST(NULL AS bigint) AS delta_physical_reads, + CAST(NULL AS bigint) AS delta_rows, + CAST(NULL AS bigint) AS delta_spills, + CAST(NULL AS int) AS min_dop, + CAST(NULL AS int) AS max_dop, + CAST(NULL AS text) AS query_plan_hash + FROM query_stats_hourly + WHERE server_id = $1 + AND database_name = $2 + AND query_hash = $3 + AND bucket >= $4 + AND bucket <= $5 + ORDER BY bucket + """; + /* ─────────────────────────── memory trend ─────────────────────────── */ /// @@ -345,11 +396,79 @@ FROM query_stats ORDER BY collection_time """; - public static async Task> GetQueryHistoryAsync( - NpgsqlDataSource postgres, int serverId, string databaseName, string queryHash, DateTime startUtc, DateTime endUtc, CancellationToken cancellationToken = default) + /// + /// Extra room inside the raw horizon before a window is handed to the aggregate (#2353). The purge is + /// periodic, so a window whose oldest point sits exactly on the four-day line may or may not still find its + /// rows depending on when the purge last ran; preferring the aggregate there trades resolution for an answer + /// that does not change with the purge schedule. Exposed so a test pins the boundary rather than restating it. + /// + public static readonly TimeSpan RawTierMargin = TimeSpan.FromHours(1); + + /// + /// Whether the raw per-collection table can serve a window, decided by the age of its OLDEST point measured + /// from WALL CLOCK (#2353). + /// + /// The second parameter is now, not the window's end, and the difference is not cosmetic. + /// Retention drops chunks by actual elapsed time, so what decides whether raw still holds a row is how long + /// ago that row happened — never where it sits inside the requested window. Measuring the start against the + /// END would call a two-hour window from ten days ago "recent", because it is recent relative to its own + /// end, and route it to a tier that dropped those rows six days earlier. ComposeSourceRouter takes a + /// caller-supplied now for exactly this reason. + /// + /// Pure so the boundary is unit-testable without a store — the tiering decision is the whole of this + /// fix, and a rule that can only be exercised against a live four-day-old hypertable is a rule nobody + /// re-checks. Width is deliberately not consulted: a narrow window sitting entirely in last week is exactly + /// as unservable from raw as a wide one. + /// + public static bool ShouldUseRawTier(DateTime startUtc, DateTime nowUtc) => + startUtc >= nowUtc - TimescaleSupport.RawRetentionSpan + RawTierMargin; + + /// + /// Reads a query's history from the tier that can actually serve the window (#2353). + /// + /// The bug this replaces. This read went to the raw query_stats table only, and the raw + /// tier of a ROLLED table is physically dropped at — four + /// days — independently of the collector's much longer advertised retention. So a caller asking for 168 + /// hours got whatever had not aged out, under a label saying 168 hours, with nothing in the response + /// marking the difference. + /// + /// Tier by the age of the window's oldest point, not by its width — the same rule + /// ComposeSourceRouter applies, and for the same reason: retention drops chunks by wall-clock age, so + /// the oldest point is the only thing that decides whether raw can answer. A window that reaches past the + /// raw horizon is served ENTIRELY from the hourly aggregate rather than stitched, because a series whose + /// bucket width changes partway is a worse answer than a coarser consistent one. + /// + /// defaults to the real clock and exists so a test can pin the boundary. + /// It is deliberately NOT defaulted to : a caller asking for a historical window + /// would then have its start measured against its own end, which makes a two-hour window from ten days ago + /// look recent and routes it to a tier that dropped those rows six days earlier. + /// + public static async Task GetQueryHistoryAsync( + NpgsqlDataSource postgres, int serverId, string databaseName, string queryHash, DateTime startUtc, DateTime endUtc, DateTime? nowUtc = null, CancellationToken cancellationToken = default) + { + var useRaw = ShouldUseRawTier(startUtc, nowUtc ?? DateTime.UtcNow); + + var items = useRaw + ? await ReadQueryHistoryAsync(postgres, QueryHistorySql, serverId, databaseName, queryHash, startUtc, endUtc, cancellationToken) + : await ReadQueryHistoryAsync(postgres, QueryHistoryHourlySql, serverId, databaseName, queryHash, startUtc, endUtc, cancellationToken); + + /* What was actually covered, which is what the caller gets told. An empty result says nothing about + coverage, so the requested start stands rather than being narrowed to a window we cannot describe. */ + var effectiveStart = items.Count > 0 ? items[0].CollectionTime : startUtc; + + return new QueryHistoryResult( + items, + useRaw ? "raw" : "hourly", + effectiveStart, + Truncated: items.Count > 0 && effectiveStart > startUtc.AddMinutes(90)); + } + + private static async Task> ReadQueryHistoryAsync( + NpgsqlDataSource postgres, string sql, int serverId, string databaseName, string queryHash, + DateTime startUtc, DateTime endUtc, CancellationToken cancellationToken) { var items = new List(); - await using var command = postgres.CreateCommand(QueryHistorySql); + await using var command = postgres.CreateCommand(sql); DarlingMcpReadParameters.AddInt(command, serverId); DarlingMcpReadParameters.AddText(command, databaseName); DarlingMcpReadParameters.AddText(command, queryHash); diff --git a/Darling/PerformanceMonitor.Darling.Service/Program.cs b/Darling/PerformanceMonitor.Darling.Service/Program.cs index 1300bcaf2..b46c94ba3 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Program.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Program.cs @@ -156,6 +156,24 @@ the running service only verifies. Reads darling.json only (no store, no credent return await DarlingCliCommands.ConfigureFirewallAsync(configPath, Console.Out, Console.Error, CancellationToken.None); } +/* CLI verb: re-apply the secret-file ACLs, elevated (#2352). The running service computes the correct DACL and + detects when the real one is wrong, but cannot apply it — re-ACLing a file it does not own needs WRITE_DAC and + taking ownership needs a privilege a virtual service account is not granted — so it logs the remedy and carries + on. This is the actor with the authority. Verifies every target after the attempt and exits non-zero if + anything is still readable, so it is usable in a provisioning script. Windows-only: ACLs. + Optional second arg = an explicit config path. */ +if (args.Length > 0 && DarlingCliCommands.IsHardenFilesVerb(args[0])) +{ + if (!OperatingSystem.IsWindows()) + { + Console.Error.WriteLine("--harden-files requires Windows (NTFS ACLs)."); + return 1; + } + + var configPath = args.Length > 1 ? args[1] : null; + return DarlingCliCommands.HardenFiles(configPath, Console.Out, Console.Error); +} + /* CLI verbs: enable/disable the embedded MCP + web-dashboard endpoints on a HEADLESS managed deployment. Each flips the live switch in config.config_service (mcp_enabled/web_enabled — the store is authoritative after the first run; darling.json's enabled is only the seed) via a targeted UPDATE whose self-bump trigger makes the diff --git a/Darling/PerformanceMonitor.Darling.Service/StoreConfigProvider.cs b/Darling/PerformanceMonitor.Darling.Service/StoreConfigProvider.cs index b29a03f9c..1c79789c7 100644 --- a/Darling/PerformanceMonitor.Darling.Service/StoreConfigProvider.cs +++ b/Darling/PerformanceMonitor.Darling.Service/StoreConfigProvider.cs @@ -334,7 +334,7 @@ private static async Task SeedServiceRowAsync(NpgsqlConnection connection, Darli /* config_version starts at 0; the four desired-state seed writes below bump it via the trigger, so the worker's post-seed baseline read reflects the seeded state and triggers no spurious reload. */ using var command = new NpgsqlCommand(@" -INSERT INTO config_service (id, paused, capture_plans, query_store_backfill_enabled, query_store_text_budget_mb, max_concurrent_sweeps, plan_xml_compression, mcp_enabled, mcp_port, web_enabled, web_port, plan_content_retention_days, config_version, updated_at, updated_by) +INSERT INTO config_service (id, paused, capture_plans, query_store_backfill_enabled, query_store_text_budget_mb, max_concurrent_sweeps, plan_xml_compression, mcp_enabled, mcp_port, web_enabled, web_port, plan_content_retention_days, compose_statement_timeout_seconds, config_version, updated_at, updated_by) VALUES (1, FALSE, $1, $7, $8, $9, $10, $2, $3, $4, $5, $11, 0, $6, 'seed') ON CONFLICT (id) DO NOTHING", connection); command.Parameters.AddWithValue(config.CapturePlans); @@ -352,6 +352,7 @@ INSERT INTO config_service (id, paused, capture_plans, query_store_backfill_enab the seed is the last step of first contact, and a cosmetic casing choice must not fail it. */ command.Parameters.AddWithValue(NormalizePlanXmlCompression(config.PlanXmlCompression)); command.Parameters.AddWithValue(ClampPlanContentRetentionDays(config.PlanContentRetentionDays)); + command.Parameters.AddWithValue(ClampComposeStatementTimeoutSeconds(config.ComposeStatementTimeoutSeconds)); await command.ExecuteNonQueryAsync(ct); } @@ -377,10 +378,12 @@ INSERT INTO config_alert_settings ( pvs_floor_gb, modified_at, database_state_enabled, self_disk_free_warn_percent, collection_stale_minutes, collection_failure_threshold, disk_critical_free_percent, disk_critical_free_gb, analysis_notify_cooldown_minutes, - store_job_cadence_warn_percent) + store_job_cadence_warn_percent, + /* #2349 appended LAST so no existing placeholder ordinal moves. */ + file_growth_enabled, file_growth_rise_mb, file_growth_volume_percent, file_growth_lookback_minutes) VALUES (1, $1, $2, $3, $4, $5, $6, $7, $8, $9, $10, $11, $12, $13, $14, $15, $16, $17, $18, $19, $20, $21, $22, $23, $24, $25, $26, $27, $28, $29, $30, $31, $32, $33, $34, $35, $36, $37, $38, $39, $40, $41, $42, - $43, $44, $45, $46, $47, $48, $49, $50, $51, $52, $53, $54, $55) + $43, $44, $45, $46, $47, $48, $49, $50, $51, $52, $53, $54, $55, $56, $57, $58, $59) ON CONFLICT (id) DO NOTHING", connection); command.Parameters.AddWithValue(a.Enabled); command.Parameters.AddWithValue(a.CpuEnabled); @@ -447,6 +450,11 @@ INSERT INTO config_alert_settings ( command.Parameters.AddWithValue(a.DiskCriticalFreeGb); command.Parameters.AddWithValue(a.AnalysisNotifyCooldownMinutes); command.Parameters.AddWithValue(a.StoreJobCadenceWarnPercent); + /* #2349, bound in the same order the columns were appended. */ + command.Parameters.AddWithValue(a.FileGrowthEnabled); + command.Parameters.AddWithValue(a.FileGrowthRiseMb); + command.Parameters.AddWithValue(a.FileGrowthVolumePercent); + command.Parameters.AddWithValue(a.FileGrowthLookbackMinutes); await command.ExecuteNonQueryAsync(ct); } @@ -542,7 +550,7 @@ the right host. */ { await using var connection = await _postgres.OpenConnectionAsync(cancellationToken); - var (paused, capturePlans, backfillEnabled, textBudgetMb, maxSweeps, planXmlCompression, mcpEnabled, mcpPort, webEnabled, webPort, planContentRetentionDays, configVersion) = await ReadServiceRowAsync(connection, cancellationToken); + var (paused, capturePlans, backfillEnabled, textBudgetMb, maxSweeps, planXmlCompression, mcpEnabled, mcpPort, webEnabled, webPort, planContentRetentionDays, composeStatementTimeoutSeconds, configVersion) = await ReadServiceRowAsync(connection, cancellationToken); var (alerts, analysis) = await ReadAlertSettingsAsync(connection, cancellationToken); /* The notification row is the ONLY read here that touches secret columns — the SMTP password and @@ -569,6 +577,7 @@ the skip parameter is gone with its last caller. */ QueryStoreBackfillEnabled = backfillEnabled, QueryStoreTextBudgetMb = textBudgetMb, PlanContentRetentionDays = planContentRetentionDays, + ComposeStatementTimeoutSeconds = composeStatementTimeoutSeconds, MaxConcurrentSweeps = maxSweeps, PlanXmlCompression = planXmlCompression, McpEnabled = mcpEnabled, @@ -610,6 +619,15 @@ the skip parameter is gone with its last caller. */ internal static int ClampPlanContentRetentionDays(int value) => value <= 0 ? 0 : Math.Clamp(value, MinPlanContentRetentionDays, MaxPlanContentRetentionDays); + /* #2357: the compose statement_timeout bounds WORK -- a LIMIT bounds output, a group-by scans and sorts + before it -- so it is clamped rather than trusted. A floor of 5s keeps the backstop meaningful; a + ceiling of 600s keeps a hand-edited absurdity from turning "hard backstop" into "no backstop". */ + internal const int MinComposeStatementTimeoutSeconds = 5; + internal const int MaxComposeStatementTimeoutSeconds = 600; + + internal static int ClampComposeStatementTimeoutSeconds(int value) => + Math.Clamp(value <= 0 ? 15 : value, MinComposeStatementTimeoutSeconds, MaxComposeStatementTimeoutSeconds); + /// #2171: unknown values normalize to 'gzip' (fail to the shipped default) so a hand-edited /// row cannot switch the writer into an undefined mode; the V62 CHECK constraint enforces the same /// set DB-side, and this guard covers pre-constraint rows and direct writes with the constraint @@ -617,18 +635,18 @@ internal static int ClampPlanContentRetentionDays(int value) => internal static string NormalizePlanXmlCompression(string? value) => string.Equals(value?.Trim(), "none", StringComparison.OrdinalIgnoreCase) ? "none" : "gzip"; - private static async Task<(bool Paused, bool CapturePlans, bool QueryStoreBackfillEnabled, int QueryStoreTextBudgetMb, int MaxConcurrentSweeps, string PlanXmlCompression, bool McpEnabled, int McpPort, bool WebEnabled, int WebPort, int PlanContentRetentionDays, long ConfigVersion)> + private static async Task<(bool Paused, bool CapturePlans, bool QueryStoreBackfillEnabled, int QueryStoreTextBudgetMb, int MaxConcurrentSweeps, string PlanXmlCompression, bool McpEnabled, int McpPort, bool WebEnabled, int WebPort, int PlanContentRetentionDays, int ComposeStatementTimeoutSeconds, long ConfigVersion)> ReadServiceRowAsync(NpgsqlConnection connection, CancellationToken ct) { using var command = new NpgsqlCommand( - "SELECT paused, capture_plans, query_store_backfill_enabled, query_store_text_budget_mb, max_concurrent_sweeps, plan_xml_compression, mcp_enabled, mcp_port, web_enabled, web_port, plan_content_retention_days, config_version FROM config_service WHERE id = 1", connection); + "SELECT paused, capture_plans, query_store_backfill_enabled, query_store_text_budget_mb, max_concurrent_sweeps, plan_xml_compression, mcp_enabled, mcp_port, web_enabled, web_port, plan_content_retention_days, compose_statement_timeout_seconds, config_version FROM config_service WHERE id = 1", connection); using var reader = await command.ExecuteReaderAsync(ct); if (!await reader.ReadAsync(ct)) { /* Row missing (unseeded) — treat as defaults; capture and backfill stay on, the memory knobs reproduce the pre-V59 compile-time constants (64 MB budget, 4-wide sweep), and plan content keeps the V75 default 21-day horizon. */ - return (false, true, true, 64, 4, "gzip", false, 5152, false, 5153, 21, 0); + return (false, true, true, 64, 4, "gzip", false, 5152, false, 5153, 21, 15, 0); } return (reader.GetBoolean(0), reader.GetBoolean(1), reader.GetBoolean(2), @@ -636,7 +654,8 @@ internal static string NormalizePlanXmlCompression(string? value) => NormalizePlanXmlCompression(reader.GetString(5)), reader.GetBoolean(6), reader.GetInt32(7), reader.GetBoolean(8), reader.GetInt32(9), - ClampPlanContentRetentionDays(reader.GetInt32(10)), reader.GetInt64(11)); + ClampPlanContentRetentionDays(reader.GetInt32(10)), + ClampComposeStatementTimeoutSeconds(reader.GetInt32(11)), reader.GetInt64(12)); } private static async Task<(AlertsConfig Alerts, AnalysisConfig Analysis)> ReadAlertSettingsAsync(NpgsqlConnection connection, CancellationToken ct) @@ -658,7 +677,8 @@ internal static string NormalizePlanXmlCompression(string? value) => pvs_floor_gb, database_state_enabled, self_disk_free_warn_percent, collection_stale_minutes, collection_failure_threshold, disk_critical_free_percent, disk_critical_free_gb, analysis_notify_cooldown_minutes, - store_job_cadence_warn_percent + store_job_cadence_warn_percent, + file_growth_enabled, file_growth_rise_mb, file_growth_volume_percent, file_growth_lookback_minutes FROM config_alert_settings WHERE id = 1", connection); using var reader = await command.ExecuteReaderAsync(ct); if (!await reader.ReadAsync(ct)) @@ -744,6 +764,15 @@ and the wholesale ApplyToConfig replacement never resets a knob. */ reachability rule as every appended knob above: ApplyToConfig replaces config.Alerts wholesale, so a column missing here would silently reset the knob on every worker start. */ StoreJobCadenceWarnPercent = reader.GetInt32(53), + + /* #2349 file-growth knobs appended (V79) at ordinals 54-57, with the same reachability rule as + every appended knob above: ApplyToConfig replaces config.Alerts wholesale, so a column read + here but not selected -- or selected but not read -- silently resets the knob on every worker + start rather than failing. */ + FileGrowthEnabled = reader.GetBoolean(54), + FileGrowthRiseMb = reader.GetInt32(55), + FileGrowthVolumePercent = reader.GetInt32(56), + FileGrowthLookbackMinutes = reader.GetInt32(57), }; var analysis = new AnalysisConfig { @@ -1073,6 +1102,13 @@ public sealed class StoreConfigView /// 0 = disabled (the fact-coupled dimension horizon stands alone). public int PlanContentRetentionDays { get; init; } = 21; + /// + /// The per-session statement_timeout for the viewer and mcp roles, in seconds (#2357). Read + /// clamped to [5,600]; 15 reproduces the constant it replaced. The provisioning DDL applies it, and that + /// DDL re-runs on every managed start, so a change here reaches an existing install on its next restart. + /// + public int ComposeStatementTimeoutSeconds { get; init; } = 15; + /// /// The #2171 plan-XML storage codec (config_service, V62), already normalized to 'gzip' or 'none'. /// 'gzip' (the default, unchanged behavior): the plan dim stores gzip bytes in query_plan_gz, diff --git a/Darling/PerformanceMonitor.Darling.Service/darling.sample.json b/Darling/PerformanceMonitor.Darling.Service/darling.sample.json index bda738e1b..88b533d38 100644 --- a/Darling/PerformanceMonitor.Darling.Service/darling.sample.json +++ b/Darling/PerformanceMonitor.Darling.Service/darling.sample.json @@ -260,5 +260,50 @@ // "allowFrom": "192.168.1.0/24", // in-app RemoteIpAddress check + firewall CIDR (loopback always allowed). // "encryptedToken": "" // REQUIRED to expose; or a plaintext "token" (dev only, warned). // } + }, + + // DECLARED PEER STORES (optional, off by default) -- for a fleet split across SEVERAL Darling boxes, + // one store each: SQL Server primaries on one box, their readable replicas on another, PostgreSQL on a + // third. Each box's MCP server answers over ITS OWN store only, so a server monitored by a sibling + // resolves as not-found -- which an agent cannot tell apart from "nobody monitors this server". Declaring + // the siblings here fixes that at the three places an agent forms its picture of the fleet: the MCP + // instructions gain a coverage paragraph, list_servers gains a "peer_fleets" block, and the + // server-resolution miss message names the peer whose declared coverage matches instead of just failing. + // + // DISCLOSURE ONLY -- there is NO address and NO credential in this block, and nothing behind it. This + // service never contacts a peer, cannot read a peer's data, and cannot tell whether a peer is even + // running: a peer is a NAME plus a sentence, so an agent (or its human) can pick the right endpoint. + // Everything here is sent VERBATIM to every connected MCP client, so the service REFUSES to start if any + // peer text looks like a connection string or credential ("password=", "connectionString", ...). + // + // "name" REQUIRED. Whatever an operator would recognize -- the box name, "the use1 store". + // "covers" A short sentence naming what that store monitors. Human prose; never parsed, only shown. + // "matches" OPTIONAL server-name substrings that store monitors, matched case-insensitively. The only + // machine-checked field: it is what lets a resolution miss name the ONE peer that owns the + // server instead of listing them all. No globbing, no regex -- plain substrings. Blank + // entries are dropped (an empty substring would match every name). A peer with no "matches" + // is still disclosed everywhere; it just cannot be singled out on a miss. + // + // A file-only block (not seeded into the control plane): it describes the deployment topology of THIS + // box, which must not be editable from a peer's Viewer. An edit takes effect on the next service restart. + // + // "peers": { + // "thisStoreCovers": "the 42 us-east-1 SQL Server primaries", + // "stores": [ + // { + // "name": "prod-pos-use2-monitor-01", + // "covers": "the readable replicas of those same 42 primaries, in-region from us-east-2", + // "matches": ["use2"] + // }, + // { + // "name": "prod-pos-pg-monitor-01", + // "covers": "the Aurora PostgreSQL clusters", + // "matches": ["-aurora-", ".cluster-"] + // } + // ] + // } + "peers": { + "thisStoreCovers": "", + "stores": [] } } diff --git a/Darling/PerformanceMonitor.Darling.Storage/PgMigrations.cs b/Darling/PerformanceMonitor.Darling.Storage/PgMigrations.cs index 26ffdaac8..db4cea428 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/PgMigrations.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/PgMigrations.cs @@ -133,6 +133,9 @@ here costs a fresh-through-this-rung store nothing and rung 54's own copy no-ops new Migration(74, "query-store-text", V74Sql), new Migration(75, "plan-content-retention-knob", V75Sql), new Migration(76, "query-store-health", V76Sql), + new Migration(77, "activity-driven-plan-fetch", V77Sql), + new Migration(78, "compose-statement-timeout", V78Sql), + new Migration(79, "file-growth-alert", V79Sql), }; /// @@ -1691,6 +1694,85 @@ PRIMARY KEY (server_id, queryid) CREATE INDEX IF NOT EXISTS idx_pg_statement_text_last_seen ON collect.pg_statement_text(last_seen);"; + /// + /// V79 — the database file-growth alert (#2349). Between tempdb Space and Volume Free Space + /// sits a file that has grown large but has not yet filled its disk, and neither existing alert can express + /// it: the first fires on reserved ÷ (reserved + unallocated), whose denominator GROWS with autogrowth so + /// the percentage FALLS as tempdb balloons; the second fires on the consequence, too late to act on and + /// unable to attribute the space to any one file. + /// + /// Two gates, both graded per server so ONE global setting works across a heterogeneous fleet — the + /// constraint that shapes this, since config_alert_settings is a single global row and an absolute MB + /// threshold is unusable when normal tempdb sizes differ by an order of magnitude: set it low enough for the + /// small instances and the large ones alert constantly. The RISE gate is primary (this file grew N MB inside + /// the window), following #2157's reasoning that a level alone pages forever about a size that has been true + /// since Tuesday; the LEVEL gate is the file as a share of its VOLUME, which self-scales to each server's + /// disk layout whether or not the file has a dedicated one. + /// + /// Ships OFF. A new alert that starts firing on upgrade is a bad citizen, and the right thresholds are + /// a property of the fleet rather than of the product. + /// + private const string V79Sql = @" +ALTER TABLE config.config_alert_settings + ADD COLUMN IF NOT EXISTS file_growth_enabled boolean NOT NULL DEFAULT false, + ADD COLUMN IF NOT EXISTS file_growth_rise_mb integer NOT NULL DEFAULT 10240, + ADD COLUMN IF NOT EXISTS file_growth_volume_percent integer NOT NULL DEFAULT 60, + ADD COLUMN IF NOT EXISTS file_growth_lookback_minutes integer NOT NULL DEFAULT 60;"; + + /// + /// V78 — the compose statement-timeout knob (#2357). The per-session statement_timeout on the + /// viewer and mcp roles is the hard backstop a composed query can never exceed, and it shipped as a bare + /// 15-second constant. Fifteen seconds is a judgement about how big a store is and how fast its disk is, + /// and the product knows neither for anyone else's deployment: a fleet-wide aggregate over a wide window + /// on a large store can exceed it with nothing wrong. + /// + /// Applied in the role PROVISIONING DDL rather than a migration, which is why the constant could + /// not simply be raised — an existing install already has the old value baked into its roles. The + /// provisioning SQL is re-run on every managed start ("idempotent + self-healing: re-run every managed + /// start, converging role state"), so a store picks this up on its next restart without any new + /// machinery. + /// + /// Default 15 preserves today's behaviour exactly for anyone who never touches it. Clamped on READ + /// like the V59/V75 knobs, so a hand-edited absurdity cannot remove the backstop the whole design leans + /// on — a LIMIT bounds output, a group-by scans and sorts before it, and something has to bound WORK. + /// + private const string V78Sql = @" +ALTER TABLE config.config_service + ADD COLUMN IF NOT EXISTS compose_statement_timeout_seconds integer NOT NULL DEFAULT 15;"; + + /// + /// V77 — the activity-driven plan/text fetch (#2312 Finding 2). Three small strokes for one shape + /// change: the fetch stops walking the target's plan catalog by watermark and instead fetches exactly + /// the plans/texts the cycle's collected rows reference that the store does not hold, making the store + /// itself the watermark. + /// + /// digest goes nullable so a plan whose XML the engine cannot persist (too large, certain + /// forced-failure paths) gets a map row with a NULL digest — the content-less MARKER. Without it the + /// missing-set probe would re-select those plans on every cycle forever; with it, "seen, and the content + /// will never exist" is a stored fact. Readers are unaffected: a NULL digest joins to no dimension row, + /// which renders exactly like the absent content it records. DROP NOT NULL is metadata-only and + /// idempotent, so this rung stays instant on the largest maps. + /// + /// query_store_text.query_hash is the reset detector: query_id is only unique until + /// a Query Store reset renumbers it, and the retired design's answer was a daily watermark expiry that + /// re-walked the whole catalog. The stored hash lets the per-cycle probe see that an id now names a + /// DIFFERENT statement and refetch just that text. Nullable and unbackfilled: legacy rows adopt the + /// live hash on their first touch, which converges the fleet with zero refetches. + /// + /// The DELETEs retire the planwm:/textwm: watermark state rows wholesale — the + /// machinery that wrote them is gone, collector_state has no retention (it is state, not facts), + /// and rows nobody will ever read again should not wait for a dropped-database prune that no longer + /// iterates their prefixes. Bare table name resolves via the migrate session's + /// search_path = collect, config, public, like every rung since V8. + /// + private const string V77Sql = @" +ALTER TABLE collect.query_store_plan_map ALTER COLUMN digest DROP NOT NULL; + +ALTER TABLE collect.query_store_text ADD COLUMN IF NOT EXISTS query_hash text; + +DELETE FROM collector_state WHERE collector_name = 'query_store_plan_xml' AND state_key LIKE 'planwm:%'; +DELETE FROM collector_state WHERE collector_name = 'query_store_text' AND state_key LIKE 'textwm:%';"; + /// /// V76 — the per-database Query Store health table (#2319): what database_config's single /// is_query_store_on bit cannot say — actual vs desired state (the cap-hit READ_ONLY transition diff --git a/Darling/PerformanceMonitor.Darling.Storage/QueryStoreFetchProbe.cs b/Darling/PerformanceMonitor.Darling.Storage/QueryStoreFetchProbe.cs new file mode 100644 index 000000000..bbf3ccf19 --- /dev/null +++ b/Darling/PerformanceMonitor.Darling.Storage/QueryStoreFetchProbe.cs @@ -0,0 +1,110 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Threading; +using System.Threading.Tasks; +using Npgsql; + +namespace PerformanceMonitor.Darling.Storage; + +/// One referenced id's probe verdict: does the store resolve it, and is the stored hash stale. +/// plan_id for the plan probe, query_id for the text probe. +/// A store row exists — including the plan side's NULL-digest content-less markers, +/// which must read as known or they ride every cycle's fetch list forever. +/// The stored hash and the batch's live hash both exist and DIFFER — an in-place +/// rewrite (plans) or a post-reset renumbering (texts). Stale ids refetch even though they resolve. +public readonly record struct FetchProbeVerdict(long Id, bool Resolved, bool HashStale); + +/// +/// Executes the two touch-and-probe statements for one database's cycle (#2312): the liveness refresh the +/// dimension GC depends on and the missing-set answer the activity-driven fetch runs on, one round trip +/// each. The runner hands in the cycle's distinct referenced ids with their live hashes; what comes back is +/// the fetch list — !Resolved || HashStale — and nothing else needs to be consulted, because the +/// store IS the watermark. +/// +public static class QueryStoreFetchProbe +{ + public static Task> TouchAndProbePlansAsync( + NpgsqlConnection connection, + int serverId, + string databaseName, + IReadOnlyList<(long PlanId, string? PlanHash)> references, + DateTime collectionTimeUtc, + CancellationToken cancellationToken = default) + => ExecuteAsync(connection, QueryStorePlanMap.TouchAndProbeSql, serverId, databaseName, references, collectionTimeUtc, cancellationToken); + + public static Task> TouchAndProbeTextsAsync( + NpgsqlConnection connection, + int serverId, + string databaseName, + IReadOnlyList<(long QueryId, string? QueryHash)> references, + DateTime collectionTimeUtc, + CancellationToken cancellationToken = default) + => ExecuteAsync(connection, QueryStoreTextStore.TouchAndProbeSql, serverId, databaseName, references, collectionTimeUtc, cancellationToken); + + private static async Task> ExecuteAsync( + NpgsqlConnection connection, + string sql, + int serverId, + string databaseName, + IReadOnlyList<(long Id, string? Hash)> references, + DateTime collectionTimeUtc, + CancellationToken cancellationToken) + { + if (connection is null) + { + throw new ArgumentNullException(nameof(connection)); + } + + if (references is null) + { + throw new ArgumentNullException(nameof(references)); + } + + var verdicts = new List(references.Count); + if (references.Count == 0) + { + return verdicts; + } + + var serverIds = new int[references.Count]; + var databases = new string[references.Count]; + var ids = new long[references.Count]; + var hashes = new string?[references.Count]; + for (var i = 0; i < references.Count; i++) + { + serverIds[i] = serverId; + databases[i] = databaseName; + ids[i] = references[i].Id; + hashes[i] = references[i].Hash; + } + + using var command = new NpgsqlCommand(sql, connection); + command.Parameters.AddWithValue(serverIds); + command.Parameters.AddWithValue(databases); + command.Parameters.AddWithValue(ids); + command.Parameters.AddWithValue(hashes); + /* Naive(), the #1969 trap: a Kind=Utc value infers timestamptz and Postgres converts it into the + session zone on the way into the naive last_seen columns — hours of silent skew on the exact + stamp the GC sweeps. */ + command.Parameters.AddWithValue(QueryStorePlanMap.Naive(collectionTimeUtc)); + + await using var reader = await command.ExecuteReaderAsync(cancellationToken); + while (await reader.ReadAsync(cancellationToken)) + { + verdicts.Add(new FetchProbeVerdict( + reader.GetInt64(2), + !reader.IsDBNull(3) && reader.GetBoolean(3), + !reader.IsDBNull(4) && reader.GetBoolean(4))); + } + + return verdicts; + } +} diff --git a/Darling/PerformanceMonitor.Darling.Storage/QueryStorePlanMap.cs b/Darling/PerformanceMonitor.Darling.Storage/QueryStorePlanMap.cs index 7669f5c72..e32fc9749 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/QueryStorePlanMap.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/QueryStorePlanMap.cs @@ -19,26 +19,28 @@ namespace PerformanceMonitor.Darling.Storage; /// each plan once, in plan_id order, and this map is how a fact row finds its content. /// /// Facts deliberately do NOT gain a digest column, which is why this table exists rather than a -/// entry: a fact row is written when the runtime stats arrive, and under the -/// new ordering that is potentially several budgeted cycles BEFORE its plan's XML is fetched, so there is no -/// digest to write at fact time. The map's absence of a row IS the pending state — distinguishable from "never -/// collected" by whether the plan_id sits above the database's watermark — and a reader with no map row renders -/// "plan not yet collected" instead of resolving to nothing. +/// entry: a fact row is written when the runtime stats arrive, potentially +/// budgeted cycles BEFORE its plan's XML is fetched, so there is no digest to write at fact time. The map's +/// absence of a row IS the pending state, and a reader with no map row renders "plan not yet collected" +/// instead of resolving to nothing. Since #2312 the map is also the FETCH's source of truth — the +/// activity-driven fetch asks which of the cycle's referenced plans lack a +/// row and fetches exactly those, which is what retired the per-database watermark and its +/// daily-expiry catalog walk. /// /// THIS TABLE'S last_seen IS LOAD-BEARING, AND IT IS THE ONLY PROTECTION QUERY STORE DIGESTS HAVE. /// The dimension GC does not enumerate references — an anti-join per dim row against two hypertables is not -/// affordable at this size — so it sweeps on last_seen, which the write path refreshes on every cycle -/// that references a digest. That worked precisely BECAUSE plan XML was re-shipped every pass; ending the -/// re-shipping ends the liveness signal, and a plan fetched once would have its dim row collected while live -/// facts still referenced it. Hence , which asserts liveness for plans the batch no -/// longer carries. +/// affordable at this size — so it sweeps on last_seen, which must be refreshed on every cycle +/// that references a digest. Ending plan re-shipping ended the accidental refresh the walk provided, and a +/// plan fetched once would have its dim row collected while live facts still referenced it. Hence +/// , which asserts liveness for every plan the cycle's batch references — +/// designed for exactly this in #2210, unwired until #2312 Finding 3. /// /// The GC's second belt does not cover this either: its cutoff is clamped to one day before the oldest /// surviving DIGEST-CARRYING fact (DarlingRetention.ComputeDimensionCutoff), and Query Store facts carry /// no digest, so the measured floor is blind to them. Do not add a Query Store entry to /// to "fix" that — the entry means "this fact column holds a digest", /// which is not true here, and the clamp would still be measuring a column that does not exist. -/// is the whole of the protection. +/// is the whole of the protection. /// public static class QueryStorePlanMap { @@ -47,9 +49,13 @@ public static class QueryStorePlanMap /// than time-series and is pruned on rather than by drop_chunks. public const string TableName = "collect.query_store_plan_map"; - /// The liveness column, swept by the prune and stamped by . + /// The liveness column, swept by the prune and stamped by . public const string LastSeenColumn = "last_seen"; + /* This const mirrors the IMMUTABLE migration rung that created the table and stays byte-frozen with it. + The LIVE shape differs in one place: V77 relaxed digest to nullable for the #2312 content-less marker + rows (a plan whose XML the engine cannot persist gets a map row with a NULL digest, so the probe reads + it as known instead of refetching it forever). */ public const string CreateTableSql = @"CREATE TABLE IF NOT EXISTS collect.query_store_plan_map ( server_id integer NOT NULL, database_name text NOT NULL, @@ -69,18 +75,25 @@ index nothing queries is pure write tax on a table every plan fetch upserts into /// /// Records what the plan fetch landed: one row per plan, carrying the digest of the content written to - /// query_plan_dim. The conflict arm advances digest as well as last_seen, because a - /// plan whose XML is rewritten in place keeps its plan_id while its content digest changes — the - /// case the watermark's refresh horizon exists to catch, and this is where the corrected content gets - /// pointed at. + /// query_plan_dim — or a NULL digest for a plan whose XML the engine itself reports as absent + /// (too large to persist, certain forced-plan-failure paths). The NULL-digest row is the #2312 + /// content-less MARKER: under store-as-watermark, "seen, and the content will never exist" has to be a + /// stored fact, or the probe would re-select those plans as missing on every cycle forever — the old + /// watermark's stall reborn in miniature. V77 relaxed the column for exactly this row shape. /// - /// plan_hash is what makes re-verification cheap, and it is why it is stored here rather than - /// derived: sys.query_store_plan.query_plan_hash reads WITHOUT decompressing the plan, so the - /// re-verify cursor can walk [0..watermark] comparing hashes on cheap columns alone and re-fetch XML - /// only where a hash DIFFERS or a map row is ABSENT. That turns in-place rewrites from a full catalog walk - /// per horizon into per-changed-plan work — 0 of 38,420 plan_ids changed hash across a day of fleet data — - /// and dormant plans fall out of the same pass with no heuristic to separate them from a reset, because mass - /// absence is caught wholesale by the runtime stream's reset arm within one cycle. + /// Both COALESCEs in the conflict arm point the same direction — never replace knowledge with + /// absence. A refetch that comes back NULL for a plan whose content the store already holds keeps the + /// content (digest); a fetch path that did not carry a hash keeps the stored one + /// (plan_hash). A refetch that carries REAL content or a REAL hash still advances both, which is + /// how an in-place rewrite's corrected content gets pointed at. + /// + /// plan_hash is the in-place-rewrite detector, stored rather than derived because + /// sys.query_store_plan.query_plan_hash reads WITHOUT decompressing the plan: + /// compares it against the batch's live hash on every cycle, so an + /// active plan whose XML was rewritten in place (same plan_id, new content) is refetched within + /// one cycle — where the retired re-verify cursor would have taken up to a day to reach it, had it ever + /// been wired (#2312 Finding 4: it was not). Measured base rate: 0 of 38,420 plan_ids changed hash in a + /// day of fleet data. /// /// Ordered by the conflict key. Same reason as : concurrent /// batch upserts that take row locks in different relative orders deadlock (#1801), and a plan fetch runs @@ -94,41 +107,40 @@ FROM unnest($1::integer[], $2::text[], $3::bigint[], $4::bytea[], $5::text[], $6 AS batch(server_id, database_name, plan_id, digest, plan_hash, stamped) ORDER BY server_id, database_name, plan_id ON CONFLICT (server_id, database_name, plan_id) DO UPDATE SET - digest = EXCLUDED.digest, - plan_hash = EXCLUDED.plan_hash, + digest = COALESCE(EXCLUDED.digest, query_store_plan_map.digest), + plan_hash = COALESCE(EXCLUDED.plan_hash, query_store_plan_map.plan_hash), last_seen = EXCLUDED.last_seen WHERE EXCLUDED.last_seen >= query_store_plan_map.last_seen"; /// - /// The liveness assertion, and the reason this whole design is safe: for the distinct - /// (database_name, plan_id) a runtime-stats batch just wrote, refresh BOTH this map row's - /// last_seen and the dimension row's, so neither can age out while facts still point at the plan. - /// The batch already carries those two columns, so nothing extra is collected to make this work. + /// The liveness assertion AND the missing-set probe, one round trip (#2312): for the distinct + /// (database_name, plan_id, plan_hash) a cycle's runtime batch references, refresh BOTH this map + /// row's last_seen and the dimension row's — so neither can age out while facts still point at + /// the plan — and return, per batch row, whether the store already resolves it and whether its stored + /// hash still matches the engine's live one. The batch already carries all three columns, so nothing + /// extra is collected to make this work. /// /// Both timestamps are stamped by the SAME pass, which is what makes the map-prune-versus-dim-GC race /// structurally impossible rather than carefully avoided: a map row's last_seen can never be older /// than the newest fact batch that referenced it, so the prune cannot take a row that live facts are - /// touching. - /// - /// It also returns the RESOLVED-ness of every batch row, in the same round trip, because the batch - /// join it already does is where the reset signal lives: a plan the store has never resolved cannot be - /// produced by "no new plans this window". What it deliberately does NOT do is decide that a reset happened. - /// Two reasons, both of which bit the first version of this query: + /// touching. This is also what makes the V75 plan-content horizon mean what it says for Query Store + /// plans: content ages out N-days-since-last-REFERENCE, not since-last-refetch — before #2312 wired + /// this, the perpetual daily catalog walk was accidentally standing in for it. /// - /// • One dormant plan is not a reset. Filtering to absent rows at or below a watermark fires on - /// a SINGLE dormant plan resuming execution, which would zero that database's watermark and trigger a full - /// refetch — the opposite of what this design is for. The reset case is MASS absence, and "mass" is a - /// judgement the caller makes across the batch. A lone absence is the CURSOR's job (it fetches that plan and - /// moves on), which is what RefreshAfter's comment already says owns dormancy.
- /// • Watermarks are per database. These array parameters can carry rows for several databases in one - /// call, so comparing them all against one scalar watermark is wrong for every database but one. The caller - /// already holds the per-database watermarks; it applies them.
+ /// The probe columns are the activity-driven fetch's entire input. resolved is "a map row + /// exists" — INCLUDING the NULL-digest content-less markers, which is the point of storing them: a plan + /// the engine cannot persist must read as known, or it rides every cycle's fetch list forever. + /// hash_stale is the in-place-rewrite signal: a stored hash that differs from the batch's live + /// one means the plan kept its id and changed its content, and the caller refetches it. A stored hash + /// of NULL is never stale — it is adopted from the batch in the same statement (legacy rows from before + /// the fetch carried hashes), so the fleet's hash coverage backfills organically with zero refetches. /// - /// So this returns facts — (server_id, database_name, plan_id, resolved) — and the host decides. - /// When it does conclude a reset it zeroes that database's watermark and logs loudly, recovering in one - /// cycle rather than waiting on a refresh sweep. + /// What this deliberately does NOT do is detect Query Store resets, because nothing needs to any + /// more: a reset renumbers plans, the new ids come back unresolved, and the fetch list picks them up + /// budget-bounded — recovery is the normal path rather than a special arm. (The retired watermark design + /// needed the mass-absence judgement precisely because it had a watermark to zero; see #2312.) /// - /// The three preceding CTEs still run. Postgres executes data-modifying WITH statements exactly + /// The preceding CTEs still run. Postgres executes data-modifying WITH statements exactly /// once and to completion whether or not the primary query reads their output, so making the final statement /// a SELECT does not turn the liveness stamping into a no-op. That is a load-bearing detail: if it were not /// true, this restructure would silently stop refreshing last_seen and reintroduce the GC hazard the @@ -136,31 +148,34 @@ ON CONFLICT (server_id, database_name, plan_id) DO UPDATE SET /// /// Guarded at one hour like the dimension upsert's own conflict arm, and for the same reason — the /// horizons are multi-day, so an update per row per hour is enough freshness and the write amplification - /// stays bounded on a hot catalog. The margin arithmetic already accounts for this trailing hour. + /// stays bounded on a hot catalog. The margin arithmetic already accounts for this trailing hour. Hash + /// adoption rides the same guard: it is a backfill, not a correctness deadline, and un-guarding it would + /// re-write every legacy row on every cycle until the first touch landed. /// - /// The CTE is ordered by the map's primary key for the #1801 reason above — this is the one statement - /// in the design that touches many rows across two tables on every cycle of every server, so it is the most - /// likely place for an unordered-batch deadlock to form. Being precise about how much that buys, because - /// can claim more than this can: an ORDER BY inside a CTE - /// feeding UPDATE ... FROM is NOT a guaranteed lock-acquisition order in Postgres the way ordering an - /// INSERT ... ON CONFLICT's input is — the planner may reorder. It makes the common plan deterministic - /// rather than making the deadlock impossible. If one is observed, the fix is to drive the update from an - /// explicitly ordered SELECT ... FOR UPDATE, not to widen this comment. + /// The CTE is ordered by the map's primary key for the #1801 reason on — + /// this is the one statement in the design that touches many rows across two tables on every cycle of + /// every server, so it is the most likely place for an unordered-batch deadlock to form. Being precise + /// about how much that buys: an ORDER BY inside a CTE feeding UPDATE ... FROM is NOT a + /// guaranteed lock-acquisition order in Postgres the way ordering an INSERT ... ON CONFLICT's + /// input is — the planner may reorder. It makes the common plan deterministic rather than making the + /// deadlock impossible. If one is observed, the fix is to drive the update from an explicitly ordered + /// SELECT ... FOR UPDATE, not to widen this comment. ///
- public const string TouchSql = @"WITH touched AS ( - SELECT m.server_id, m.database_name, m.plan_id, m.digest + public const string TouchAndProbeSql = @"WITH touched AS ( + SELECT m.server_id, m.database_name, m.plan_id, m.digest, batch.plan_hash AS live_hash FROM collect.query_store_plan_map AS m - JOIN unnest($1::integer[], $2::text[], $3::bigint[]) - AS batch(server_id, database_name, plan_id) + JOIN unnest($1::integer[], $2::text[], $3::bigint[], $4::text[]) + AS batch(server_id, database_name, plan_id, plan_hash) ON batch.server_id = m.server_id AND batch.database_name = m.database_name AND batch.plan_id = m.plan_id - WHERE m.last_seen < $4::timestamp - interval '1 hour' + WHERE m.last_seen < $5::timestamp - interval '1 hour' ORDER BY m.server_id, m.database_name, m.plan_id ), map_touch AS ( UPDATE collect.query_store_plan_map AS m - SET last_seen = $4::timestamp + SET last_seen = $5::timestamp, + plan_hash = COALESCE(m.plan_hash, t.live_hash) FROM touched AS t WHERE m.server_id = t.server_id AND m.database_name = t.database_name @@ -169,14 +184,20 @@ RETURNING t.digest ), dim_touch AS ( UPDATE collect.query_plan_dim AS d - SET last_seen = $4::timestamp - WHERE d.digest IN (SELECT digest FROM map_touch) - AND d.last_seen < $4::timestamp - interval '1 hour' + SET last_seen = $5::timestamp + WHERE d.digest IN (SELECT digest FROM map_touch WHERE digest IS NOT NULL) + AND d.last_seen < $5::timestamp - interval '1 hour' RETURNING d.digest ) -SELECT batch.server_id, batch.database_name, batch.plan_id, (m.plan_id IS NOT NULL) AS resolved -FROM unnest($1::integer[], $2::text[], $3::bigint[]) - AS batch(server_id, database_name, plan_id) +SELECT + batch.server_id, + batch.database_name, + batch.plan_id, + (m.plan_id IS NOT NULL) AS resolved, + (m.plan_id IS NOT NULL AND m.plan_hash IS NOT NULL AND batch.plan_hash IS NOT NULL + AND m.plan_hash <> batch.plan_hash) AS hash_stale +FROM unnest($1::integer[], $2::text[], $3::bigint[], $4::text[]) + AS batch(server_id, database_name, plan_id, plan_hash) LEFT JOIN collect.query_store_plan_map AS m ON m.server_id = batch.server_id AND m.database_name = batch.database_name @@ -186,7 +207,7 @@ LEFT JOIN collect.query_store_plan_map AS m /// /// Strips the off a timestamp before it is bound to any of this class's /// ::timestamp parameters. Every call site that binds a here must go through - /// this — 's stamp array and 's $4, plus the prune's + /// this — 's stamp array and 's $5, plus the prune's /// cutoff. /// /// This is the #1969 trap, and it is silent. Npgsql infers the parameter type from the value's Kind: @@ -203,69 +224,6 @@ LEFT JOIN collect.query_store_plan_map AS m /// public static DateTime Naive(DateTime utc) => DateTime.SpecifyKind(utc, DateTimeKind.Unspecified); - /// - /// The map rows a re-verify cursor slice needs to judge, for one database over a bounded - /// plan_id range: what content the store believes each plan has. - /// - /// Returns plan_id and plan_hash only — never the digest, never content. The caller - /// pairs this against the same id range read from sys.query_store_plan (also hash-only, which reads - /// without decompressing) and re-fetches XML for exactly three cases: a hash that DIFFERS (the plan was - /// rewritten in place while keeping its id), a map row that is ABSENT (a plan dormant through every - /// collected window, so the watermark passed it without its content ever landing), and a stored - /// plan_hash that is NULL (written by a build before the hash column existed — re-verify once, then - /// it self-heals). - /// - /// This is the whole reason the horizon stopped being a full refetch. The old expiry dropped the - /// watermark to zero and re-walked every plan's XML, which the walk-cost measurement showed cannot even - /// complete inside a day on the larger catalogs (2.2-15.1 GB of plan XML per catalog; 15.9 to 107.5 hours at - /// a 12 MB budget and 5-minute cadence), so those catalogs restarted forever and never reached their own - /// newest plans. A hash-only sweep over the same id range is bounded by ROW count instead of BYTE volume — - /// 77k ids at ~270 per pass — and re-fetches only what actually changed, which across a day of fleet data - /// was 0 of 38,420 plans. - /// - public const string CursorSliceSql = @"SELECT m.plan_id, m.plan_hash -FROM collect.query_store_plan_map AS m -WHERE m.server_id = $1 - AND m.database_name = $2 - AND m.plan_id > $3 - AND m.plan_id <= $4 -ORDER BY m.plan_id"; - - /// - /// The cursor's slice width for one pass: the id range divided by how many passes fit in the sweep period. - /// is no longer an expiry — it is the target period for ONE full - /// re-verification sweep — and this is where that meaning is applied. - /// - /// Floored at one so a cursor always makes progress, and floored again by - /// so a tiny catalog does not crawl an id at a time. Bounded by the range - /// itself, so a sweep never claims to cover ids that do not exist. - /// - public static long CursorSliceWidth(long watermark, TimeSpan refreshAfter, TimeSpan cadence, long minimumSlice = 64) - { - if (watermark <= 0) - { - return 0; - } - - var passes = cadence > TimeSpan.Zero ? refreshAfter.Ticks / cadence.Ticks : 1; - if (passes < 1) - { - passes = 1; - } - - /* CEILING, not floor. Truncating divides a sweep that never completes: Redstone's 77,176 ids over 288 - five-minute passes floors to 267, and 267 * 288 = 76,896 — 280 ids short, every sweep, forever. The - cursor would walk almost the whole catalog and then restart, which is a quieter version of the exact - failure this design replaced. */ - var slice = (watermark + passes - 1) / passes; - if (slice < minimumSlice) - { - slice = minimumSlice; - } - - return slice > watermark ? watermark : slice; - } - /// /// Days of margin the map prune adds past the fact-retention horizon. **Strictly less than the dimension /// GC's margin**, which is ChunkIntervalDays + 1 — see for why @@ -296,7 +254,7 @@ public static bool MarginOrderingHolds(int chunkIntervalDays) => /// /// Timestamp-driven, NOT an existence check against query_store_stats. An anti-join against a /// 43 GB hypertable per map row is exactly the cost this architecture avoids, and it is unnecessary here - /// because keeps last_seen current for anything live — the same argument the + /// because keeps last_seen current for anything live — the same argument the /// dimension GC already rests on, applied to one more timestamped table. /// /// A plan whose query goes quiet needs no special handling: it stops being touched, its facts age out diff --git a/Darling/PerformanceMonitor.Darling.Storage/QueryStorePlanWriter.cs b/Darling/PerformanceMonitor.Darling.Storage/QueryStorePlanWriter.cs index 1e89638af..50b919880 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/QueryStorePlanWriter.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/QueryStorePlanWriter.cs @@ -17,9 +17,10 @@ namespace PerformanceMonitor.Darling.Storage; /// One plan the plan-XML fetch landed, before it has been given a content digest. /// The Query Store plan id, unique within its database. /// The plan XML, or null — a plan too large to persist reads NULL and still ships, so the -/// watermark can advance past a plan whose content will never exist. -/// SQL Server's query_plan_hash, readable without decompressing the plan, which is -/// what lets the re-verify cursor detect in-place rewrites on cheap columns alone. +/// store can record the content-less marker instead of re-selecting the plan as missing forever (#2312). +/// SQL Server's query_plan_hash, readable without decompressing the plan — the +/// stored baseline QueryStorePlanMap.TouchAndProbeSql compares live hashes against to catch in-place +/// rewrites on cheap columns alone. public readonly record struct FetchedPlan(long PlanId, string? PlanXml, string? PlanHash); /// @@ -41,15 +42,17 @@ namespace PerformanceMonitor.Darling.Storage; public static class QueryStorePlanWriter { /// - /// Lands a fetch's plans for one database. Returns the plan_ids whose content actually stored, in the order - /// they were supplied, which is what the caller feeds to - /// QueryStorePlanXmlState.AdvanceWatermark — the watermark must reflect what LANDED, not what was - /// selected, or a torn pass advances past content that never arrived. + /// Lands a fetch's plans for one database. Returns the plan_ids that landed, in the order they were + /// supplied — the caller uses them to clear its budget-carry-over set, since anything landed is no + /// longer missing. /// - /// A plan with NULL XML counts as landed and gets NO dimension row and NO map row: there is no content - /// to key, and inventing a digest for absent content would make the map point at nothing. The watermark - /// still advances past it, which is correct — that plan's XML will never exist, and stalling on it forever is - /// the failure the budget predicate already had to be fixed for twice. + /// A plan with NULL XML gets NO dimension row (there is no content to key, and inventing a digest + /// for absent content would make the map point at nothing) but it DOES get a map row with a NULL digest + /// — the #2312 content-less marker. Under store-as-watermark the map row IS the fact that the plan was + /// fetched and the engine had nothing to give: without it the probe reads the plan as missing and the + /// fetch re-selects it every cycle forever, which is the old oversized-plan stall reborn through the + /// probe. Readers are unaffected — a NULL digest joins to no dimension row, which renders exactly like + /// the absent content it records. /// public static async Task> WriteAsync( NpgsqlConnection connection, @@ -79,7 +82,7 @@ public static async Task> WriteAsync( var mapServerIds = new List(plans.Count); var mapDatabases = new List(plans.Count); var mapPlanIds = new List(plans.Count); - var mapDigests = new List(plans.Count); + var mapDigests = new List(plans.Count); var mapHashes = new List(plans.Count); /* Naive() on the stamp, not the raw UTC value: these are ::timestamp parameters, and Npgsql would infer @@ -92,14 +95,13 @@ timestamptz from a Utc Kind and let Postgres convert into the session zone on th { landed.Add(plan.PlanId); - if (string.IsNullOrEmpty(plan.PlanXml)) + byte[]? digest = null; + if (!string.IsNullOrEmpty(plan.PlanXml)) { - continue; + digest = PayloadDimensions.Digest(plan.PlanXml!); + batch.Add(PayloadDimensions.QueryPlanDimTable, digest, plan.PlanXml!); } - var digest = PayloadDimensions.Digest(plan.PlanXml!); - batch.Add(PayloadDimensions.QueryPlanDimTable, digest, plan.PlanXml!); - mapServerIds.Add(serverId); mapDatabases.Add(databaseName); mapPlanIds.Add(plan.PlanId); @@ -107,11 +109,6 @@ timestamptz from a Utc Kind and let Postgres convert into the session zone on th mapHashes.Add(plan.PlanHash); } - if (mapPlanIds.Count == 0) - { - return landed; - } - using var transaction = await connection.BeginTransactionAsync(cancellationToken); /* Dimension FIRST — see the class comment on why the torn-write side matters. */ diff --git a/Darling/PerformanceMonitor.Darling.Storage/QueryStoreTextStore.cs b/Darling/PerformanceMonitor.Darling.Storage/QueryStoreTextStore.cs index 3039ddc94..c52d4d8bc 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/QueryStoreTextStore.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/QueryStoreTextStore.cs @@ -71,27 +71,76 @@ CREATE INDEX IF NOT EXISTS idx_query_store_text_last_seen /// The conflict arm overwrites the TEXT, not just the stamp, and that is load-bearing /// rather than defensive. query_id is unique within a database only until Query Store is reset: /// a reset renumbers from the start, so id 5 afterwards is a DIFFERENT statement than id 5 before. The - /// refresh horizon on the watermark is what brings us back to re-read it, and this is where the - /// corrected text has to land. Touching only last_seen would leave the old statement's text - /// attached to the new id forever, which reads as a plausible wrong answer rather than as missing - /// data. + /// hash comparison is what brings us back to re-read it (#2312 — the + /// retired watermark's daily expiry used to, eventually), and this is where the corrected text has to + /// land. Touching only last_seen would leave the old statement's text attached to the new id + /// forever, which reads as a plausible wrong answer rather than as missing data. /// /// ORDER BY on the conflict key because concurrent batches that touch overlapping keys in /// different orders deadlock (#1801) — the same reason the plan map's upsert carries one. The /// WHERE EXCLUDED.last_seen >= guard keeps the stamp monotonic so an out-of-order write - /// cannot age a row backwards into the prune's reach. + /// cannot age a row backwards into the prune's reach. query_hash takes EXCLUDED when the fetch + /// carried one and keeps the stored value otherwise — never replace knowledge with absence. /// public const string UpsertSql = @"INSERT INTO collect.query_store_text - (server_id, database_name, query_id, query_sql_text, last_seen) -SELECT server_id, database_name, query_id, query_sql_text, stamped -FROM unnest($1::integer[], $2::text[], $3::bigint[], $4::text[], $5::timestamp[]) - AS batch(server_id, database_name, query_id, query_sql_text, stamped) + (server_id, database_name, query_id, query_sql_text, query_hash, last_seen) +SELECT server_id, database_name, query_id, query_sql_text, query_hash, stamped +FROM unnest($1::integer[], $2::text[], $3::bigint[], $4::text[], $5::text[], $6::timestamp[]) + AS batch(server_id, database_name, query_id, query_sql_text, query_hash, stamped) ORDER BY server_id, database_name, query_id ON CONFLICT (server_id, database_name, query_id) DO UPDATE SET query_sql_text = EXCLUDED.query_sql_text, + query_hash = COALESCE(EXCLUDED.query_hash, query_store_text.query_hash), last_seen = EXCLUDED.last_seen WHERE EXCLUDED.last_seen >= query_store_text.last_seen"; + /// + /// The text side's liveness touch and missing-set probe (#2312), the single-table sibling of + /// : refresh last_seen for every statement the + /// cycle's batch references (hourly-guarded, same write-amplification argument), adopt the batch's + /// query_hash where the stored one is NULL (legacy rows from before the column existed), and + /// return per batch row whether the store already holds the text and whether the stored hash still + /// matches the live one. hash_stale is the Query Store RESET detector: ids renumber, so id 5 + /// carrying a different hash means it now names a different statement and its text must be refetched — + /// per-id, within one cycle, where the retired watermark design re-walked the whole catalog daily to + /// eventually notice. + /// + public const string TouchAndProbeSql = @"WITH touched AS ( + SELECT t.server_id, t.database_name, t.query_id, batch.query_hash AS live_hash + FROM collect.query_store_text AS t + JOIN unnest($1::integer[], $2::text[], $3::bigint[], $4::text[]) + AS batch(server_id, database_name, query_id, query_hash) + ON batch.server_id = t.server_id + AND batch.database_name = t.database_name + AND batch.query_id = t.query_id + WHERE t.last_seen < $5::timestamp - interval '1 hour' + ORDER BY t.server_id, t.database_name, t.query_id +), +text_touch AS ( + UPDATE collect.query_store_text AS t + SET last_seen = $5::timestamp, + query_hash = COALESCE(t.query_hash, x.live_hash) + FROM touched AS x + WHERE t.server_id = x.server_id + AND t.database_name = x.database_name + AND t.query_id = x.query_id + RETURNING t.query_id +) +SELECT + batch.server_id, + batch.database_name, + batch.query_id, + (t.query_id IS NOT NULL) AS resolved, + (t.query_id IS NOT NULL AND t.query_hash IS NOT NULL AND batch.query_hash IS NOT NULL + AND t.query_hash <> batch.query_hash) AS hash_stale +FROM unnest($1::integer[], $2::text[], $3::bigint[], $4::text[]) + AS batch(server_id, database_name, query_id, query_hash) +LEFT JOIN collect.query_store_text AS t + ON t.server_id = batch.server_id + AND t.database_name = batch.database_name + AND t.query_id = batch.query_id +ORDER BY batch.server_id, batch.database_name, batch.query_id"; + /// /// Retires text whose facts have all aged out, bounded to roughly one chunk-width of the oldest rows /// per call so a single sweep cannot take an unbounded row lock — the same shape and the same reason as diff --git a/Darling/PerformanceMonitor.Darling.Storage/QueryStoreTextWriter.cs b/Darling/PerformanceMonitor.Darling.Storage/QueryStoreTextWriter.cs index 09aad9edc..264e81b1d 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/QueryStoreTextWriter.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/QueryStoreTextWriter.cs @@ -15,7 +15,10 @@ namespace PerformanceMonitor.Darling.Storage; /// One statement's text as the fetch returned it. -public readonly record struct FetchedQueryText(long QueryId, string? QueryText); +/// SQL Server's query_hash, the renumbering detector (#2312): query_id is +/// only unique until a Query Store reset, so the stored hash is what lets the probe see that an id now +/// names a DIFFERENT statement and refetch its text. +public readonly record struct FetchedQueryText(long QueryId, string? QueryText, string? QueryHash); /// /// Lands what the query-text fetch returned into (#2150). @@ -29,16 +32,13 @@ public static class QueryStoreTextWriter { /// /// Lands a fetch's statement text for one database, returning the query_ids that stored, in the - /// order supplied — which is what the caller feeds to - /// . The watermark must - /// reflect what LANDED rather than what was selected, or a torn pass advances past text that never - /// arrived. + /// order supplied — the caller uses them to clear its budget-carry-over set, since anything landed is + /// no longer missing (#2312). /// /// Rows with null text are stored as null rather than skipped. Query Store does not produce them /// in practice, so this is about not having a special case to get wrong: a null in this store means "we /// fetched and there was nothing", the readers already COALESCE onto the fact row's own column, - /// and the watermark advances either way — stalling on a statement whose text will never exist is the - /// failure the budget predicate had to be fixed for twice on the plan side. + /// and the stored row is what stops the probe from re-selecting the id as missing forever. /// public static async Task> WriteAsync( NpgsqlConnection connection, @@ -68,6 +68,7 @@ public static async Task> WriteAsync( var databases = new string[texts.Count]; var queryIds = new long[texts.Count]; var bodies = new string?[texts.Count]; + var hashes = new string?[texts.Count]; var stamps = new DateTime[texts.Count]; /* Naive() on the stamp, not the raw UTC value: last_seen is a ::timestamp parameter, and Npgsql @@ -85,6 +86,7 @@ way in (#1969). A last_seen written at the wrong hour ages rows out ahead of the databases[i] = databaseName; queryIds[i] = text.QueryId; bodies[i] = text.QueryText; + hashes[i] = text.QueryHash; stamps[i] = stamp; } @@ -93,6 +95,7 @@ way in (#1969). A last_seen written at the wrong hour ages rows out ahead of the upsert.Parameters.AddWithValue(databases); upsert.Parameters.AddWithValue(queryIds); upsert.Parameters.AddWithValue(bodies); + upsert.Parameters.AddWithValue(hashes); upsert.Parameters.AddWithValue(stamps); await upsert.ExecuteNonQueryAsync(cancellationToken); diff --git a/Darling/PerformanceMonitor.Darling.Storage/StorageVersion.cs b/Darling/PerformanceMonitor.Darling.Storage/StorageVersion.cs index 13c4693ab..727d3b68d 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/StorageVersion.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/StorageVersion.cs @@ -16,5 +16,5 @@ namespace PerformanceMonitor.Darling.Storage; /// public static class StorageVersion { - public const int SchemaVersion = 76; + public const int SchemaVersion = 79; } diff --git a/Darling/PerformanceMonitor.Darling.Viewer/FinOpsTab.xaml b/Darling/PerformanceMonitor.Darling.Viewer/FinOpsTab.xaml index 41a8371f3..a87d180ea 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/FinOpsTab.xaml +++ b/Darling/PerformanceMonitor.Darling.Viewer/FinOpsTab.xaml @@ -972,7 +972,13 @@ - internal static int MapProbedSchemaVersion(bool hasConfigControlPlane, bool hasAlertDeliveryOverride, bool hasAnalysisState, bool hasAlertTuningKnobs, bool hasDefaultTraceEvents, bool hasIndexObjectStatsLatestIndex, bool hasCollectionLogHypertableOrPlainPg, bool hasJobHistory, bool hasAgentStatus, bool hasGenericWebhook, bool hasDeadlocksDatabaseName, bool hasQueryStoreReplicaRole, bool hasLongQueryCompletions, bool hasWebDashboardConfig, bool hasCustomViews, bool hasServerTags, bool hasConnectionRefireKnobs = false, bool hasAgCollectors = false, bool hasAgAlertKnobs = false, bool hasAgLatencyColumns = false, bool hasAgDisconnectRefire = false, bool hasPayloadDimensions = false, bool hasDimFloorIndexes = false, bool hasBlockingWaitThreshold = false, bool hasQueryStoreIntervalIdentity = false, bool hasPagerDutyWebhook = false, bool hasPagerDutyProxy = false, bool hasCollectorState = false, bool hasPlanCorrection = false, bool hasPvsStats = false, bool hasPvsPressureKnobs = false, bool hasDatabaseStateAlert = false, bool hasServerTagColour = false, bool hasQueryStatsHostObject = false, bool hasFindingDrillDown = false, bool hasStoreMetrics = false, bool hasPlanDimGzip = false, bool hasSelfAlertKnobs = false, bool hasJobMetricsColumns = false, bool hasJobCadenceKnob = false, bool hasBackfillSwitch = false, bool hasCollectorMemoryKnobs = false, bool hasDatabaseStateEdgeMemory = false, bool hasIncidentOccurrences = false, bool hasPlanXmlCompressionKnob = false, bool hasMonitoredServerEngine = false, bool hasPgBlockingEdges = false, bool hasQueryStorePlanMap = false, bool hasPgStatementText = false, bool hasQueryStoreText = false, bool hasPlanContentRetentionKnob = false, bool hasQueryStoreHealth = false) + internal static int MapProbedSchemaVersion(bool hasConfigControlPlane, bool hasAlertDeliveryOverride, bool hasAnalysisState, bool hasAlertTuningKnobs, bool hasDefaultTraceEvents, bool hasIndexObjectStatsLatestIndex, bool hasCollectionLogHypertableOrPlainPg, bool hasJobHistory, bool hasAgentStatus, bool hasGenericWebhook, bool hasDeadlocksDatabaseName, bool hasQueryStoreReplicaRole, bool hasLongQueryCompletions, bool hasWebDashboardConfig, bool hasCustomViews, bool hasServerTags, bool hasConnectionRefireKnobs = false, bool hasAgCollectors = false, bool hasAgAlertKnobs = false, bool hasAgLatencyColumns = false, bool hasAgDisconnectRefire = false, bool hasPayloadDimensions = false, bool hasDimFloorIndexes = false, bool hasBlockingWaitThreshold = false, bool hasQueryStoreIntervalIdentity = false, bool hasPagerDutyWebhook = false, bool hasPagerDutyProxy = false, bool hasCollectorState = false, bool hasPlanCorrection = false, bool hasPvsStats = false, bool hasPvsPressureKnobs = false, bool hasDatabaseStateAlert = false, bool hasServerTagColour = false, bool hasQueryStatsHostObject = false, bool hasFindingDrillDown = false, bool hasStoreMetrics = false, bool hasPlanDimGzip = false, bool hasSelfAlertKnobs = false, bool hasJobMetricsColumns = false, bool hasJobCadenceKnob = false, bool hasBackfillSwitch = false, bool hasCollectorMemoryKnobs = false, bool hasDatabaseStateEdgeMemory = false, bool hasIncidentOccurrences = false, bool hasPlanXmlCompressionKnob = false, bool hasMonitoredServerEngine = false, bool hasPgBlockingEdges = false, bool hasQueryStorePlanMap = false, bool hasPgStatementText = false, bool hasQueryStoreText = false, bool hasPlanContentRetentionKnob = false, bool hasQueryStoreHealth = false, bool hasQueryStoreTextHash = false, bool hasComposeTimeoutKnob = false, bool hasFileGrowthAlert = false) { /* V71 (the PostgreSQL blocking-edges rung): a table-existence sentinel and now the newest-first arm. A collector table would ordinarily get no arm at all — see the V63-V69 note below — but the TOP @@ -614,6 +617,32 @@ from it. */ rather than falling through. The table is named only in the probe line, not in this prose, per the V71 finding: the coverage ratchet strips information_schema lines but cannot strip a comment, so a prose mention would exempt it. */ + /* #2312: newest first. The arm below STAYS — a store migrated to exactly 76 must map to 76 + rather than falling through. The column is named only in the probe line, not in this prose, + per the V71 finding: the coverage ratchet strips information_schema lines but cannot strip + a comment, so a prose mention would exempt it. */ + /* #2357: newest first. The arm below STAYS — a store migrated to exactly 77 must map to 77 rather + than falling through. The column is named only in the probe line, not in this prose, per the V71 + finding: the coverage ratchet strips information_schema lines but cannot strip a comment, so a + prose mention would exempt it. */ + /* #2349: newest first. The arm below STAYS — a store migrated to exactly 78 must map to 78 rather + than falling through. The column is named only in the probe line, not in this prose, per the V71 + finding: the coverage ratchet strips information_schema lines but cannot strip a comment. */ + if (hasFileGrowthAlert) + { + return 79; + } + + if (hasComposeTimeoutKnob) + { + return 78; + } + + if (hasQueryStoreTextHash) + { + return 77; + } + if (hasQueryStoreHealth) { return 76; diff --git a/Darling/PerformanceMonitor.Darling.Viewer/packages.lock.json b/Darling/PerformanceMonitor.Darling.Viewer/packages.lock.json index 88ae7f5d7..729d8fc7f 100644 --- a/Darling/PerformanceMonitor.Darling.Viewer/packages.lock.json +++ b/Darling/PerformanceMonitor.Darling.Viewer/packages.lock.json @@ -108,8 +108,8 @@ }, "Microsoft.Extensions.DependencyInjection.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "z/2xXlFw2aLGjHyEm6E0tQ+In6VfzQzTrtArbQ2c0TQE16ZbyDCMGPvaUT9I0s8rgy9sRWlU2P9waW37qV04qA==" + "resolved": "10.0.11", + "contentHash": "/a1aJz4m7ylhEDf25ugQChLQoN5XwoGjWw/BoR/ZWWKsO1v4DdJElS1uyngahz4B/eOzjFk1KNTkarRLE5wsIg==" }, "Microsoft.Extensions.Diagnostics.Abstractions": { "type": "Transitive", @@ -208,8 +208,8 @@ }, "ModelContextProtocol.Core": { "type": "Transitive", - "resolved": "2.1.0", - "contentHash": "cU/urrhRxE4/iSyBIJI7QOaFqSP1FOEnwEHsct9n6t6/XluCAFD9iqnrPkBAsEYr+f/G4tVQ21U+6wN/6fQvOg==", + "resolved": "2.2.0", + "contentHash": "FeBfXU6T8k+jw4afg4sfxdEX2rL/e5oKOk9ROOGztu9k47+7Bz08sdaToYt2XvMY1opNbwxYQOFMj6wH9TInhA==", "dependencies": { "Microsoft.Extensions.AI.Abstractions": "10.8.3", "Microsoft.Extensions.Logging.Abstractions": "10.0.10" @@ -399,19 +399,21 @@ "type": "Project", "dependencies": { "CredentialManagement": "[1.0.2, )", - "Microsoft.Extensions.Logging.Abstractions": "[10.0.10, )", - "ModelContextProtocol": "[2.1.0, )" + "Microsoft.Extensions.Logging.Abstractions": "[10.0.11, )", + "ModelContextProtocol": "[2.2.0, )" } }, "performancemonitor.darling.analysis": { "type": "Project", "dependencies": { "Microsoft.Data.SqlClient": "[7.0.2, )", + "ModelContextProtocol": "[2.2.0, )", "PerformanceMonitor.Analysis": "[1.0.0, )", "PerformanceMonitor.Collectors": "[1.0.0, )", "PerformanceMonitor.Darling.Storage": "[1.0.0, )", "PerformanceMonitor.Notifications": "[1.0.0, )", - "PerformanceMonitor.PlanAnalysis": "[1.0.0, )" + "PerformanceMonitor.PlanAnalysis": "[1.0.0, )", + "System.Security.Cryptography.ProtectedData": "[10.0.11, )" } }, "performancemonitor.darling.storage": { @@ -424,7 +426,7 @@ "performancemonitor.notifications": { "type": "Project", "dependencies": { - "Microsoft.Extensions.Logging.Abstractions": "[10.0.10, )", + "Microsoft.Extensions.Logging.Abstractions": "[10.0.11, )", "PerformanceMonitor.Analysis": "[1.0.0, )" } }, @@ -432,7 +434,8 @@ "type": "Project", "dependencies": { "Microsoft.Data.SqlClient": "[7.0.2, )", - "PerformanceMonitor.Common": "[1.0.0, )" + "PerformanceMonitor.Common": "[1.0.0, )", + "System.Security.Cryptography.ProtectedData": "[10.0.11, )" } }, "performancemonitor.ui": { @@ -467,22 +470,22 @@ }, "Microsoft.Extensions.Logging.Abstractions": { "type": "CentralTransitive", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "zkFxGYUvdxAvIKTyXHrmW+Sux53D4SezD9dMyZ6hrwwzPQJNuwCRy1f5W7AvYTqacEGhWF2XderRQG1OvbV8og==", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "Ljd0Uxoq5XpScD2Bg0nM/r3mwx7Ao5Uq24eo2ARxbGvqJ7Zht6rt2cJtwVRH4Cv+1ZVMdXz6TB43KbpmsxRrvQ==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11" } }, "ModelContextProtocol": { "type": "CentralTransitive", - "requested": "[2.1.0, )", - "resolved": "2.1.0", - "contentHash": "Oa4rU7EL9C2qyFjQj1dx+ysGMzfWDRpM8RRaUMmLGs5vPvfJ9xyz4ZtyF4ychY+Nx1b/auGCqIQLqSz/IpPkKA==", + "requested": "[2.2.0, )", + "resolved": "2.2.0", + "contentHash": "4Pb9u02Nwsp0poueDsqNdyGRojFxOYpljB7zDBsq+aHL+Afou3OgxlBc3GWFVnsRMRJrUtWqDh3s6k2JgPzmrQ==", "dependencies": { "Microsoft.Extensions.Caching.Abstractions": "10.0.10", "Microsoft.Extensions.Hosting.Abstractions": "10.0.10", - "ModelContextProtocol.Core": "[2.1.0]" + "ModelContextProtocol.Core": "[2.2.0]" } }, "Npgsql": { @@ -493,6 +496,12 @@ "dependencies": { "Microsoft.Extensions.Logging.Abstractions": "10.0.0" } + }, + "System.Security.Cryptography.ProtectedData": { + "type": "CentralTransitive", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "PNoxCTPb+Tlux+GJyq4c89ddYdpioVSqfGx8pqOF6shCSKwUNNctXhQtRkCICjbJmGrJJsW8NY52kVYO/b8mlQ==" } } } diff --git a/Darling/README.md b/Darling/README.md index 77243e15c..97ad4e15e 100644 --- a/Darling/README.md +++ b/Darling/README.md @@ -169,7 +169,7 @@ Extract to a local, machine-scoped path — `C:\PerformanceMonitorDarling` is th The service runs as the unprivileged virtual account `NT SERVICE\PerformanceMonitor Darling`, never LocalSystem, because the bundled PostgreSQL refuses to run with administrative privileges. That account is not you, not SYSTEM, and not Administrators — and a user profile grants access to about those three and nobody else, so the service cannot read its own program files there. It installs cleanly and then fails: `initdb.exe` dies at `0xC0000135` (STATUS_DLL_NOT_FOUND) before it can report anything (#2185). A folder created under `C:\` inherits read + execute for `BUILTIN\Users` instead, which the virtual account is a member of, which is why the documented location works. Network paths fail for a related reason: a virtual account [reaches the network as the computer account](https://learn.microsoft.com/en-us/sql/database-engine/configure-windows/configure-windows-service-accounts-and-permissions#virtual-accounts) rather than as you, and a mapped drive letter belongs to your logon session, which a service does not share. -`install-darling.ps1` refuses a fresh install in any of these locations rather than leaving you a service that cannot start. To move an existing install, stop the service, move the folder, and re-run `install-darling.ps1` from the new location — it updates the service's binPath in place and leaves your `darling.json`, store data, and credentials alone. +`install-darling.ps1` refuses a fresh install in any of these locations rather than leaving you a service that cannot start. A service registered by hand instead — the manual `sc create` path below, which the installer never sees — gets the same diagnosis from the service itself: on start, ahead of reading `darling.json` and long before the store bootstrap, it logs one critical line naming the path, why its own account cannot read it, and where to move it, so the cause is above the failure rather than three messages downstream of it. To move an existing install, stop the service, move the folder, and re-run `install-darling.ps1` from the new location — it updates the service's binPath in place and leaves your `darling.json`, store data, and credentials alone. **Manual:** publish (or copy the build output) to a stable path, put `darling.json` next to the exe (or set `DARLING_CONFIG` as a machine environment variable), then register it: @@ -413,7 +413,7 @@ The embedded MCP server, over Streamable HTTP bound to `localhost` by default (s - **Fifteen core data-read tools** — the diagnostic reads an assistant needs to investigate a server, each a stored read of the collected data (never a live query against the monitored server): - *Resource metrics* — `get_cpu_utilization`, `get_wait_stats`, `get_wait_trend`, `get_wait_types` (the distinct observed wait types, to pick one for `get_wait_trend`), `get_memory_stats`, `get_memory_clerks`, `get_file_io_stats`, `get_tempdb_trend`, `get_perfmon_stats`. - *Query performance* — `get_top_queries_by_cpu`, `get_top_procedures_by_cpu`, `get_query_store_top` (these hand back the `query_hash` / `sql_handle` / `query_id` + `plan_id` keys the plan-analysis tools consume). - - *Discovery / health* — `list_servers` (with collection-freshness status), `get_collection_health`, `get_server_properties`. + - *Discovery / health* — `list_servers` (with collection-freshness status, and the [declared peer stores](#peers) when the fleet is split across several Darling boxes), `get_collection_health`, `get_server_properties`. These are the tools the analysis findings' `next_tools` recommendations point at, so a client following a finding's advice resolves them on this same server. Result shapes match Lite's (the store is Lite's collector schema); where Lite and the Dashboard's shapes diverge, Darling follows Lite — the shape its collector-mirror store can serve faithfully. - **Twenty diagnostic-depth data-read tools** — deeper reads for a blocking / deadlock / session / configuration / storage investigation, each a stored read: @@ -492,6 +492,47 @@ Once enabled, open `http://localhost:5153/` in a browser on the service host. Li **What you see.** The dashboard opens on a **Fleet Overview**: a card per enabled server with a status dot, six per-metric health bands (CPU, threads, memory, blocking, deadlocks, collectors), and its last collection time — all banded server-side, so the browser only renders (a server that has never reported shows an amber "Awaiting first collection", never a red offline). Above the cards a worst-first "Needs attention" list surfaces the servers to look at, or an all-healthy line when there is nothing to chase. Click a card to **drill into one server**: an overview, wait stats with a trend for the heaviest wait, active queries, a CPU chart, memory and file-I/O trends, and collection health — the same collected data the viewer shows, over inline charts. A fleet-wide **Alert History** page (with a server filter box) rounds out phase 1. It is a read-only view — no settings, no write paths, no live-server queries — and refreshes every 60 seconds (pausing while the tab is hidden). The frontend ships fully self-contained (no CDN, no fonts, no remote anything), so it works on an air-gapped host with no internet access. +### peers + +**Declared peer stores** — optional, and only relevant when the fleet is split across **several Darling boxes**, one store each (SQL Server primaries on one box, their readable replicas on another, PostgreSQL on a third). Each box's MCP server answers over **its own** store only, so a server monitored by a sibling resolves as not-found — which an agent cannot tell apart from *"nobody monitors this server."* Declaring the siblings fixes that at the three places an agent forms its picture of the fleet. + +**Disclosure only.** There is no address and no credential in this block, and nothing behind it: the service never contacts a peer, cannot read a peer's data, and cannot tell whether a peer is even running. A peer is a **name** plus a **sentence**, so an agent (or its human) can pick the right endpoint. Everything here is sent verbatim to every connected MCP client, so the service **refuses to start** if any peer text looks like a connection string or credential. + +| Key | Default | Notes | +|---|---|---| +| `thisStoreCovers` | `""` | One sentence naming what THIS store monitors — the anchor the peer list is relative to | +| `stores[].name` | — | **Required.** Whatever an operator would recognize (the box name, "the use1 store") | +| `stores[].covers` | `""` | A short sentence naming what that store monitors. Human prose — never parsed, only shown | +| `stores[].matches` | `[]` | Optional server-name **substrings** that store monitors, case-insensitive. The only machine-checked field | + +```jsonc +"peers": { + "thisStoreCovers": "the 42 us-east-1 SQL Server primaries", + "stores": [ + { + "name": "prod-pos-use2-monitor-01", + "covers": "the readable replicas of those same 42 primaries, in-region from us-east-2", + "matches": ["use2"] + }, + { "name": "prod-pos-pg-monitor-01", "covers": "the Aurora PostgreSQL clusters", "matches": ["-aurora-"] } + ] +} +``` + +What it changes, with peers declared: + +- **The MCP instructions** gain a Fleet Coverage section, high enough that an agent reads which store it is talking to before it reads the tool census. +- **`list_servers`** gains `this_store_covers`, a `peer_fleets` array, and a `peer_note`. Both are always present: an *empty* `peer_fleets` has two very different meanings (this really is the only store, or nobody declared the siblings) and the service cannot tell them apart, so `peer_note` says exactly that rather than letting an empty array read as "this is the whole fleet." An **empty registry** answers in prose rather than JSON, and carries the peer list too — a store with nothing registered is a fresh or just-restarted box, which is the worst place to drop the disclosure. +- **The server-resolution miss** appends the disclosure to the existing "Could not resolve server. Available servers:" listing, naming the peer whose declared coverage matches — so *not monitored here* stops looking like *not monitored anywhere*. + +`matches` is deliberately plain substrings, no globbing and no regex: it exists to answer "which region/role prefix is this name?", and a pattern language would be a config surface with its own failure modes. Blank entries are dropped — an empty substring matches every name, which would make one peer claim the whole fleet. A peer with no `matches` is still disclosed everywhere; it just cannot be singled out on a miss, and the miss message says so instead of implying the server is unmonitored. + +A **file-only** block (not seeded into the control plane): it describes the deployment topology of *this* box, which must not be editable from a peer's Viewer. An edit takes effect on the next service restart. There is deliberately **no cross-store connectivity** here — actual federated reads (auth between stores, latency, partial failures) are a much larger surface, and may never be worth building if disclosure alone makes the split legible. + +**Declaring nothing changes nothing, with one exception worth knowing about on upgrade.** The instructions, the resolution-miss message, and `list_servers`' empty-registry sentence are byte-for-byte what they were. But `list_servers`' JSON envelope carries `this_store_covers`, `peer_fleets` and `peer_note` on *every* response, declared or not — so a script comparing that tool's exact shape sees three new keys even if you never write a `peers` block. That is deliberate: an empty `peer_fleets` means *either* "this is the only store" *or* "nobody declared the siblings", and a note that only appeared when peers were declared would say nothing in precisely the case that produces the wrong conclusion. + +**A `peers` block that fails validation is refused whole, and nothing is disclosed** — not the valid subset. An unfinished block that asserts coverage which may be wrong is worse than no block, and the service logs each problem at Critical. The check runs inside the publish rather than only in config validation, because the MCP host loads its own config and deliberately never validates it (its fail-closed checks are host-local), so validation alone would leave the one path that actually broadcasts uncovered. + ### No Schedule Knobs, by Design There are deliberately **no collection-schedule or retention settings** in `darling.json`. The service consumes the shared per-collector defaults (`CollectorScheduleDefaults`) — the same cadences and retention horizons a fresh Lite install uses, identity-pinned by tests so the two editions cannot drift. If a schedule knob is ever genuinely needed, it will be added then, not speculatively. @@ -621,6 +662,7 @@ The notable rungs are below. For the **complete** current schema, read `Darling/ | **V72** — Query Store plan map | `collect.query_store_plan_map` — `(server_id, database_name, plan_id)` → digest, so Query Store facts can reference plan XML they no longer carry once that content moves into the shared `query_plan_dim`. Plan XML was stored INLINE on `query_store_stats` at roughly 5x redundancy. Not a hypertable: one row per distinct plan per database, so it is dimension-shaped and pruned on `last_seen` rather than by `drop_chunks`. Its `last_seen` is load-bearing — the dimension GC sweeps on timestamps rather than counting references, so ending the re-shipping also ends the liveness signal that used to keep those dim rows alive | | **V73** — PostgreSQL statement text | `collect.pg_statement_text` — `(server_id, queryid)` → statement text, refreshed hourly, so `get_pg_top_queries` returns something readable (#2219). `pg_statement_stats` stores no text because `showtext` is a real per-collection cost and normalized text is highly repetitive; but `queryid` is NOT stable across a major version upgrade, so without this the stored history joins to nothing after one — a list of integers that used to be your slowest queries, unrecoverable because the live view no longer holds the old ids. Text is INLINE rather than a `query_text_dim` digest: the dimension route needs the GC liveness interlock whose failure mode is silently missing text, and inline cannot dangle. Not a hypertable and not a collector table, exactly like V72 — a bespoke upsert path, pruned on `last_seen` with a margin that makes text OUTLIVE the statistics referencing it | | **V76** — Query Store health | `collect.query_store_health` + its index + the `v_query_store_health` passthrough view — the #2319 per-database `sys.database_query_store_options` collector's store table: actual vs desired state (the cap-hit READ_ONLY transition and its readonly_reason), current vs max storage, cleanup thresholds, and the runtime-stats interval length. A fresh store gets the table from V1's generated schema; V76 is what an already-existing store gets | +| **V77** — Activity-driven plan fetch | Three strokes behind #2312's reshape of the Query Store plan/text fetch: `query_store_plan_map.digest` goes **nullable** (a plan whose XML the engine cannot persist gets a NULL-digest map row — the content-less marker that stops the probe re-selecting it forever), `query_store_text` gains `query_hash` (the Query Store reset detector: an id whose stored hash differs from the live one names a DIFFERENT statement now and its text refetches within one cycle), and the retired `planwm:`/`textwm:` watermark state rows are deleted wholesale. The fetch itself no longer walks the plan catalog by watermark — the cycle's collected rows name their plans, the store answers which are missing, and only those are fetched | All timestamps in the store are **naive-UTC** `timestamp` columns — the product-wide cross-store contract (Lite's DuckDB does the same). diff --git a/Darling/tools/install-darling.ps1 b/Darling/tools/install-darling.ps1 index 69a155d59..805b35dc9 100644 --- a/Darling/tools/install-darling.ps1 +++ b/Darling/tools/install-darling.ps1 @@ -106,15 +106,33 @@ function Get-ProfilesDirectory { return (Join-Path $env:SystemDrive 'Users') } +# Rewrites an extended-length path to its ordinary spelling (#2348), leaving anything else alone: +# \\?\UNC\server\share -> \\server\share, and \\?\C:\dir -> C:\dir. The C# twin is +# DarlingInstallLocation.StripExtendedLengthPrefix and the two must stay identical. +# +# The UNC form is tested FIRST because it is the longer, more specific prefix - checking \\?\ first would +# strip four characters off a share and leave the nonsense UNC\server\share, which is neither a share nor a +# local path. 'UNC' is matched case-insensitively because Windows accepts \\?\unc\ too, and a miss there +# would silently re-open the hole this closes. +function Convert-ExtendedLengthPath([string]$path) { + if ([string]::IsNullOrEmpty($path)) { return $path } + if ($path.StartsWith('\\?\UNC\', [StringComparison]::OrdinalIgnoreCase)) { return '\\' + $path.Substring(8) } + if ($path.StartsWith('\\?\', [StringComparison]::Ordinal)) { return $path.Substring(4) } + return $path +} + # 'UNC', 'mapped drive', or $null. Named separately from the profile case because the reason differs: # a virtual service account reaches the network as the COMPUTER account rather than as the operator who # typed the path, and a mapped drive letter belongs to one logon session, which a service never shares. +# +# Callers pass a path that Convert-ExtendedLengthPath has already normalized, so there is no \\?\ carve-out +# here any more: an extended-length LOCAL root has become C:\... and cannot reach the UNC test, while an +# extended-length SHARE has become \\server\share and correctly does (#2348 - the old wholesale exclusion +# waved real shares through, because skipping the check is not the same as passing it). function Get-NetworkPathKind([string]$path) { if ([string]::IsNullOrWhiteSpace($path)) { return $null } - # \\?\ is the long-path prefix on a LOCAL path, not a server name - excluded so an extended-length - # local path is not mistaken for a share. - if ($path.StartsWith('\\', [StringComparison]::Ordinal) -and -not $path.StartsWith('\\?\', [StringComparison]::Ordinal)) { + if ($path.StartsWith('\\', [StringComparison]::Ordinal)) { return 'UNC' } @@ -188,10 +206,15 @@ if (-not (Test-Path $serviceExe)) { # This runs BEFORE the pre-flight, the Event Log source, and service creation, so a doomed location costs # nothing and leaves nothing behind. $existing = Get-Service -Name $serviceName -ErrorAction SilentlyContinue -$networkKind = Get-NetworkPathKind $root +# Classify the NORMALIZED spelling, but keep installing to $root exactly as given (#2348). The \\?\ prefix +# instructs the path parser and is not part of where the install lives, so stripping it for the decision +# changes which rules see the path and nothing about where files land. +$classifyRoot = Convert-ExtendedLengthPath $root + +$networkKind = Get-NetworkPathKind $classifyRoot # $env:USERPROFILE as well as the machine's profile root: a profile redirected outside ProfilesDirectory # is still a profile, and it is the profile whose owner is most likely to be running this script. -$underProfile = (Test-PathIsAtOrUnder $root (Get-ProfilesDirectory)) -or (Test-PathIsAtOrUnder $root $env:USERPROFILE) +$underProfile = (Test-PathIsAtOrUnder $classifyRoot (Get-ProfilesDirectory)) -or (Test-PathIsAtOrUnder $classifyRoot $env:USERPROFILE) if ($underProfile -or $networkKind) { if ($underProfile) { diff --git a/Directory.Packages.props b/Directory.Packages.props index cd477d4d3..e6fc27315 100644 --- a/Directory.Packages.props +++ b/Directory.Packages.props @@ -1,34 +1,32 @@ - - - - true - - - - - - - - - - - - - - - - - - - - - - - - - - + + + + true + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/Lite.Tests/DatabaseStateExpectedStoreTests.cs b/Lite.Tests/DatabaseStateExpectedStoreTests.cs index fb0214c3d..beee2f459 100644 --- a/Lite.Tests/DatabaseStateExpectedStoreTests.cs +++ b/Lite.Tests/DatabaseStateExpectedStoreTests.cs @@ -120,6 +120,40 @@ private static async Task> SweepUntilAsync( return result; } + /// + /// Establishes the auto-baseline and does not return until it actually LANDED (#2374). + /// + /// Same root cause as and the same remedy, one step earlier: the seed + /// rides the best-effort maintenance block, so a single bare sweep can complete having written no + /// expectation at all. A test that then acts as though it has a baseline is asserting on a precondition it + /// never established — and it fails far from the cause, because the deviation read still succeeds and simply + /// returns nothing to deviate FROM. That is the whole of #2374: Assert.Single on an empty collection, + /// three lines after a sweep that quietly did nothing. + /// + /// Settling on the recorded STATE rather than on row count is load-bearing. + /// LEFT JOINs the newest snapshot, so every + /// current database comes back whether or not anything was ever seeded for it — an unseeded database is a + /// present row with an EMPTY expected state, and "did I get rows?" is therefore always true and never the + /// question. It re-runs the seed on its own ("reuse the seed/prune side-effect"), which is what makes + /// re-reading it a retry rather than just a re-check. + /// + /// It cannot mask a regression, for 's reason: a seed that is genuinely + /// broken never records the state, every cycle runs, and the caller's own assertion fails on the same empty + /// result it sees today. + /// + private static async Task BaselineAsync( + LocalDataService service, string database, string expectedState, int cycles = 5) + { + for (var attempt = 0; attempt < cycles; attempt++) + { + var rows = await service.GetDatabaseStateExpectationsAsync(ServerId); + if (rows.Any(r => r.DatabaseName == database && r.ExpectedState == expectedState)) + { + return; + } + } + } + /// /// Writes an expectation row directly, bypassing the service's writers — the only way to reproduce a /// row the OLD seed wrote (#2189), since no current code path can produce one any more. @@ -161,7 +195,7 @@ public async Task StateChange_DoesNotFireOnASingleSample_ThenFiresOnTwoConsecuti { await SeedSnapshotAsync(T0, ("App", "ONLINE", false)); var service = new LocalDataService(_duckDb); - await service.GetDatabaseStateDeviationsAsync(ServerId); // baseline App = ONLINE + await BaselineAsync(service, "App", "ONLINE"); // One OFFLINE sample (a transient — e.g. mid-restart) must NOT fire. await SeedSnapshotAsync(T0.AddMinutes(1), ("App", "OFFLINE", false)); @@ -180,7 +214,7 @@ public async Task StateChange_DoesNotFireOnASingleSample_ThenFiresOnTwoConsecuti private async Task DriveAppToStableStateAsync(LocalDataService service, string state) { await SeedSnapshotAsync(T0, ("App", "ONLINE", false)); - await service.GetDatabaseStateDeviationsAsync(ServerId); // baseline ONLINE + await BaselineAsync(service, "App", "ONLINE"); await SeedSnapshotAsync(T0.AddMinutes(1), ("App", state, false)); await SeedSnapshotAsync(T0.AddMinutes(2), ("App", state, false)); } @@ -317,7 +351,12 @@ database still baselines on its first observation. */ await SeedSnapshotAsync(T0, ("Restoring", "RESTORING", false), ("Recovering", "RECOVERING", false), ("Healthy", "ONLINE", false)); var service = new LocalDataService(_duckDb); - await service.GetDatabaseStateDeviationsAsync(ServerId); + + /* Settle on Healthy, and the two pending assertions below become meaningful. A skipped maintenance + block seeds NOTHING, which reads as all three being pending — so on a bare sweep the two "" + assertions pass for entirely the wrong reason and only Healthy reports the problem. Waiting for the + one database that MUST be learned proves the seed ran before asking what it declined to learn. */ + await BaselineAsync(service, "Healthy", "ONLINE"); var rows = await service.GetDatabaseStateExpectationsAsync(ServerId); Assert.Equal("", rows.Single(r => r.DatabaseName == "Restoring").ExpectedState); @@ -401,7 +440,7 @@ public async Task AutoBaselinedOffline_IsNeverHealed_SoParkingItAgainStaysQuiet( is silence, and the baseline is the same one it started with. */ await SeedSnapshotAsync(T0, ("Parked", "OFFLINE", false)); var service = new LocalDataService(_duckDb); - await service.GetDatabaseStateDeviationsAsync(ServerId); // learns OFFLINE + await BaselineAsync(service, "Parked", "OFFLINE"); await SeedSnapshotAsync(T0.AddMinutes(1), ("Parked", "ONLINE", false)); await SeedSnapshotAsync(T0.AddMinutes(2), ("Parked", "ONLINE", false)); @@ -427,7 +466,7 @@ up truly ONLINE (is_in_standby now 0) has been RECOVERED - log shipping is broke when the operator re-established standby, announcing the repair instead of the break. */ await SeedSnapshotAsync(T0, ("LogShip", "ONLINE", true)); var service = new LocalDataService(_duckDb); - await service.GetDatabaseStateDeviationsAsync(ServerId); // learns STANDBY + await BaselineAsync(service, "LogShip", "STANDBY"); await SeedSnapshotAsync(T0.AddMinutes(1), ("LogShip", "ONLINE", false)); await SeedSnapshotAsync(T0.AddMinutes(2), ("LogShip", "ONLINE", false)); @@ -499,7 +538,7 @@ so a heal written against the RAW column would re-baseline every log-shipping se database family #1986 works hardest to keep quiet. Matching the EFFECTIVE state is what prevents it. */ await SeedSnapshotAsync(T0, ("LogShip", "ONLINE", true)); var service = new LocalDataService(_duckDb); - await service.GetDatabaseStateDeviationsAsync(ServerId); // baselines STANDBY + await BaselineAsync(service, "LogShip", "STANDBY"); await SeedSnapshotAsync(T0.AddMinutes(1), ("LogShip", "ONLINE", true)); await SeedSnapshotAsync(T0.AddMinutes(2), ("LogShip", "RESTORING", true)); diff --git a/Lite.Tests/FileGrowthAlertTests.cs b/Lite.Tests/FileGrowthAlertTests.cs new file mode 100644 index 000000000..888f9a176 --- /dev/null +++ b/Lite.Tests/FileGrowthAlertTests.cs @@ -0,0 +1,222 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Linq; +using PerformanceMonitor.Alerting; +using Xunit; + +namespace Lite.Tests; + +/// +/// #2349: the database file-growth alert — the gap between the two alerts that already look at disk. +/// +/// tempdb Space fires on reserved ÷ (reserved + unallocated). Autogrowth adds unallocated extents, +/// so the denominator grows with the file and the percentage FALLS as tempdb balloons — it answers "is tempdb +/// internally full right now", which is a real question and structurally not this one. Volume Free Space +/// fires on the consequence, by which point a restart is overdue, and cannot attribute the space to one file. +/// Between them sits a file that has grown large but has not yet filled its disk. +/// +public class FileGrowthAlertTests +{ + private const string Server = "SQLPROD01"; + + private static DatabaseFileGrowthInfo File( + string db = "tempdb", string name = "tempdev", double sizeMb = 100_000, double growthMb = 0, + double windowMinutes = 60, double volumeTotalMb = 500_000, double volumeFreeMb = 200_000) => + new() + { + DatabaseName = db, + FileName = name, + PhysicalName = $@"D:\data\{name}.mdf", + FileTypeDesc = "ROWS", + TotalSizeMb = sizeMb, + GrowthMb = growthMb, + GrowthWindowMinutes = windowMinutes, + VolumeMountPoint = @"D:\", + VolumeTotalMb = volumeTotalMb, + VolumeFreeMb = volumeFreeMb, + }; + + /// + /// The RISE gate is the point of the alert: an event, not a level. #2157's reasoning applies exactly — a + /// level alone re-pages every cooldown about a size that has been true since Tuesday, which trains people + /// to mute it, while "80 GB in the last hour" is the thing worth waking up for. + /// + [Fact] + public void TheRiseGate_FiresOnGrowth_EvenWhenTheVolumeIsRoomy() + { + var files = new List { File(sizeMb: 90_000, growthMb: 40_000, volumeTotalMb: 4_000_000) }; + + /* 2% of a 4 TB volume — the level gate cannot see this, and it is exactly the case the issue is about. */ + Assert.True(files[0].VolumePercent < 5); + + var breached = AlertContextBuilders.GetBreachedFiles(files, riseMb: 10_240, volumePercent: 60); + + Assert.Single(breached); + } + + /// + /// The LEVEL gate catches the file that is already large and stopped moving — the state the rise gate goes + /// quiet about by design. Self-scaling, which is what makes ONE global setting usable across a fleet: the + /// same 60% catches a 128 GB file on a small volume and a 1.6 TB file on a large one. + /// + [Fact] + public void TheLevelGate_FiresOnAFileThatIsLargeButNoLongerGrowing() + { + var files = new List { File(sizeMb: 400_000, growthMb: 0, volumeTotalMb: 500_000) }; + + var breached = AlertContextBuilders.GetBreachedFiles(files, riseMb: 10_240, volumePercent: 60); + + Assert.Single(breached); + Assert.Equal(80, breached[0].VolumePercent); + } + + /// A file breaching neither gate is silent, which is most files most of the time. + [Fact] + public void AQuietFile_DoesNotFire() + { + var files = new List { File(sizeMb: 50_000, growthMb: 100, volumeTotalMb: 500_000) }; + + Assert.Empty(AlertContextBuilders.GetBreachedFiles(files, riseMb: 10_240, volumePercent: 60)); + } + + /// + /// Zero disables ONE gate rather than being nonsense, so an operator can run rise-only or level-only + /// without a second switch — and disabling one must not silently disable the other. + /// + [Fact] + public void ZeroDisablesOneGate_NotBoth() + { + var grew = new List { File(sizeMb: 90_000, growthMb: 40_000, volumeTotalMb: 4_000_000) }; + var large = new List { File(sizeMb: 400_000, growthMb: 0, volumeTotalMb: 500_000) }; + + /* level off: the rise still fires, the large-but-static file does not */ + Assert.Single(AlertContextBuilders.GetBreachedFiles(grew, riseMb: 10_240, volumePercent: 0)); + Assert.Empty(AlertContextBuilders.GetBreachedFiles(large, riseMb: 10_240, volumePercent: 0)); + + /* rise off: the level still fires, the growing-but-small-share file does not */ + Assert.Empty(AlertContextBuilders.GetBreachedFiles(grew, riseMb: 0, volumePercent: 60)); + Assert.Single(AlertContextBuilders.GetBreachedFiles(large, riseMb: 0, volumePercent: 60)); + } + + /// + /// A file on a volume with no size reported (Azure SQL DB has no volume stats) must not divide by zero and + /// must not fire the level gate on a fabricated 0%. + /// + [Fact] + public void AFileWithNoVolumeStats_IsNotLevelGated() + { + var files = new List { File(sizeMb: 400_000, growthMb: 0, volumeTotalMb: 0) }; + + Assert.Equal(0, files[0].VolumePercent); + Assert.Empty(AlertContextBuilders.GetBreachedFiles(files, riseMb: 10_240, volumePercent: 60)); + } + + /// + /// Ordered by share of volume, because that is how close this is to becoming a Volume Free Space + /// page. A 40 GB rise on a 4 TB volume is less urgent than a 10 GB file that is now 80% of a small one. + /// + [Fact] + public void BreachedFiles_AreOrderedByHowCloseTheyAreToFillingTheirVolume() + { + var files = new List + { + File(db: "big", name: "f1", sizeMb: 90_000, growthMb: 80_000, volumeTotalMb: 4_000_000), + File(db: "tight", name: "f2", sizeMb: 400_000, growthMb: 20_000, volumeTotalMb: 500_000), + }; + + var breached = AlertContextBuilders.GetBreachedFiles(files, riseMb: 10_240, volumePercent: 60); + + Assert.Equal(2, breached.Count); + Assert.Equal("tight", breached[0].DatabaseName); + } + + /// + /// Fingerprinted per FILE, not per database. Eight tempdb data files growing together are eight files and + /// one problem, but a log file running away while its data files sit still is a different incident — and + /// collapsing on database name would merge the two and pool their totals. + /// + [Fact] + public void IncidentsAreFingerprintedPerFile_AndCarryTheDatabase() + { + var incidents = AlertContextBuilders.FileGrowthIncidents(Server, new List + { + File(db: "tempdb", name: "tempdev"), + File(db: "tempdb", name: "templog"), + }); + + Assert.Equal(2, incidents.Count); + Assert.Equal(2, incidents.Select(i => i.DedupKey).Distinct(StringComparer.Ordinal).Count()); + Assert.All(incidents, i => Assert.Equal("tempdb", i.Database)); + } + + /// + /// #2362's rule: the observation list is UNCAPPED while the card renders a subset. Observing only what is + /// displayed would reset the total of any file that fell out of the top N. + /// + [Fact] + public void TheIncidentListIsUncapped() + { + var files = Enumerable.Range(0, 12).Select(i => File(db: "db", name: $"f{i}")).ToList(); + + Assert.Equal(12, AlertContextBuilders.FileGrowthIncidents(Server, files).Count); + } + + /// + /// The card names what an operator needs to act without opening the Viewer, including the percent-autogrowth + /// misconfiguration — each growth bigger than the last is exactly how a file gets away from someone, and the + /// WS3 advisory knows the pattern but does not alert on it. + /// + [Fact] + public void TheCardNamesTheFileTheVolumeAndAPercentAutogrowth() + { + var f = File(sizeMb: 400_000, growthMb: 40_000, volumeTotalMb: 500_000); + f.IsPercentGrowth = true; + f.GrowthPct = 10; + + var context = AlertContextBuilders.BuildFileGrowthContext(Server, new List { f }); + + Assert.NotNull(context); + var fields = context!.Details.SelectMany(d => d.Fields).ToList(); + + Assert.Contains(fields, x => x.Item1 == "Database" && x.Item2 == "tempdb"); + Assert.Contains(fields, x => x.Item1 == "Physical Name"); + Assert.Contains(fields, x => x.Item1 == "Volume Free"); + Assert.Contains(fields, x => x.Item1 == "Autogrowth" && x.Item2.Contains("percent growth", StringComparison.Ordinal)); + } + + /// + /// A window holding one sample reports zero growth, not a rise of the whole file — the difference between + /// "no rise observed" and "this file appeared from nothing", which is what a freshly-collecting server + /// would otherwise look like. + /// + [Fact] + public void ASingleSampleWindow_ReportsNoRise() + { + var f = File(sizeMb: 400_000, growthMb: 0, windowMinutes: 0); + + Assert.Equal(0, f.GrowthMb); + Assert.Equal(0, f.GrowthMbPerHour); + Assert.Empty(AlertContextBuilders.GetBreachedFiles(new List { f }, riseMb: 10_240, volumePercent: 0)); + } + + /// The rate is derived from the MEASURED window, so a collection gap cannot make a slow rise + /// look fast. + [Theory] + [InlineData(6000, 60, 6000)] + [InlineData(6000, 30, 12000)] + [InlineData(6000, 120, 3000)] + public void TheRateUsesTheMeasuredWindow(double growthMb, double windowMinutes, double expectedPerHour) + { + var f = File(growthMb: growthMb, windowMinutes: windowMinutes); + + Assert.Equal(expectedPerHour, f.GrowthMbPerHour, precision: 3); + } +} diff --git a/Lite.Tests/IncidentObservationCoverageTests.cs b/Lite.Tests/IncidentObservationCoverageTests.cs new file mode 100644 index 000000000..9404d034b --- /dev/null +++ b/Lite.Tests/IncidentObservationCoverageTests.cs @@ -0,0 +1,146 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Linq; +using PerformanceMonitor.Alerting; +using Xunit; + +namespace Lite.Tests; + +/// +/// #2362: every fingerprinted alert accumulates a monotonic total, not just blocking and deadlocks. +/// +/// Why the observation list must be UNCAPPED. Each context builder renders a capped subset — 3 for +/// long-running queries and anomalous jobs, 5 for volumes, PVS and failed jobs — because a card with fifty +/// entries helps nobody. The render cap is a display budget. A fingerprint outside it still has a live incident, +/// and observing only the displayed subset would reset the total of anything that fell out of the top N the +/// moment it surfaced again: a subtler version of the undercount #2216 exists to fix, reintroduced by the fix +/// for it. BlockingIncidents already carries that reasoning; these five now share it. +/// +public class IncidentObservationCoverageTests +{ + private const string Server = "SQLPROD01"; + + /* Render caps, from each context builder's own GetRange(0, Math.Min(N, ...)). The point of every test + below is that the OBSERVATION list is larger than these. */ + private const int LongRunningRenderCap = 3; + private const int VolumeRenderCap = 5; + private const int PvsRenderCap = 5; + private const int AnomalousJobRenderCap = 3; + private const int FailedJobRenderCap = 5; + + /// + /// The whole point, stated once per alert: hand each builder MORE items than its card renders and get an + /// incident for every one of them. A regression that reintroduced the cap would fail here by count. + /// + [Fact] + public void EveryBuilder_ObservesBeyondItsRenderCap() + { + var lrq = AlertContextBuilders.LongRunningQueryIncidents(Server, + Enumerable.Range(0, LongRunningRenderCap + 4) + .Select(i => new LongRunningQueryInfo { QueryHash = $"0xHASH{i}", DatabaseName = "OrdersDB" }).ToList()); + Assert.Equal(LongRunningRenderCap + 4, lrq.Count); + + var vol = AlertContextBuilders.VolumeFreeSpaceIncidents(Server, + Enumerable.Range(0, VolumeRenderCap + 4) + .Select(i => new VolumeFreeSpaceInfo { MountPoint = $"{(char)('D' + i)}:\\" }).ToList()); + Assert.Equal(VolumeRenderCap + 4, vol.Count); + + var pvs = AlertContextBuilders.PvsPressureIncidents(Server, + Enumerable.Range(0, PvsRenderCap + 4) + .Select(i => new PvsPressureInfo { DatabaseName = $"DB{i}" }).ToList()); + Assert.Equal(PvsRenderCap + 4, pvs.Count); + + var anomalous = AlertContextBuilders.AnomalousJobIncidents(Server, + Enumerable.Range(0, AnomalousJobRenderCap + 4) + .Select(i => new AnomalousJobInfo { JobName = $"Job{i}" }).ToList()); + Assert.Equal(AnomalousJobRenderCap + 4, anomalous.Count); + + var failed = AlertContextBuilders.FailedJobIncidents(Server, + Enumerable.Range(0, FailedJobRenderCap + 4) + .Select(i => new FailedJobInfo { JobName = $"Job{i}" }).ToList()); + Assert.Equal(FailedJobRenderCap + 4, failed.Count); + } + + /// + /// Distinct fingerprints, because the accumulator keys on DedupKey: if two different volumes or + /// databases collapsed to one key their totals would merge, which is worse than not counting them. + /// + [Fact] + public void Incidents_CarryDistinctDedupKeys() + { + var vol = AlertContextBuilders.VolumeFreeSpaceIncidents(Server, + new List + { + new() { MountPoint = @"C:\" }, new() { MountPoint = @"D:\" }, new() { MountPoint = @"E:\" }, + }); + + Assert.Equal(3, vol.Select(i => i.DedupKey).Distinct(StringComparer.Ordinal).Count()); + Assert.DoesNotContain(vol, i => string.IsNullOrEmpty(i.DedupKey)); + } + + /// + /// A null or empty source is an empty list, never a throw. These run on every sweep, including the ones + /// where a fetch came back with nothing, and an observation path that throws would take the check down. + /// + [Fact] + public void EveryBuilder_ToleratesNullAndEmpty() + { + Assert.Empty(AlertContextBuilders.LongRunningQueryIncidents(Server, null)); + Assert.Empty(AlertContextBuilders.VolumeFreeSpaceIncidents(Server, null)); + Assert.Empty(AlertContextBuilders.PvsPressureIncidents(Server, null)); + Assert.Empty(AlertContextBuilders.AnomalousJobIncidents(Server, null)); + Assert.Empty(AlertContextBuilders.FailedJobIncidents(Server, null)); + + Assert.Empty(AlertContextBuilders.LongRunningQueryIncidents(Server, new List())); + Assert.Empty(AlertContextBuilders.VolumeFreeSpaceIncidents(Server, new List())); + Assert.Empty(AlertContextBuilders.PvsPressureIncidents(Server, new List())); + Assert.Empty(AlertContextBuilders.AnomalousJobIncidents(Server, new List())); + Assert.Empty(AlertContextBuilders.FailedJobIncidents(Server, new List())); + } + + /// + /// A long-running query with no query_hash produces no incident rather than an incident keyed on the + /// empty string — every hashless query would otherwise share one fingerprint and pool its total. + /// + [Fact] + public void ALongRunningQueryWithNoHash_ProducesNoIncident() + { + var incidents = AlertContextBuilders.LongRunningQueryIncidents(Server, + new List + { + new() { QueryHash = null, DatabaseName = "OrdersDB" }, + new() { QueryHash = "0xABC", DatabaseName = "OrdersDB" }, + }); + + Assert.Single(incidents); + } + + /// + /// The render path and the observation path must agree on what a fingerprint IS, or a total would attach to + /// a key the card never shows. Same builder, same server, the capped input the card renders: the keys it + /// produces are a prefix of the uncapped set. + /// + [Fact] + public void TheRenderedSubset_UsesTheSameFingerprintsAsTheObservation() + { + var all = Enumerable.Range(0, PvsRenderCap + 3) + .Select(i => new PvsPressureInfo { DatabaseName = $"DB{i}" }).ToList(); + + var observed = AlertContextBuilders.PvsPressureIncidents(Server, all); + var rendered = AlertContextBuilders.PvsPressureIncidents(Server, all.GetRange(0, PvsRenderCap)); + + Assert.Equal(PvsRenderCap, rendered.Count); + Assert.True(observed.Count > rendered.Count, "the observation must be wider than the render"); + Assert.Equal( + rendered.Select(i => i.DedupKey).ToList(), + observed.Take(PvsRenderCap).Select(i => i.DedupKey).ToList()); + } +} diff --git a/Lite.Tests/IncidentProjectionTests.cs b/Lite.Tests/IncidentProjectionTests.cs new file mode 100644 index 000000000..670a8a5e8 --- /dev/null +++ b/Lite.Tests/IncidentProjectionTests.cs @@ -0,0 +1,181 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Linq; +using System.Text.Json; +using PerformanceMonitor.Notifications; +using Xunit; + +namespace Lite.Tests; + +/// +/// #2361: Database and LastEventUtc on the incident projection. +/// +/// Why Database belongs on the incident. A consumer previously string-searched Details[] for +/// a section whose fields contain a Database label. That is exact only for deadlocks — they are +/// self-contained, built with includeDetailFields: true — while every other fingerprinted alert appends a +/// BARE Incident item beside its data item, so the incident's own section carries no Database and the fallback +/// degrades to "any Database anywhere in the payload". On a multi-incident alert spanning databases that is not +/// an approximation; it is the wrong value with nothing marking it wrong. +/// +/// Both members are shared: AlertIncident and AlertIncidentDto live in +/// PerformanceMonitor.Notifications, so Lite and Darling get this by construction rather than by parity +/// maintenance. +/// +public class IncidentProjectionTests +{ + /// + /// The compatibility guarantee. Both members are trailing and optional, so contextJson persisted + /// before they existed still round-trips — the same property the DTO's own comment claims for + /// RemediationActionDto. Alert history is durable; a deserialization failure here would break the + /// reading of alerts that were written correctly at the time. + /// + [Fact] + public void LegacyContextJson_WithoutTheNewMembers_StillDeserializes() + { + const string legacy = """ + {"DedupKey":"abc123","InvolvedObjects":["dbo.Orders"],"OccurrenceCount":3, + "WaitRange":"1s-4s","TotalOccurrences":11,"IncidentStartedUtc":"2026-08-19T07:59:10Z"} + """; + + var dto = JsonSerializer.Deserialize(legacy); + + Assert.NotNull(dto); + Assert.Equal("abc123", dto!.DedupKey); + Assert.Equal(11, dto.TotalOccurrences); + Assert.Null(dto.Database); + Assert.Null(dto.LastEventUtc); + } + + /// Both members survive a full serialize/deserialize cycle, which is what the webhook token and the + /// persisted alert-history ContextJson both depend on. + [Fact] + public void TheNewMembers_RoundTrip() + { + var dto = new AlertIncidentDto( + "key", new List { "dbo.Orders" }, 2, "1s-4s", 7, + new DateTime(2026, 8, 19, 7, 59, 10, DateTimeKind.Utc), + "OrdersDB", + new DateTime(2026, 8, 19, 8, 42, 0, DateTimeKind.Utc)); + + var back = JsonSerializer.Deserialize(JsonSerializer.Serialize(dto)); + + Assert.Equal("OrdersDB", back!.Database); + Assert.Equal(new DateTime(2026, 8, 19, 8, 42, 0, DateTimeKind.Utc), back.LastEventUtc); + } + + /// + /// A fingerprint kind that is not database-scoped carries null rather than an empty string — a disk or a job + /// has no database, and "" would read downstream as a database whose name is blank. + /// + [Fact] + public void ANonDatabaseScopedIncident_CarriesNull() + { + var disk = AlertFingerprint.ForKey("SQLPROD01", AlertFingerprint.Disk, @"C:\", new[] { @"C:\" }); + Assert.NotNull(disk); + Assert.Null(disk!.Database); + + var blank = AlertFingerprint.ForKey("SQLPROD01", AlertFingerprint.Query, "0xABC", null, database: " "); + Assert.NotNull(blank); + Assert.Null(blank!.Database); + } + + /// A database-scoped incident carries it. + [Fact] + public void ADatabaseScopedIncident_CarriesTheDatabase() + { + var pvs = AlertFingerprint.ForKey( + "SQLPROD01", AlertFingerprint.Database, "OrdersDB", new[] { "OrdersDB" }, database: "OrdersDB"); + + Assert.NotNull(pvs); + Assert.Equal("OrdersDB", pvs!.Database); + } + + /// + /// Blocking incidents pick up the group's database, and the grouper's unknown sentinel becomes null. + /// That sentinel is fine as a display string and wrong as a data member: a consumer routing on + /// Database would file tickets against a database literally named "unknown". + /// + [Fact] + public void BlockingIncidents_CarryTheirDatabase_AndTheUnknownSentinelBecomesNull() + { + var withDb = BlockingIncidentGrouper.Group("SQLPROD01", new[] + { + new BlockingIncidentGrouper.BlockedEvent("OrdersDB", "dbo.Orders", "UPDATE", "SELECT", 1200, "LCK_M_X"), + }); + Assert.Single(withDb); + Assert.Equal("OrdersDB", withDb[0].Incident.Database); + + var noDb = BlockingIncidentGrouper.Group("SQLPROD01", new[] + { + new BlockingIncidentGrouper.BlockedEvent(null, "dbo.Orders", "UPDATE", "SELECT", 1200, "LCK_M_X"), + }); + Assert.Single(noDb); + Assert.Null(noDb[0].Incident.Database); + } + + /// + /// Deadlocks carry the database as a #2109 discrete fact on their detail fields, so the projection reads it + /// from there — the same lookup a consumer was doing by hand, done once where it is exact. + /// + [Fact] + public void DeadlockIncidents_TakeTheirDatabaseFromTheDetailFact() + { + var groups = DeadlockIncidentGrouper.Group("SQLPROD01", new[] + { + new DeadlockIncidentGrouper.DeadlockEvent( + new[] { "dbo.Orders", "dbo.Items" }, + new List { new("Database", "OrdersDB"), new("Victim SQL", "UPDATE ...") }), + }); + + Assert.Single(groups); + Assert.Equal("OrdersDB", groups[0].Incident.Database); + } + + /// A deadlock with no Database fact carries null rather than guessing from another field. + [Fact] + public void ADeadlockWithNoDatabaseFact_CarriesNull() + { + var groups = DeadlockIncidentGrouper.Group("SQLPROD01", new[] + { + new DeadlockIncidentGrouper.DeadlockEvent( + new[] { "dbo.Orders" }, + new List { new("Victim SQL", "UPDATE ...") }), + }); + + Assert.Single(groups); + Assert.Null(groups[0].Incident.Database); + } + + /// + /// The DTO projection carries both through. Serialize and SerializeIncidents share one private + /// ToDto, so the webhook token and the persisted ContextJson cannot disagree — a member added to the + /// record but not the mapping would silently serialize as null everywhere. + /// + [Fact] + public void TheSharedProjection_CarriesBothMembers() + { + var context = new AlertContext + { + Incidents = new List + { + new("key", new[] { "dbo.Orders" }, 2, null, null, 9, + new DateTime(2026, 8, 19, 7, 0, 0, DateTimeKind.Utc), + "OrdersDB", + new DateTime(2026, 8, 19, 9, 30, 0, DateTimeKind.Utc)), + }, + }; + + var json = AlertContextSerializer.SerializeIncidents(context); + + Assert.Contains("OrdersDB", json, StringComparison.Ordinal); + Assert.Contains("LastEventUtc", json, StringComparison.Ordinal); + } +} diff --git a/Lite.Tests/Lite.Tests.csproj b/Lite.Tests/Lite.Tests.csproj index c058a8f42..458d72963 100644 --- a/Lite.Tests/Lite.Tests.csproj +++ b/Lite.Tests/Lite.Tests.csproj @@ -1,10 +1,12 @@ - + net10.0-windows enable true false true + + Exe @@ -12,11 +14,6 @@ - - - all - runtime; build; native; contentfiles; analyzers; buildtransitive - diff --git a/Lite.Tests/LiteAlertForwardingTests.cs b/Lite.Tests/LiteAlertForwardingTests.cs index d270ee12b..9809a7300 100644 --- a/Lite.Tests/LiteAlertForwardingTests.cs +++ b/Lite.Tests/LiteAlertForwardingTests.cs @@ -134,6 +134,12 @@ public Task> GetLongRunningQueriesAsync( public Task> GetVolumeFreeSpaceAsync(string serverKey, CancellationToken cancellationToken = default) => Task.FromResult(new List(Volumes)); + /* #2349: empty on purpose. These tests exercise other alerts, and a fabricated file would + make the file-growth gate fire inside an unrelated scenario. */ + public Task> GetDatabaseFileGrowthAsync( + string serverKey, int lookbackMinutes, CancellationToken cancellationToken = default) => + Task.FromResult(new List()); + public Task GetTempDbSpaceAsync(string serverKey, CancellationToken cancellationToken = default) => Task.FromResult(TempDb); diff --git a/Lite.Tests/McpOutputCompactionTests.cs b/Lite.Tests/McpOutputCompactionTests.cs new file mode 100644 index 000000000..4b333f421 --- /dev/null +++ b/Lite.Tests/McpOutputCompactionTests.cs @@ -0,0 +1,105 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Linq; +using System.Text.Json; +using PerformanceMonitor.Common; +using Xunit; + +namespace Lite.Tests; + +/// +/// #2350: MCP tool results serialize COMPACT, and the config files people hand-edit do not. +/// +/// Both halves are pinned here because the change is one property on one shared object, and the risk is +/// entirely in scope rather than in mechanism — flipping WriteIndented on something that turns out to +/// write a config file would make servers.json a single unreadable line, which is the kind of damage +/// nobody notices until they open the file by hand at an awkward moment. +/// +/// The saving is real but should not be oversold: measured on a 15-field blocking-event shape it is +/// ~23% of the BYTES, and the token saving is smaller than that because BPE tokenizers pack runs of spaces +/// efficiently. It costs nothing, which is the argument — not the headline percentage. +/// +public class McpOutputCompactionTests +{ + private sealed record Row(int BlockedSessionId, string WaitType, int WaitMs); + + private static string SampleToolResult(int rows) => + JsonSerializer.Serialize( + new + { + server = "SQLPROD01", + total_events = rows, + events = Enumerable.Range(0, rows).Select(i => new Row(60 + i, "LCK_M_X", 1000 + i)).ToList(), + }, + McpHelpers.JsonOptions); + + /// The property itself, so a well-meaning "make the output readable" edit has to argue with a test. + [Fact] + public void McpJsonOptions_AreCompact() + { + Assert.False(McpHelpers.JsonOptions.WriteIndented); + } + + /// + /// The observable consequence, not just the flag: a record array serializes with no newline and no run of + /// indent spaces anywhere in it. + /// + [Fact] + public void AToolResult_CarriesNoLayoutWhitespace() + { + var json = SampleToolResult(30); + + Assert.DoesNotContain('\n', json); + Assert.DoesNotContain('\r', json); + Assert.DoesNotContain(" ", json, StringComparison.Ordinal); + } + + /// + /// Compaction must not change the DATA — the whole case for it is that the only consumer is a parser, so + /// the parsed value has to be identical to what the indented form produced. + /// + [Fact] + public void Compaction_ChangesLayoutOnly_NotContent() + { + var compact = SampleToolResult(10); + var indented = JsonSerializer.Serialize( + JsonSerializer.Deserialize(compact), + new JsonSerializerOptions { WriteIndented = true }); + + using var a = JsonDocument.Parse(compact); + using var b = JsonDocument.Parse(indented); + + Assert.Equal( + a.RootElement.GetProperty("events").GetArrayLength(), + b.RootElement.GetProperty("events").GetArrayLength()); + Assert.Equal( + a.RootElement.GetProperty("server").GetString(), + b.RootElement.GetProperty("server").GetString()); + + /* And it is genuinely smaller, which is the only reason to do it at all. */ + Assert.True(compact.Length < indented.Length, "compact output must be smaller than indented output"); + } + + /// + /// The boundary, pinned structurally the way this repo pins every invariant it cannot compile: the + /// managers that persist files a human opens keep indenting. Their options are private statics, so the + /// source is the assertable surface — the same idiom GridPayloadColumnOrderPinTests uses. + /// + [Theory] + [InlineData("Lite/Services/ServerManager.cs")] + [InlineData("Lite/Services/ProfileManager.cs")] + [InlineData("Lite/Services/ScheduleManager.cs")] + public void ConfigFileWriters_StayIndented(string relativePath) + { + var source = ParitySource.ReadFile(relativePath); + + Assert.Contains("WriteIndented = true", source, StringComparison.Ordinal); + } +} diff --git a/Lite.Tests/QueryStoreCollectorDefinitionTests.cs b/Lite.Tests/QueryStoreCollectorDefinitionTests.cs index ea0d5892a..4f9246afa 100644 --- a/Lite.Tests/QueryStoreCollectorDefinitionTests.cs +++ b/Lite.Tests/QueryStoreCollectorDefinitionTests.cs @@ -566,55 +566,59 @@ public void WithTheFlag_TheTextIsNulledAtTheSameOrdinal() } /// - /// #2150: the text fetch resumes from a query_id watermark, is cut by an exact byte budget, and - /// ships in query_id order — the ordering being what makes a budget cut a SUFFIX, so the highest - /// stored id resumes with no hole. + /// #2150's split, driven the #2312 way: the text fetch selects exactly the query_ids the caller names — + /// the cycle's collected rows whose text the store does not hold — cut by an exact byte budget in + /// query_id order. There is no watermark to resume from any more; the STORE answers what is + /// missing, and an empty missing set issues no query at all. /// [Fact] - public void TextFetch_ResumesFromTheWatermark_AndIsBudgetCutInQueryIdOrder() + public void TextFetchByIds_SelectsTheNamedIds_AndIsBudgetCutInQueryIdOrder() { - var sql = QueryStoreCollector.Instance.BuildTextFetchQuery( - "SO", MakeContext(fetchQueryTextSeparately: true), watermark: 4242, - candidateTexts: QueryStoreTextState.CandidateTexts, budgetBytes: 12 * 1024 * 1024).Text; + var sql = QueryStoreCollector.Instance.BuildTextFetchByIdsQuery( + "SO", MakeContext(fetchQueryTextSeparately: true), new long[] { 4242, 4243 }, + budgetBytes: 12 * 1024 * 1024).Text; Assert.Contains("EXECUTE [SO].sys.sp_executesql", sql, StringComparison.Ordinal); - Assert.Contains("WHERE qsq.query_id > 4242", sql, StringComparison.Ordinal); - Assert.Contains($"TOP ({QueryStoreTextState.CandidateTexts})", sql, StringComparison.Ordinal); - Assert.Contains("ORDER BY qsq.query_id", sql, StringComparison.Ordinal); + Assert.Contains("WHERE qsq.query_id IN (4242, 4243)", sql, StringComparison.Ordinal); + Assert.DoesNotContain("qsq.query_id > ", sql, StringComparison.Ordinal); + Assert.DoesNotContain("SELECT TOP", sql, StringComparison.Ordinal); Assert.Contains("ORDER BY b.query_id", sql, StringComparison.Ordinal); Assert.Contains("b.running_bytes - b.text_bytes < 12582912", sql, StringComparison.Ordinal); + /* query_id is only unique until a Query Store reset renumbers it; the hash is the reset detector. */ + Assert.Contains("query_hash = CONVERT(varchar(64), qsq.query_hash, 1)", sql, StringComparison.Ordinal); /* ROWS, not the RANGE default: RANGE tie-groups peers and forces a spool, and the frame has to be per-row because the cut falls BETWEEN two statements. */ Assert.Contains("ROWS UNBOUNDED PRECEDING", sql, StringComparison.Ordinal); Assert.Contains("OPTION(RECOMPILE)", sql, StringComparison.Ordinal); - /* It fetches text and nothing else — plan XML has its own fetch, with its own watermark. */ + /* It fetches text and nothing else — plan XML has its own fetch. */ Assert.DoesNotContain("query_plan", sql, StringComparison.Ordinal); } /// - /// #2150: every input that would make the fetch ship nothing and silently stall the watermark throws - /// instead. A stalled watermark looks exactly like a quiet database, which is why these are exceptions - /// rather than no-ops — the plan fetch learned this the hard way from several directions. + /// Every input that would make the fetch ship nothing — and leave the ids missing forever — throws + /// instead. A permanent missing set looks exactly like a quiet database, which is why these are + /// exceptions rather than no-ops; the plan fetch learned this the hard way from several directions. /// [Fact] - public void TextFetch_RefusesInputsThatWouldStallTheWatermark() + public void TextFetchByIds_RefusesInputsThatWouldStall() { var enabled = MakeContext(fetchQueryTextSeparately: true); + var ids = new long[] { 1 }; /* Issuing it while the host still ships text inline would fetch and store text nobody reads. */ Assert.Throws(() => - QueryStoreCollector.Instance.BuildTextFetchQuery("SO", MakeContext(), 0, 5_000, 1024)); + QueryStoreCollector.Instance.BuildTextFetchByIdsQuery("SO", MakeContext(), ids, 1024)); /* `running_bytes - text_bytes < 0` excludes even the first candidate, so the pass ships nothing. */ Assert.Throws(() => - QueryStoreCollector.Instance.BuildTextFetchQuery("SO", enabled, 0, 5_000, 0)); + QueryStoreCollector.Instance.BuildTextFetchByIdsQuery("SO", enabled, ids, 0)); - /* TOP (0) returns no rows; a negative literal is a syntax error. */ - Assert.Throws(() => - QueryStoreCollector.Instance.BuildTextFetchQuery("SO", enabled, 0, 0, 1024)); + /* Empty means "nothing missing" — the caller must skip, not build IN (). */ + Assert.Throws(() => + QueryStoreCollector.Instance.BuildTextFetchByIdsQuery("SO", enabled, Array.Empty(), 1024)); Assert.Throws(() => - QueryStoreCollector.Instance.BuildTextFetchQuery("SO", null!, 0, 5_000, 1024)); + QueryStoreCollector.Instance.BuildTextFetchByIdsQuery("SO", null!, ids, 1024)); } /// diff --git a/Lite.Tests/QueryStoreStatePruneTests.cs b/Lite.Tests/QueryStoreStatePruneTests.cs index c4a813325..151039dbd 100644 --- a/Lite.Tests/QueryStoreStatePruneTests.cs +++ b/Lite.Tests/QueryStoreStatePruneTests.cs @@ -215,19 +215,18 @@ await SeedStateAsync(ServerId, QueryStoreBackfillState.StateCollectorName, [Fact] public async Task Prune_RunsTheWatermarkStatementToo_EvenThoughLiteWritesNone() { - /* Lite iterates the SHARED QueryStorePerDatabaseState.PrunableKeys, which carries planwm: even - though Lite never writes it. That is deliberate: the day plan capture is enabled here, the prune - is already in place rather than being a thing somebody has to remember. Today it must simply be - harmless — one delete matching nothing — which is what this checks by proving a planted planwm: - row for a DROPPED database is retired by the same pass. */ + /* Lite iterates the SHARED QueryStorePerDatabaseState.PrunableKeys — including prefixes Lite may + never write itself. That is deliberate: a prefix pruned on one SKU and orphaning on the other is + the drift the shared list exists to prevent, and the cost is one delete matching nothing. Proven + here by planting a qsowm: row for a DROPPED database and watching the same pass retire it. */ await SeedSnapshotAsync(Newest, "Live"); - await SeedStateAsync(ServerId, QueryStorePlanXmlState.StateCollectorName, - QueryStorePlanXmlState.WatermarkKeyPrefix + "Dropped", "900000:1786449600"); + await SeedStateAsync(ServerId, QueryStoreOpenIntervalState.StateCollectorName, + QueryStoreOpenIntervalState.WatermarkKeyPrefix + "Dropped", "900000:1786449600"); await _pruner.PruneAsync(ServerId); - Assert.Null(await ValueAsync(ServerId, QueryStorePlanXmlState.StateCollectorName, - QueryStorePlanXmlState.WatermarkKeyPrefix + "Dropped")); + Assert.Null(await ValueAsync(ServerId, QueryStoreOpenIntervalState.StateCollectorName, + QueryStoreOpenIntervalState.WatermarkKeyPrefix + "Dropped")); } /* ---------------- helpers ---------------- */ @@ -417,23 +416,23 @@ public async Task ForeignPrune_WithNoOwnDatabase_RetiresNothing(string ownDataba } /// - /// Every per-database prefix is pruned, not just the backfill ones — the watermark prefix included, even - /// though Lite writes none today (it never sets CapturePlanXml). Pinned for the same reason the - /// on-prem twin pins it: the shared prefix list is what stops a prefix being pruned on one SKU and - /// orphaning on the other, and a Lite-only omission would be invisible on Darling. + /// Every per-database prefix is pruned, not just the backfill ones — the open-interval stamp included. + /// Pinned for the same reason the on-prem twin pins it: the shared prefix list is what stops a prefix + /// being pruned on one SKU and orphaning on the other, and a Lite-only omission would be invisible on + /// Darling. (#2312 retired the planwm:/textwm: families this fact previously exercised.) /// [Fact] public async Task ForeignPrune_CoversTheWatermarkPrefixToo() { - var foreignWatermark = QueryStorePlanXmlState.KeyFor("Sibling-A"); - var ownWatermark = QueryStorePlanXmlState.KeyFor("Payments"); + var foreignWatermark = QueryStoreOpenIntervalState.KeyFor("Sibling-A"); + var ownWatermark = QueryStoreOpenIntervalState.KeyFor("Payments"); - await SeedStateAsync(ServerId, QueryStorePlanXmlState.StateCollectorName, foreignWatermark, "8140"); - await SeedStateAsync(ServerId, QueryStorePlanXmlState.StateCollectorName, ownWatermark, "8150"); + await SeedStateAsync(ServerId, QueryStoreOpenIntervalState.StateCollectorName, foreignWatermark, "8140"); + await SeedStateAsync(ServerId, QueryStoreOpenIntervalState.StateCollectorName, ownWatermark, "8150"); await _pruner.PruneForeignAsync(ServerId, "Payments"); - Assert.Null(await ValueAsync(ServerId, QueryStorePlanXmlState.StateCollectorName, foreignWatermark)); - Assert.Equal("8150", await ValueAsync(ServerId, QueryStorePlanXmlState.StateCollectorName, ownWatermark)); + Assert.Null(await ValueAsync(ServerId, QueryStoreOpenIntervalState.StateCollectorName, foreignWatermark)); + Assert.Equal("8150", await ValueAsync(ServerId, QueryStoreOpenIntervalState.StateCollectorName, ownWatermark)); } } diff --git a/Lite.Tests/WatermarkPolicyTests.cs b/Lite.Tests/WatermarkPolicyTests.cs index 329ce2bf5..b610e29b4 100644 --- a/Lite.Tests/WatermarkPolicyTests.cs +++ b/Lite.Tests/WatermarkPolicyTests.cs @@ -83,4 +83,52 @@ clamp sat far above the cost tipping point on big databases and never interrupte Assert.Equal(TimeSpan.FromHours(1), WatermarkPolicy.MaxCatchup); Assert.Equal(QueryStoreBackfillState.MaxSliceSpan, WatermarkPolicy.MaxCatchup); } + + /// + /// #2344: the read floor must sit STRICTLY OLDER than the clamp horizon, because that ordering is the + /// whole safety argument. A floor at or newer than the horizon could hide a row the clamp would have + /// honoured; older by any margin cannot, since every outcome is max(stored, now - MaxCatchup) and a row + /// below the horizon produces the same answer found or not. + /// + [Fact] + public void ReadFloor_SitsStrictlyOlderThanTheClampHorizon() + { + var floor = WatermarkPolicy.ReadFloor(Now); + + Assert.NotNull(floor); + Assert.True(floor < Now - WatermarkPolicy.MaxCatchup, + "the read floor must be older than the clamp horizon, or the bound could hide a row the clamp would honour"); + Assert.Equal(Now - WatermarkPolicy.MaxCatchup - WatermarkPolicy.ReadFloorMargin, floor); + } + + /// + /// The equivalence the bound rests on, stated as a test rather than a comment: for any watermark at or + /// below the read floor, the CLAMPED result is the horizon — identical to what the caller derives when + /// the bounded read returns nothing at all (null falls back to query_store's 60-minute window, which is + /// the same instant as the horizon). So bounding the read cannot change a single caller's outcome. + /// + [Theory] + [InlineData(4)] + [InlineData(6)] + [InlineData(48)] + [InlineData(24 * 90)] + public void AnyWatermarkBelowTheReadFloor_ClampsToTheSameInstantAsFindingNothing(int hoursOld) + { + var floor = WatermarkPolicy.ReadFloor(Now)!.Value; + var buried = Now.AddHours(-hoursOld); + Assert.True(buried <= floor, "fixture must sit at or below the read floor"); + + /* Found-but-old and not-found-at-all reach the same place. */ + var clampedIfFound = WatermarkPolicy.ClampCatchup(buried, Now); + Assert.Equal(Now - WatermarkPolicy.MaxCatchup, clampedIfFound); + Assert.Equal(Now.AddMinutes(-60), clampedIfFound); + Assert.Null(WatermarkPolicy.ClampCatchup(null, Now)); + } + + /// A default input yields no floor — callers pass it straight through as "unbounded". + [Fact] + public void ReadFloor_OnDefault_IsNull() + { + Assert.Null(WatermarkPolicy.ReadFloor(default)); + } } diff --git a/Lite.Tests/packages.lock.json b/Lite.Tests/packages.lock.json index dcc0b23bc..94d312335 100644 --- a/Lite.Tests/packages.lock.json +++ b/Lite.Tests/packages.lock.json @@ -2,22 +2,6 @@ "version": 2, "dependencies": { "net10.0-windows7.0": { - "Microsoft.NET.Test.Sdk": { - "type": "Direct", - "requested": "[18.8.1, )", - "resolved": "18.8.1", - "contentHash": "dknJL3/9Y3t4XuCBqnc0PevPxgLsUMmVhjwup/b1HNovA8zWcj3XsfIf7c6p05363DWcqL7X/YhDL9B+Zymv1w==", - "dependencies": { - "Microsoft.CodeCoverage": "18.8.1", - "Microsoft.TestPlatform.TestHost": "18.8.1" - } - }, - "xunit.runner.visualstudio": { - "type": "Direct", - "requested": "[3.1.5, )", - "resolved": "3.1.5", - "contentHash": "tKi7dSTwP4m5m9eXPM2Ime4Kn7xNf4x4zT9sdLO/G4hZVnQCRiMTWoSZqI/pYTVeI27oPPqHBKYI/DjJ9GsYgA==" - }, "xunit.v3": { "type": "Direct", "requested": "[3.2.2, )", @@ -106,11 +90,6 @@ "resolved": "9.0.13", "contentHash": "5T+bH3Lb1nEe8Hf/ixMxLmhlrx5wRi53wv7OhVwG2F1ZviW1ejFRS1NHur3uqPpJRGtkQwUchtY6zhVK2R+v+w==" }, - "Microsoft.CodeCoverage": { - "type": "Transitive", - "resolved": "18.8.1", - "contentHash": "Eclse/ZZjr4lmWzZFNN9h/OluhKL+SK/QbUyKUewgX139aGeyMEO/DkMPwuFs2MixvanTnz6891rF8UHDg+W4Q==" - }, "Microsoft.Data.SqlClient.Extensions.Abstractions": { "type": "Transitive", "resolved": "7.0.2", @@ -156,215 +135,215 @@ }, "Microsoft.Extensions.Configuration.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "5Vnd2I75DmZCVEjSynIdJ/0EGafgnLQwgR3t2C2/fkjx/nRG+cLwxLLdInoHeCEpkD5K4Ov/g9ZCRYrl4TRsaA==", + "resolved": "10.0.11", + "contentHash": "fVi053xdpda9Em7vSkmgVxO/PtgC2m78ekReKWsgcyskqY0U82Bz/MONwxpGzI0hElYKJfw+fupqMVeKW3fSaA==", "dependencies": { - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Configuration.Binder": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "GqmN2o1CkJvk7uWp+p4CwBYW0w/zfoEbvsiFDbO2G8l1Uz+mrDAbAcZiXhU2lufKPby1cjAUdd5GTWpebYOkOA==", + "resolved": "10.0.11", + "contentHash": "rFn8RuszZn3qquPVkDytMUlPc2+rXl9MCoygwc1XmAgC5vg5/oXJ8hkOosOrLoBLsqdTy4lFwP6iQdPS9uSYOA==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Configuration.CommandLine": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "33cBeR2HRbzHUTtmcmLdNOApneNGcymwwL4arHuotgVK9Frba8kcDTrvVTj7cSCmF1R9OiSbZH0KxNOwab3HUg==", + "resolved": "10.0.11", + "contentHash": "1KHr/1L56llwQ/yI0tAisEA31UpPsn8aasjASIwELOaN4JIUcbjuQBMdFOIzfNBBeULoUa0XfBe5QDtRRUY+fg==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Configuration.EnvironmentVariables": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "KRfFSSCV58vEdU7mPED/YMzeovIWF5P0g8s9K8n9HEfy0/WzMq37SrPdXdFN5/dFT/rPMHpF7AvpoXHckbcBFg==", + "resolved": "10.0.11", + "contentHash": "KICyU3eVi5jvloKm01EXV69L97H/zkhISVtV98cIuzuFOxNx3xTUVcXqvWTz3aq7OvUuDB/MFlPFjmxRaKF7/A==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Configuration.FileExtensions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "ZOhZYwvbXGTgGVRwswIirofEMVHuWdxjdh0JeUZXwaF9cgcjXdz/t0ELtgaevw7ezTyv47yPNCgGreWtLkn3IQ==", + "resolved": "10.0.11", + "contentHash": "mDW7KVFB05M6jiRUyaZiOMWhS31n5HlSZwoYctHAZAucD4sMDJ70IxOmkGDt6RpstchD+keWBjhdzcMpSkWvWQ==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10", - "Microsoft.Extensions.FileProviders.Physical": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11", + "Microsoft.Extensions.FileProviders.Physical": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Configuration.UserSecrets": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "1s1sKFTk/Foam64JY6+m/diH8drL3Wx6V3gtSd5v1IEZtszZYyc1pW8uRnMblzpNiR0l0t8gGk7tXj3xHzFgdg==", + "resolved": "10.0.11", + "contentHash": "BRliLdUowglV8GS+J1G/QsSofCJYYFg3U8QZx0ACRn+a91az/Qnpy+h6PyHS94WgV2TSazX5D/cuuk6wnCJatw==", "dependencies": { - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Configuration.Json": "10.0.10", - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10", - "Microsoft.Extensions.FileProviders.Physical": "10.0.10" + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Configuration.Json": "10.0.11", + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11", + "Microsoft.Extensions.FileProviders.Physical": "10.0.11" } }, "Microsoft.Extensions.DependencyInjection": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "ANyvsgkNBRvcJh2XLgn8veGmajf+8m0AbKK+HPWdRL1yraSNVVSmQhFntLtdz/C795jxqqup+k05cs/3jZQPOA==", + "resolved": "10.0.11", + "contentHash": "PSmotV19c7E3lKed++uYo1kSiXFI+uTl37CBSrhq+CfLC3FCHjG7R91+xPnNehQfHS1b0Tzo/CCLPWH3qaEheg==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11" } }, "Microsoft.Extensions.DependencyInjection.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "z/2xXlFw2aLGjHyEm6E0tQ+In6VfzQzTrtArbQ2c0TQE16ZbyDCMGPvaUT9I0s8rgy9sRWlU2P9waW37qV04qA==" + "resolved": "10.0.11", + "contentHash": "/a1aJz4m7ylhEDf25ugQChLQoN5XwoGjWw/BoR/ZWWKsO1v4DdJElS1uyngahz4B/eOzjFk1KNTkarRLE5wsIg==" }, "Microsoft.Extensions.Diagnostics": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "Kr/e7lUf4+N8tacbqJ2Ctwe/HarKdAc9ZkgKVVqvtJDBKbez+T/KnUwu82KSlnBp/SrpBcxc7u7xkE2oUZT/5Q==", + "resolved": "10.0.11", + "contentHash": "HT70uGPxMLqqnOzKMcnQtDmeV4r0KHr4qVCLhP7SXil9jMEm8sQXwcybxVVFGXZJ1V44xV0mLqQ54aZbcR2OiQ==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Diagnostics.Abstractions": "10.0.10", - "Microsoft.Extensions.Options.ConfigurationExtensions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Diagnostics.Abstractions": "10.0.11", + "Microsoft.Extensions.Options.ConfigurationExtensions": "10.0.11" } }, "Microsoft.Extensions.Diagnostics.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "9uWiKpeOVac355STyChWR/pliFX/5CeLqChW9kKsaxyDH4EUTZxMkT4Jwp/J/peLm0GBFmSX5c0WCse3yCnq1Q==", + "resolved": "10.0.11", + "contentHash": "se7Kx8QpJEt+nf26L4qIVAofGTDr1wbexxsh/Fm3Xc04xUkqUXK06KUS7FLwSQYSjqb7q9n+T7MEcXYBhI1Y5g==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11" } }, "Microsoft.Extensions.FileProviders.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "c5zqFCY9DiIpMovLd7/d/CTiEtrMOuQ639dhv3PABtKQIKNQikSHwQt8+N679uii9q+B55lgK28Uv64FOwEu8w==", + "resolved": "10.0.11", + "contentHash": "JOjac6SQQgZmdmB8WGEw61/7siqMZoWJMkmq2p1goJGxqI59lO6oB4bOl0jNsbaPBdYy5Mlkb+6U7T4+CjnD8Q==", "dependencies": { - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.FileProviders.Physical": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "jhJAyo38kSrH3ARvWUk0h8itogVnQu2DCZuPo+s0Z+tXes0ugTxMPaHYzap85785eHQmPFqD9TYERqBbtGxn/w==", + "resolved": "10.0.11", + "contentHash": "Tq/UqMaczePv9yWwSsJZRgKtgA46djVR5xHj/lZBCueQ3ag8f9v5mu0EdhrNx7tXxNk+Y9OurG2oKuSKINjr0A==", "dependencies": { - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10", - "Microsoft.Extensions.FileSystemGlobbing": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11", + "Microsoft.Extensions.FileSystemGlobbing": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.FileSystemGlobbing": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "jSOCVxEwCd4Aq925kJVz1kSO1EpX2OHYKL04qVREXkDU7Ce3pVDdHPYm+fEy8y/th2kJf/DAstRHpJAqoNWP8w==" + "resolved": "10.0.11", + "contentHash": "2i6rtW/B5rCnWCnhdmWWEmaM9O0HD0zsPY9eRqa++y4tclI3Uw8zvGbBvhY/LjAdtf8gUHhUPcAWj3DRlWMXmQ==" }, "Microsoft.Extensions.Hosting.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "5LugpYGHk+mkn0a8IZgcyfBca8PCTAU9RQFoMrTdtOOidq88M2SI5f3px6ugnzgxC+eTkvYYJi8pzlUnG5xdAQ==", + "resolved": "10.0.11", + "contentHash": "pwtpF7iF/NNaOBcX+pvMZ7y2+JAVbH5KkNrH9uMZtuVxVJsFTDiWiCR7Tk3HVptsAijaetipUZVRVK2LLq+nvA==", "dependencies": { - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Diagnostics.Abstractions": "10.0.10", - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10" + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Diagnostics.Abstractions": "10.0.11", + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Logging.Configuration": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "cLrqxkuEfcilZ8SjK+9KAnpLk9lOoMPaOokF+wRUYie+iUEcdX4/p/+gJkt0BYgWLthjpBUCkVTBI6Kxg0nsOw==", + "resolved": "10.0.11", + "contentHash": "S7LvLeVHKNPaY2NMyxW7c2TBGsLgxoSUBCV5Ev5iN8kgC7EPR2UB7eW7vHsElGMcIUDwRmoxLfvGDynCn3q6EA==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Configuration.Binder": "10.0.10", - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10", - "Microsoft.Extensions.Options.ConfigurationExtensions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Configuration.Binder": "10.0.11", + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11", + "Microsoft.Extensions.Options.ConfigurationExtensions": "10.0.11" } }, "Microsoft.Extensions.Logging.Console": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "VIlNzPwPS0GeQVSmCqqo36ugryX3LpE9ul6gEkks5VLET3weH/XMLeWmclwfoGn4Nxi2mwVibB+OZBVJ9tDqvg==", + "resolved": "10.0.11", + "contentHash": "dFc0yDudyD1iIg6z9XT7ofsT3hVO7Y4ylrxGHIVRR0GaZ4CUk4ujOrMoy7wWEdNHZhvJooySg6hZpOxyS8zEVA==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging.Configuration": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging.Configuration": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11" } }, "Microsoft.Extensions.Logging.Debug": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "8+TZBnV5fgBXoVNJ5ROSErUwYogk4hOgV7c2HWK1u5cqKGmiUTUn7+KqZ35iQu8e/B7Ykccyz5OTjdXcidNZ9g==", + "resolved": "10.0.11", + "contentHash": "wr+j1bjdFXhc8lKTLoq+RbwFM8M+orcMS9xrcqLmDGOxJcXpKizEeE5h6v/GKwCZV02FmhaA7OlNjoq072jZpQ==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Logging.EventLog": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "0RE4951AzQ+YD4gVrvbq0BhdsiBgSDo44yM7+QBZ2mrmMJeNjY+teCIYfUjqDPVYnKs0HR6SkkhgrX1YgXZq3Q==", + "resolved": "10.0.11", + "contentHash": "Eck9GpCCpvZ3f6L7IUlN+mPtRVefnf7PsiIG5vi61QawPtLNCEAv2TPD/M3SojcU0PFaef+BxiVGPOShFHtDog==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10", - "System.Diagnostics.EventLog": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11", + "System.Diagnostics.EventLog": "10.0.11" } }, "Microsoft.Extensions.Logging.EventSource": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "85SAPwXhJtdBInzN2k7SChiFiBGh3KOWay5AfoY+GREF6P7oZA98+ST2p7Z9384iLKYjkZSKIZ/FqIO5aojtNw==", + "resolved": "10.0.11", + "contentHash": "hs6QWECLLohi2VKqUvSGRUvrg7eXR1DqKL95Jrtz3cdD2g2nBA+yJPdRQLZ7SLmnTZWycxfMDK2s0ho+rfst5w==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Options": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "srnhnk7nE8krBiIXp71LvBmKBtraBONWSRzdjJgRv1Ko9Mp8IVNqv4vIS9hGeVteBig8aQkva9ZG+sC+o5sVcA==", + "resolved": "10.0.11", + "contentHash": "eY1GAKcTfD2maP27J84X9IovT3yjHJ2dVDzPmDg6/XqYvt3jMzJhtfQCLjG9pVsZGAd+8DQ2QrjaDcs2+VQLGw==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Options.ConfigurationExtensions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "tnBmu/LwF25ZQK+HBNCu2xrwnkKoB/XEbJyooGGoYxHrhvxbSKi7eOFiJ4AXBy/QU4vtCvCJfoi8k9Ej72qzOQ==", + "resolved": "10.0.11", + "contentHash": "syEhXQ/sEaSBFaqzlp9gDGHX/nk6gkQkh1sIUpBO1mlBj3Phu1rmb4ML1uCiyPW9N6Kxfxv3y5FGObC+bV01Qw==", "dependencies": { - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Configuration.Binder": "10.0.10", - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Configuration.Binder": "10.0.11", + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Primitives": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "5wu/GrYVd8mG2DVUw3vFJzF+O336TyTGg/Kmcgw9bfwYhCoFiV5lR5QeEmKecJyrW4W54nMfD3p3589E8a7czQ==" + "resolved": "10.0.11", + "contentHash": "SXcz+kF+4Oo9b1+55zntpJFYfwb1jw66ioxptyNOOTDc8g2FHnBFWjZpsWfCvZIhzr0x+4e2trVTs4OKwQfBtw==" }, "Microsoft.Identity.Client": { "type": "Transitive", @@ -478,19 +457,6 @@ "Microsoft.Testing.Platform": "1.9.1" } }, - "Microsoft.TestPlatform.ObjectModel": { - "type": "Transitive", - "resolved": "18.8.1", - "contentHash": "qLbktNB1+b1XZLNJBTzaWVVJAd6PEzD7cgD406geMb6PcFZhp3EDNa1tctWx1+mtMU6MP/6ozVvFPC9vs2a9rw==" - }, - "Microsoft.TestPlatform.TestHost": { - "type": "Transitive", - "resolved": "18.8.1", - "contentHash": "FaQHPDTUOcE+SFTjssNPfrub2lT9Zyon4J2W/KLHt/efLJACb1TCeWXyOgh0D/4Q1e4n+S3E6mOKud+9nLZlEA==", - "dependencies": { - "Microsoft.TestPlatform.ObjectModel": "18.8.1" - } - }, "Microsoft.Win32.Registry": { "type": "Transitive", "resolved": "5.0.0", @@ -498,8 +464,8 @@ }, "ModelContextProtocol.Core": { "type": "Transitive", - "resolved": "2.1.0", - "contentHash": "cU/urrhRxE4/iSyBIJI7QOaFqSP1FOEnwEHsct9n6t6/XluCAFD9iqnrPkBAsEYr+f/G4tVQ21U+6wN/6fQvOg==", + "resolved": "2.2.0", + "contentHash": "FeBfXU6T8k+jw4afg4sfxdEX2rL/e5oKOk9ROOGztu9k47+7Bz08sdaToYt2XvMY1opNbwxYQOFMj6wH9TInhA==", "dependencies": { "Microsoft.Extensions.AI.Abstractions": "10.8.3", "Microsoft.Extensions.Logging.Abstractions": "10.0.10" @@ -677,8 +643,8 @@ }, "System.Diagnostics.EventLog": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "OvGz3PrzuAI/Sj7LTcXcCe3FClRI1IyRMZjNONcZtFh+Ww7nAtSh4kh08r8KVe/xxkXJPjR0Y1jF7H+N42d4xQ==" + "resolved": "10.0.11", + "contentHash": "QTXEoQBzz00SFWbo7nAg1Ogd4f99lwqcO9uAJ7MYSLEUR28f6As32QktrqG2Fr9cfAfd1GjLyGYspE7Ipj7P6w==" }, "System.IdentityModel.Tokens.Jwt": { "type": "Transitive", @@ -777,14 +743,14 @@ "type": "Project", "dependencies": { "CredentialManagement": "[1.0.2, )", - "Microsoft.Extensions.Logging.Abstractions": "[10.0.10, )", - "ModelContextProtocol": "[2.1.0, )" + "Microsoft.Extensions.Logging.Abstractions": "[10.0.11, )", + "ModelContextProtocol": "[2.2.0, )" } }, "performancemonitor.notifications": { "type": "Project", "dependencies": { - "Microsoft.Extensions.Logging.Abstractions": "[10.0.10, )", + "Microsoft.Extensions.Logging.Abstractions": "[10.0.11, )", "PerformanceMonitor.Analysis": "[1.0.0, )" } }, @@ -792,7 +758,8 @@ "type": "Project", "dependencies": { "Microsoft.Data.SqlClient": "[7.0.2, )", - "PerformanceMonitor.Common": "[1.0.0, )" + "PerformanceMonitor.Common": "[1.0.0, )", + "System.Security.Cryptography.ProtectedData": "[10.0.11, )" } }, "performancemonitor.ui": { @@ -812,10 +779,10 @@ "Hardcodet.NotifyIcon.Wpf": "[2.0.1, )", "Microsoft.Data.SqlClient": "[7.0.2, )", "Microsoft.Data.SqlClient.Extensions.Azure": "[7.0.2, )", - "Microsoft.Extensions.Hosting": "[10.0.10, )", - "Microsoft.Extensions.Logging": "[10.0.10, )", - "ModelContextProtocol": "[2.1.0, )", - "ModelContextProtocol.AspNetCore": "[2.1.0, )", + "Microsoft.Extensions.Hosting": "[10.0.11, )", + "Microsoft.Extensions.Logging": "[10.0.11, )", + "ModelContextProtocol": "[2.2.0, )", + "ModelContextProtocol.AspNetCore": "[2.2.0, )", "PerformanceMonitor.Alerting": "[1.0.0, )", "PerformanceMonitor.Analysis": "[1.0.0, )", "PerformanceMonitor.Collectors": "[1.0.0, )", @@ -888,94 +855,94 @@ }, "Microsoft.Extensions.Configuration": { "type": "CentralTransitive", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "plJWK2zpWuuyxI8F8s2scx6Je7N1Ajjs6HvYUGKwRnDMWIVIz9FHwAkiT7ASgrvAOd10T0FPVlh9BzAJJME+jg==", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "wlhRqZW8LcJPa+vk2oLAc/REXDItHtkFQdf/QcXYGZbZOO13izcsKY1pCvuFQYwUiZD+hwSZwsKASjqT+BNaVg==", "dependencies": { - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Configuration.Json": { "type": "CentralTransitive", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "uvJ6sHwjgrkMEJOgiC76G0mcZGXerwyyWkwX34EOjCbxKG6TCtfAoqDKAMsCvEBf9HxjlGQEgqsSMOGCmGBf+A==", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "nSPrT8U/cNoB4coqkmnanAMK9PsL7lsjG+LLUKEwHRFwS6E78b8S1wdv/y88EOxBhasWov1rLd7RTHmmsYPOLg==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Configuration.FileExtensions": "10.0.10", - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Configuration.FileExtensions": "10.0.11", + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Hosting": { "type": "CentralTransitive", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "tL9FkfV64GPUDSPvwrgyw42LVzsnVAnyrqJEuZVJbODgrQ3eL63zmzEcVWoCHzfgqUhWggzbgAyUCnz/zfI3Pg==", - "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Configuration.Binder": "10.0.10", - "Microsoft.Extensions.Configuration.CommandLine": "10.0.10", - "Microsoft.Extensions.Configuration.EnvironmentVariables": "10.0.10", - "Microsoft.Extensions.Configuration.FileExtensions": "10.0.10", - "Microsoft.Extensions.Configuration.Json": "10.0.10", - "Microsoft.Extensions.Configuration.UserSecrets": "10.0.10", - "Microsoft.Extensions.DependencyInjection": "10.0.10", - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Diagnostics": "10.0.10", - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10", - "Microsoft.Extensions.FileProviders.Physical": "10.0.10", - "Microsoft.Extensions.Hosting.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging.Configuration": "10.0.10", - "Microsoft.Extensions.Logging.Console": "10.0.10", - "Microsoft.Extensions.Logging.Debug": "10.0.10", - "Microsoft.Extensions.Logging.EventLog": "10.0.10", - "Microsoft.Extensions.Logging.EventSource": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10" + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "eIDa/Rl+93aj17gMlFsJJx+LhBvb3CP0Mu1PeVYkDp2Y3S4Jock8UynfGQEcx7lrlq+gKW+ECQJHbro/LTPDEQ==", + "dependencies": { + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Configuration.Binder": "10.0.11", + "Microsoft.Extensions.Configuration.CommandLine": "10.0.11", + "Microsoft.Extensions.Configuration.EnvironmentVariables": "10.0.11", + "Microsoft.Extensions.Configuration.FileExtensions": "10.0.11", + "Microsoft.Extensions.Configuration.Json": "10.0.11", + "Microsoft.Extensions.Configuration.UserSecrets": "10.0.11", + "Microsoft.Extensions.DependencyInjection": "10.0.11", + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Diagnostics": "10.0.11", + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11", + "Microsoft.Extensions.FileProviders.Physical": "10.0.11", + "Microsoft.Extensions.Hosting.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging.Configuration": "10.0.11", + "Microsoft.Extensions.Logging.Console": "10.0.11", + "Microsoft.Extensions.Logging.Debug": "10.0.11", + "Microsoft.Extensions.Logging.EventLog": "10.0.11", + "Microsoft.Extensions.Logging.EventSource": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11" } }, "Microsoft.Extensions.Logging": { "type": "CentralTransitive", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "Tf6z5HsL0VDYRTfvsoNrTGHGheCwkTsZBA2FFh5ATJUbkAwug+FFNISJK2gjpUNemlAOoWllAK52HOWCjto3EQ==", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "nUOJwgFkSiLHiVGFpU22pIJtuWYewuSYQ3JVuP/gdK8ASMT807Px+TYQiRWs6uSsOmoyFTaVCwKXTasczV6BpA==", "dependencies": { - "Microsoft.Extensions.DependencyInjection": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10" + "Microsoft.Extensions.DependencyInjection": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11" } }, "Microsoft.Extensions.Logging.Abstractions": { "type": "CentralTransitive", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "zkFxGYUvdxAvIKTyXHrmW+Sux53D4SezD9dMyZ6hrwwzPQJNuwCRy1f5W7AvYTqacEGhWF2XderRQG1OvbV8og==", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "Ljd0Uxoq5XpScD2Bg0nM/r3mwx7Ao5Uq24eo2ARxbGvqJ7Zht6rt2cJtwVRH4Cv+1ZVMdXz6TB43KbpmsxRrvQ==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11" } }, "ModelContextProtocol": { "type": "CentralTransitive", - "requested": "[2.1.0, )", - "resolved": "2.1.0", - "contentHash": "Oa4rU7EL9C2qyFjQj1dx+ysGMzfWDRpM8RRaUMmLGs5vPvfJ9xyz4ZtyF4ychY+Nx1b/auGCqIQLqSz/IpPkKA==", + "requested": "[2.2.0, )", + "resolved": "2.2.0", + "contentHash": "4Pb9u02Nwsp0poueDsqNdyGRojFxOYpljB7zDBsq+aHL+Afou3OgxlBc3GWFVnsRMRJrUtWqDh3s6k2JgPzmrQ==", "dependencies": { "Microsoft.Extensions.Caching.Abstractions": "10.0.10", "Microsoft.Extensions.Hosting.Abstractions": "10.0.10", - "ModelContextProtocol.Core": "[2.1.0]" + "ModelContextProtocol.Core": "[2.2.0]" } }, "ModelContextProtocol.AspNetCore": { "type": "CentralTransitive", - "requested": "[2.1.0, )", - "resolved": "2.1.0", - "contentHash": "yhJ8bBXIgrX0mAgRYRgzcbH3bLdv3MDSkG52utRW9EAAtQrPw/g7Q/T6EurxKV+L+Zefv8VVUYcNbXEOd9GgfA==", + "requested": "[2.2.0, )", + "resolved": "2.2.0", + "contentHash": "3JelDMuFIwFzXybsh6K30G6wXu5gmqKnuRnieqgBMuR0FkqqXRv4B+NYZuPgthSyNq5UwRyvv2U2VJBE3RD3PQ==", "dependencies": { - "ModelContextProtocol": "[2.1.0]" + "ModelContextProtocol": "[2.2.0]" } }, "ScottPlot.WPF": { @@ -990,6 +957,12 @@ "SkiaSharp.Views.WPF": "3.119.0" } }, + "System.Security.Cryptography.ProtectedData": { + "type": "CentralTransitive", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "PNoxCTPb+Tlux+GJyq4c89ddYdpioVSqfGx8pqOF6shCSKwUNNctXhQtRkCICjbJmGrJJsW8NY52kVYO/b8mlQ==" + }, "Velopack": { "type": "CentralTransitive", "requested": "[1.2.0, )", diff --git a/Lite/App.xaml.cs b/Lite/App.xaml.cs index 512f0a133..3941539ae 100644 --- a/Lite/App.xaml.cs +++ b/Lite/App.xaml.cs @@ -156,6 +156,10 @@ being handed back the stale in-memory version. The coordinator owns the mutex + public static bool AlertPvsEnabled { get; set; } = true; // #1984 ADR persistent version store pressure public static int AlertPvsThresholdPercent { get; set; } = 40; // Alert when an ADR database's PVS >= X% of its data files (0 disables this check) public static int AlertPvsFloorGb { get; set; } = 1; // AND-qualifier: the PVS must also be >= X GB (0 removes the floor) + public static bool AlertFileGrowthEnabled { get; set; } // #2349 database file growth -- OFF by default + public static int AlertFileGrowthRiseMb { get; set; } = 10240; // RISE gate: a file grew >= X MB in the window (0 disables this gate) + public static int AlertFileGrowthVolumePercent { get; set; } = 60; // LEVEL gate: a file is >= X% of its volume (0 disables this gate) + public static int AlertFileGrowthLookbackMinutes { get; set; } = 60;// how far back the rise is measured public static bool AlertLongRunningJobEnabled { get; set; } = true; public static int AlertLongRunningJobMultiplier { get; set; } = 3; public static bool AlertFailedJobEnabled { get; set; } = true; @@ -738,6 +742,10 @@ cannot drive a nonsense threshold in either app. */ if (root.TryGetProperty("alert_disk_critical_free_gb", out v)) AlertDiskCriticalFreeGb = (int)Math.Max(0, v.GetInt64()); if (root.TryGetProperty("alert_pvs_enabled", out v)) AlertPvsEnabled = v.GetBoolean(); if (root.TryGetProperty("alert_pvs_threshold_percent", out v)) AlertPvsThresholdPercent = (int)Math.Clamp(v.GetInt64(), 0, 100); + if (root.TryGetProperty("alert_file_growth_enabled", out v)) AlertFileGrowthEnabled = v.GetBoolean(); + if (root.TryGetProperty("alert_file_growth_rise_mb", out v)) AlertFileGrowthRiseMb = (int)Math.Max(0, v.GetInt64()); + if (root.TryGetProperty("alert_file_growth_volume_percent", out v)) AlertFileGrowthVolumePercent = (int)Math.Clamp(v.GetInt64(), 0, 100); + if (root.TryGetProperty("alert_file_growth_lookback_minutes", out v)) AlertFileGrowthLookbackMinutes = (int)Math.Clamp(v.GetInt64(), 5, 1440); if (root.TryGetProperty("alert_pvs_floor_gb", out v)) AlertPvsFloorGb = (int)Math.Max(0, v.GetInt64()); if (root.TryGetProperty("alert_long_running_job_enabled", out v)) AlertLongRunningJobEnabled = v.GetBoolean(); if (root.TryGetProperty("alert_long_running_job_multiplier", out v)) AlertLongRunningJobMultiplier = v.GetInt32(); diff --git a/Lite/Services/AppAlertEngineSettings.cs b/Lite/Services/AppAlertEngineSettings.cs index 38a102256..cef175fde 100644 --- a/Lite/Services/AppAlertEngineSettings.cs +++ b/Lite/Services/AppAlertEngineSettings.cs @@ -83,6 +83,14 @@ would be silently reset on the first config_version bump. If this needs to be co public int CollectionFailureThreshold => 10; public int PvsThresholdPercent => App.AlertPvsThresholdPercent; public int PvsFloorGb => App.AlertPvsFloorGb; + + /* #2349: the file-growth gates, same clamps as Darling's adapter so the two SKUs cannot disagree about + what a threshold means. Zero disables one gate rather than being nonsense, so rise-only or level-only + needs no second switch. */ + public bool FileGrowthEnabled => App.AlertFileGrowthEnabled; + public int FileGrowthRiseMb => Math.Max(0, App.AlertFileGrowthRiseMb); + public int FileGrowthVolumePercent => Math.Clamp(App.AlertFileGrowthVolumePercent, 0, 100); + public int FileGrowthLookbackMinutes => Math.Clamp(App.AlertFileGrowthLookbackMinutes, 5, 1440); public int LongRunningJobMultiplier => App.AlertLongRunningJobMultiplier; public int FailedJobLookbackMinutes => App.AlertFailedJobLookbackMinutes; public int CooldownMinutes => App.AlertCooldownMinutes; diff --git a/Lite/Services/LiteAlertReadAdapter.cs b/Lite/Services/LiteAlertReadAdapter.cs index 3df7f8a9a..7d6567f82 100644 --- a/Lite/Services/LiteAlertReadAdapter.cs +++ b/Lite/Services/LiteAlertReadAdapter.cs @@ -141,6 +141,15 @@ public async Task> GetVolumeFreeSpaceAsync( return await Task.Run(() => _dataService.GetVolumeFreeSpaceAsync(serverId), cancellationToken); } + /// #2349: the file-growth read, delegated like every other adapter member so the alert engine + /// stays free of DuckDB. + public async Task> GetDatabaseFileGrowthAsync( + string serverKey, int lookbackMinutes, CancellationToken cancellationToken = default) + { + var serverId = ParseServerKey(serverKey); + return await Task.Run(() => _dataService.GetDatabaseFileGrowthAsync(serverId, lookbackMinutes), cancellationToken); + } + public async Task> GetPvsPressureAsync( string serverKey, CancellationToken cancellationToken = default) { diff --git a/Lite/Services/LocalDataService.FileGrowth.cs b/Lite/Services/LocalDataService.FileGrowth.cs new file mode 100644 index 000000000..7fee8c73c --- /dev/null +++ b/Lite/Services/LocalDataService.FileGrowth.cs @@ -0,0 +1,103 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Threading.Tasks; +using DuckDB.NET.Data; +using PerformanceMonitor.Alerting; + +namespace PerformanceMonitorLite.Services; + +/// +/// Lite's half of the file-growth read (#2349) — the DuckDB twin of +/// DarlingAlertReadAdapter.DatabaseFileGrowthSql, so the alert behaves identically in both SKUs. +/// +/// Same shape as the Postgres side and for the same reasons: newest row per (database, file), a baseline +/// from the oldest sample inside the window, and growth as the difference — 0 when the window holds one sample, +/// which reads as "no rise observed" rather than as a rise of the whole file. +/// +/// DuckDB has no DISTINCT ON, so the same selection is expressed with ROW_NUMBER() +/// partitioned by the file key — the idiom the rest of Lite's store SQL already uses where Postgres would use +/// DISTINCT ON. +/// +public partial class LocalDataService +{ + public async Task> GetDatabaseFileGrowthAsync(int serverId, int lookbackMinutes) + { + using var connection = await OpenConnectionAsync(); + using var command = connection.CreateCommand(); + + command.CommandText = @" +WITH windowed AS ( + SELECT + database_name, file_name, physical_name, file_type_desc, collection_time, + total_size_mb, auto_growth_mb, is_percent_growth, growth_pct, max_size_mb, + volume_mount_point, volume_total_mb, volume_free_mb, + ROW_NUMBER() OVER (PARTITION BY database_name, file_name ORDER BY collection_time DESC) AS rn_new, + ROW_NUMBER() OVER (PARTITION BY database_name, file_name ORDER BY collection_time ASC) AS rn_old + FROM v_database_size_stats + WHERE server_id = $1 + AND collection_time >= $2 +) +SELECT + c.database_name, + c.file_name, + COALESCE(c.physical_name, '') AS physical_name, + COALESCE(c.file_type_desc, '') AS file_type_desc, + COALESCE(c.total_size_mb, 0) AS total_size_mb, + COALESCE(c.total_size_mb, 0) - COALESCE(b.total_size_mb, c.total_size_mb, 0) AS growth_mb, + COALESCE(date_diff('second', b.collection_time, c.collection_time) / 60.0, 0) AS growth_window_minutes, + COALESCE(c.volume_mount_point, '') AS volume_mount_point, + COALESCE(c.volume_total_mb, 0) AS volume_total_mb, + COALESCE(c.volume_free_mb, 0) AS volume_free_mb, + c.auto_growth_mb, + COALESCE(c.is_percent_growth, false) AS is_percent_growth, + c.growth_pct, + c.max_size_mb +FROM windowed c +LEFT JOIN windowed b + ON b.database_name = c.database_name + AND b.file_name = c.file_name + AND b.rn_old = 1 +WHERE c.rn_new = 1 +AND c.total_size_mb IS NOT NULL +ORDER BY c.database_name, c.file_name"; + + command.Parameters.Add(new DuckDBParameter { Value = serverId }); + command.Parameters.Add(new DuckDBParameter + { + Value = DateTime.UtcNow.AddMinutes(-Math.Max(1, lookbackMinutes)) + }); + + var items = new List(); + using var reader = await command.ExecuteReaderAsync(); + while (await reader.ReadAsync()) + { + items.Add(new DatabaseFileGrowthInfo + { + DatabaseName = reader.IsDBNull(0) ? "" : reader.GetString(0), + FileName = reader.IsDBNull(1) ? "" : reader.GetString(1), + PhysicalName = reader.IsDBNull(2) ? "" : reader.GetString(2), + FileTypeDesc = reader.IsDBNull(3) ? "" : reader.GetString(3), + TotalSizeMb = reader.IsDBNull(4) ? 0 : ToDouble(reader.GetValue(4)), + GrowthMb = reader.IsDBNull(5) ? 0 : ToDouble(reader.GetValue(5)), + GrowthWindowMinutes = reader.IsDBNull(6) ? 0 : ToDouble(reader.GetValue(6)), + VolumeMountPoint = reader.IsDBNull(7) ? "" : reader.GetString(7), + VolumeTotalMb = reader.IsDBNull(8) ? 0 : ToDouble(reader.GetValue(8)), + VolumeFreeMb = reader.IsDBNull(9) ? 0 : ToDouble(reader.GetValue(9)), + AutoGrowthMb = reader.IsDBNull(10) ? null : ToDouble(reader.GetValue(10)), + IsPercentGrowth = !reader.IsDBNull(11) && Convert.ToBoolean(reader.GetValue(11)), + GrowthPct = reader.IsDBNull(12) ? null : ToDouble(reader.GetValue(12)), + MaxSizeMb = reader.IsDBNull(13) ? null : ToDouble(reader.GetValue(13)), + }); + } + + return items; + } +} diff --git a/Lite/Services/RemoteCollectorService.DefinitionRunner.cs b/Lite/Services/RemoteCollectorService.DefinitionRunner.cs index a8ab74417..9df6ec313 100644 --- a/Lite/Services/RemoteCollectorService.DefinitionRunner.cs +++ b/Lite/Services/RemoteCollectorService.DefinitionRunner.cs @@ -249,9 +249,14 @@ store round-trips on cancellationToken made THIS loop — the one the field repo specifically: a budget expiry abandons the whole pass, so the watermark does not advance, the clamp is re-derived next cycle, and the hole is re-recorded (merged wider with any already pending) rather than lost. */ + /* #2344: same bound as the enumerated arm, safe by the other route — this + branch does not clamp itself, but query_store's BuildCutoffParameters does. */ + var azureReadFloor = string.Equals(definition.Name, QueryStoreCollector.Instance.Name, StringComparison.Ordinal) + ? WatermarkPolicy.ReadFloor(collectionTime) + : null; context.Watermark = await GetLastCollectedTimeForDatabaseAsync( serverId, definition.TargetTable, definition.WatermarkColumn!, - definition.PerDatabaseWatermarkColumn!, databaseName, dbToken); + definition.PerDatabaseWatermarkColumn!, databaseName, dbToken, azureReadFloor); /* #2111 adaptive shrink, Azure arm — tighten BEFORE BuildQuery: the definition's own clamp only floors OLDER watermarks, so a tighter one @@ -559,9 +564,16 @@ Only query_store (the sole enumeration collector with a per-database timestamp ? null : async (item, ct) => { + /* #2344: bound the read for the ONE collector whose value is clamped on the + next line. Name-guarded rather than applied to every enumerating definition: + a ring-buffer source whose legitimate catch-up spans days must keep reading + its whole history, so the clamp and the bound travel together. */ + var readFloor = string.Equals(definition.Name, QueryStoreCollector.Instance.Name, StringComparison.Ordinal) + ? WatermarkPolicy.ReadFloor(collectionTime) + : null; var raw = await GetLastCollectedTimeForDatabaseAsync( serverId, definition.TargetTable, definition.WatermarkColumn!, - definition.PerDatabaseWatermarkColumn!, item, ct); + definition.PerDatabaseWatermarkColumn!, item, ct, readFloor); var clamped = WatermarkPolicy.ClampCatchup(raw, collectionTime); if (raw.HasValue && clamped != raw) { diff --git a/Lite/Services/RemoteCollectorService.QueryStoreBackfill.cs b/Lite/Services/RemoteCollectorService.QueryStoreBackfill.cs index e9076cfc4..af8f17727 100644 --- a/Lite/Services/RemoteCollectorService.QueryStoreBackfill.cs +++ b/Lite/Services/RemoteCollectorService.QueryStoreBackfill.cs @@ -563,7 +563,7 @@ protected async Task DeleteCollectorStateKeyAsync( /// collected database_states prunes nothing rather than everything. /// /// Which keys comes from the SHARED , - /// deliberately including planwm: that Lite never writes: running one no-op delete is what + /// deliberately including prefixes Lite never writes: running one no-op delete is what /// guarantees that enabling plan capture here later cannot quietly create an orphan class this forgot /// about. Best-effort like its siblings — a failed prune leaves the rows and the next tick retries. /// diff --git a/Lite/Services/RemoteCollectorService.cs b/Lite/Services/RemoteCollectorService.cs index 13bfac501..4b95f344f 100644 --- a/Lite/Services/RemoteCollectorService.cs +++ b/Lite/Services/RemoteCollectorService.cs @@ -1237,18 +1237,35 @@ protected internal static int GetServerId(ServerConnection server) /// (Azure SQL DB per-database XE capture): the newest already-collected value for ONE database, /// so each database's ring buffer dedups against its own history. Null on first run for that /// database or on failure — the caller falls back to the definition's documented window. + /// + /// bounds the read on collection_time (#2344). Null + /// keeps the unbounded behaviour, correct for any reader whose watermark is NOT clamped; pass + /// only from a caller whose value is, and read that method's + /// remarks for why the bound provably changes no answer. Unbounded, this is a MAX over the + /// whole of a database's history every cycle — measured at multiple seconds per database on the + /// Darling twin's larger store, and the same shape here. DuckDB does not partition the way the + /// Postgres store's hypertables do, so the win is min-max index pruning and a smaller scan rather + /// than chunk exclusion, but the predicate is the same and so is the argument for it. /// protected async Task GetLastCollectedTimeForDatabaseAsync( - int serverId, string tableName, string columnName, string databaseColumnName, string databaseName, CancellationToken cancellationToken) + int serverId, string tableName, string columnName, string databaseColumnName, string databaseName, + CancellationToken cancellationToken, DateTime? collectedSince = null) { try { using var conn = _duckDb.CreateConnection(); await conn.OpenAsync(cancellationToken); using var cmd = conn.CreateCommand(); - cmd.CommandText = $"SELECT MAX({columnName}) FROM {tableName} WHERE server_id = $1 AND {databaseColumnName} = $2"; + cmd.CommandText = collectedSince is null + ? $"SELECT MAX({columnName}) FROM {tableName} WHERE server_id = $1 AND {databaseColumnName} = $2" + : $"SELECT MAX({columnName}) FROM {tableName} WHERE server_id = $1 AND {databaseColumnName} = $2 AND collection_time > $3"; cmd.Parameters.Add(new DuckDB.NET.Data.DuckDBParameter { Value = serverId }); cmd.Parameters.Add(new DuckDB.NET.Data.DuckDBParameter { Value = databaseName }); + if (collectedSince is DateTime floor) + { + cmd.Parameters.Add(new DuckDB.NET.Data.DuckDBParameter { Value = floor }); + } + var result = await cmd.ExecuteScalarAsync(cancellationToken); if (result is DateTime dt) return dt; diff --git a/Lite/packages.lock.json b/Lite/packages.lock.json index 271d0b77c..5d7661377 100644 --- a/Lite/packages.lock.json +++ b/Lite/packages.lock.json @@ -63,63 +63,63 @@ }, "Microsoft.Extensions.Hosting": { "type": "Direct", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "tL9FkfV64GPUDSPvwrgyw42LVzsnVAnyrqJEuZVJbODgrQ3eL63zmzEcVWoCHzfgqUhWggzbgAyUCnz/zfI3Pg==", - "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Configuration.Binder": "10.0.10", - "Microsoft.Extensions.Configuration.CommandLine": "10.0.10", - "Microsoft.Extensions.Configuration.EnvironmentVariables": "10.0.10", - "Microsoft.Extensions.Configuration.FileExtensions": "10.0.10", - "Microsoft.Extensions.Configuration.Json": "10.0.10", - "Microsoft.Extensions.Configuration.UserSecrets": "10.0.10", - "Microsoft.Extensions.DependencyInjection": "10.0.10", - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Diagnostics": "10.0.10", - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10", - "Microsoft.Extensions.FileProviders.Physical": "10.0.10", - "Microsoft.Extensions.Hosting.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging.Configuration": "10.0.10", - "Microsoft.Extensions.Logging.Console": "10.0.10", - "Microsoft.Extensions.Logging.Debug": "10.0.10", - "Microsoft.Extensions.Logging.EventLog": "10.0.10", - "Microsoft.Extensions.Logging.EventSource": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10" + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "eIDa/Rl+93aj17gMlFsJJx+LhBvb3CP0Mu1PeVYkDp2Y3S4Jock8UynfGQEcx7lrlq+gKW+ECQJHbro/LTPDEQ==", + "dependencies": { + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Configuration.Binder": "10.0.11", + "Microsoft.Extensions.Configuration.CommandLine": "10.0.11", + "Microsoft.Extensions.Configuration.EnvironmentVariables": "10.0.11", + "Microsoft.Extensions.Configuration.FileExtensions": "10.0.11", + "Microsoft.Extensions.Configuration.Json": "10.0.11", + "Microsoft.Extensions.Configuration.UserSecrets": "10.0.11", + "Microsoft.Extensions.DependencyInjection": "10.0.11", + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Diagnostics": "10.0.11", + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11", + "Microsoft.Extensions.FileProviders.Physical": "10.0.11", + "Microsoft.Extensions.Hosting.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging.Configuration": "10.0.11", + "Microsoft.Extensions.Logging.Console": "10.0.11", + "Microsoft.Extensions.Logging.Debug": "10.0.11", + "Microsoft.Extensions.Logging.EventLog": "10.0.11", + "Microsoft.Extensions.Logging.EventSource": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11" } }, "Microsoft.Extensions.Logging": { "type": "Direct", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "Tf6z5HsL0VDYRTfvsoNrTGHGheCwkTsZBA2FFh5ATJUbkAwug+FFNISJK2gjpUNemlAOoWllAK52HOWCjto3EQ==", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "nUOJwgFkSiLHiVGFpU22pIJtuWYewuSYQ3JVuP/gdK8ASMT807Px+TYQiRWs6uSsOmoyFTaVCwKXTasczV6BpA==", "dependencies": { - "Microsoft.Extensions.DependencyInjection": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10" + "Microsoft.Extensions.DependencyInjection": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11" } }, "ModelContextProtocol": { "type": "Direct", - "requested": "[2.1.0, )", - "resolved": "2.1.0", - "contentHash": "Oa4rU7EL9C2qyFjQj1dx+ysGMzfWDRpM8RRaUMmLGs5vPvfJ9xyz4ZtyF4ychY+Nx1b/auGCqIQLqSz/IpPkKA==", + "requested": "[2.2.0, )", + "resolved": "2.2.0", + "contentHash": "4Pb9u02Nwsp0poueDsqNdyGRojFxOYpljB7zDBsq+aHL+Afou3OgxlBc3GWFVnsRMRJrUtWqDh3s6k2JgPzmrQ==", "dependencies": { "Microsoft.Extensions.Caching.Abstractions": "10.0.10", "Microsoft.Extensions.Hosting.Abstractions": "10.0.10", - "ModelContextProtocol.Core": "[2.1.0]" + "ModelContextProtocol.Core": "[2.2.0]" } }, "ModelContextProtocol.AspNetCore": { "type": "Direct", - "requested": "[2.1.0, )", - "resolved": "2.1.0", - "contentHash": "yhJ8bBXIgrX0mAgRYRgzcbH3bLdv3MDSkG52utRW9EAAtQrPw/g7Q/T6EurxKV+L+Zefv8VVUYcNbXEOd9GgfA==", + "requested": "[2.2.0, )", + "resolved": "2.2.0", + "contentHash": "3JelDMuFIwFzXybsh6K30G6wXu5gmqKnuRnieqgBMuR0FkqqXRv4B+NYZuPgthSyNq5UwRyvv2U2VJBE3RD3PQ==", "dependencies": { - "ModelContextProtocol": "[2.1.0]" + "ModelContextProtocol": "[2.2.0]" } }, "ScottPlot.WPF": { @@ -259,215 +259,215 @@ }, "Microsoft.Extensions.Configuration.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "5Vnd2I75DmZCVEjSynIdJ/0EGafgnLQwgR3t2C2/fkjx/nRG+cLwxLLdInoHeCEpkD5K4Ov/g9ZCRYrl4TRsaA==", + "resolved": "10.0.11", + "contentHash": "fVi053xdpda9Em7vSkmgVxO/PtgC2m78ekReKWsgcyskqY0U82Bz/MONwxpGzI0hElYKJfw+fupqMVeKW3fSaA==", "dependencies": { - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Configuration.Binder": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "GqmN2o1CkJvk7uWp+p4CwBYW0w/zfoEbvsiFDbO2G8l1Uz+mrDAbAcZiXhU2lufKPby1cjAUdd5GTWpebYOkOA==", + "resolved": "10.0.11", + "contentHash": "rFn8RuszZn3qquPVkDytMUlPc2+rXl9MCoygwc1XmAgC5vg5/oXJ8hkOosOrLoBLsqdTy4lFwP6iQdPS9uSYOA==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Configuration.CommandLine": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "33cBeR2HRbzHUTtmcmLdNOApneNGcymwwL4arHuotgVK9Frba8kcDTrvVTj7cSCmF1R9OiSbZH0KxNOwab3HUg==", + "resolved": "10.0.11", + "contentHash": "1KHr/1L56llwQ/yI0tAisEA31UpPsn8aasjASIwELOaN4JIUcbjuQBMdFOIzfNBBeULoUa0XfBe5QDtRRUY+fg==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Configuration.EnvironmentVariables": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "KRfFSSCV58vEdU7mPED/YMzeovIWF5P0g8s9K8n9HEfy0/WzMq37SrPdXdFN5/dFT/rPMHpF7AvpoXHckbcBFg==", + "resolved": "10.0.11", + "contentHash": "KICyU3eVi5jvloKm01EXV69L97H/zkhISVtV98cIuzuFOxNx3xTUVcXqvWTz3aq7OvUuDB/MFlPFjmxRaKF7/A==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Configuration.FileExtensions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "ZOhZYwvbXGTgGVRwswIirofEMVHuWdxjdh0JeUZXwaF9cgcjXdz/t0ELtgaevw7ezTyv47yPNCgGreWtLkn3IQ==", + "resolved": "10.0.11", + "contentHash": "mDW7KVFB05M6jiRUyaZiOMWhS31n5HlSZwoYctHAZAucD4sMDJ70IxOmkGDt6RpstchD+keWBjhdzcMpSkWvWQ==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10", - "Microsoft.Extensions.FileProviders.Physical": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11", + "Microsoft.Extensions.FileProviders.Physical": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Configuration.UserSecrets": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "1s1sKFTk/Foam64JY6+m/diH8drL3Wx6V3gtSd5v1IEZtszZYyc1pW8uRnMblzpNiR0l0t8gGk7tXj3xHzFgdg==", + "resolved": "10.0.11", + "contentHash": "BRliLdUowglV8GS+J1G/QsSofCJYYFg3U8QZx0ACRn+a91az/Qnpy+h6PyHS94WgV2TSazX5D/cuuk6wnCJatw==", "dependencies": { - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Configuration.Json": "10.0.10", - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10", - "Microsoft.Extensions.FileProviders.Physical": "10.0.10" + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Configuration.Json": "10.0.11", + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11", + "Microsoft.Extensions.FileProviders.Physical": "10.0.11" } }, "Microsoft.Extensions.DependencyInjection": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "ANyvsgkNBRvcJh2XLgn8veGmajf+8m0AbKK+HPWdRL1yraSNVVSmQhFntLtdz/C795jxqqup+k05cs/3jZQPOA==", + "resolved": "10.0.11", + "contentHash": "PSmotV19c7E3lKed++uYo1kSiXFI+uTl37CBSrhq+CfLC3FCHjG7R91+xPnNehQfHS1b0Tzo/CCLPWH3qaEheg==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11" } }, "Microsoft.Extensions.DependencyInjection.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "z/2xXlFw2aLGjHyEm6E0tQ+In6VfzQzTrtArbQ2c0TQE16ZbyDCMGPvaUT9I0s8rgy9sRWlU2P9waW37qV04qA==" + "resolved": "10.0.11", + "contentHash": "/a1aJz4m7ylhEDf25ugQChLQoN5XwoGjWw/BoR/ZWWKsO1v4DdJElS1uyngahz4B/eOzjFk1KNTkarRLE5wsIg==" }, "Microsoft.Extensions.Diagnostics": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "Kr/e7lUf4+N8tacbqJ2Ctwe/HarKdAc9ZkgKVVqvtJDBKbez+T/KnUwu82KSlnBp/SrpBcxc7u7xkE2oUZT/5Q==", + "resolved": "10.0.11", + "contentHash": "HT70uGPxMLqqnOzKMcnQtDmeV4r0KHr4qVCLhP7SXil9jMEm8sQXwcybxVVFGXZJ1V44xV0mLqQ54aZbcR2OiQ==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Diagnostics.Abstractions": "10.0.10", - "Microsoft.Extensions.Options.ConfigurationExtensions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Diagnostics.Abstractions": "10.0.11", + "Microsoft.Extensions.Options.ConfigurationExtensions": "10.0.11" } }, "Microsoft.Extensions.Diagnostics.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "9uWiKpeOVac355STyChWR/pliFX/5CeLqChW9kKsaxyDH4EUTZxMkT4Jwp/J/peLm0GBFmSX5c0WCse3yCnq1Q==", + "resolved": "10.0.11", + "contentHash": "se7Kx8QpJEt+nf26L4qIVAofGTDr1wbexxsh/Fm3Xc04xUkqUXK06KUS7FLwSQYSjqb7q9n+T7MEcXYBhI1Y5g==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11" } }, "Microsoft.Extensions.FileProviders.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "c5zqFCY9DiIpMovLd7/d/CTiEtrMOuQ639dhv3PABtKQIKNQikSHwQt8+N679uii9q+B55lgK28Uv64FOwEu8w==", + "resolved": "10.0.11", + "contentHash": "JOjac6SQQgZmdmB8WGEw61/7siqMZoWJMkmq2p1goJGxqI59lO6oB4bOl0jNsbaPBdYy5Mlkb+6U7T4+CjnD8Q==", "dependencies": { - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.FileProviders.Physical": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "jhJAyo38kSrH3ARvWUk0h8itogVnQu2DCZuPo+s0Z+tXes0ugTxMPaHYzap85785eHQmPFqD9TYERqBbtGxn/w==", + "resolved": "10.0.11", + "contentHash": "Tq/UqMaczePv9yWwSsJZRgKtgA46djVR5xHj/lZBCueQ3ag8f9v5mu0EdhrNx7tXxNk+Y9OurG2oKuSKINjr0A==", "dependencies": { - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10", - "Microsoft.Extensions.FileSystemGlobbing": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11", + "Microsoft.Extensions.FileSystemGlobbing": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.FileSystemGlobbing": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "jSOCVxEwCd4Aq925kJVz1kSO1EpX2OHYKL04qVREXkDU7Ce3pVDdHPYm+fEy8y/th2kJf/DAstRHpJAqoNWP8w==" + "resolved": "10.0.11", + "contentHash": "2i6rtW/B5rCnWCnhdmWWEmaM9O0HD0zsPY9eRqa++y4tclI3Uw8zvGbBvhY/LjAdtf8gUHhUPcAWj3DRlWMXmQ==" }, "Microsoft.Extensions.Hosting.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "5LugpYGHk+mkn0a8IZgcyfBca8PCTAU9RQFoMrTdtOOidq88M2SI5f3px6ugnzgxC+eTkvYYJi8pzlUnG5xdAQ==", + "resolved": "10.0.11", + "contentHash": "pwtpF7iF/NNaOBcX+pvMZ7y2+JAVbH5KkNrH9uMZtuVxVJsFTDiWiCR7Tk3HVptsAijaetipUZVRVK2LLq+nvA==", "dependencies": { - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Diagnostics.Abstractions": "10.0.10", - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10" + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Diagnostics.Abstractions": "10.0.11", + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Logging.Configuration": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "cLrqxkuEfcilZ8SjK+9KAnpLk9lOoMPaOokF+wRUYie+iUEcdX4/p/+gJkt0BYgWLthjpBUCkVTBI6Kxg0nsOw==", + "resolved": "10.0.11", + "contentHash": "S7LvLeVHKNPaY2NMyxW7c2TBGsLgxoSUBCV5Ev5iN8kgC7EPR2UB7eW7vHsElGMcIUDwRmoxLfvGDynCn3q6EA==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Configuration.Binder": "10.0.10", - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10", - "Microsoft.Extensions.Options.ConfigurationExtensions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Configuration.Binder": "10.0.11", + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11", + "Microsoft.Extensions.Options.ConfigurationExtensions": "10.0.11" } }, "Microsoft.Extensions.Logging.Console": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "VIlNzPwPS0GeQVSmCqqo36ugryX3LpE9ul6gEkks5VLET3weH/XMLeWmclwfoGn4Nxi2mwVibB+OZBVJ9tDqvg==", + "resolved": "10.0.11", + "contentHash": "dFc0yDudyD1iIg6z9XT7ofsT3hVO7Y4ylrxGHIVRR0GaZ4CUk4ujOrMoy7wWEdNHZhvJooySg6hZpOxyS8zEVA==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging.Configuration": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging.Configuration": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11" } }, "Microsoft.Extensions.Logging.Debug": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "8+TZBnV5fgBXoVNJ5ROSErUwYogk4hOgV7c2HWK1u5cqKGmiUTUn7+KqZ35iQu8e/B7Ykccyz5OTjdXcidNZ9g==", + "resolved": "10.0.11", + "contentHash": "wr+j1bjdFXhc8lKTLoq+RbwFM8M+orcMS9xrcqLmDGOxJcXpKizEeE5h6v/GKwCZV02FmhaA7OlNjoq072jZpQ==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Logging.EventLog": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "0RE4951AzQ+YD4gVrvbq0BhdsiBgSDo44yM7+QBZ2mrmMJeNjY+teCIYfUjqDPVYnKs0HR6SkkhgrX1YgXZq3Q==", + "resolved": "10.0.11", + "contentHash": "Eck9GpCCpvZ3f6L7IUlN+mPtRVefnf7PsiIG5vi61QawPtLNCEAv2TPD/M3SojcU0PFaef+BxiVGPOShFHtDog==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10", - "System.Diagnostics.EventLog": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11", + "System.Diagnostics.EventLog": "10.0.11" } }, "Microsoft.Extensions.Logging.EventSource": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "85SAPwXhJtdBInzN2k7SChiFiBGh3KOWay5AfoY+GREF6P7oZA98+ST2p7Z9384iLKYjkZSKIZ/FqIO5aojtNw==", + "resolved": "10.0.11", + "contentHash": "hs6QWECLLohi2VKqUvSGRUvrg7eXR1DqKL95Jrtz3cdD2g2nBA+yJPdRQLZ7SLmnTZWycxfMDK2s0ho+rfst5w==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Options": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "srnhnk7nE8krBiIXp71LvBmKBtraBONWSRzdjJgRv1Ko9Mp8IVNqv4vIS9hGeVteBig8aQkva9ZG+sC+o5sVcA==", + "resolved": "10.0.11", + "contentHash": "eY1GAKcTfD2maP27J84X9IovT3yjHJ2dVDzPmDg6/XqYvt3jMzJhtfQCLjG9pVsZGAd+8DQ2QrjaDcs2+VQLGw==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Options.ConfigurationExtensions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "tnBmu/LwF25ZQK+HBNCu2xrwnkKoB/XEbJyooGGoYxHrhvxbSKi7eOFiJ4AXBy/QU4vtCvCJfoi8k9Ej72qzOQ==", + "resolved": "10.0.11", + "contentHash": "syEhXQ/sEaSBFaqzlp9gDGHX/nk6gkQkh1sIUpBO1mlBj3Phu1rmb4ML1uCiyPW9N6Kxfxv3y5FGObC+bV01Qw==", "dependencies": { - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Configuration.Binder": "10.0.10", - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Configuration.Binder": "10.0.11", + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Primitives": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "5wu/GrYVd8mG2DVUw3vFJzF+O336TyTGg/Kmcgw9bfwYhCoFiV5lR5QeEmKecJyrW4W54nMfD3p3589E8a7czQ==" + "resolved": "10.0.11", + "contentHash": "SXcz+kF+4Oo9b1+55zntpJFYfwb1jw66ioxptyNOOTDc8g2FHnBFWjZpsWfCvZIhzr0x+4e2trVTs4OKwQfBtw==" }, "Microsoft.Identity.Client": { "type": "Transitive", @@ -553,8 +553,8 @@ }, "ModelContextProtocol.Core": { "type": "Transitive", - "resolved": "2.1.0", - "contentHash": "cU/urrhRxE4/iSyBIJI7QOaFqSP1FOEnwEHsct9n6t6/XluCAFD9iqnrPkBAsEYr+f/G4tVQ21U+6wN/6fQvOg==", + "resolved": "2.2.0", + "contentHash": "FeBfXU6T8k+jw4afg4sfxdEX2rL/e5oKOk9ROOGztu9k47+7Bz08sdaToYt2XvMY1opNbwxYQOFMj6wH9TInhA==", "dependencies": { "Microsoft.Extensions.AI.Abstractions": "10.8.3", "Microsoft.Extensions.Logging.Abstractions": "10.0.10" @@ -732,8 +732,8 @@ }, "System.Diagnostics.EventLog": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "OvGz3PrzuAI/Sj7LTcXcCe3FClRI1IyRMZjNONcZtFh+Ww7nAtSh4kh08r8KVe/xxkXJPjR0Y1jF7H+N42d4xQ==" + "resolved": "10.0.11", + "contentHash": "QTXEoQBzz00SFWbo7nAg1Ogd4f99lwqcO9uAJ7MYSLEUR28f6As32QktrqG2Fr9cfAfd1GjLyGYspE7Ipj7P6w==" }, "System.IdentityModel.Tokens.Jwt": { "type": "Transitive", @@ -765,14 +765,14 @@ "type": "Project", "dependencies": { "CredentialManagement": "[1.0.2, )", - "Microsoft.Extensions.Logging.Abstractions": "[10.0.10, )", - "ModelContextProtocol": "[2.1.0, )" + "Microsoft.Extensions.Logging.Abstractions": "[10.0.11, )", + "ModelContextProtocol": "[2.2.0, )" } }, "performancemonitor.notifications": { "type": "Project", "dependencies": { - "Microsoft.Extensions.Logging.Abstractions": "[10.0.10, )", + "Microsoft.Extensions.Logging.Abstractions": "[10.0.11, )", "PerformanceMonitor.Analysis": "[1.0.0, )" } }, @@ -780,7 +780,8 @@ "type": "Project", "dependencies": { "Microsoft.Data.SqlClient": "[7.0.2, )", - "PerformanceMonitor.Common": "[1.0.0, )" + "PerformanceMonitor.Common": "[1.0.0, )", + "System.Security.Cryptography.ProtectedData": "[10.0.11, )" } }, "performancemonitor.ui": { @@ -793,34 +794,40 @@ }, "Microsoft.Extensions.Configuration": { "type": "CentralTransitive", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "plJWK2zpWuuyxI8F8s2scx6Je7N1Ajjs6HvYUGKwRnDMWIVIz9FHwAkiT7ASgrvAOd10T0FPVlh9BzAJJME+jg==", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "wlhRqZW8LcJPa+vk2oLAc/REXDItHtkFQdf/QcXYGZbZOO13izcsKY1pCvuFQYwUiZD+hwSZwsKASjqT+BNaVg==", "dependencies": { - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Configuration.Json": { "type": "CentralTransitive", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "uvJ6sHwjgrkMEJOgiC76G0mcZGXerwyyWkwX34EOjCbxKG6TCtfAoqDKAMsCvEBf9HxjlGQEgqsSMOGCmGBf+A==", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "nSPrT8U/cNoB4coqkmnanAMK9PsL7lsjG+LLUKEwHRFwS6E78b8S1wdv/y88EOxBhasWov1rLd7RTHmmsYPOLg==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Configuration.FileExtensions": "10.0.10", - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Configuration.FileExtensions": "10.0.11", + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Logging.Abstractions": { "type": "CentralTransitive", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "zkFxGYUvdxAvIKTyXHrmW+Sux53D4SezD9dMyZ6hrwwzPQJNuwCRy1f5W7AvYTqacEGhWF2XderRQG1OvbV8og==", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "Ljd0Uxoq5XpScD2Bg0nM/r3mwx7Ao5Uq24eo2ARxbGvqJ7Zht6rt2cJtwVRH4Cv+1ZVMdXz6TB43KbpmsxRrvQ==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11" } + }, + "System.Security.Cryptography.ProtectedData": { + "type": "CentralTransitive", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "PNoxCTPb+Tlux+GJyq4c89ddYdpioVSqfGx8pqOF6shCSKwUNNctXhQtRkCICjbJmGrJJsW8NY52kVYO/b8mlQ==" } } } diff --git a/PerformanceMonitor.Alerting/AlertContextBuilders.cs b/PerformanceMonitor.Alerting/AlertContextBuilders.cs index b5e78efa5..6cc61cdd8 100644 --- a/PerformanceMonitor.Alerting/AlertContextBuilders.cs +++ b/PerformanceMonitor.Alerting/AlertContextBuilders.cs @@ -228,6 +228,175 @@ public static IReadOnlyList DeadlockIncidents( return GroupDeadlocks(serverName, filtered).Select(g => g.Incident).ToList(); } + /// + /// #2362: the observation lists for the five remaining fingerprinted alerts, mirroring + /// and . + /// + /// Uncapped, and that is the whole point. Each context builder renders a capped subset + /// (3 for long-running queries and anomalous jobs, 5 for the rest) because a card with fifty entries + /// helps nobody. The render cap is a display budget; a fingerprint outside it still has a live incident, + /// and observing only the displayed subset would reset the total of anything that fell out of the top N — + /// a subtler version of the undercount #2216 exists to fix, reintroduced by the fix for it. + /// + /// Each is a pure function of a list, so the same builder serves both callers: the check passes the + /// FULL list to observe, the context builder passes its capped shown to render. One grouping rule, + /// two inputs, no way for the two to disagree about what a fingerprint is. + /// + public static IReadOnlyList LongRunningQueryIncidents( + string serverName, IReadOnlyList? queries) + { + if (queries is null || queries.Count == 0) return Array.Empty(); + + /* #1140: dedup key = query_hash (stable across literals/plans). Null hash -> no incident. */ + return queries + .Select(q => AlertFingerprint.ForKey(serverName, AlertFingerprint.Query, q.QueryHash ?? "", + string.IsNullOrEmpty(q.DatabaseName) ? Array.Empty() : new[] { q.DatabaseName }, + database: q.DatabaseName)) + .Where(i => i is not null).Select(i => i!).ToList(); + } + + /// + public static IReadOnlyList VolumeFreeSpaceIncidents( + string serverName, IReadOnlyList? volumes) + { + if (volumes is null || volumes.Count == 0) return Array.Empty(); + + /* #1140: dedup key per volume (the drive/mount point). */ + return volumes + .Select(v => AlertFingerprint.ForKey(serverName, AlertFingerprint.Disk, v.MountPoint, new[] { v.MountPoint })) + .Where(i => i is not null).Select(i => i!).ToList(); + } + + /// + public static IReadOnlyList PvsPressureIncidents( + string serverName, IReadOnlyList? databases) + { + if (databases is null || databases.Count == 0) return Array.Empty(); + + return databases + .Select(d => AlertFingerprint.ForKey(serverName, AlertFingerprint.Database, d.DatabaseName, new[] { d.DatabaseName }, + database: d.DatabaseName)) + .Where(i => i is not null).Select(i => i!).ToList(); + } + + /// + public static IReadOnlyList AnomalousJobIncidents( + string serverName, IReadOnlyList? jobs) + { + if (jobs is null || jobs.Count == 0) return Array.Empty(); + + /* #1140: dedup key per job (job name, scoped to the instance via serverName). */ + return jobs + .Select(j => AlertFingerprint.ForKey(serverName, AlertFingerprint.Job, j.JobName, new[] { j.JobName })) + .Where(i => i is not null).Select(i => i!).ToList(); + } + + /// + public static IReadOnlyList FailedJobIncidents( + string serverName, IReadOnlyList? jobs) + { + if (jobs is null || jobs.Count == 0) return Array.Empty(); + + return jobs + .Select(j => AlertFingerprint.ForKey(serverName, AlertFingerprint.Job, j.JobName, new[] { j.JobName })) + .Where(i => i is not null).Select(i => i!).ToList(); + } + + /// + /// #2349: the file-growth observation list. UNCAPPED, like every other *Incidents builder and for + /// the reason #2362 established — the card renders a capped subset, and observing only what is displayed + /// resets the total of any file that fell out of the top N. + /// + /// Fingerprinted on the FILE, not the database: a database with eight tempdb data files that all grow + /// together is eight files and one problem, but a log file that runs away while its data files sit still is + /// a different incident from its neighbours, and collapsing them on database name would merge the two. + /// + public static IReadOnlyList FileGrowthIncidents( + string serverName, IReadOnlyList? files) + { + if (files is null || files.Count == 0) return Array.Empty(); + + return files + .Select(f => AlertFingerprint.ForKey( + serverName, AlertFingerprint.Disk, $"{f.DatabaseName}|{f.FileName}", + new[] { $"{f.DatabaseName}.{f.FileName}" }, + database: f.DatabaseName)) + .Where(i => i is not null).Select(i => i!).ToList(); + } + + /// + /// #2349: the files breaching either gate, worst first. Both gates are applied HERE rather than in the + /// engine so the render path, the observation path and the decision can never disagree about which files + /// are involved. + /// + /// Ordered by how much of its volume the file occupies, because that is the one number that says how + /// close this is to becoming a Volume Free Space page — a 40 GB rise on a 4 TB volume is less urgent + /// than a 10 GB file that is now 80% of a small one. + /// + public static List GetBreachedFiles( + IReadOnlyList? files, int riseMb, int volumePercent) + { + if (files is null || files.Count == 0) return new List(); + + var breached = files + .Where(f => + (riseMb > 0 && f.GrowthMb >= riseMb) + || (volumePercent > 0 && f.VolumeTotalMb > 0 && f.VolumePercent >= volumePercent)) + .OrderByDescending(f => f.VolumePercent) + .ThenByDescending(f => f.GrowthMb) + .ToList(); + + return breached; + } + + /// #2349: the alert card. Renders the top few by the same order + /// produced, and names the fields an operator needs to act without opening the Viewer — including + /// is_percent_growth, which surfaces a percent-autogrowth misconfiguration for free. + public static AlertContext? BuildFileGrowthContext( + string serverName, List files, + Func, IReadOnlyList>? decorateIncidents = null) + { + if (files.Count == 0) return null; + + var context = new AlertContext(); + var shown = files.GetRange(0, Math.Min(5, files.Count)); + foreach (var f in shown) + { + var fields = new List<(string, string)> + { + /* #2109: the database as a discrete fact, not only in the heading. */ + ("Database", f.DatabaseName), + ("File", f.FileName), + ("Physical Name", f.PhysicalName), + ("Size", $"{f.TotalSizeGb:F1} GB"), + ("Growth", $"{f.GrowthGb:F1} GB in {f.GrowthWindowMinutes:F0} min ({f.GrowthMbPerHour:F0} MB/hr)"), + ("Volume", string.IsNullOrEmpty(f.VolumeMountPoint) ? "(unknown)" : f.VolumeMountPoint), + ("Volume Free", $"{f.VolumeFreeMb / 1024.0:F1} GB"), + ("File % of Volume", $"{f.VolumePercent:F0}%"), + /* A percent autogrowth on a large file is its own finding: each growth is bigger than the last, + which is exactly how a file gets away from someone. WS3 knows about the pattern and does not + alert on it. */ + ("Autogrowth", f.IsPercentGrowth + ? $"{f.GrowthPct:F0}% (percent growth)" + : f.AutoGrowthMb is double mb ? $"{mb:F0} MB" : "(unknown)"), + }; + + if (f.MaxSizeMb is double max) + { + fields.Add(("Max Size", max < 0 ? "Unlimited" : $"{max / 1024.0:F1} GB")); + } + + context.Details.Add(new AlertDetailItem + { + Heading = $"{f.DatabaseName}.{f.FileName} — {f.TotalSizeGb:F1} GB ({f.VolumePercent:F0}% of {f.VolumeMountPoint})", + Fields = fields + }); + } + + AlertIncidentRenderer.Apply(context, Decorate(FileGrowthIncidents(serverName, shown).ToList(), decorateIncidents)); + return context; + } + /* Excluded databases drop their rows; rows with no database always pass. Shared by the render path and #2216's observation path so the two can never disagree about which rows exist. */ private static IReadOnlyList FilterBlocking( @@ -368,7 +537,9 @@ public static bool IsDeadlockExcluded(DeadlockAlertRow row, IReadOnlyList queries) + public static AlertContext? BuildLongRunningQueryContext( + string serverName, List queries, + Func, IReadOnlyList>? decorateIncidents = null) { if (queries.Count == 0) return null; @@ -400,10 +571,7 @@ public static bool IsDeadlockExcluded(DeadlockAlertRow row, IReadOnlyList no incident. */ - AlertIncidentRenderer.Apply(context, shown - .Select(q => AlertFingerprint.ForKey(serverName, AlertFingerprint.Query, q.QueryHash ?? "", - string.IsNullOrEmpty(q.DatabaseName) ? System.Array.Empty() : new[] { q.DatabaseName })) - .Where(i => i is not null).Select(i => i!).ToList()); + AlertIncidentRenderer.Apply(context, Decorate(LongRunningQueryIncidents(serverName, shown).ToList(), decorateIncidents)); return context; } @@ -427,7 +595,9 @@ public static string FormatLowDiskThreshold(double thresholdPercent, double thre return parts.Count > 0 ? string.Join(" / ", parts) : "—"; } - public static AlertContext? BuildVolumeFreeSpaceContext(string serverName, List volumes) + public static AlertContext? BuildVolumeFreeSpaceContext( + string serverName, List volumes, + Func, IReadOnlyList>? decorateIncidents = null) { if (volumes.Count == 0) return null; @@ -448,9 +618,7 @@ public static string FormatLowDiskThreshold(double thresholdPercent, double thre } /* #1140: dedup key per volume (the drive/mount point). */ - AlertIncidentRenderer.Apply(context, shown - .Select(v => AlertFingerprint.ForKey(serverName, AlertFingerprint.Disk, v.MountPoint, new[] { v.MountPoint })) - .Where(i => i is not null).Select(i => i!).ToList()); + AlertIncidentRenderer.Apply(context, Decorate(VolumeFreeSpaceIncidents(serverName, shown).ToList(), decorateIncidents)); return context; } @@ -476,7 +644,9 @@ public static string FormatPvsThreshold(double thresholdPercent, double floorGb) : $"{thresholdPercent}% of database"; } - public static AlertContext? BuildPvsPressureContext(string serverName, List databases) + public static AlertContext? BuildPvsPressureContext( + string serverName, List databases, + Func, IReadOnlyList>? decorateIncidents = null) { if (databases.Count == 0) return null; @@ -508,9 +678,7 @@ public static string FormatPvsThreshold(double thresholdPercent, double floorGb) }); } - AlertIncidentRenderer.Apply(context, shown - .Select(d => AlertFingerprint.ForKey(serverName, AlertFingerprint.Database, d.DatabaseName, new[] { d.DatabaseName })) - .Where(i => i is not null).Select(i => i!).ToList()); + AlertIncidentRenderer.Apply(context, Decorate(PvsPressureIncidents(serverName, shown).ToList(), decorateIncidents)); return context; } @@ -535,7 +703,9 @@ public static string FormatPvsThreshold(double thresholdPercent, double floorGb) return context; } - public static AlertContext? BuildAnomalousJobContext(string serverName, List jobs) + public static AlertContext? BuildAnomalousJobContext( + string serverName, List jobs, + Func, IReadOnlyList>? decorateIncidents = null) { if (jobs.Count == 0) return null; @@ -558,13 +728,13 @@ public static string FormatPvsThreshold(double thresholdPercent, double floorGb) } /* #1140: dedup key per job (job name, scoped to the instance via serverName). */ - AlertIncidentRenderer.Apply(context, shown - .Select(j => AlertFingerprint.ForKey(serverName, AlertFingerprint.Job, j.JobName, new[] { j.JobName })) - .Where(i => i is not null).Select(i => i!).ToList()); + AlertIncidentRenderer.Apply(context, Decorate(AnomalousJobIncidents(serverName, shown).ToList(), decorateIncidents)); return context; } - public static AlertContext? BuildFailedJobContext(string serverName, List jobs) + public static AlertContext? BuildFailedJobContext( + string serverName, List jobs, + Func, IReadOnlyList>? decorateIncidents = null) { if (jobs.Count == 0) return null; @@ -585,9 +755,7 @@ public static string FormatPvsThreshold(double thresholdPercent, double floorGb) /* #1140: dedup key per job (job name, scoped to the instance via serverName) — mirrors BuildAnomalousJobContext so two distinct failed jobs are distinct incidents under the #1154 per-fingerprint cooldown instead of coalescing on the metric key. */ - AlertIncidentRenderer.Apply(context, shown - .Select(j => AlertFingerprint.ForKey(serverName, AlertFingerprint.Job, j.JobName, new[] { j.JobName })) - .Where(i => i is not null).Select(i => i!).ToList()); + AlertIncidentRenderer.Apply(context, Decorate(FailedJobIncidents(serverName, shown).ToList(), decorateIncidents)); return context; } diff --git a/PerformanceMonitor.Alerting/AlertEngine.cs b/PerformanceMonitor.Alerting/AlertEngine.cs index 2e9975fea..edf409a25 100644 --- a/PerformanceMonitor.Alerting/AlertEngine.cs +++ b/PerformanceMonitor.Alerting/AlertEngine.cs @@ -67,6 +67,17 @@ shared so Lite's existing config_edge_trigger_watermarks rows seed this engine u public const string BlockingWatermarkMetric = "Blocking Detected"; public const string DeadlockWatermarkMetric = "Deadlocks Detected"; + /* #2362: the remaining fingerprinted alerts. Same names their FireAsync/mute contexts use, so the + accumulator's per-fingerprint state lives under the metric an operator already knows. Forced Plan + Failing is deliberately absent: it builds a bare context and never calls AlertIncidentRenderer.Apply, + so it carries no dedup keys for the accumulator to key on. */ + public const string LongRunningQueryWatermarkMetric = "Long-Running Query"; + public const string VolumeFreeSpaceWatermarkMetric = "Volume Free Space"; + public const string PvsWatermarkMetric = "Version Store (PVS)"; + public const string FileGrowthWatermarkMetric = "Database File Growth"; + public const string AnomalousJobWatermarkMetric = "Long-Running Job"; + public const string FailedJobWatermarkMetric = "Failed Agent Job"; + /// /// The rolling window both count gates read, in hours (#1091's "in the last hour"). Named because /// #2216's occurrence accumulator has to agree with it: its staleness horizon is what stops a row @@ -134,6 +145,8 @@ shared so Lite's existing config_edge_trigger_watermarks rows seed this engine u for the resolved transition, and the PvsAlertGate worsening watermark. */ private readonly ConcurrentDictionary _lastPvsAlert = new(); private readonly ConcurrentDictionary _activePvsAlert = new(); + private readonly ConcurrentDictionary _lastFileGrowthAlert = new(); + private readonly ConcurrentDictionary _activeFileGrowthAlert = new(); private readonly ConcurrentDictionary _lastAlertedPvsPercent = new(); /* Rolling-count edge-trigger watermarks (#1091) — Lite's MainWindow.xaml.cs:103-104; @@ -262,6 +275,7 @@ private async Task EvaluateCoreAsync(AlertServerSnapshot snaps await CheckTempDbSpaceAsync(key, serverName, now, alertCooldown, suppressed, ct); bool lowDiskConditionPresent = await CheckLowDiskAsync(key, serverName, now, alertCooldown, suppressed, ct); await CheckPvsPressureAsync(key, serverName, now, alertCooldown, suppressed, ct); + await CheckFileGrowthAsync(key, serverName, now, alertCooldown, suppressed, ct); await CheckAnomalousJobsAsync(key, serverName, now, alertCooldown, suppressed, ct); bool failedJobConditionPresent = await CheckFailedJobsAsync(snapshot, key, serverName, now, alertCooldown, suppressed, ct); await CheckDatabaseStateAsync(key, serverName, now, alertCooldown, suppressed, ct); @@ -620,6 +634,11 @@ incident is not null { TotalOccurrences = state.TotalOccurrences, IncidentStartedUtc = state.IncidentStartedUtc, + /* #2361: LastObservedUtc already exists on the state -- it is the value the + staleness horizon compares against so a flat incident does not expire itself. + It simply never reached the incident. This is a projection, not a new + measurement, which is why it rides the same hook as the two above it. */ + LastEventUtc = state.LastObservedUtc, } : incident!); } @@ -913,6 +932,12 @@ private async Task CheckLongRunningQueriesAsync( _settings.ExcludedDatabases, ct); + /* #2362: observe every sweep, OUTSIDE the fire branch — the #2216 reasoning, which applies + identically here: counting only at delivery lets an event that ages out during a cooldown mask + an arrival. The list is UNCAPPED while the render below is capped, so a fingerprint outside the + displayed top N keeps its total instead of restarting. */ + var lrqOccurrences = await ObserveOccurrencesAsync( + key, LongRunningQueryWatermarkMetric, AlertContextBuilders.LongRunningQueryIncidents(serverName, longRunning), now); if (longRunning.Count > 0) { _activeLongRunningQueryAlert[key] = true; /* :350 */ @@ -934,7 +959,7 @@ private async Task CheckLongRunningQueriesAsync( bool isMuted = _isAlertMuted(muteCtx); /* :365 */ _lastLongRunningQueryAlert[key] = now; /* :366 */ - var lrqContext = AlertContextBuilders.BuildLongRunningQueryContext(serverName, longRunning); /* :379 */ + var lrqContext = AlertContextBuilders.BuildLongRunningQueryContext(serverName, longRunning, lrqOccurrences.Decorate); /* :379 */ var detailText = AlertContextBuilders.ContextToDetailText(lrqContext); /* :380 */ /* :382-392. ShortMessage = the toast body of :374. */ @@ -951,7 +976,8 @@ await FireAsync(new AlertOutcome( } else if (_activeLongRunningQueryAlert.TryGetValue(key, out var wasLongRunning) && wasLongRunning) /* :395 */ { - _activeLongRunningQueryAlert[key] = false; /* :397 */ + _activeLongRunningQueryAlert[key] = false; + await ClearOccurrencesAsync(key, LongRunningQueryWatermarkMetric); /* :397 */ if (!suppressed) /* :398 */ { await NotifyResolutionAsync(new AlertResolution( @@ -1054,6 +1080,12 @@ private async Task CheckLowDiskAsync( var breached = AlertContextBuilders.GetBreachedVolumes(volumes, _settings.LowDiskThresholdPercent, _settings.LowDiskThresholdGb); /* :481 */ conditionPresent = breached.Count > 0; /* :487 — feeds the sweep result */ + /* #2362: observe every sweep, OUTSIDE the fire branch — the #2216 reasoning, which applies + identically here: counting only at delivery lets an event that ages out during a cooldown mask + an arrival. The list is UNCAPPED while the render below is capped, so a fingerprint outside the + displayed top N keeps its total instead of restarting. */ + var lowDiskOccurrences = await ObserveOccurrencesAsync( + key, VolumeFreeSpaceWatermarkMetric, AlertContextBuilders.VolumeFreeSpaceIncidents(serverName, breached), now); if (breached.Count > 0) { var worst = breached[0]; /* :489 */ @@ -1070,7 +1102,7 @@ private async Task CheckLowDiskAsync( _lastLowDiskAlert[key] = now; /* :501 */ _lastAlertedLowDiskPercent[key] = worst.FreePercent; /* :502 */ - var lowDiskContext = AlertContextBuilders.BuildVolumeFreeSpaceContext(serverName, breached); /* :515 */ + var lowDiskContext = AlertContextBuilders.BuildVolumeFreeSpaceContext(serverName, breached, lowDiskOccurrences.Decorate); /* :515 */ /* :516-522 — #1136: grade WARNING normally, CRITICAL when critically low. */ if (lowDiskContext is not null && LowDiskAlertGate.IsCriticallyLow( worst.FreePercent, worst.FreeGb, _settings.DiskCriticalFreePercent, _settings.DiskCriticalFreeGb)) @@ -1093,7 +1125,8 @@ await FireAsync(new AlertOutcome( } else if (_activeLowDiskAlert.TryGetValue(key, out var wasLowDisk) && wasLowDisk) /* :538 */ { - _activeLowDiskAlert[key] = false; /* :540 */ + _activeLowDiskAlert[key] = false; + await ClearOccurrencesAsync(key, VolumeFreeSpaceWatermarkMetric); /* :540 */ _lastAlertedLowDiskPercent.TryRemove(key, out _); /* :541 */ if (!suppressed) /* :542 */ { @@ -1141,6 +1174,12 @@ private async Task CheckPvsPressureAsync( var databases = await _readAdapter.GetPvsPressureAsync(key, ct); var breached = AlertContextBuilders.GetBreachedPvsDatabases(databases, _settings.PvsThresholdPercent, _settings.PvsFloorGb); + /* #2362: observe every sweep, OUTSIDE the fire branch — the #2216 reasoning, which applies + identically here: counting only at delivery lets an event that ages out during a cooldown mask + an arrival. The list is UNCAPPED while the render below is capped, so a fingerprint outside the + displayed top N keeps its total instead of restarting. */ + var pvsOccurrences = await ObserveOccurrencesAsync( + key, PvsWatermarkMetric, AlertContextBuilders.PvsPressureIncidents(serverName, breached), now); if (breached.Count > 0) { var worst = breached[0]; @@ -1156,7 +1195,7 @@ private async Task CheckPvsPressureAsync( _lastPvsAlert[key] = now; _lastAlertedPvsPercent[key] = worst.PvsPercent; - var pvsContext = AlertContextBuilders.BuildPvsPressureContext(serverName, breached); + var pvsContext = AlertContextBuilders.BuildPvsPressureContext(serverName, breached, pvsOccurrences.Decorate); var detailText = AlertContextBuilders.ContextToDetailText(pvsContext); await FireAsync(new AlertOutcome( @@ -1173,6 +1212,7 @@ await FireAsync(new AlertOutcome( else if (_activePvsAlert.TryGetValue(key, out var wasPvs) && wasPvs) { _activePvsAlert[key] = false; + await ClearOccurrencesAsync(key, PvsWatermarkMetric); _lastAlertedPvsPercent.TryRemove(key, out _); if (!suppressed) { @@ -1193,6 +1233,104 @@ await NotifyResolutionAsync(new AlertResolution( } } + /* ---------------- database file growth (#2349) ---------------- */ + + /// + /// The gap between tempdb Space and Volume Free Space: a file that has grown large but has + /// not yet filled its disk. + /// + /// Why neither existing alert can express it. tempdb Space fires on + /// reserved ÷ (reserved + unallocated) — autogrowth adds unallocated extents, so the denominator grows with + /// the file and the percentage FALLS as tempdb balloons. It answers "is tempdb internally full right now", + /// which is a real question and structurally not this one. Volume Free Space fires on the + /// consequence, by which point a restart is already overdue, and cannot attribute the space to one file. + /// + /// Two gates, both graded per server. config_alert_settings is a single global row, so + /// an absolute MB threshold is unusable across a fleet whose normal tempdb sizes differ by an order of + /// magnitude. The RISE gate is the event (#2157's reasoning: a level alone re-pages every cooldown about a + /// size that has been true since Tuesday, which trains people to mute it); the LEVEL gate is the file as a + /// share of its volume, which self-scales to each server's disk layout. + /// + /// Observation sits OUTSIDE the fire branch, like blocking's (#2216/#2362): counting only at delivery + /// lets a file that stops breaching during a cooldown mask the next one. + /// + private async Task CheckFileGrowthAsync( + string key, string serverName, DateTime now, TimeSpan alertCooldown, bool suppressed, CancellationToken ct) + { + if (!_settings.FileGrowthEnabled) + { + return; + } + + try + { + var files = await _readAdapter.GetDatabaseFileGrowthAsync( + key, _settings.FileGrowthLookbackMinutes, ct); + + var breached = AlertContextBuilders.GetBreachedFiles( + files, _settings.FileGrowthRiseMb, _settings.FileGrowthVolumePercent); + + var fileGrowthOccurrences = await ObserveOccurrencesAsync( + key, FileGrowthWatermarkMetric, + AlertContextBuilders.FileGrowthIncidents(serverName, breached), now); + + if (breached.Count > 0) + { + var worst = breached[0]; + _activeFileGrowthAlert[key] = true; + + if (!suppressed && CooldownElapsed(_lastFileGrowthAlert, key, now, alertCooldown)) + { + var muteCtx = new AlertMuteContext { ServerName = serverName, MetricName = "Database File Growth" }; + bool isMuted = _isAlertMuted(muteCtx); + _lastFileGrowthAlert[key] = now; + + var context = AlertContextBuilders.BuildFileGrowthContext( + serverName, breached, fileGrowthOccurrences.Decorate); + var detailText = AlertContextBuilders.ContextToDetailText(context); + + /* The headline names the file, its size and its share of the volume — the three facts that + decide whether this is worth getting up for. The rise is in the card. */ + var headline = + $"{worst.DatabaseName}.{worst.FileName} is {worst.TotalSizeGb:F1} GB " + + $"({worst.VolumePercent:F0}% of {worst.VolumeMountPoint}), " + + $"grew {worst.GrowthGb:F1} GB in {worst.GrowthWindowMinutes:F0} min"; + + await FireAsync(new AlertOutcome( + key, serverName, "Database File Growth", + headline, + $"rise ≥ {_settings.FileGrowthRiseMb} MB or file ≥ {_settings.FileGrowthVolumePercent}% of volume", + context, detailText, + NumericCurrentValue: worst.VolumePercent, + NumericThresholdValue: _settings.FileGrowthVolumePercent, + Muted: isMuted, Severity: null, + ShortMessage: headline), ct); + } + } + else if (_activeFileGrowthAlert.TryGetValue(key, out var wasGrowing) && wasGrowing) + { + _activeFileGrowthAlert[key] = false; + await ClearOccurrencesAsync(key, FileGrowthWatermarkMetric); + + if (!suppressed) + { + await NotifyResolutionAsync(new AlertResolution( + key, serverName, "Database File Growth", + "Database File Growth Resolved", + $"{serverName}: no file is growing past the threshold or filling its volume"), ct); + } + } + } + catch (OperationCanceledException) when (ct.IsCancellationRequested) + { + throw; + } + catch (Exception ex) + { + _logger?.LogError("Failed to check database file growth for {Server}: {Message}", serverName, ex.Message); + } + } + /* ---------------- anomalous Agent jobs (Lite AlertEngine.cs:557-632) ---------------- */ private async Task CheckAnomalousJobsAsync( @@ -1230,6 +1368,12 @@ the cooldown each pass (scans ALL servers' entries, exactly like Lite). */ _lastLongRunningJobAlert.TryRemove(staleJobKey, out _); } + /* #2362: observe every sweep, OUTSIDE the fire branch — the #2216 reasoning, which applies + identically here: counting only at delivery lets an event that ages out during a cooldown mask + an arrival. The list is UNCAPPED while the render below is capped, so a fingerprint outside the + displayed top N keeps its total instead of restarting. */ + var jobOccurrences = await ObserveOccurrencesAsync( + key, AnomalousJobWatermarkMetric, AlertContextBuilders.AnomalousJobIncidents(serverName, anomalousJobs), now); if (anomalousJobs.Count > 0) { _activeLongRunningJobAlert[key] = true; /* :577 */ @@ -1243,7 +1387,7 @@ the cooldown each pass (scans ALL servers' entries, exactly like Lite). */ bool isMuted = _isAlertMuted(muteCtx); /* :586 */ _lastLongRunningJobAlert[jobKey] = now; /* :587 */ - var jobContext = AlertContextBuilders.BuildAnomalousJobContext(serverName, anomalousJobs); /* :600 */ + var jobContext = AlertContextBuilders.BuildAnomalousJobContext(serverName, anomalousJobs, jobOccurrences.Decorate); /* :600 */ var detailText = AlertContextBuilders.ContextToDetailText(jobContext); /* :601 */ /* :603-613. ShortMessage = the toast body of :595. */ @@ -1260,7 +1404,8 @@ await FireAsync(new AlertOutcome( } else if (_activeLongRunningJobAlert.TryGetValue(key, out var wasJob) && wasJob) /* :616 */ { - _activeLongRunningJobAlert[key] = false; /* :618 */ + _activeLongRunningJobAlert[key] = false; + await ClearOccurrencesAsync(key, AnomalousJobWatermarkMetric); /* :618 */ if (!suppressed) /* :619 */ { await NotifyResolutionAsync(new AlertResolution( @@ -1313,6 +1458,17 @@ msdb read to an empty list inside the fetcher instead. Failures are point-in-tim var failedJobs = await _failedJobsFetcher(key, _settings.FailedJobLookbackMinutes, ct); /* :657 */ conditionPresent = failedJobs.Count > 0; /* :663 — feeds the sweep result */ + /* #2362: observe every sweep, OUTSIDE the fire branch — the #2216 reasoning, which applies + identically here: counting only at delivery lets an event that ages out during a cooldown mask + an arrival. The list is UNCAPPED while the render below is capped, so a fingerprint outside the + displayed top N keeps its total instead of restarting. */ + var failedJobOccurrences = await ObserveOccurrencesAsync( + key, FailedJobWatermarkMetric, AlertContextBuilders.FailedJobIncidents(serverName, failedJobs), now); + + /* No ClearOccurrencesAsync counterpart, and that is not an omission: a failed job is an EVENT, + not a condition that resolves, so this check has no else-branch to clear from. The accumulator's + staleness horizon is the cleanup path here -- a fingerprint whose gauge stops moving for longer + than the horizon expires itself, which is exactly the shape an event stream needs. */ if (failedJobs.Count > 0) { var newestFailure = failedJobs.Max(j => j.RunDateTime); /* :665 */ @@ -1332,7 +1488,7 @@ msdb read to an empty list inside the fetcher instead. Failures are point-in-tim /* :679-682 — persist the SERVER-LOCAL watermark on-change only (#1145 parity). */ await _stateStore.SaveFailedJobWatermarkAsync(key, newestFailure); - var failedJobContext = AlertContextBuilders.BuildFailedJobContext(serverName, failedJobs); /* :695 */ + var failedJobContext = AlertContextBuilders.BuildFailedJobContext(serverName, failedJobs, failedJobOccurrences.Decorate); /* :695 */ var detailText = AlertContextBuilders.ContextToDetailText(failedJobContext); /* :696 */ /* :698-708. ShortMessage = the toast body of :690. */ diff --git a/PerformanceMonitor.Alerting/DatabaseFileGrowthInfo.cs b/PerformanceMonitor.Alerting/DatabaseFileGrowthInfo.cs new file mode 100644 index 000000000..4b1b2bdd8 --- /dev/null +++ b/PerformanceMonitor.Alerting/DatabaseFileGrowthInfo.cs @@ -0,0 +1,68 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +namespace PerformanceMonitor.Alerting; + +/// +/// One database file's current size, its growth over a lookback window, and the volume it sits on (#2349). +/// +/// Why this exists between the two alerts that already look at disk. tempdb Space fires on +/// reserved ÷ (reserved + unallocated), and autogrowth adds unallocated extents — so the percentage FALLS as +/// tempdb balloons. It answers "is tempdb internally full right now", which is useful and structurally cannot +/// answer "has this file grown". Volume Free Space catches the consequence: it fires when the drive is +/// nearly full, by which point a restart is already overdue, and cannot attribute the space to one file. +/// Between them sits a file that has grown large but has not yet filled its disk. +/// +/// Both gates come from one read. The store already holds the time series, so +/// is measured against a sample from the lookback window rather than tracked in memory +/// — no per-file state to keep, survive a restart, or leak. +/// +public class DatabaseFileGrowthInfo +{ + public string DatabaseName { get; set; } = ""; + public string FileName { get; set; } = ""; + public string PhysicalName { get; set; } = ""; + public string FileTypeDesc { get; set; } = ""; + + /// Current size. Since #2169 this is the in-database current size where the probe got it, so it + /// does not lag autogrowth — which tempdb, the motivating case, is guaranteed to do. + public double TotalSizeMb { get; set; } + + /// Growth over the lookback window: current minus the oldest sample in it. Zero when the window + /// holds only one sample, which reads as "no rise observed" rather than as a rise of the whole file. + public double GrowthMb { get; set; } + + /// How wide the window actually was, so a rise can be reported as a rate rather than a bare number + /// and a short window cannot masquerade as a slow one. + public double GrowthWindowMinutes { get; set; } + + public string VolumeMountPoint { get; set; } = ""; + public double VolumeTotalMb { get; set; } + public double VolumeFreeMb { get; set; } + + /// Null when growth is by PERCENT — the collector reports it that way on purpose, and a percent + /// autogrowth on a large file is itself the misconfiguration worth surfacing. + public double? AutoGrowthMb { get; set; } + public bool IsPercentGrowth { get; set; } + public double? GrowthPct { get; set; } + + /// -1 means unlimited, which the collector normalizes; carried so the payload can say so. + public double? MaxSizeMb { get; set; } + + /// The file as a share of its volume — the self-scaling level gate. One global threshold behaves + /// correctly across a fleet whose servers have very different normal sizes, which an absolute MB threshold + /// cannot: set it low enough for the small instances and the large ones alert constantly. + public double VolumePercent => VolumeTotalMb > 0 ? TotalSizeMb / VolumeTotalMb * 100 : 0; + + public double TotalSizeGb => TotalSizeMb / 1024.0; + public double GrowthGb => GrowthMb / 1024.0; + + /// Growth per hour, for a message that distinguishes "80 GB in an hour" from "80 GB since Tuesday". + public double GrowthMbPerHour => + GrowthWindowMinutes > 0 ? GrowthMb / (GrowthWindowMinutes / 60.0) : 0; +} diff --git a/PerformanceMonitor.Alerting/IAlertEngineSettings.cs b/PerformanceMonitor.Alerting/IAlertEngineSettings.cs index dc083ea2e..19523b240 100644 --- a/PerformanceMonitor.Alerting/IAlertEngineSettings.cs +++ b/PerformanceMonitor.Alerting/IAlertEngineSettings.cs @@ -73,6 +73,13 @@ public interface IAlertEngineSettings bool LongRunningJobEnabled { get; } bool FailedJobEnabled { get; } bool PvsEnabled { get; } + + /// + /// The database file-growth alert (#2349) — OFF by default. It sits between tempdb Space, whose + /// denominator grows with autogrowth so its percentage FALLS as tempdb balloons, and Volume Free + /// Space, which fires on the consequence and cannot attribute it to a file. + /// + bool FileGrowthEnabled { get; } bool DatabaseStateEnabled { get; } /// @@ -188,6 +195,24 @@ public interface IAlertEngineSettings /// int PvsFloorGb { get; } + /// + /// The RISE gate: a file that grew at least this many MB inside the lookback window (#2349). Primary + /// rather than the level, for #2157's reason — a level alone re-pages every cooldown about a size that has + /// been true for a week, which trains people to mute it, while a rise is an event. + /// + int FileGrowthRiseMb { get; } + + /// + /// The LEVEL gate: a file occupying at least this share of its volume (#2349). Self-scaling, which is what + /// makes ONE global setting usable across a fleet whose servers have very different normal sizes — an + /// absolute MB threshold cannot be set low enough for the small instances without deafening the large ones. + /// + int FileGrowthVolumePercent { get; } + + /// How far back the rise is measured (#2349). The window is MEASURED from the samples rather than + /// assumed, so a gap in collection cannot make a slow rise look fast. + int FileGrowthLookbackMinutes { get; } + /// Fire when a running job exceeds this multiple of its historical average duration. int LongRunningJobMultiplier { get; } diff --git a/PerformanceMonitor.Alerting/IAlertReadAdapter.cs b/PerformanceMonitor.Alerting/IAlertReadAdapter.cs index 2e4b754ef..ad58398a4 100644 --- a/PerformanceMonitor.Alerting/IAlertReadAdapter.cs +++ b/PerformanceMonitor.Alerting/IAlertReadAdapter.cs @@ -114,6 +114,17 @@ Task> GetLongRunningQueriesAsync( Task> GetVolumeFreeSpaceAsync( string serverKey, CancellationToken cancellationToken = default); + /// + /// Per-file size, growth over , and the volume each file sits on (#2349). + /// + /// One read serves both gates the file-growth alert applies — a RISE (this file grew N MB) and a + /// LEVEL (this file is N% of its volume) — because the store already holds the time series and computing + /// the rise there costs one more join rather than per-file state the engine would have to keep, persist + /// and expire. + /// + Task> GetDatabaseFileGrowthAsync( + string serverKey, int lookbackMinutes, CancellationToken cancellationToken = default); + /// /// The latest tempdb space snapshot, or null when the store has none for this server. /// Threshold evaluation (UsedPercent) stays engine-side. diff --git a/PerformanceMonitor.Collectors/CollectorContext.cs b/PerformanceMonitor.Collectors/CollectorContext.cs index 7f0194ff1..596517e42 100644 --- a/PerformanceMonitor.Collectors/CollectorContext.cs +++ b/PerformanceMonitor.Collectors/CollectorContext.cs @@ -127,7 +127,7 @@ public sealed class CollectorContext /// /// When true, the query_store payload leaves query_sql_text NULL and the host is responsible - /// for resolving statement text through instead + /// for resolving statement text through instead /// (#2150). Default false, which keeps the text inline exactly as it ships today. /// /// Why the text has to leave that projection. The payload selects it inside a diff --git a/PerformanceMonitor.Collectors/QueryStoreCollector.cs b/PerformanceMonitor.Collectors/QueryStoreCollector.cs index 18b4497f4..737d30e8e 100644 --- a/PerformanceMonitor.Collectors/QueryStoreCollector.cs +++ b/PerformanceMonitor.Collectors/QueryStoreCollector.cs @@ -10,6 +10,7 @@ using System.Collections.Generic; using System.Data.Common; using System.Globalization; +using System.Linq; using System.Threading; using System.Threading.Tasks; @@ -684,44 +685,21 @@ stays DESC even though the outer sort is now ASC (#1960): under oldest-first shi plans live, and Darling's stored-plan readers all guard `query_plan_text IS NOT NULL`. Not mirrored into the Dashboard proc: its "Download Plan" reads by exact collection_id, where per-row NULLs would break a real reader. */ - /* #2164: skip the XML for plans the store already holds. 97% of the plan XML shipped in a - three-hour fleet window was for plans held over an hour — the ROW_NUMBER gate ships each plan once - per PASS but re-ships it every pass forever, and since drain is 94-97% of a pass and is per-row LOB - cost, NOT fetching is worth far more than fetching less. The watermark is the highest plan_id whose - XML was actually STORED for this database; plan_id is monotonic within a database, so a higher id - is a plan we have never stored. Inlined as a parsed long (never operator input) because the body - nests inside sp_executesql on three paths and threading another parameter through all of them buys - nothing. Zero — absent, malformed, or expired — renders no predicate, so the conservative path is - byte-identical to the pre-#2164 query. - - NEVER on the backfill path. The watermark tracks the plans the LIVE window has stored, and backfill - digs the other way — into intervals older than anything collected, whose rows reference plans - compiled long ago and therefore numbered BELOW the live watermark. Applying it there would suppress - essentially every plan the backfill exists to fetch, silently: the slices would still ship runtime - stats, so a filled range would look complete while carrying no plan XML at all. - - KNOWN GAP, bounded by QueryStorePlanXmlState.RefreshAfter: plan_id is monotonic in COMPILE order, which is - not the same as "we have stored it". A plan compiled before monitoring began, dormant through every - collected window, then executed again, arrives with an id below the watermark and has its XML - suppressed until the refresh horizon expires. Bounding it is the reason that horizon exists. The - exact fix is a store-DERIVED watermark (the host asking its own plan dimension for the lowest - plan_id missing XML) rather than this collector-derived one; that needs host plumbing on both - products and is tracked separately. */ - /* #2210: the watermark now belongs to BuildPlanFetchQuery (the `watermark` parameter there, - resolved by the host via QueryStorePlanXmlState.Resolve). It no longer narrows anything in - this runtime-stats query, so there is nothing to compute here. */ - - /* #2210: this runtime-stats query no longer carries plan XML at all — the ROW_NUMBER-gated - CASE and its in-stream watermark predicate are DELETED, not reworked. BuildPlanFetchQuery is - the only thing that reads plan XML now: it fetches plans in plan_id order under a byte - budget and lands each plan ONCE per database LIFETIME instead of once per PASS. The shape - being replaced re-shipped every plan on every pass forever — measured at 5.0x redundancy - (871,196 plan-XML rows against 175,328 distinct database/plan pairs in a day, on a 33 GB - table). Both branches below now emit the same placeholder, so the payload is byte-identical - to Lite's regardless of the flag, and CapturePlanXml gates the separate BuildPlanFetchQuery - fetch rather than this query. Existing inline rows are NOT migrated by this change and stay - readable via the reader's existing NULL-guarded fallback; dropping the query_plan_text column - itself is a separate, later migration. */ + /* #2312: this runtime-stats query no longer carries plan XML at all — the #2164 ROW_NUMBER + gate and its in-stream watermark predicate were DELETED by #2210, and #2312 then retired + the watermark itself: the #2164 KNOWN GAP's "exact fix is a store-DERIVED watermark" is + what the activity-driven fetch now IS. BuildPlanFetchByIdsQuery is the only thing that + reads plan XML: the host probes its own store for the plans this cycle's rows reference + and fetches exactly the missing ones, so each plan lands ONCE per database lifetime, a + dormant plan resuming execution is fetched the cycle it resumes (no refresh horizon to + wait out), and a caught-up database issues no fetch at all. The shape #2210 replaced + re-shipped every plan on every pass forever — measured at 5.0x redundancy (871,196 + plan-XML rows against 175,328 distinct database/plan pairs in a day, on a 33 GB table). + Both branches below emit the same placeholder, so the payload is byte-identical to Lite's + regardless of the flag, and CapturePlanXml gates the separate by-ids fetch rather than + this query. Existing inline rows are NOT migrated and stay readable via the reader's + NULL-guarded fallback; backfill plan XML stays on its own rows for the same reason it + always did — those intervals' plans are never in the live cycle's reference set. */ const string planTextCol = "query_plan_text = CONVERT(nvarchar(1), NULL),"; /* #2150: the LAST nvarchar(max) in this projection, and now the whole remaining cost of it. The cap @@ -1128,53 +1106,58 @@ EXECUTE [{escapedDbName}].sys.sp_executesql } /// - /// The plan-XML fetch for one database (#2210): plans above the watermark, in plan_id order, bounded - /// twice — coarsely by and exactly by a running byte total. + /// The plan-XML fetch for one database (#2312 Finding 2): exactly the plans the caller names — the cycle's + /// collected runtime rows whose XML the store does not already hold — in plan_id order, cut exactly + /// by a running byte total. The STORE is the watermark: a caught-up database has an empty missing set and + /// no fetch runs at all, which is the property the #2210 catalog walk lacked (measured 23s to discover + /// "nothing new" on a warm catalog, every cycle, because the walk re-read the catalog to find out). /// - /// SEPARATE from the runtime-stats query on purpose, and that separation is the fix rather than a - /// refactor. The runtime query ships ORDER BY qsrs.last_execution_time, so a budget cut truncates it - /// in TIME order and the plans whose XML landed are an arbitrary SUBSET of plan_ids — against which no - /// watermark value is safe, because receiving plan 500 while missing 300 skips 300 forever. That is why the - /// previous shape could not advance on a cut, and 97.8% of production passes are cut. Here rows arrive in - /// plan_id order, so a cut truncates a SUFFIX and the highest landed id is safe by construction. + /// SEPARATE from the runtime-stats query on purpose, and that separation is still the fix rather than + /// a refactor. The runtime query ships ORDER BY qsrs.last_execution_time, so a budget cut truncates + /// it in TIME order — plan XML inline there re-shipped the same plans at 5.0x (measured, #2210). Here the + /// caller hands an explicit id list, so a budget cut leaves ids that are simply STILL MISSING from the + /// store, and the next cycle that references them (or the caller's own carry-over) re-selects them. No + /// watermark, no suffix-safety argument, no out-of-order hazard. /// - /// The two bounds are not redundant. The running total is exact but expensive to compute: it needs - /// DATALENGTH, and sys.query_store_plan.query_plan is decompressed BY the view on access, so an - /// unbounded candidate set pays a whole catalog's decompression to enforce a budget meant to prevent exactly - /// that. TOP (@candidate_plans) is evaluated on plan_id alone — no XML touched to sort or - /// filter — so the decompression is capped at K, sized per database by - /// from the previous pass's own bytes-per-plan. + /// The candidate bound is the ID LIST ITSELF, which the caller caps via + /// before building: the running total needs + /// DATALENGTH, and sys.query_store_plan.query_plan is decompressed BY the view on access, so + /// handing the whole missing set of a first-contact database in one statement would pay its entire + /// decompression to enforce a budget meant to prevent exactly that. Chunking and capping are caller + /// decisions; this builder's contract is only "the list you hand me is what I decompress". + /// + /// query_plan_hash rides along in the SELECT — CONVERT(varchar(64), ..., 1), the same + /// rendering the runtime payload uses — because it reads WITHOUT decompressing the plan and the map stores + /// it as the in-place-rewrite detector: a batch whose live hash differs from the stored one is the one case + /// activity-driven fetch cannot see on its own. /// /// The budget test is running_bytes - plan_bytes < budget, i.e. admit a plan when the total /// BEFORE it was still under. The obvious running_bytes <= budget is a per-database STALL: a single /// plan larger than the whole budget has a running total that already exceeds it on its own row, so it is - /// excluded, every later row is excluded too (the total is monotonic), the pass ships nothing, the watermark - /// holds, and the next pass re-selects the same plan first — forever. One 13 MB plan against the 12 MB - /// default is enough, and it is the same never-advances failure this change exists to end, reached through - /// plan SIZE instead of cut ordering. Admitting the offender ships it alone, cuts after it, and moves the - /// watermark past it. + /// excluded, every later row is excluded too (the total is monotonic), the pass ships nothing, the plan + /// stays missing, and the next pass re-selects it first — forever. One 13 MB plan against the 12 MB + /// default is enough. Admitting the offender ships it alone and cuts after it; once landed it is never + /// selected again. /// /// The honest cost of that: worst-case bytes for one pass are budget + largest single plan, /// not budget. The runtime-stats budget a few hundred lines up pays exactly the same price for the /// same reason (measured: 19.6 MB shipped against a 12 MB budget when one very large plan carried a pass /// past it), so "12 MB" is a floor on ship volume in both paths rather than a cap. /// - /// Both bounds are inlined as parsed integers rather than parameters, matching the watermark predicate - /// above and for the same reason: the body nests inside sp_executesql, and the values are host-computed - /// longs that never touch operator input. + /// The budget and the id list are inlined as parsed integers rather than parameters, and for the same + /// reason as each other: the body nests inside sp_executesql, and the values are host-computed longs + /// that never touch operator input. /// /// A NULL query_plan — a plan too large to persist, or certain forced-plan-failure paths — /// counts as ZERO bytes and STILL SHIPS, as a row with NULL text. Letting the NULL propagate through the - /// arithmetic instead would make the budget predicate NULL and filter the row out, and a window whose plans - /// are all NULL would then return nothing, hold the watermark, and re-select the same plans forever: the - /// same permanent stall as the oversized-plan case, reached through a different mechanism. Shipping the row - /// lets the watermark advance past a plan whose XML will never exist, which is correct — the store's readers - /// already guard query_plan_text IS NOT NULL because the runtime path has always been able to write - /// per-row NULLs there. + /// arithmetic instead would make the budget predicate NULL and filter the row out, and the plan would be + /// re-selected as missing forever. Shipping the row lets the writer record a content-less map row (the + /// NULL-digest marker), which is what makes "the engine says this XML will never exist" a stored fact + /// instead of a per-cycle rediscovery — the store's readers already guard + /// query_plan_text IS NOT NULL, so absent content renders as absent either way. /// - /// NEVER on the backfill path, for the reason the watermark itself is not: backfill reads intervals - /// older than anything collected, whose plans are numbered BELOW the watermark, so a plan_id-ascending fetch - /// above the watermark would return nothing the backfill needs. Backfill plan XML stays on its own rows. + /// NEVER on the backfill path: backfill plan XML stays on its own rows, exactly as before — this + /// fetch serves the live cycle's referenced plans and nothing else. /// /// The CONVERT happens ONCE, inside the candidate window, and the running total sums /// DATALENGTH of that converted text rather than of the view column. The alternative — measure with @@ -1184,7 +1167,7 @@ EXECUTE [{escapedDbName}].sys.sp_executesql /// form took 274ms cold / 262ms warm against 133ms for this one. Plan-id-only with no XML touched was 114ms, /// so this shape sits 19ms above the floor while the join-back form pays for the decompression twice. /// - public CollectorQuery BuildPlanFetchQuery(string item, CollectorContext context, long watermark, int candidatePlans, long budgetBytes) + public CollectorQuery BuildPlanFetchByIdsQuery(string item, CollectorContext context, IReadOnlyList planIds, long budgetBytes) { /* The invariant the doc comment spends a paragraph on, actually enforced rather than left to the caller: this query exists only to fetch plan XML, so building it with plan capture off is a caller bug, not a @@ -1198,48 +1181,49 @@ runtime query's plan-text CASE already reads the same flag. */ if (!context.CapturePlanXml) { throw new InvalidOperationException( - "BuildPlanFetchQuery requires CapturePlanXml; a host that does not capture plan XML must not issue the plan fetch."); + "BuildPlanFetchByIdsQuery requires CapturePlanXml; a host that does not capture plan XML must not issue the plan fetch."); } /* A non-positive budget would make the predicate `running_bytes - plan_bytes < 0`, which excludes even the FIRST candidate (its running total before it is 0, and 0 < 0 is false) — the pass ships nothing, - the watermark holds, and the next pass re-selects the same plans. The oversized-plan stall for a third - time, from a third direction. CandidatePlanCount already floors a non-positive budget for its own - sizing; this method has to guard its own input rather than assume the caller passed that value through. */ + and because the ids stay missing from the store, the next pass re-selects the same plans forever. + The oversized-plan stall, reached through the budget input rather than through cut ordering. */ if (budgetBytes <= 0) { throw new ArgumentOutOfRangeException( - nameof(budgetBytes), budgetBytes, "The plan-fetch byte budget must be positive; a zero or negative budget ships nothing and stalls the watermark."); + nameof(budgetBytes), budgetBytes, "The plan-fetch byte budget must be positive; a zero or negative budget ships nothing and the ids stay missing forever."); } - /* Same failure, fourth route: TOP (0) returns no rows and TOP with a negative literal is a syntax error, - so a bad candidate count ships nothing and holds the watermark exactly like a bad budget. Every caller - today sources this from CandidatePlanCount, which floors at MinCandidatePlans — but "the only caller - happens to be safe" is the assumption this method has already been wrong about once. */ - if (candidatePlans <= 0) + /* An empty id list means the store already holds every plan this cycle referenced — the steady state + whose whole point is that NO target query runs (#2312 Finding 2). Reaching this method with one is a + caller bug, not a no-op to swallow: an `IN ()` is a syntax error anyway, and silently returning a + no-op query would hide the caller's missing skip. */ + if (planIds is null || planIds.Count == 0) { - throw new ArgumentOutOfRangeException( - nameof(candidatePlans), candidatePlans, "The candidate plan count must be positive; TOP (0) ships nothing and stalls the watermark."); + throw new ArgumentException( + "The plan id list must be non-empty; an empty missing set means no fetch should be issued at all.", nameof(planIds)); } var escapedDbName = item.Replace("]", "]]", StringComparison.Ordinal); - var k = candidatePlans.ToString(System.Globalization.CultureInfo.InvariantCulture); var budget = budgetBytes.ToString(System.Globalization.CultureInfo.InvariantCulture); - var floor = watermark.ToString(System.Globalization.CultureInfo.InvariantCulture); + /* Host-computed longs, inlined like the budget and for the same reason: the body nests inside + sp_executesql, and none of these values ever touch operator input. */ + var idList = string.Join(", ", planIds.Select(id => id.ToString(System.Globalization.CultureInfo.InvariantCulture))); /* ROWS UNBOUNDED PRECEDING, not the RANGE default: RANGE would tie-group peers and, more to the point, forces a spool. The frame is per-row precisely because the cut has to fall between two plans. */ var body = $@"WITH candidates AS ( - SELECT TOP ({k}) + SELECT plan_id = qsp.plan_id, + query_plan_hash = CONVERT(varchar(64), qsp.query_plan_hash, 1), query_plan_text = CONVERT(nvarchar(max), qsp.query_plan) FROM sys.query_store_plan AS qsp - WHERE qsp.plan_id > {floor} - ORDER BY qsp.plan_id + WHERE qsp.plan_id IN ({idList}) ), budgeted AS ( SELECT plan_id = c.plan_id, + query_plan_hash = c.query_plan_hash, query_plan_text = c.query_plan_text, plan_bytes = COALESCE(DATALENGTH(c.query_plan_text), 0), running_bytes = SUM(COALESCE(DATALENGTH(c.query_plan_text), 0)) OVER (ORDER BY c.plan_id ROWS UNBOUNDED PRECEDING) @@ -1247,6 +1231,7 @@ FROM candidates AS c ) SELECT plan_id = b.plan_id, + query_plan_hash = b.query_plan_hash, query_plan_text = b.query_plan_text FROM budgeted AS b WHERE b.running_bytes - b.plan_bytes < {budget} @@ -1263,28 +1248,28 @@ EXECUTE [{escapedDbName}].sys.sp_executesql } /// - /// Statement text for one database, resumed from a query_id watermark and cut by a byte budget - /// (#2150) — the sibling of , and the other half of taking - /// query_sql_text out of the runtime stream. - /// - /// Ordered by query_id, which is what makes a budget cut safe. The cut falls between - /// two statements, so everything up to it is stored and the highest stored id is a resume point with no - /// hole — the same suffix argument the plan fetch rests on. query_id is also already a stored - /// payload column on the runtime row, so this needs no new fact-table column and no migration to be - /// joinable. + /// Statement text for one database (#2312 Finding 2, applying #2150's split): exactly the query_ids the + /// caller names — the cycle's collected rows whose text the store does not already hold — cut by a byte + /// budget. The sibling of , with the same store-as-watermark + /// contract: an empty missing set means no fetch runs at all. /// /// Simpler than the plan fetch on purpose. There is no candidate-window estimator here /// because DATALENGTH(query_sql_text) is cheap: sys.query_store_plan.query_plan is - /// decompressed BY the view on access, which is what forces the plan side to bound how many plans a - /// windowed running total may touch, and query_sql_text has no such cost. A flat coarse bound - /// plus the exact running total is enough. There is no content hash either — plan XML can be rewritten - /// in place, whereas a statement's text is fixed for the life of its id. + /// decompressed BY the view on access, which is what forces the plan side to cap its id list, and + /// query_sql_text has no such cost — the caller may hand the whole missing set (chunked only for + /// statement-size sanity). + /// + /// query_hash rides along — CONVERT(varchar(64), ..., 1), the runtime payload's own + /// rendering — because query_id is only unique until a Query Store reset renumbers it: a stored + /// hash that differs from the batch's live one is how the store detects that id 5 is now a DIFFERENT + /// statement and refetches, where the old design relied on a daily watermark expiry to eventually + /// re-read everything. /// /// ROWS UNBOUNDED PRECEDING rather than the RANGE default, for the same reason as /// the plan fetch: RANGE would tie-group peers and force a spool, and the frame has to be per-row /// because the cut falls between two rows. /// - public CollectorQuery BuildTextFetchQuery(string item, CollectorContext context, long watermark, int candidateTexts, long budgetBytes) + public CollectorQuery BuildTextFetchByIdsQuery(string item, CollectorContext context, IReadOnlyList queryIds, long budgetBytes) { if (context is null) { @@ -1297,44 +1282,43 @@ caller bug rather than a harmless extra round trip — it would fetch and store if (!context.FetchQueryTextSeparately) { throw new InvalidOperationException( - "BuildTextFetchQuery requires FetchQueryTextSeparately; a host that still ships query_sql_text inline must not issue the text fetch."); + "BuildTextFetchByIdsQuery requires FetchQueryTextSeparately; a host that still ships query_sql_text inline must not issue the text fetch."); } /* A non-positive budget makes the predicate `running_bytes - text_bytes < 0` exclude even the FIRST - candidate (its running total before it is 0, and 0 < 0 is false), so the pass ships nothing, the - watermark holds, and the next pass re-selects the same statements — a stall that looks like a - quiet database. */ + candidate (its running total before it is 0, and 0 < 0 is false), so the pass ships nothing and the + ids stay missing forever — a stall that looks like a quiet database. */ if (budgetBytes <= 0) { throw new ArgumentOutOfRangeException( - nameof(budgetBytes), budgetBytes, "The text-fetch byte budget must be positive; a zero or negative budget ships nothing and stalls the watermark."); + nameof(budgetBytes), budgetBytes, "The text-fetch byte budget must be positive; a zero or negative budget ships nothing and the ids stay missing forever."); } - /* Same stall, other route: TOP (0) returns no rows and a negative literal is a syntax error. */ - if (candidateTexts <= 0) + /* Same contract as the plan fetch: empty means the caller should not have called. */ + if (queryIds is null || queryIds.Count == 0) { - throw new ArgumentOutOfRangeException( - nameof(candidateTexts), candidateTexts, "The candidate text count must be positive; TOP (0) ships nothing and stalls the watermark."); + throw new ArgumentException( + "The query id list must be non-empty; an empty missing set means no fetch should be issued at all.", nameof(queryIds)); } var escapedDbName = item.Replace("]", "]]", StringComparison.Ordinal); - var k = candidateTexts.ToString(System.Globalization.CultureInfo.InvariantCulture); var budget = budgetBytes.ToString(System.Globalization.CultureInfo.InvariantCulture); - var floor = watermark.ToString(System.Globalization.CultureInfo.InvariantCulture); + var idList = string.Join(", ", queryIds.Select(id => id.ToString(System.Globalization.CultureInfo.InvariantCulture))); var body = $@"WITH candidates AS ( - SELECT TOP ({k}) + SELECT query_id = qsq.query_id, + query_hash = CONVERT(varchar(64), qsq.query_hash, 1), query_sql_text = qst.query_sql_text FROM sys.query_store_query AS qsq JOIN sys.query_store_query_text AS qst ON qst.query_text_id = qsq.query_text_id - WHERE qsq.query_id > {floor} - ORDER BY qsq.query_id + WHERE qsq.query_id IN ({idList}) ), budgeted AS ( SELECT query_id = c.query_id, + query_hash = c.query_hash, query_sql_text = c.query_sql_text, text_bytes = COALESCE(DATALENGTH(c.query_sql_text), 0), running_bytes = SUM(COALESCE(DATALENGTH(c.query_sql_text), 0)) OVER (ORDER BY c.query_id ROWS UNBOUNDED PRECEDING) @@ -1342,6 +1326,7 @@ FROM candidates AS c ) SELECT query_id = b.query_id, + query_hash = b.query_hash, query_sql_text = b.query_sql_text FROM budgeted AS b WHERE b.running_bytes - b.text_bytes < {budget} @@ -1449,12 +1434,6 @@ last_execution_time still ship (they are adjacent under the query's ASC order), var budgetSpent = false; DateTime? cutBoundary = null; - /* #2164 watermark bookkeeping: counts ONLY plans whose XML actually landed in this batch, so a - budget-cut pass cannot claim coverage it does not have. Plans observed but not stored are - deliberately not tracked — see QueryStorePlanXmlState.RefreshAfter for why the observed maximum cannot be - used to detect a Query Store reset. */ - long maxStoredPlanId = 0; - while (await reader.ReadAsync(cancellationToken)) { var row = new Row @@ -1550,11 +1529,6 @@ and finish the boundary tie group — the host surfaces the WARNING. Rows are re a bounded cycle costs latency, never data. */ textBytes += ((long)(row.QueryText?.Length ?? 0) + (row.QueryPlanText?.Length ?? 0)) * 2L; - if (row.QueryPlanText is not null && row.PlanId > maxStoredPlanId) - { - maxStoredPlanId = row.PlanId; - } - if (!budgetSpent && textBytes >= budget) { budgetSpent = true; @@ -1566,38 +1540,9 @@ and finish the boundary tie group — the host surfaces the WARNING. Rows are re context.PerItemTextBytesShipped = textBytes; context.PerItemShippedBoundary = rows.Count > 0 ? rows[^1].LastExecutionTime : null; - /* #2164: persist the plan-XML watermark for this database. - - Advance to the highest plan_id whose XML actually stored, never past it. - - Never move BACKWARD: a window whose newest-executing plan is older than the newest-COMPILED one - is an ordinary quiet window, not a reset, and lowering the watermark there would refetch the - whole catalog next cycle. (Treating it as a reset is the trap documented on - QueryStorePlanXmlState.RefreshAfter — it holds in most steady-state windows.) - - Never advance AT ALL on a budget-cut pass. Rows ship ordered by last_execution_time, NOT by - plan_id, so the cut drops an arbitrary set of plan_ids from the tail of the window — including - ids BELOW the highest one that did store. Advancing past them would suppress their XML on every - later pass (the ids no longer clear the watermark) even though it never shipped once. The cut is - already resumable on the time watermark, so declining to advance costs one repeated fetch and - nothing else. */ - if (context.CapturePlanXml && !string.IsNullOrEmpty(databaseName) && !budgetSpent && maxStoredPlanId > 0) - { - var standing = QueryStorePlanXmlState.Resolve(context.State, databaseName, context.CollectionTime); - - if (maxStoredPlanId > standing) - { - /* The stamp dates the last FULL fetch, and is carried FORWARD across advances rather than - renewed on each one. Re-stamping here would push the refresh horizon out every time a new - plan compiled, so on any database that keeps compiling — the busy ones, where a stale plan - is most likely to matter — the horizon would never fire and the watermark would effectively - be permanent. A standing watermark of 0 means this pass WAS the full fetch (absent or just - expired), so that is the one case that stamps now. */ - var stamp = standing > 0 - ? QueryStorePlanXmlState.ResolveStamp(context.State, databaseName) ?? context.CollectionTime - : context.CollectionTime; - - context.PendingState[QueryStorePlanXmlState.KeyFor(databaseName)] = - QueryStorePlanXmlState.Format(maxStoredPlanId, stamp); - } - } + /* #2312: no plan-XML watermark write-back any more. Inline-shipped plan XML (the backfill path) + lands on its own rows; the LIVE fetch is activity-driven against the store's own map, so there + is no resume point to persist here and nothing for a budget cut to corrupt. */ } diff --git a/PerformanceMonitor.Collectors/QueryStorePerDatabaseState.cs b/PerformanceMonitor.Collectors/QueryStorePerDatabaseState.cs index 949dd6121..7e8a47147 100644 --- a/PerformanceMonitor.Collectors/QueryStorePerDatabaseState.cs +++ b/PerformanceMonitor.Collectors/QueryStorePerDatabaseState.cs @@ -14,21 +14,20 @@ namespace PerformanceMonitor.Collectors; /// Every collector_state key query_store owns that is keyed by DATABASE NAME (#2188) — the set both /// hosts prune when a database is dropped or renamed. /// -/// Nothing ever retired these. The #2164 plan-XML watermark writes one planwm: row per database -/// and the #2022/#2058 backfill worker writes done: and hole:, and while the worker deletes a -/// hole when it SERVICES or expires it, a dropped database will never service one — its hole can never be -/// dug and its tail can never drain. collector_state is a keyed registry rather than a hypertable -/// (pinned by CollectorStateContractTests), so no retention policy caught them either. +/// Nothing ever retired these. The #2022/#2058 backfill worker writes done: and hole:, +/// and while the worker deletes a hole when it SERVICES or expires it, a dropped database will never +/// service one — its hole can never be dug and its tail can never drain. collector_state is a keyed +/// registry rather than a hypertable (pinned by CollectorStateContractTests), so no retention policy +/// caught them either. (The #2164 planwm: and #2150 textwm: watermarks were members of this +/// list until #2312 retired the watermarks themselves — the fetches are activity-driven now, the store is +/// the watermark, and V77 deleted the orphaned rows once.) /// /// Shared rather than one list per host, which is the whole reason this file exists. The two /// stores prune with different dialects (Postgres anti-join, DuckDB NOT IN) and the SKUs write -/// different subsets — Lite never sets CollectorContext.CapturePlanXml, so it writes no -/// planwm: at all, while both write the backfill pair. A per-host list would make a fourth prefix a -/// two-place edit whose omission fails nothing: the rows would simply orphan on one SKU, invisibly, which is -/// the drift this product keeps paying for. Both hosts iterate THIS, so a prefix is pruned everywhere or -/// nowhere. Lite running the planwm: statement against rows it never writes costs one no-op delete -/// and buys the guarantee that enabling plan capture there cannot quietly create an unpruned orphan -/// class. +/// different subsets. A per-host list would make a new prefix a two-place edit whose omission fails +/// nothing: the rows would simply orphan on one SKU, invisibly, which is the drift this product keeps +/// paying for. Both hosts iterate THIS, so a prefix is pruned everywhere or nowhere — a no-op delete on +/// the SKU that never writes a prefix is the cheap price of that guarantee. /// /// Membership is a real decision, not a listing of every key: a key must be /// <prefix><databaseName>, because both prunes reconstruct it that way to test it against @@ -45,16 +44,10 @@ public static class QueryStorePerDatabaseState /// public static readonly IReadOnlyList<(string Owner, string Prefix)> PrunableKeys = new[] { - (QueryStorePlanXmlState.StateCollectorName, QueryStorePlanXmlState.WatermarkKeyPrefix), (QueryStoreBackfillState.StateCollectorName, QueryStoreBackfillState.DoneKeyPrefix), (QueryStoreBackfillState.StateCollectorName, QueryStoreBackfillState.HoleKeyPrefix), - /* #2150: the text watermark is keyed prefix + databaseName exactly like the plan watermark above, - so a dropped database's key must go with it. Paired with its OWN collector name rather than the - plan fetch's — the two watermarks are stored separately on purpose, and a prefix pruned under - the wrong owner silently deletes nothing. */ - (QueryStoreTextState.StateCollectorName, QueryStoreTextState.WatermarkKeyPrefix), - /* #2312: the open-interval refresh stamp, per database like the three above, under its own owner - for the same never-prune-under-the-wrong-name reason. */ + /* #2312: the open-interval refresh stamp, per database like the pair above, under its own owner + because a prefix pruned under the wrong collector name silently deletes nothing. */ (QueryStoreOpenIntervalState.StateCollectorName, QueryStoreOpenIntervalState.WatermarkKeyPrefix), }; diff --git a/PerformanceMonitor.Collectors/QueryStorePlanXmlState.cs b/PerformanceMonitor.Collectors/QueryStorePlanXmlState.cs index ef38f6203..058e724f9 100644 --- a/PerformanceMonitor.Collectors/QueryStorePlanXmlState.cs +++ b/PerformanceMonitor.Collectors/QueryStorePlanXmlState.cs @@ -6,160 +6,66 @@ * Licensed under the MIT License. See LICENSE file in the project root for full license information. */ -using System; -using System.Collections.Generic; -using System.Globalization; - namespace PerformanceMonitor.Collectors; /// -/// What one plan-fetch pass earned: the watermark to persist, and whether the pass's rows actually arrived in -/// the plan_id order its ORDER BY promises (#2210). One value rather than two calls so a caller cannot -/// take the watermark without being handed the reason it may not have moved — the ordering guard is only -/// useful if the violation gets LOGGED, and a signal a caller can forget to ask for is one that eventually -/// nobody asks for. -/// -/// The plan_id to persist; the standing value when the pass earned no advance. -/// False when a descent was seen, meaning the advance was abandoned and the -/// caller should log a precondition violation rather than treat a static watermark as a quiet pass. -public readonly record struct PlanWatermarkAdvance(long Watermark, bool ArrivedInPlanIdOrder); - -/// -/// The persisted per-database plan-XML watermark (#2164) — the highest plan_id whose execution-plan -/// XML has actually been stored for a database, so collection stops re-shipping plans the store already -/// holds. 97% of the plan XML shipped in a three-hour fleet window was for plans held for over an hour, and -/// because streaming rows is 94-97% of a pass and costs per-row LOB bytes, not fetching beats fetching less. +/// Per-database SIZING for the plan-XML fetch (#2312 Finding 1, wired in #2322): how many plans one pass may +/// hand , learned from each database's own +/// bytes-per-plan rather than a fleet constant, because measured plan size spans 11x across databases +/// (162 KB to 15 KB by quartile) and no single value is right at both ends. /// -/// Owned by the HOST under its own , exactly like -/// and for the same reason: the query_store DEFINITION keeps declaring -/// no state keys, so CollectorStateContractTests stays honest and adding per-database state does not -/// silently become a two-host contract change. The keys are dynamic (one per database), which the host's -/// state read supports because it loads every row for a collector name rather than a declared key list — the -/// definition's StateKeys could not express these anyway. +/// What this class no longer is. Through #2210 it owned the persisted per-database plan-id +/// WATERMARK (planwm: under the query_store_plan_xml state owner) — the resume point for a +/// budgeted whole-catalog walk, with a daily refresh expiry standing in for a re-verify cursor that was +/// designed but never wired. #2312 Finding 4 measured what that actually did in production: catalogs whose +/// full walk needs more than a day expired MID-walk, restarted from plan_id 0, and looped the full catalog +/// fetch forever. The fetch is now activity-driven — the cycle's collected rows name their plans, the STORE +/// answers which are missing (QueryStorePlanMap's touch-and-probe), and only those are fetched — so +/// there is no watermark, no expiry, and no state rows. The retired planwm:/textwm: rows are +/// deleted once by the V77 migration. /// -/// Lives in the shared collectors project rather than either host because it is watermark-shaped state -/// that must decode identically wherever it is read: a row written by Darling today has to keep meaning the -/// same thing after an upgrade, and Lite reads the same definition. +/// What remains is the sizing estimator, which still matters: the missing set of a first-contact +/// database is its whole catalog, and the fetch's running byte total has to DECOMPRESS every plan it +/// considers to measure it, so the id list handed to one pass must be capped near what the byte budget can +/// actually ship. Lives in the shared collectors project because it is pure arithmetic pinned by tests in +/// both hosts' suites. /// public static class QueryStorePlanXmlState { - /// - /// The collector_state owner name for these rows — deliberately NOT the query_store definition's name, - /// which is the seam that lets the definition declare no state keys while the host still persists - /// per-database state for it. - /// - public const string StateCollectorName = "query_store_plan_xml"; - - /// - /// State key prefix; the remainder is the database name, because plan_id is only unique within one - /// database's Query Store and means nothing across databases. - /// - public const string WatermarkKeyPrefix = "planwm:"; - - /// - /// The target period for ONE FULL RE-VERIFICATION SWEEP of a database's plans — not an expiry, and - /// emphatically not a refetch trigger. QueryStorePlanMap.CursorSliceWidth derives the cursor's - /// per-pass id slice from it, so this constant sets the PACE of re-verification rather than a deadline - /// anything has to beat. - /// - /// It used to mean "after this long, drop the watermark to zero and refetch every plan's XML", and - /// that was measured to be unreachable on the catalogs it mattered most for: 2.2-15.1 GB of plan XML per - /// catalog on the production fleet, which at a 12 MB budget and 5-minute cadence is 15.9 to 107.5 HOURS of - /// walking — so a 1-day expiry meant the biggest catalogs restarted from their lowest plan_id forever and - /// never once reached their newest plans. The optimization could not converge on exactly the databases it - /// existed for. Raising the number does not fix that shape; the sweep has to stop being a byte-volume walk. - /// It now is one: hash-only, bounded by ROW count (77k ids at ~270 per pass), re-fetching XML solely where - /// something actually changed. - /// - /// THREE MECHANISMS, each owning one failure, none of them this constant on its own: - /// - /// • A Query Store reset — the map's absent-content signal on the runtime stream - /// (TouchSql), recovering in one cycle. The ONLY thing permitted to zero a watermark.
- /// • Dormant plans — the cursor finds a map row ABSENT at an id the watermark already passed, and - /// fetches it. No heuristic separates dormancy from a reset, because it does not have to: mass absence is - /// caught wholesale by the reset arm within a cycle.
- /// • In-place XML rewrites — the cursor finds a stored plan_hash that DIFFERS from the live - /// one and re-fetches that plan alone. Across a day of fleet data this was 0 of 38,420 plan_ids, which is - /// why paying for it with a full walk was the wrong trade.
- /// - /// ONE DAY remains the right pace for a hash-only sweep, for the reason the old value was chosen and - /// for a new one: the redundancy removed is per-pass, and a sweep bounded by rows rather than bytes finishes - /// comfortably inside a day on every catalog measured. - /// - /// Historical note on the three guarantees the old expiry claimed, kept because the reasoning still - /// explains why each mechanism above exists: - /// - /// 1. In-place XML rewrites. plan_id is monotonic and a plan's identity is stable (0 of 38,420 - /// plan_ids changed their plan hash in a day of fleet data), but nothing guarantees a feature like - /// memory-grant feedback never edits grant values inside the XML of a plan that keeps its id. The - /// expiry means that question does not have to be load-bearing. - /// - /// 2. A Query Store RESET. Clearing Query Store restarts plan_id at 1, so every new plan sorts - /// below a stale watermark and its XML would be suppressed. This is NOT what covers that any more — the - /// tempting detection test ("the highest plan_id seen this pass is below the standing watermark") is TRUE in - /// any ordinary quiet window, so it would drop the watermark constantly; but the map gives the payload a - /// signal it never had on its own. A plan_id at or below the watermark whose content the store has never - /// resolved is a RENUMBERED plan, which "no new plans this window" cannot produce, and - /// QueryStorePlanMap.TouchSql surfaces exactly those rows from the batch join it already performs. - /// That is the reset mechanism, it recovers in ONE CYCLE, and it is the only thing permitted to zero the - /// watermark. - /// - /// 3. The dormant-plan gap: plan_id is monotonic in COMPILE order, which is not the same as "we - /// have stored it", so a plan compiled before monitoring began and dormant through every collected - /// window arrives below the watermark. - /// - /// ONE DAY, not a week: the redundancy removed is per-pass (a 15-minute cadence re-ships a plan - /// ~96 times a day), so a daily full fetch already eliminates ~99% of it and a weekly one adds almost - /// nothing — while buying 7x the exposure on all three guarantees above, including a reset blackout - /// measured in days. - /// - /// That trade omits a term, named here because it is the one that will move this number: expiry - /// resets the watermark to zero, so "one expensive pass" is really a full budgeted catalog WALK. At a 12 MB - /// ship budget an 82k-plan catalog spends most of a day walking, which means the largest catalogs — the ones - /// this optimization matters most for — are close to continuously refetching, and shortening the horizon - /// makes that worse rather than safer. Once the stream signal above covers resets, the walk buys only the - /// in-place-rewrite case (speculative: 0 of 38,420 plan_ids changed hash in a day of fleet data) and dormant - /// plans (real, small), and a longer horizon is likely correct. Measure the walk cost on the worst catalog - /// before changing it. - ///
- public static readonly TimeSpan RefreshAfter = TimeSpan.FromDays(1); - - /// The state key for one database. - public static string KeyFor(string databaseName) => WatermarkKeyPrefix + databaseName; - /// /// The average plan size assumed for a database with no previous pass to learn from. Deliberately near the /// LARGE end of the measured fleet range (per-quartile averages of 162 / 80 / 39 / 15 KB across 2,166 /// budget-cut passes on a 52-server fleet), because the estimate feeds a DIVISOR: over-estimating plan size - /// yields a SMALL candidate window, and small is the safe direction. A window that is too small merely - /// advances the watermark more slowly; one that is too large decompresses plans it will never ship, which - /// is the exact cost the window exists to bound. + /// yields a SMALL candidate cap, and small is the safe direction. A cap that is too small merely spreads + /// the catch-up across more cycles; one that is too large decompresses plans it will never ship, which is + /// the exact cost the cap exists to bound. /// public const long FirstContactAvgPlanBytes = 160L * 1024L; /// - /// Floor on the candidate window, so progress is always possible. Even if the observed average is wildly - /// over-stated — one enormous plan in a quiet pass — a database must still be able to walk its catalog. + /// Floor on the candidate cap, so progress is always possible. Even if the observed average is wildly + /// over-stated — one enormous plan in a quiet pass — a database must still be able to work off its + /// missing set. /// public const int MinCandidatePlans = 32; /// - /// Ceiling on the candidate window. The smallest measured quartile average (15 KB) puts a 12 MB budget at + /// Ceiling on the candidate cap. The smallest measured quartile average (15 KB) puts a 12 MB budget at /// ~820 plans, so this leaves headroom for genuinely tiny plans while refusing to let a near-zero estimate - /// turn the window back into "the whole catalog" — which is the first-contact trap this window exists to - /// prevent. + /// turn one pass back into "decompress the whole catalog" — which is the first-contact trap the cap exists + /// to prevent. /// public const int MaxCandidatePlans = 2048; /// - /// How far past the budget the window reaches, in expected plans. The window is the COARSE bound and the + /// How far past the budget the cap reaches, in expected plans. The cap is the COARSE bound and the /// running byte total is the exact one, so the margin only has to cover the estimate being wrong in the /// "plans are smaller than expected" direction — where extra plans genuinely fit the budget. /// /// Kept modest at 1.5x because margin is not free: a windowed running total is evaluated over every - /// row IN the window, so the server decompresses all K plans to compute it whether the budget is reached at - /// plan 5 or plan 500. Margin buys reachability and costs decompression, which is why the estimate errs - /// large and the margin stays small. + /// row handed to the fetch, so the server decompresses all of them to compute it whether the budget is + /// reached at plan 5 or plan 500. Margin buys reachability and costs decompression, which is why the + /// estimate errs large and the margin stays small. /// public const double CandidatePlanMargin = 1.5; @@ -174,8 +80,8 @@ public static class QueryStorePlanXmlState /// /// One database's carried plan-size estimate (#2312 Finding 1): the observed average the next pass - /// sizes its candidate window from, and whether the walk is mid-backlog (which biases the sample - /// small — the overload floors it). + /// caps its id list from, and whether the fetch is mid-backlog (which biases the sample small — the + /// overload floors it). /// AvgBytes of zero means "never learned"; callers pass null to CandidatePlanCount then. /// public readonly record struct PlanSizeEstimate(long AvgBytes, bool CatchUpInProgress); @@ -183,16 +89,16 @@ public static class QueryStorePlanXmlState /// /// Folds one pass's outcome into the carried estimate. The rules, each load-bearing: /// a pass that shipped nothing teaches nothing about size (previous average stands) but DOES - /// prove the walk is caught up (nothing qualified past the watermark), so catch-up clears; - /// a pass cut by either bound — the candidate window consumed or the byte budget reached — + /// prove the fetch is caught up (nothing was missing, or nothing fit), so catch-up clears; + /// a pass cut by either bound — the candidate cap consumed or the byte budget reached — /// proves a backlog remains, so catch-up sets; an ordinary partial pass learns its average and /// clears catch-up. Pure so the table is pinnable; the runner owns only the dictionary. /// /// Two counts on purpose (the review catch): is the RAW - /// row count — NULL-XML plans deliberately ship as rows so the watermark can pass unpersistable - /// plans, and the window/catch-up comparison wants exactly that count. But the average's divisor + /// row count — NULL-XML plans deliberately ship as rows so the store can record the content-less + /// marker, and the cap/catch-up comparison wants exactly that count. But the average's divisor /// is , the rows that actually carried XML: dividing real bytes - /// by a NULL-inflated count would understate the average, which INFLATES the next window — the + /// by a NULL-inflated count would understate the average, which INFLATES the next cap — the /// unsafe direction the whole estimator errs away from. /// public static PlanSizeEstimate Learn( @@ -214,32 +120,32 @@ public static PlanSizeEstimate Learn( /// /// This is the trap mitigation. SUM(DATALENGTH(query_plan)) OVER (ORDER BY plan_id) has to /// materialize the XML to measure it — query_store_plan.query_plan is decompressed BY the TVF on - /// access — so an unbounded candidate set pays the whole catalog's decompression to enforce a budget meant - /// to prevent exactly that. Bounding the window first on the cheap columns costs nothing and caps it. + /// access — so an unbounded id list pays the whole missing set's decompression to enforce a budget meant + /// to prevent exactly that. Capping the list first on the cheap side costs nothing. /// /// Per-database rather than one fleet constant because measured plan size spans 11x (162 KB to 15 KB /// by quartile). A constant sized for the small-plan end (~820) would decompress ~134 MB to ship 12 MB on /// the large-plan end; one sized for the large end would never reach the budget on the small end. No single /// value is both, which is what makes this adaptive rather than tunable. /// - /// reports that a bound was applied, so the caller can LOG it. A window + /// reports that a bound was applied, so the caller can LOG it. A cap /// silently pinned at its ceiling looks identical to one that fit, and that is how a cap becomes invisible. ///
public static int CandidatePlanCount(long? observedAvgPlanBytes, long budgetBytes, out bool clamped) => CandidatePlanCount(observedAvgPlanBytes, budgetBytes, catchUpInProgress: false, out clamped); /// - /// As above, with the catch-up guard: while — the watermark still below - /// the server's newest plan_id — the observed average is FLOORED at + /// As above, with the catch-up guard: while — the missing set still + /// larger than one pass can ship — the observed average is FLOORED at /// rather than trusted. /// /// The estimator is biased during exactly that window, and measurably so: the average is computed over /// the plans a pass actually shipped, which under plan_id-ascending shipping are the OLDEST ids in the - /// catalog. On one production catalog the plans the fetch shipped averaged 15 KB while the newest 300 plans - /// in the same catalog averaged 46 KB — a 3x under-estimate, which inflates K threefold and decompresses - /// that much more than the budget can ship. Flooring at the seed applies the same over-estimate-is-safe - /// logic the seed itself rests on, for the one window where the sample is known to be unrepresentative. - /// Once the first walk has converged the sample spans the catalog and the observed average is trusted. + /// missing set. On one production catalog the plans the fetch shipped averaged 15 KB while the newest 300 + /// plans in the same catalog averaged 46 KB — a 3x under-estimate, which inflates the cap threefold and + /// decompresses that much more than the budget can ship. Flooring at the seed applies the same + /// over-estimate-is-safe logic the seed itself rests on, for the one window where the sample is known to be + /// unrepresentative. Once caught up the sample spans the catalog and the observed average is trusted. /// public static int CandidatePlanCount(long? observedAvgPlanBytes, long budgetBytes, bool catchUpInProgress, out bool clamped) { @@ -261,133 +167,14 @@ public static int CandidatePlanCount(long? observedAvgPlanBytes, long budgetByte comparison below unable to tell a clamp from a natural landing, which is the false positive this reports on. int.MaxValue only guards the cast itself, since the budget is operator input. */ var wanted = (double)budgetBytes / avg * CandidatePlanMargin; - var unclamped = wanted >= int.MaxValue ? int.MaxValue : (int)Math.Ceiling(wanted); - var bounded = Math.Clamp(unclamped, MinCandidatePlans, MaxCandidatePlans); + var unclamped = wanted >= int.MaxValue ? int.MaxValue : (int)System.Math.Ceiling(wanted); + var bounded = System.Math.Clamp(unclamped, MinCandidatePlans, MaxCandidatePlans); - /* Reports that a bound CHANGED the answer, not that the answer happens to equal one. A window whose + /* Reports that a bound CHANGED the answer, not that the answer happens to equal one. A cap whose measured size lands naturally on 32 or 2048 was sized by the measurement and needs no log line; saying "clamped" there is a false positive against this contract, and a caller that logs on it teaches its reader to ignore the message. */ clamped = bounded != unclamped; return bounded; } - - /// - /// The watermark a pass earned, given the plan_ids whose XML actually landed. Under plan_id-ordered - /// shipping a budget cut truncates a SUFFIX, so the highest landed id is safe to keep even from a cut pass - /// — which is the whole point of the reordering (#2210): the previous design shipped in - /// last_execution_time order, where a cut left an arbitrary SUBSET and no value was safe, so the - /// watermark could not advance on 97.8% of passes and therefore never advanced at all. - /// - /// Defensive on the precondition rather than trusting it: a DESCENT anywhere in - /// abandons the advance entirely and reports itself through - /// . Honouring the leading ascending run instead - /// looks safer and is not — given {105, 101} it would advance to 105, and once ordering is broken - /// there is no longer any basis for inferring that every SELECTED plan below 105 landed, so a plan whose - /// XML never arrived gets suppressed until the refresh horizon. Ordering is what makes a cut a suffix; with - /// it gone the pass has earned nothing, and one lost pass of progress is the cheap side of that trade. - /// - /// The verdict and the signal come back TOGETHER, in one value, deliberately. Two separate functions - /// would let a caller take the watermark and never ask whether ordering held — a watermark that quietly - /// stops moving with nothing logged, which is precisely the failure this whole redesign exists to correct - /// and would be a poor thing to reintroduce one level up. - /// - /// Never moves backward: a pass that lands nothing, or only ids at or below the standing watermark, - /// returns the standing value. Lowering it would refetch the catalog, and "no new plans this window" is an - /// ordinary quiet pass, not a reset — the reset signal lives on the runtime stream, where a plan at or below - /// the watermark that the store has never resolved can actually be observed. - /// - public static PlanWatermarkAdvance AdvanceWatermark(long standing, IReadOnlyList landedPlanIdsInOrder) - { - if (landedPlanIdsInOrder is null || landedPlanIdsInOrder.Count == 0) - { - return new PlanWatermarkAdvance(standing, true); - } - - var advanced = standing; - var previous = long.MinValue; - - foreach (var planId in landedPlanIdsInOrder) - { - if (planId < previous) - { - return new PlanWatermarkAdvance(standing, false); - } - - previous = planId; - - if (planId > advanced) - { - advanced = planId; - } - } - - return new PlanWatermarkAdvance(advanced, true); - } - - /// - /// The watermark to apply for one database, or 0 — meaning "fetch every plan's XML" — for an absent, - /// malformed, EXPIRED or future-stamped one. Zero is the documented conservative path: absent is what a - /// first run, a restarted host and a broken store all look like, and all three must refetch rather than - /// skip. A future stamp means the clock moved backwards, which would otherwise pin the watermark for as - /// long as the skew lasts. - /// - public static long Resolve(IReadOnlyDictionary state, string databaseName, DateTime utcNow) - { - if (!TryParse(state, databaseName, out var planId, out var stamped)) - { - return 0; - } - - if (stamped > utcNow || utcNow - stamped >= RefreshAfter) - { - return 0; - } - - return planId; - } - - /// - /// The stored stamp — when this database last did a FULL plan-XML fetch — with no expiry applied, so a - /// write-back can carry it forward across an advance instead of renewing the refresh horizon. Null when - /// there is nothing parseable to carry, which the caller treats as "stamp now". - /// - public static DateTime? ResolveStamp(IReadOnlyDictionary state, string databaseName) => - TryParse(state, databaseName, out _, out var stamped) ? stamped : null; - - /// - /// Formats a watermark for storage: highest stored plan_id plus the stamp dating the last FULL fetch. - /// The stamp is a parameter rather than "now" precisely because it must survive advances — re-stamping - /// on every advance would push the horizon out forever on any database that keeps compiling plans, which - /// is the busy ones where a stale plan matters most, and the bounded refresh would never fire. - /// - public static string Format(long planId, DateTime fullFetchAtUtc) => - planId.ToString(CultureInfo.InvariantCulture) + ":" + - new DateTimeOffset(DateTime.SpecifyKind(fullFetchAtUtc, DateTimeKind.Utc)).ToUnixTimeSeconds() - .ToString(CultureInfo.InvariantCulture); - - private static bool TryParse( - IReadOnlyDictionary state, string databaseName, out long planId, out DateTime stamped) - { - planId = 0; - stamped = default; - - if (state is null || !state.TryGetValue(KeyFor(databaseName), out var raw) || string.IsNullOrWhiteSpace(raw)) - { - return false; - } - - var parts = raw.Split(':'); - if (parts.Length != 2 - || !long.TryParse(parts[0], NumberStyles.Integer, CultureInfo.InvariantCulture, out planId) - || !long.TryParse(parts[1], NumberStyles.Integer, CultureInfo.InvariantCulture, out var stampedUnix) - || planId <= 0) - { - planId = 0; - return false; - } - - stamped = DateTimeOffset.FromUnixTimeSeconds(stampedUnix).UtcDateTime; - return true; - } } diff --git a/PerformanceMonitor.Collectors/QueryStoreTextState.cs b/PerformanceMonitor.Collectors/QueryStoreTextState.cs deleted file mode 100644 index 10b4ec7fd..000000000 --- a/PerformanceMonitor.Collectors/QueryStoreTextState.cs +++ /dev/null @@ -1,220 +0,0 @@ -/* - * Copyright (c) 2026 Erik Darling, Darling Data LLC - * - * This file is part of the SQL Server Performance Monitor. - * - * Licensed under the MIT License. See LICENSE file in the project root for full license information. - */ - -using System; -using System.Collections.Generic; -using System.Globalization; - -namespace PerformanceMonitor.Collectors; - -/// The result of advancing a text watermark: see . -public readonly record struct TextWatermarkAdvance(long Watermark, bool ArrivedInQueryIdOrder); - -/// -/// Per-database watermark for the query-text fetch (#2150), the sibling of -/// — same encoding, same conservative-zero rules, same -/// never-backward advance. -/// -/// Why this exists. The runtime-stats payload carried query_sql_text -/// (nvarchar(max)) inside a TOP ... WITH TIES ... ORDER BY last_execution_time projection. -/// A Top-N Sort carries every output column through the sort and reads ALL of its input before emitting -/// a row, so choosing the rows to ship materialized the text for the entire qualifying set. Measured on -/// a purpose-built Azure SQL DB store with the plan XML already removed by #2210 — the only difference -/// being that one column — time-to-first-row was 4.67s vs 0.45s at 1,505 rows / 12.8 MB of text -/// and 5.02s vs 0.57s at 4,037 rows / 34 MB, with full drain 8.06s vs 0.50s and -/// 16.95s vs 1.45s. Neither the row cap nor the client byte budget can bound that: TOP (500) -/// measured identical to TOP (50000), and wall time was flat across a 4 MB → 256 MB budget sweep, -/// because the server finishes before the client sees a byte. -/// -/// Why a watermarked fetch rather than a per-pass dedupe. That path was already tried on the -/// plan side and abandoned: #1556's ROW_NUMBER gate shipped each plan once per PASS, and #2164 -/// replaced it precisely because "the ROW_NUMBER gate ships each plan once per pass but re-ships it every -/// pass forever, and since drain is 94-97% of a pass and is per-row LOB cost, NOT fetching is worth far -/// more than fetching less." #2210 then took the column out of the stream entirely. -/// query_id is an identity, monotonic within a database, so the same shape applies: fetch a -/// statement's text ONCE, ever. -/// -/// Keyed on query_id, not query_text_id, and that is what keeps this cheap. -/// query_id is ALREADY a stored payload column on the runtime row, so readers get the join key for -/// free and the fact table needs no new column and no migration. Keying on query_text_id would have -/// required adding it to the payload — a schema change — to buy de-duplication across the handful of -/// query_ids that share one text (a query_id is per text PLUS context settings, so the two -/// are close to 1:1 in practice). Storing a rare duplicate is the cheaper side of that trade. -/// -/// What this deliberately does NOT mirror, and why. The plan side carries a whole candidate- -/// window estimator (FirstContactAvgPlanBytes, min/max clamps, an observed-average learning loop) -/// because SUM(DATALENGTH(query_plan)) OVER (ORDER BY plan_id) forces the server to DECOMPRESS -/// every plan in the window — sys.query_store_plan.query_plan is decompressed by the view on -/// access. sys.query_store_query_text.query_sql_text is not, so its DATALENGTH is cheap and -/// the window needs no estimate at all: a flat coarse bound plus the exact running-byte total is enough. -/// The plan side also re-verifies content hashes because plan XML can be rewritten in place; a -/// query_text_id maps to fixed text forever — a changed statement is a new id — so there is -/// nothing to re-verify and no content digest to track. -/// -public static class QueryStoreTextState -{ - /// - /// The collector name the watermark is stored under. Separate from the plan fetch's own state so the - /// two advance independently: they walk different catalogs at different rates, and sharing a key would - /// let a plan-side reset drop the text watermark (and vice versa) for no reason. - /// - public const string StateCollectorName = "query_store_text"; - - /// Prefix for the per-database state key. - public const string WatermarkKeyPrefix = "textwm:"; - - /// - /// How long a watermark stands before a full re-walk. Matched to the plan side's one day rather than - /// tuned separately, so an operator reasoning about one fetch reasons about both — and the term that - /// made the plan side's choice tight does not apply here: expiry means a budgeted catalog walk, and - /// text is roughly an order of magnitude smaller per row than plan XML (8.5 KB against 195 KB on the - /// measured store), so the walk this horizon triggers is correspondingly cheaper. - /// - /// The re-walk is not decoration. query_id is monotonic in FIRST-SEEN order, not in "we - /// have stored it", so two things arrive below a standing watermark: a statement first seen before - /// monitoring began and only executed later, and — the one that matters — a Query Store reset, which - /// renumbers ids from the start. Without a bounded horizon a reset would suppress every text forever. - /// - public static readonly TimeSpan RefreshAfter = TimeSpan.FromDays(1); - - /// - /// How many texts one pass may CONSIDER. A flat bound, not an estimate: the running byte total is the - /// exact constraint and DATALENGTH(query_sql_text) is cheap to evaluate, so this only has to be - /// large enough that the budget binds first and small enough that a pass never windows an entire - /// catalog. At the 12 MB default ship budget this covers texts averaging under ~2.5 KB, which is - /// comfortably below what a fragmenting literal-heavy statement produces. - /// - public const int CandidateTexts = 5_000; - - /// The state key for one database. - public static string KeyFor(string databaseName) => WatermarkKeyPrefix + databaseName; - - /// - /// The highest query_id landed, or the standing watermark when a pass lands nothing. - /// - /// Reports whether the ids arrived in query_id order, because that ordering is what - /// makes a budget cut a SUFFIX — everything up to the cut is stored, so the highest stored id is a - /// safe resume point. Out of order, that argument collapses and the caller must hold the watermark - /// rather than advance past statements whose text it never stored. - /// - /// Never moves backward. A pass landing nothing, or only ids at or below the standing watermark, - /// is an ordinary quiet pass — not a reset — and lowering the watermark would refetch the catalog. - /// - public static TextWatermarkAdvance AdvanceWatermark(long standing, IReadOnlyList landedQueryIdsInOrder) - { - if (landedQueryIdsInOrder is null || landedQueryIdsInOrder.Count == 0) - { - return new TextWatermarkAdvance(standing, true); - } - - var advanced = standing; - var previous = long.MinValue; - - foreach (var queryId in landedQueryIdsInOrder) - { - if (queryId < previous) - { - return new TextWatermarkAdvance(standing, false); - } - - previous = queryId; - - if (queryId > advanced) - { - advanced = queryId; - } - } - - return new TextWatermarkAdvance(advanced, true); - } - - /// - /// The watermark to apply for one database, or 0 — meaning "fetch every text" — for an absent, - /// malformed, EXPIRED or future-stamped one. Zero is the conservative path, and all three of a first - /// run, a restarted host and a broken store look identical from here: every one of them must refetch - /// rather than skip. A future stamp means the clock moved backwards, which would otherwise pin the - /// watermark for as long as the skew lasts. - /// - public static long Resolve(IReadOnlyDictionary state, string databaseName, DateTime utcNow) - { - if (!TryParse(state, databaseName, out var textId, out var stamped)) - { - return 0; - } - - if (stamped > utcNow || utcNow - stamped >= RefreshAfter) - { - return 0; - } - - return textId; - } - - /// - /// The stored stamp — when this database last did a FULL text fetch — with no expiry applied, so a - /// write-back can carry it forward across an advance instead of renewing the refresh horizon. Null - /// when there is nothing parseable to carry, which the caller treats as "stamp now". - /// - public static DateTime? ResolveStamp(IReadOnlyDictionary state, string databaseName) => - TryParse(state, databaseName, out _, out var stamped) ? stamped : null; - - /// - /// Formats a watermark for storage: highest stored query_id plus the stamp dating the last - /// FULL fetch. The stamp is a parameter rather than "now" precisely because it must survive advances — - /// re-stamping on every advance would push the horizon out forever on any database that keeps seeing - /// new statements, which is exactly where a reset would hurt most, and the bounded re-walk would never - /// fire. - /// - public static string Format(long textId, DateTime fullFetchAtUtc) => - textId.ToString(CultureInfo.InvariantCulture) + ":" + - new DateTimeOffset(DateTime.SpecifyKind(fullFetchAtUtc, DateTimeKind.Utc)).ToUnixTimeSeconds() - .ToString(CultureInfo.InvariantCulture); - - private static bool TryParse( - IReadOnlyDictionary state, string databaseName, out long textId, out DateTime stamped) - { - textId = 0; - stamped = default; - - if (state is null || !state.TryGetValue(KeyFor(databaseName), out var raw) || string.IsNullOrWhiteSpace(raw)) - { - return false; - } - - var split = raw.IndexOf(':'); - if (split <= 0 || split == raw.Length - 1) - { - return false; - } - - if (!long.TryParse(raw.AsSpan(0, split), NumberStyles.Integer, CultureInfo.InvariantCulture, out textId) - || textId < 0) - { - textId = 0; - return false; - } - - if (!long.TryParse(raw.AsSpan(split + 1), NumberStyles.Integer, CultureInfo.InvariantCulture, out var unix)) - { - textId = 0; - return false; - } - - try - { - stamped = DateTimeOffset.FromUnixTimeSeconds(unix).UtcDateTime; - } - catch (ArgumentOutOfRangeException) - { - textId = 0; - return false; - } - - return true; - } -} diff --git a/PerformanceMonitor.Collectors/WatermarkPolicy.cs b/PerformanceMonitor.Collectors/WatermarkPolicy.cs index 31881a315..16be1e958 100644 --- a/PerformanceMonitor.Collectors/WatermarkPolicy.cs +++ b/PerformanceMonitor.Collectors/WatermarkPolicy.cs @@ -72,4 +72,39 @@ public static class WatermarkPolicy var floor = now - MaxCatchup; return watermark.Value < floor ? floor : watermark; } + + /// + /// Extra history the watermark READ may look at beyond (#2344). Purely a + /// safety margin for clock disagreement between a monitored server and the store — the correctness + /// argument needs none of it, so it is generous rather than tuned. + /// + public static readonly TimeSpan ReadFloorMargin = TimeSpan.FromHours(2); + + /// + /// The oldest collection_time a clamped watermark read has to consider (#2344), or null when + /// is default — callers pass this straight through to the store read as an + /// optional bound. + /// + /// Why bounding the read changes no answer. Every consumer of a clamped watermark ends + /// up at max(stored, now - MaxCatchup): floors anything older, and a + /// NULL result falls back to query_store's documented 60-minute first-run window — the same instant as + /// the floor. So a row older than the horizon cannot move the result whether it is found or not, and + /// the unbounded MAX that used to find it was paying to confirm a value the clamp would have + /// produced anyway. Measured on the 106 GB use1 store: 25,766 buffer reads plus temp spill cold, 228 ms + /// warm, against 29 ms bounded (five chunks excluded) — and the unbounded cost scales with STORE SIZE + /// and cache residency rather than with anything the monitored server is doing, so it degrades exactly + /// where an operator is weakest. + /// + /// Bound the PARTITIONING column, not the watermark column. The hypertables partition on + /// collection_time; a predicate on the watermark column alone prunes nothing. This is safe + /// because a row's watermark value can never exceed its own collection_time — an execution + /// cannot be collected before it happens — so no qualifying row hides behind the bound. + /// + /// Only for readers whose value is clamped. The clamp is scoped to query_store (see the + /// class remarks); a ring-buffer collector whose legitimate catch-up spans days must keep reading its + /// full history, and handing it this floor would silently truncate that. A future definition wanting + /// the bound has to adopt first — the two travel together. + /// + public static DateTime? ReadFloor(DateTime now) => + now == default ? null : now - MaxCatchup - ReadFloorMargin; } diff --git a/PerformanceMonitor.Common/Mcp/McpHelpers.cs b/PerformanceMonitor.Common/Mcp/McpHelpers.cs index bee0c8070..903032f82 100644 --- a/PerformanceMonitor.Common/Mcp/McpHelpers.cs +++ b/PerformanceMonitor.Common/Mcp/McpHelpers.cs @@ -27,9 +27,26 @@ internal static class McpHelpers public const int MaxTop = 1000; /// - /// Shared JSON serializer options with indented formatting. + /// Shared JSON serializer options for MCP tool results — compact, not indented (#2350). + /// + /// The only consumer of an MCP tool result is a language model, and indentation buys a model + /// nothing. It was costing roughly 23% of the bytes of a record-heavy result (measured on a 15-field + /// blocking-event shape: 2,977 → 2,297 at 10 rows, 29,082 → 22,462 at 100). The token saving is smaller + /// than the byte saving — BPE tokenizers pack runs of spaces efficiently — so this is not the 23% + /// win it looks like in bytes. It is still free, and it compounds where it matters: tool results are the + /// bulk of what fills an agent's context on a real incident, and the fleet-wide reads are the widest + /// results we return. + /// + /// Deliberately NOT applied to the config files (servers.json, profiles, schedules, alert state). + /// Those are read and hand-edited by people, and ServerManager/ProfileManager/ + /// ScheduleManager keep their own indented options for that reason. This object is MCP output only + /// — every one of its ~78 call sites serializes a tool result or the web endpoint twin of one. + /// + /// Nothing parses our output positionally: it is JSON to a JSON reader on both sides, and the tests + /// that touch this object assert field NAMES (there is no naming policy here, so snake_case comes from + /// [JsonPropertyName] attributes) rather than layout. /// - public static readonly JsonSerializerOptions JsonOptions = new() { WriteIndented = true }; + public static readonly JsonSerializerOptions JsonOptions = new() { WriteIndented = false }; /// /// Truncates a string to the specified maximum length, adding a truncation suffix. diff --git a/PerformanceMonitor.Notifications/AlertContext.cs b/PerformanceMonitor.Notifications/AlertContext.cs index bf42507d8..fe9d72261 100644 --- a/PerformanceMonitor.Notifications/AlertContext.cs +++ b/PerformanceMonitor.Notifications/AlertContext.cs @@ -74,7 +74,22 @@ public sealed record AlertIncident( string? WaitRange = null, IReadOnlyList? DetailFields = null, long? TotalOccurrences = null, - DateTime? IncidentStartedUtc = null); + DateTime? IncidentStartedUtc = null, + /* #2361: the incident's database scope, as a discrete member rather than something a consumer has to + string-search Details[] for. That search is exact only for deadlocks -- they are self-contained, built + with includeDetailFields: true -- while every other fingerprinted alert appends a BARE Incident item + beside its data item, so the incident's own section carries no Database at all and the fallback becomes + "any Database anywhere in the payload". On a multi-incident alert spanning databases that is not an + approximation, it is the wrong value with nothing marking it wrong. + + Distinct from InvolvedObjects, which is what the incident is ABOUT (tables, mount points, job names). + This is where it lives. Null when the alert is not database-scoped -- a volume or a job is not. */ + string? Database = null, + /* #2361: the newest event in this incident, the counterpart to IncidentStartedUtc. Populated from + IncidentOccurrenceState.LastObservedUtc, which the accumulator already computes and persists as the + value its staleness horizon compares against -- so this is a projection of something that existed, not + a new measurement. Null on any alert the accumulator does not run for. */ + DateTime? LastEventUtc = null); /// /// A forensic label/value pair carried on an for #1141 Per-event delivery @@ -148,7 +163,11 @@ public record AlertIncidentDto( int OccurrenceCount = 1, string? WaitRange = null, long? TotalOccurrences = null, - DateTime? IncidentStartedUtc = null); + DateTime? IncidentStartedUtc = null, + /* #2361. Trailing and optional for the same reason the rest of this DTO is: legacy contextJson written + before these existed deserializes them to null rather than failing the round trip. */ + string? Database = null, + DateTime? LastEventUtc = null); /// /// JSON mirror of / @@ -356,7 +375,9 @@ public static string SerializeIncidents(AlertContext? context) i.OccurrenceCount, i.WaitRange, i.TotalOccurrences, - i.IncidentStartedUtc)); + i.IncidentStartedUtc, + i.Database, + i.LastEventUtc)); /// /// Serializes a single to JSON for persistence on a diff --git a/PerformanceMonitor.Notifications/AlertFingerprint.cs b/PerformanceMonitor.Notifications/AlertFingerprint.cs index 5a8d9c533..aaf2faf8d 100644 --- a/PerformanceMonitor.Notifications/AlertFingerprint.cs +++ b/PerformanceMonitor.Notifications/AlertFingerprint.cs @@ -89,14 +89,20 @@ public static class AlertFingerprint string naturalKey, IReadOnlyList? displayObjects = null, int occurrenceCount = 1, - string? waitRange = null) + string? waitRange = null, + /* #2361: the incident's database scope. Optional because most fingerprint kinds are not + database-scoped -- a disk or a job is not -- and a caller that has no database says so by omission + rather than by passing an empty string that would read as "no database" downstream. */ + string? database = null) { var normalized = Normalize(naturalKey); if (normalized.Length == 0) return null; var key = Hash(BuildInput(serverName, incidentType, new[] { normalized })); - return new AlertIncident(key, displayObjects ?? Array.Empty(), occurrenceCount, waitRange); + return new AlertIncident( + key, displayObjects ?? Array.Empty(), occurrenceCount, waitRange, + Database: string.IsNullOrWhiteSpace(database) ? null : database); } /// SHA-256 of as lowercase hex (64 chars). Public for tests. diff --git a/PerformanceMonitor.Notifications/IncidentGrouping.cs b/PerformanceMonitor.Notifications/IncidentGrouping.cs index cf9500d6b..73aedc342 100644 --- a/PerformanceMonitor.Notifications/IncidentGrouping.cs +++ b/PerformanceMonitor.Notifications/IncidentGrouping.cs @@ -295,7 +295,16 @@ the INCIDENT IDENTITY is normalized. */ // #1141: carry the chain's forensic detail on the incident so per-event cards keep it // (Summary already shows it via the builder's items; this travels for the per-event split). - var enriched = incident with { DetailFields = BlockingDetail(representative) }; + /* #2361: the group already knows its database, so the incident carries it rather than leaving a + consumer to string-search Details[] for it. Attached at the `with` so both the ForObjects and + ForKey branches above get it from one place. UnknownDatabase is ContentiousObjectLabel's sentinel for + "the row had none" and becomes null here -- a literal "unknown" in a Database field reads as a + database actually called that. */ + var enriched = incident with + { + DetailFields = BlockingDetail(representative), + Database = IncidentDatabaseHelpers.NormalizeDatabase(representative.Database), + }; groups.Add(new BlockingGroup( representative.Database, representative.ContentiousObject, @@ -354,6 +363,43 @@ private static List BlockingDetail(BlockedEvent e) private static string Truncate(string s) => s.Length <= 300 ? s : s.Substring(0, 300) + "…"; } +internal static class IncidentDatabaseHelpers +{ + /// + /// The grouper's UnknownDatabase sentinel means "the row carried none" (#2361). It is fine as a + /// display string and wrong as a data member: a consumer routing on Database would file tickets + /// against a database literally named "unknown". + /// + internal static string? NormalizeDatabase(string? database) => + string.IsNullOrWhiteSpace(database) + || string.Equals(database, ContentiousObjectLabel.UnknownDatabase, StringComparison.OrdinalIgnoreCase) + ? null + : database; + + /// + /// Pulls the #2109 Database fact off an incident's detail fields (#2361) — the value a consumer was + /// otherwise string-searching the payload for. Label match is case-insensitive; the first wins, because a + /// deadlock graph spanning several databases already ships them as one CSV rather than repeated fields. + /// + internal static string? DatabaseFromFields(IReadOnlyList? fields) + { + if (fields is null) + { + return null; + } + + foreach (var field in fields) + { + if (string.Equals(field.Label, "Database", StringComparison.OrdinalIgnoreCase)) + { + return NormalizeDatabase(field.Value); + } + } + + return null; + } +} + /// /// Groups deadlock events by the sorted set of fully-qualified objects involved (#1140). One /// deadlock spanning multiple databases/objects is a single incident listing them all; recurrences @@ -408,7 +454,15 @@ public static List Group(string serverName, IEnumerable - - net10.0 - enable - disable - latest - PerformanceMonitor.PlanAnalysis - PerformanceMonitor.PlanAnalysis - Darling Data, LLC - Copyright © 2026 Darling Data, LLC - true - latest-recommended - CA1849;CA2007;CA1508;CA1822;CA1805;CA1510;CA1816;CA1861;CA1845;CA2201;CA1848;CA1852;CA1305;CA1860;CA1707;CA1507;CA1806 - - - - - - - - - - - - - - - - - - - - + + + net10.0 + enable + disable + latest + PerformanceMonitor.PlanAnalysis + PerformanceMonitor.PlanAnalysis + Darling Data, LLC + Copyright © 2026 Darling Data, LLC + true + latest-recommended + CA1849;CA2007;CA1508;CA1822;CA1805;CA1510;CA1816;CA1861;CA1845;CA2201;CA1848;CA1852;CA1305;CA1860;CA1707;CA1507;CA1806 + + + + + + + + + + + + + + + + + + + + + diff --git a/deprecated/Dashboard.Tests/Dashboard.Tests.csproj b/deprecated/Dashboard.Tests/Dashboard.Tests.csproj index b36d3f555..de6a137f0 100644 --- a/deprecated/Dashboard.Tests/Dashboard.Tests.csproj +++ b/deprecated/Dashboard.Tests/Dashboard.Tests.csproj @@ -5,15 +5,12 @@ true false true + + Exe CA1849;CA2007;CA1508;CA1822;CA1805;CA1510;CA1816;CA1861;CA1845;CA2201;CS4014;NU1701;CA1001;CA1848;CA1852;CA1305;CA1860;CA1707;CA1507;CA1806 - - - all - runtime; build; native; contentfiles; analyzers; buildtransitive - diff --git a/deprecated/Dashboard.Tests/packages.lock.json b/deprecated/Dashboard.Tests/packages.lock.json index a7111190c..e5d2db0fc 100644 --- a/deprecated/Dashboard.Tests/packages.lock.json +++ b/deprecated/Dashboard.Tests/packages.lock.json @@ -2,22 +2,6 @@ "version": 2, "dependencies": { "net10.0-windows7.0": { - "Microsoft.NET.Test.Sdk": { - "type": "Direct", - "requested": "[18.8.1, )", - "resolved": "18.8.1", - "contentHash": "dknJL3/9Y3t4XuCBqnc0PevPxgLsUMmVhjwup/b1HNovA8zWcj3XsfIf7c6p05363DWcqL7X/YhDL9B+Zymv1w==", - "dependencies": { - "Microsoft.CodeCoverage": "18.8.1", - "Microsoft.TestPlatform.TestHost": "18.8.1" - } - }, - "xunit.runner.visualstudio": { - "type": "Direct", - "requested": "[3.1.5, )", - "resolved": "3.1.5", - "contentHash": "tKi7dSTwP4m5m9eXPM2Ime4Kn7xNf4x4zT9sdLO/G4hZVnQCRiMTWoSZqI/pYTVeI27oPPqHBKYI/DjJ9GsYgA==" - }, "xunit.v3": { "type": "Direct", "requested": "[3.2.2, )", @@ -88,11 +72,6 @@ "resolved": "9.0.13", "contentHash": "5T+bH3Lb1nEe8Hf/ixMxLmhlrx5wRi53wv7OhVwG2F1ZviW1ejFRS1NHur3uqPpJRGtkQwUchtY6zhVK2R+v+w==" }, - "Microsoft.CodeCoverage": { - "type": "Transitive", - "resolved": "18.8.1", - "contentHash": "Eclse/ZZjr4lmWzZFNN9h/OluhKL+SK/QbUyKUewgX139aGeyMEO/DkMPwuFs2MixvanTnz6891rF8UHDg+W4Q==" - }, "Microsoft.Data.SqlClient.Extensions.Abstractions": { "type": "Transitive", "resolved": "7.0.2", @@ -138,215 +117,215 @@ }, "Microsoft.Extensions.Configuration.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "5Vnd2I75DmZCVEjSynIdJ/0EGafgnLQwgR3t2C2/fkjx/nRG+cLwxLLdInoHeCEpkD5K4Ov/g9ZCRYrl4TRsaA==", + "resolved": "10.0.11", + "contentHash": "fVi053xdpda9Em7vSkmgVxO/PtgC2m78ekReKWsgcyskqY0U82Bz/MONwxpGzI0hElYKJfw+fupqMVeKW3fSaA==", "dependencies": { - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Configuration.Binder": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "GqmN2o1CkJvk7uWp+p4CwBYW0w/zfoEbvsiFDbO2G8l1Uz+mrDAbAcZiXhU2lufKPby1cjAUdd5GTWpebYOkOA==", + "resolved": "10.0.11", + "contentHash": "rFn8RuszZn3qquPVkDytMUlPc2+rXl9MCoygwc1XmAgC5vg5/oXJ8hkOosOrLoBLsqdTy4lFwP6iQdPS9uSYOA==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Configuration.CommandLine": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "33cBeR2HRbzHUTtmcmLdNOApneNGcymwwL4arHuotgVK9Frba8kcDTrvVTj7cSCmF1R9OiSbZH0KxNOwab3HUg==", + "resolved": "10.0.11", + "contentHash": "1KHr/1L56llwQ/yI0tAisEA31UpPsn8aasjASIwELOaN4JIUcbjuQBMdFOIzfNBBeULoUa0XfBe5QDtRRUY+fg==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Configuration.EnvironmentVariables": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "KRfFSSCV58vEdU7mPED/YMzeovIWF5P0g8s9K8n9HEfy0/WzMq37SrPdXdFN5/dFT/rPMHpF7AvpoXHckbcBFg==", + "resolved": "10.0.11", + "contentHash": "KICyU3eVi5jvloKm01EXV69L97H/zkhISVtV98cIuzuFOxNx3xTUVcXqvWTz3aq7OvUuDB/MFlPFjmxRaKF7/A==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Configuration.FileExtensions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "ZOhZYwvbXGTgGVRwswIirofEMVHuWdxjdh0JeUZXwaF9cgcjXdz/t0ELtgaevw7ezTyv47yPNCgGreWtLkn3IQ==", + "resolved": "10.0.11", + "contentHash": "mDW7KVFB05M6jiRUyaZiOMWhS31n5HlSZwoYctHAZAucD4sMDJ70IxOmkGDt6RpstchD+keWBjhdzcMpSkWvWQ==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10", - "Microsoft.Extensions.FileProviders.Physical": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11", + "Microsoft.Extensions.FileProviders.Physical": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Configuration.UserSecrets": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "1s1sKFTk/Foam64JY6+m/diH8drL3Wx6V3gtSd5v1IEZtszZYyc1pW8uRnMblzpNiR0l0t8gGk7tXj3xHzFgdg==", + "resolved": "10.0.11", + "contentHash": "BRliLdUowglV8GS+J1G/QsSofCJYYFg3U8QZx0ACRn+a91az/Qnpy+h6PyHS94WgV2TSazX5D/cuuk6wnCJatw==", "dependencies": { - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Configuration.Json": "10.0.10", - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10", - "Microsoft.Extensions.FileProviders.Physical": "10.0.10" + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Configuration.Json": "10.0.11", + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11", + "Microsoft.Extensions.FileProviders.Physical": "10.0.11" } }, "Microsoft.Extensions.DependencyInjection": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "ANyvsgkNBRvcJh2XLgn8veGmajf+8m0AbKK+HPWdRL1yraSNVVSmQhFntLtdz/C795jxqqup+k05cs/3jZQPOA==", + "resolved": "10.0.11", + "contentHash": "PSmotV19c7E3lKed++uYo1kSiXFI+uTl37CBSrhq+CfLC3FCHjG7R91+xPnNehQfHS1b0Tzo/CCLPWH3qaEheg==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11" } }, "Microsoft.Extensions.DependencyInjection.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "z/2xXlFw2aLGjHyEm6E0tQ+In6VfzQzTrtArbQ2c0TQE16ZbyDCMGPvaUT9I0s8rgy9sRWlU2P9waW37qV04qA==" + "resolved": "10.0.11", + "contentHash": "/a1aJz4m7ylhEDf25ugQChLQoN5XwoGjWw/BoR/ZWWKsO1v4DdJElS1uyngahz4B/eOzjFk1KNTkarRLE5wsIg==" }, "Microsoft.Extensions.Diagnostics": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "Kr/e7lUf4+N8tacbqJ2Ctwe/HarKdAc9ZkgKVVqvtJDBKbez+T/KnUwu82KSlnBp/SrpBcxc7u7xkE2oUZT/5Q==", + "resolved": "10.0.11", + "contentHash": "HT70uGPxMLqqnOzKMcnQtDmeV4r0KHr4qVCLhP7SXil9jMEm8sQXwcybxVVFGXZJ1V44xV0mLqQ54aZbcR2OiQ==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Diagnostics.Abstractions": "10.0.10", - "Microsoft.Extensions.Options.ConfigurationExtensions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Diagnostics.Abstractions": "10.0.11", + "Microsoft.Extensions.Options.ConfigurationExtensions": "10.0.11" } }, "Microsoft.Extensions.Diagnostics.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "9uWiKpeOVac355STyChWR/pliFX/5CeLqChW9kKsaxyDH4EUTZxMkT4Jwp/J/peLm0GBFmSX5c0WCse3yCnq1Q==", + "resolved": "10.0.11", + "contentHash": "se7Kx8QpJEt+nf26L4qIVAofGTDr1wbexxsh/Fm3Xc04xUkqUXK06KUS7FLwSQYSjqb7q9n+T7MEcXYBhI1Y5g==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11" } }, "Microsoft.Extensions.FileProviders.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "c5zqFCY9DiIpMovLd7/d/CTiEtrMOuQ639dhv3PABtKQIKNQikSHwQt8+N679uii9q+B55lgK28Uv64FOwEu8w==", + "resolved": "10.0.11", + "contentHash": "JOjac6SQQgZmdmB8WGEw61/7siqMZoWJMkmq2p1goJGxqI59lO6oB4bOl0jNsbaPBdYy5Mlkb+6U7T4+CjnD8Q==", "dependencies": { - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.FileProviders.Physical": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "jhJAyo38kSrH3ARvWUk0h8itogVnQu2DCZuPo+s0Z+tXes0ugTxMPaHYzap85785eHQmPFqD9TYERqBbtGxn/w==", + "resolved": "10.0.11", + "contentHash": "Tq/UqMaczePv9yWwSsJZRgKtgA46djVR5xHj/lZBCueQ3ag8f9v5mu0EdhrNx7tXxNk+Y9OurG2oKuSKINjr0A==", "dependencies": { - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10", - "Microsoft.Extensions.FileSystemGlobbing": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11", + "Microsoft.Extensions.FileSystemGlobbing": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.FileSystemGlobbing": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "jSOCVxEwCd4Aq925kJVz1kSO1EpX2OHYKL04qVREXkDU7Ce3pVDdHPYm+fEy8y/th2kJf/DAstRHpJAqoNWP8w==" + "resolved": "10.0.11", + "contentHash": "2i6rtW/B5rCnWCnhdmWWEmaM9O0HD0zsPY9eRqa++y4tclI3Uw8zvGbBvhY/LjAdtf8gUHhUPcAWj3DRlWMXmQ==" }, "Microsoft.Extensions.Hosting.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "5LugpYGHk+mkn0a8IZgcyfBca8PCTAU9RQFoMrTdtOOidq88M2SI5f3px6ugnzgxC+eTkvYYJi8pzlUnG5xdAQ==", + "resolved": "10.0.11", + "contentHash": "pwtpF7iF/NNaOBcX+pvMZ7y2+JAVbH5KkNrH9uMZtuVxVJsFTDiWiCR7Tk3HVptsAijaetipUZVRVK2LLq+nvA==", "dependencies": { - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Diagnostics.Abstractions": "10.0.10", - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10" + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Diagnostics.Abstractions": "10.0.11", + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Logging.Configuration": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "cLrqxkuEfcilZ8SjK+9KAnpLk9lOoMPaOokF+wRUYie+iUEcdX4/p/+gJkt0BYgWLthjpBUCkVTBI6Kxg0nsOw==", + "resolved": "10.0.11", + "contentHash": "S7LvLeVHKNPaY2NMyxW7c2TBGsLgxoSUBCV5Ev5iN8kgC7EPR2UB7eW7vHsElGMcIUDwRmoxLfvGDynCn3q6EA==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Configuration.Binder": "10.0.10", - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10", - "Microsoft.Extensions.Options.ConfigurationExtensions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Configuration.Binder": "10.0.11", + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11", + "Microsoft.Extensions.Options.ConfigurationExtensions": "10.0.11" } }, "Microsoft.Extensions.Logging.Console": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "VIlNzPwPS0GeQVSmCqqo36ugryX3LpE9ul6gEkks5VLET3weH/XMLeWmclwfoGn4Nxi2mwVibB+OZBVJ9tDqvg==", + "resolved": "10.0.11", + "contentHash": "dFc0yDudyD1iIg6z9XT7ofsT3hVO7Y4ylrxGHIVRR0GaZ4CUk4ujOrMoy7wWEdNHZhvJooySg6hZpOxyS8zEVA==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging.Configuration": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging.Configuration": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11" } }, "Microsoft.Extensions.Logging.Debug": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "8+TZBnV5fgBXoVNJ5ROSErUwYogk4hOgV7c2HWK1u5cqKGmiUTUn7+KqZ35iQu8e/B7Ykccyz5OTjdXcidNZ9g==", + "resolved": "10.0.11", + "contentHash": "wr+j1bjdFXhc8lKTLoq+RbwFM8M+orcMS9xrcqLmDGOxJcXpKizEeE5h6v/GKwCZV02FmhaA7OlNjoq072jZpQ==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Logging.EventLog": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "0RE4951AzQ+YD4gVrvbq0BhdsiBgSDo44yM7+QBZ2mrmMJeNjY+teCIYfUjqDPVYnKs0HR6SkkhgrX1YgXZq3Q==", + "resolved": "10.0.11", + "contentHash": "Eck9GpCCpvZ3f6L7IUlN+mPtRVefnf7PsiIG5vi61QawPtLNCEAv2TPD/M3SojcU0PFaef+BxiVGPOShFHtDog==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10", - "System.Diagnostics.EventLog": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11", + "System.Diagnostics.EventLog": "10.0.11" } }, "Microsoft.Extensions.Logging.EventSource": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "85SAPwXhJtdBInzN2k7SChiFiBGh3KOWay5AfoY+GREF6P7oZA98+ST2p7Z9384iLKYjkZSKIZ/FqIO5aojtNw==", + "resolved": "10.0.11", + "contentHash": "hs6QWECLLohi2VKqUvSGRUvrg7eXR1DqKL95Jrtz3cdD2g2nBA+yJPdRQLZ7SLmnTZWycxfMDK2s0ho+rfst5w==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Options": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "srnhnk7nE8krBiIXp71LvBmKBtraBONWSRzdjJgRv1Ko9Mp8IVNqv4vIS9hGeVteBig8aQkva9ZG+sC+o5sVcA==", + "resolved": "10.0.11", + "contentHash": "eY1GAKcTfD2maP27J84X9IovT3yjHJ2dVDzPmDg6/XqYvt3jMzJhtfQCLjG9pVsZGAd+8DQ2QrjaDcs2+VQLGw==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Options.ConfigurationExtensions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "tnBmu/LwF25ZQK+HBNCu2xrwnkKoB/XEbJyooGGoYxHrhvxbSKi7eOFiJ4AXBy/QU4vtCvCJfoi8k9Ej72qzOQ==", + "resolved": "10.0.11", + "contentHash": "syEhXQ/sEaSBFaqzlp9gDGHX/nk6gkQkh1sIUpBO1mlBj3Phu1rmb4ML1uCiyPW9N6Kxfxv3y5FGObC+bV01Qw==", "dependencies": { - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Configuration.Binder": "10.0.10", - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Configuration.Binder": "10.0.11", + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Primitives": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "5wu/GrYVd8mG2DVUw3vFJzF+O336TyTGg/Kmcgw9bfwYhCoFiV5lR5QeEmKecJyrW4W54nMfD3p3589E8a7czQ==" + "resolved": "10.0.11", + "contentHash": "SXcz+kF+4Oo9b1+55zntpJFYfwb1jw66ioxptyNOOTDc8g2FHnBFWjZpsWfCvZIhzr0x+4e2trVTs4OKwQfBtw==" }, "Microsoft.Identity.Client": { "type": "Transitive", @@ -460,19 +439,6 @@ "Microsoft.Testing.Platform": "1.9.1" } }, - "Microsoft.TestPlatform.ObjectModel": { - "type": "Transitive", - "resolved": "18.8.1", - "contentHash": "qLbktNB1+b1XZLNJBTzaWVVJAd6PEzD7cgD406geMb6PcFZhp3EDNa1tctWx1+mtMU6MP/6ozVvFPC9vs2a9rw==" - }, - "Microsoft.TestPlatform.TestHost": { - "type": "Transitive", - "resolved": "18.8.1", - "contentHash": "FaQHPDTUOcE+SFTjssNPfrub2lT9Zyon4J2W/KLHt/efLJACb1TCeWXyOgh0D/4Q1e4n+S3E6mOKud+9nLZlEA==", - "dependencies": { - "Microsoft.TestPlatform.ObjectModel": "18.8.1" - } - }, "Microsoft.Win32.Registry": { "type": "Transitive", "resolved": "5.0.0", @@ -480,8 +446,8 @@ }, "ModelContextProtocol.Core": { "type": "Transitive", - "resolved": "2.1.0", - "contentHash": "cU/urrhRxE4/iSyBIJI7QOaFqSP1FOEnwEHsct9n6t6/XluCAFD9iqnrPkBAsEYr+f/G4tVQ21U+6wN/6fQvOg==", + "resolved": "2.2.0", + "contentHash": "FeBfXU6T8k+jw4afg4sfxdEX2rL/e5oKOk9ROOGztu9k47+7Bz08sdaToYt2XvMY1opNbwxYQOFMj6wH9TInhA==", "dependencies": { "Microsoft.Extensions.AI.Abstractions": "10.8.3", "Microsoft.Extensions.Logging.Abstractions": "10.0.10" @@ -659,8 +625,8 @@ }, "System.Diagnostics.EventLog": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "OvGz3PrzuAI/Sj7LTcXcCe3FClRI1IyRMZjNONcZtFh+Ww7nAtSh4kh08r8KVe/xxkXJPjR0Y1jF7H+N42d4xQ==" + "resolved": "10.0.11", + "contentHash": "QTXEoQBzz00SFWbo7nAg1Ogd4f99lwqcO9uAJ7MYSLEUR28f6As32QktrqG2Fr9cfAfd1GjLyGYspE7Ipj7P6w==" }, "System.IdentityModel.Tokens.Jwt": { "type": "Transitive", @@ -763,14 +729,14 @@ "type": "Project", "dependencies": { "CredentialManagement": "[1.0.2, )", - "Microsoft.Extensions.Logging.Abstractions": "[10.0.10, )", - "ModelContextProtocol": "[2.1.0, )" + "Microsoft.Extensions.Logging.Abstractions": "[10.0.11, )", + "ModelContextProtocol": "[2.2.0, )" } }, "performancemonitor.notifications": { "type": "Project", "dependencies": { - "Microsoft.Extensions.Logging.Abstractions": "[10.0.10, )", + "Microsoft.Extensions.Logging.Abstractions": "[10.0.11, )", "PerformanceMonitor.Analysis": "[1.0.0, )" } }, @@ -778,7 +744,8 @@ "type": "Project", "dependencies": { "Microsoft.Data.SqlClient": "[7.0.2, )", - "PerformanceMonitor.Common": "[1.0.0, )" + "PerformanceMonitor.Common": "[1.0.0, )", + "System.Security.Cryptography.ProtectedData": "[10.0.11, )" } }, "performancemonitor.ui": { @@ -797,11 +764,11 @@ "Installer.Core": "[3.3.0, )", "Microsoft.Data.SqlClient": "[7.0.2, )", "Microsoft.Data.SqlClient.Extensions.Azure": "[7.0.2, )", - "Microsoft.Extensions.Configuration": "[10.0.10, )", - "Microsoft.Extensions.Configuration.Json": "[10.0.10, )", - "Microsoft.Extensions.Hosting": "[10.0.10, )", - "ModelContextProtocol": "[2.1.0, )", - "ModelContextProtocol.AspNetCore": "[2.1.0, )", + "Microsoft.Extensions.Configuration": "[10.0.11, )", + "Microsoft.Extensions.Configuration.Json": "[10.0.11, )", + "Microsoft.Extensions.Hosting": "[10.0.11, )", + "ModelContextProtocol": "[2.2.0, )", + "ModelContextProtocol.AspNetCore": "[2.2.0, )", "PerformanceMonitor.Alerting": "[1.0.0, )", "PerformanceMonitor.Analysis": "[1.0.0, )", "PerformanceMonitor.Common": "[1.0.0, )", @@ -857,94 +824,94 @@ }, "Microsoft.Extensions.Configuration": { "type": "CentralTransitive", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "plJWK2zpWuuyxI8F8s2scx6Je7N1Ajjs6HvYUGKwRnDMWIVIz9FHwAkiT7ASgrvAOd10T0FPVlh9BzAJJME+jg==", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "wlhRqZW8LcJPa+vk2oLAc/REXDItHtkFQdf/QcXYGZbZOO13izcsKY1pCvuFQYwUiZD+hwSZwsKASjqT+BNaVg==", "dependencies": { - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Configuration.Json": { "type": "CentralTransitive", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "uvJ6sHwjgrkMEJOgiC76G0mcZGXerwyyWkwX34EOjCbxKG6TCtfAoqDKAMsCvEBf9HxjlGQEgqsSMOGCmGBf+A==", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "nSPrT8U/cNoB4coqkmnanAMK9PsL7lsjG+LLUKEwHRFwS6E78b8S1wdv/y88EOxBhasWov1rLd7RTHmmsYPOLg==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Configuration.FileExtensions": "10.0.10", - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Configuration.FileExtensions": "10.0.11", + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Hosting": { "type": "CentralTransitive", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "tL9FkfV64GPUDSPvwrgyw42LVzsnVAnyrqJEuZVJbODgrQ3eL63zmzEcVWoCHzfgqUhWggzbgAyUCnz/zfI3Pg==", - "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Configuration.Binder": "10.0.10", - "Microsoft.Extensions.Configuration.CommandLine": "10.0.10", - "Microsoft.Extensions.Configuration.EnvironmentVariables": "10.0.10", - "Microsoft.Extensions.Configuration.FileExtensions": "10.0.10", - "Microsoft.Extensions.Configuration.Json": "10.0.10", - "Microsoft.Extensions.Configuration.UserSecrets": "10.0.10", - "Microsoft.Extensions.DependencyInjection": "10.0.10", - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Diagnostics": "10.0.10", - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10", - "Microsoft.Extensions.FileProviders.Physical": "10.0.10", - "Microsoft.Extensions.Hosting.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging.Configuration": "10.0.10", - "Microsoft.Extensions.Logging.Console": "10.0.10", - "Microsoft.Extensions.Logging.Debug": "10.0.10", - "Microsoft.Extensions.Logging.EventLog": "10.0.10", - "Microsoft.Extensions.Logging.EventSource": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10" + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "eIDa/Rl+93aj17gMlFsJJx+LhBvb3CP0Mu1PeVYkDp2Y3S4Jock8UynfGQEcx7lrlq+gKW+ECQJHbro/LTPDEQ==", + "dependencies": { + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Configuration.Binder": "10.0.11", + "Microsoft.Extensions.Configuration.CommandLine": "10.0.11", + "Microsoft.Extensions.Configuration.EnvironmentVariables": "10.0.11", + "Microsoft.Extensions.Configuration.FileExtensions": "10.0.11", + "Microsoft.Extensions.Configuration.Json": "10.0.11", + "Microsoft.Extensions.Configuration.UserSecrets": "10.0.11", + "Microsoft.Extensions.DependencyInjection": "10.0.11", + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Diagnostics": "10.0.11", + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11", + "Microsoft.Extensions.FileProviders.Physical": "10.0.11", + "Microsoft.Extensions.Hosting.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging.Configuration": "10.0.11", + "Microsoft.Extensions.Logging.Console": "10.0.11", + "Microsoft.Extensions.Logging.Debug": "10.0.11", + "Microsoft.Extensions.Logging.EventLog": "10.0.11", + "Microsoft.Extensions.Logging.EventSource": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11" } }, "Microsoft.Extensions.Logging": { "type": "CentralTransitive", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "Tf6z5HsL0VDYRTfvsoNrTGHGheCwkTsZBA2FFh5ATJUbkAwug+FFNISJK2gjpUNemlAOoWllAK52HOWCjto3EQ==", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "nUOJwgFkSiLHiVGFpU22pIJtuWYewuSYQ3JVuP/gdK8ASMT807Px+TYQiRWs6uSsOmoyFTaVCwKXTasczV6BpA==", "dependencies": { - "Microsoft.Extensions.DependencyInjection": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10" + "Microsoft.Extensions.DependencyInjection": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11" } }, "Microsoft.Extensions.Logging.Abstractions": { "type": "CentralTransitive", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "zkFxGYUvdxAvIKTyXHrmW+Sux53D4SezD9dMyZ6hrwwzPQJNuwCRy1f5W7AvYTqacEGhWF2XderRQG1OvbV8og==", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "Ljd0Uxoq5XpScD2Bg0nM/r3mwx7Ao5Uq24eo2ARxbGvqJ7Zht6rt2cJtwVRH4Cv+1ZVMdXz6TB43KbpmsxRrvQ==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11" } }, "ModelContextProtocol": { "type": "CentralTransitive", - "requested": "[2.1.0, )", - "resolved": "2.1.0", - "contentHash": "Oa4rU7EL9C2qyFjQj1dx+ysGMzfWDRpM8RRaUMmLGs5vPvfJ9xyz4ZtyF4ychY+Nx1b/auGCqIQLqSz/IpPkKA==", + "requested": "[2.2.0, )", + "resolved": "2.2.0", + "contentHash": "4Pb9u02Nwsp0poueDsqNdyGRojFxOYpljB7zDBsq+aHL+Afou3OgxlBc3GWFVnsRMRJrUtWqDh3s6k2JgPzmrQ==", "dependencies": { "Microsoft.Extensions.Caching.Abstractions": "10.0.10", "Microsoft.Extensions.Hosting.Abstractions": "10.0.10", - "ModelContextProtocol.Core": "[2.1.0]" + "ModelContextProtocol.Core": "[2.2.0]" } }, "ModelContextProtocol.AspNetCore": { "type": "CentralTransitive", - "requested": "[2.1.0, )", - "resolved": "2.1.0", - "contentHash": "yhJ8bBXIgrX0mAgRYRgzcbH3bLdv3MDSkG52utRW9EAAtQrPw/g7Q/T6EurxKV+L+Zefv8VVUYcNbXEOd9GgfA==", + "requested": "[2.2.0, )", + "resolved": "2.2.0", + "contentHash": "3JelDMuFIwFzXybsh6K30G6wXu5gmqKnuRnieqgBMuR0FkqqXRv4B+NYZuPgthSyNq5UwRyvv2U2VJBE3RD3PQ==", "dependencies": { - "ModelContextProtocol": "[2.1.0]" + "ModelContextProtocol": "[2.2.0]" } }, "ScottPlot.WPF": { @@ -959,6 +926,12 @@ "SkiaSharp.Views.WPF": "3.119.0" } }, + "System.Security.Cryptography.ProtectedData": { + "type": "CentralTransitive", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "PNoxCTPb+Tlux+GJyq4c89ddYdpioVSqfGx8pqOF6shCSKwUNNctXhQtRkCICjbJmGrJJsW8NY52kVYO/b8mlQ==" + }, "Velopack": { "type": "CentralTransitive", "requested": "[1.2.0, )", diff --git a/deprecated/Dashboard/packages.lock.json b/deprecated/Dashboard/packages.lock.json index 3c7d85bd7..f2754866a 100644 --- a/deprecated/Dashboard/packages.lock.json +++ b/deprecated/Dashboard/packages.lock.json @@ -47,74 +47,74 @@ }, "Microsoft.Extensions.Configuration": { "type": "Direct", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "plJWK2zpWuuyxI8F8s2scx6Je7N1Ajjs6HvYUGKwRnDMWIVIz9FHwAkiT7ASgrvAOd10T0FPVlh9BzAJJME+jg==", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "wlhRqZW8LcJPa+vk2oLAc/REXDItHtkFQdf/QcXYGZbZOO13izcsKY1pCvuFQYwUiZD+hwSZwsKASjqT+BNaVg==", "dependencies": { - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Configuration.Json": { "type": "Direct", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "uvJ6sHwjgrkMEJOgiC76G0mcZGXerwyyWkwX34EOjCbxKG6TCtfAoqDKAMsCvEBf9HxjlGQEgqsSMOGCmGBf+A==", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "nSPrT8U/cNoB4coqkmnanAMK9PsL7lsjG+LLUKEwHRFwS6E78b8S1wdv/y88EOxBhasWov1rLd7RTHmmsYPOLg==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Configuration.FileExtensions": "10.0.10", - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Configuration.FileExtensions": "10.0.11", + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Hosting": { "type": "Direct", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "tL9FkfV64GPUDSPvwrgyw42LVzsnVAnyrqJEuZVJbODgrQ3eL63zmzEcVWoCHzfgqUhWggzbgAyUCnz/zfI3Pg==", - "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Configuration.Binder": "10.0.10", - "Microsoft.Extensions.Configuration.CommandLine": "10.0.10", - "Microsoft.Extensions.Configuration.EnvironmentVariables": "10.0.10", - "Microsoft.Extensions.Configuration.FileExtensions": "10.0.10", - "Microsoft.Extensions.Configuration.Json": "10.0.10", - "Microsoft.Extensions.Configuration.UserSecrets": "10.0.10", - "Microsoft.Extensions.DependencyInjection": "10.0.10", - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Diagnostics": "10.0.10", - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10", - "Microsoft.Extensions.FileProviders.Physical": "10.0.10", - "Microsoft.Extensions.Hosting.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging.Configuration": "10.0.10", - "Microsoft.Extensions.Logging.Console": "10.0.10", - "Microsoft.Extensions.Logging.Debug": "10.0.10", - "Microsoft.Extensions.Logging.EventLog": "10.0.10", - "Microsoft.Extensions.Logging.EventSource": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10" + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "eIDa/Rl+93aj17gMlFsJJx+LhBvb3CP0Mu1PeVYkDp2Y3S4Jock8UynfGQEcx7lrlq+gKW+ECQJHbro/LTPDEQ==", + "dependencies": { + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Configuration.Binder": "10.0.11", + "Microsoft.Extensions.Configuration.CommandLine": "10.0.11", + "Microsoft.Extensions.Configuration.EnvironmentVariables": "10.0.11", + "Microsoft.Extensions.Configuration.FileExtensions": "10.0.11", + "Microsoft.Extensions.Configuration.Json": "10.0.11", + "Microsoft.Extensions.Configuration.UserSecrets": "10.0.11", + "Microsoft.Extensions.DependencyInjection": "10.0.11", + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Diagnostics": "10.0.11", + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11", + "Microsoft.Extensions.FileProviders.Physical": "10.0.11", + "Microsoft.Extensions.Hosting.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging.Configuration": "10.0.11", + "Microsoft.Extensions.Logging.Console": "10.0.11", + "Microsoft.Extensions.Logging.Debug": "10.0.11", + "Microsoft.Extensions.Logging.EventLog": "10.0.11", + "Microsoft.Extensions.Logging.EventSource": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11" } }, "ModelContextProtocol": { "type": "Direct", - "requested": "[2.1.0, )", - "resolved": "2.1.0", - "contentHash": "Oa4rU7EL9C2qyFjQj1dx+ysGMzfWDRpM8RRaUMmLGs5vPvfJ9xyz4ZtyF4ychY+Nx1b/auGCqIQLqSz/IpPkKA==", + "requested": "[2.2.0, )", + "resolved": "2.2.0", + "contentHash": "4Pb9u02Nwsp0poueDsqNdyGRojFxOYpljB7zDBsq+aHL+Afou3OgxlBc3GWFVnsRMRJrUtWqDh3s6k2JgPzmrQ==", "dependencies": { "Microsoft.Extensions.Caching.Abstractions": "10.0.10", "Microsoft.Extensions.Hosting.Abstractions": "10.0.10", - "ModelContextProtocol.Core": "[2.1.0]" + "ModelContextProtocol.Core": "[2.2.0]" } }, "ModelContextProtocol.AspNetCore": { "type": "Direct", - "requested": "[2.1.0, )", - "resolved": "2.1.0", - "contentHash": "yhJ8bBXIgrX0mAgRYRgzcbH3bLdv3MDSkG52utRW9EAAtQrPw/g7Q/T6EurxKV+L+Zefv8VVUYcNbXEOd9GgfA==", + "requested": "[2.2.0, )", + "resolved": "2.2.0", + "contentHash": "3JelDMuFIwFzXybsh6K30G6wXu5gmqKnuRnieqgBMuR0FkqqXRv4B+NYZuPgthSyNq5UwRyvv2U2VJBE3RD3PQ==", "dependencies": { - "ModelContextProtocol": "[2.1.0]" + "ModelContextProtocol": "[2.2.0]" } }, "ScottPlot.WPF": { @@ -236,215 +236,215 @@ }, "Microsoft.Extensions.Configuration.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "5Vnd2I75DmZCVEjSynIdJ/0EGafgnLQwgR3t2C2/fkjx/nRG+cLwxLLdInoHeCEpkD5K4Ov/g9ZCRYrl4TRsaA==", + "resolved": "10.0.11", + "contentHash": "fVi053xdpda9Em7vSkmgVxO/PtgC2m78ekReKWsgcyskqY0U82Bz/MONwxpGzI0hElYKJfw+fupqMVeKW3fSaA==", "dependencies": { - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Configuration.Binder": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "GqmN2o1CkJvk7uWp+p4CwBYW0w/zfoEbvsiFDbO2G8l1Uz+mrDAbAcZiXhU2lufKPby1cjAUdd5GTWpebYOkOA==", + "resolved": "10.0.11", + "contentHash": "rFn8RuszZn3qquPVkDytMUlPc2+rXl9MCoygwc1XmAgC5vg5/oXJ8hkOosOrLoBLsqdTy4lFwP6iQdPS9uSYOA==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Configuration.CommandLine": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "33cBeR2HRbzHUTtmcmLdNOApneNGcymwwL4arHuotgVK9Frba8kcDTrvVTj7cSCmF1R9OiSbZH0KxNOwab3HUg==", + "resolved": "10.0.11", + "contentHash": "1KHr/1L56llwQ/yI0tAisEA31UpPsn8aasjASIwELOaN4JIUcbjuQBMdFOIzfNBBeULoUa0XfBe5QDtRRUY+fg==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Configuration.EnvironmentVariables": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "KRfFSSCV58vEdU7mPED/YMzeovIWF5P0g8s9K8n9HEfy0/WzMq37SrPdXdFN5/dFT/rPMHpF7AvpoXHckbcBFg==", + "resolved": "10.0.11", + "contentHash": "KICyU3eVi5jvloKm01EXV69L97H/zkhISVtV98cIuzuFOxNx3xTUVcXqvWTz3aq7OvUuDB/MFlPFjmxRaKF7/A==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Configuration.FileExtensions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "ZOhZYwvbXGTgGVRwswIirofEMVHuWdxjdh0JeUZXwaF9cgcjXdz/t0ELtgaevw7ezTyv47yPNCgGreWtLkn3IQ==", + "resolved": "10.0.11", + "contentHash": "mDW7KVFB05M6jiRUyaZiOMWhS31n5HlSZwoYctHAZAucD4sMDJ70IxOmkGDt6RpstchD+keWBjhdzcMpSkWvWQ==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10", - "Microsoft.Extensions.FileProviders.Physical": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11", + "Microsoft.Extensions.FileProviders.Physical": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Configuration.UserSecrets": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "1s1sKFTk/Foam64JY6+m/diH8drL3Wx6V3gtSd5v1IEZtszZYyc1pW8uRnMblzpNiR0l0t8gGk7tXj3xHzFgdg==", + "resolved": "10.0.11", + "contentHash": "BRliLdUowglV8GS+J1G/QsSofCJYYFg3U8QZx0ACRn+a91az/Qnpy+h6PyHS94WgV2TSazX5D/cuuk6wnCJatw==", "dependencies": { - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Configuration.Json": "10.0.10", - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10", - "Microsoft.Extensions.FileProviders.Physical": "10.0.10" + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Configuration.Json": "10.0.11", + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11", + "Microsoft.Extensions.FileProviders.Physical": "10.0.11" } }, "Microsoft.Extensions.DependencyInjection": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "ANyvsgkNBRvcJh2XLgn8veGmajf+8m0AbKK+HPWdRL1yraSNVVSmQhFntLtdz/C795jxqqup+k05cs/3jZQPOA==", + "resolved": "10.0.11", + "contentHash": "PSmotV19c7E3lKed++uYo1kSiXFI+uTl37CBSrhq+CfLC3FCHjG7R91+xPnNehQfHS1b0Tzo/CCLPWH3qaEheg==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11" } }, "Microsoft.Extensions.DependencyInjection.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "z/2xXlFw2aLGjHyEm6E0tQ+In6VfzQzTrtArbQ2c0TQE16ZbyDCMGPvaUT9I0s8rgy9sRWlU2P9waW37qV04qA==" + "resolved": "10.0.11", + "contentHash": "/a1aJz4m7ylhEDf25ugQChLQoN5XwoGjWw/BoR/ZWWKsO1v4DdJElS1uyngahz4B/eOzjFk1KNTkarRLE5wsIg==" }, "Microsoft.Extensions.Diagnostics": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "Kr/e7lUf4+N8tacbqJ2Ctwe/HarKdAc9ZkgKVVqvtJDBKbez+T/KnUwu82KSlnBp/SrpBcxc7u7xkE2oUZT/5Q==", + "resolved": "10.0.11", + "contentHash": "HT70uGPxMLqqnOzKMcnQtDmeV4r0KHr4qVCLhP7SXil9jMEm8sQXwcybxVVFGXZJ1V44xV0mLqQ54aZbcR2OiQ==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Diagnostics.Abstractions": "10.0.10", - "Microsoft.Extensions.Options.ConfigurationExtensions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Diagnostics.Abstractions": "10.0.11", + "Microsoft.Extensions.Options.ConfigurationExtensions": "10.0.11" } }, "Microsoft.Extensions.Diagnostics.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "9uWiKpeOVac355STyChWR/pliFX/5CeLqChW9kKsaxyDH4EUTZxMkT4Jwp/J/peLm0GBFmSX5c0WCse3yCnq1Q==", + "resolved": "10.0.11", + "contentHash": "se7Kx8QpJEt+nf26L4qIVAofGTDr1wbexxsh/Fm3Xc04xUkqUXK06KUS7FLwSQYSjqb7q9n+T7MEcXYBhI1Y5g==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11" } }, "Microsoft.Extensions.FileProviders.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "c5zqFCY9DiIpMovLd7/d/CTiEtrMOuQ639dhv3PABtKQIKNQikSHwQt8+N679uii9q+B55lgK28Uv64FOwEu8w==", + "resolved": "10.0.11", + "contentHash": "JOjac6SQQgZmdmB8WGEw61/7siqMZoWJMkmq2p1goJGxqI59lO6oB4bOl0jNsbaPBdYy5Mlkb+6U7T4+CjnD8Q==", "dependencies": { - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.FileProviders.Physical": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "jhJAyo38kSrH3ARvWUk0h8itogVnQu2DCZuPo+s0Z+tXes0ugTxMPaHYzap85785eHQmPFqD9TYERqBbtGxn/w==", + "resolved": "10.0.11", + "contentHash": "Tq/UqMaczePv9yWwSsJZRgKtgA46djVR5xHj/lZBCueQ3ag8f9v5mu0EdhrNx7tXxNk+Y9OurG2oKuSKINjr0A==", "dependencies": { - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10", - "Microsoft.Extensions.FileSystemGlobbing": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11", + "Microsoft.Extensions.FileSystemGlobbing": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.FileSystemGlobbing": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "jSOCVxEwCd4Aq925kJVz1kSO1EpX2OHYKL04qVREXkDU7Ce3pVDdHPYm+fEy8y/th2kJf/DAstRHpJAqoNWP8w==" + "resolved": "10.0.11", + "contentHash": "2i6rtW/B5rCnWCnhdmWWEmaM9O0HD0zsPY9eRqa++y4tclI3Uw8zvGbBvhY/LjAdtf8gUHhUPcAWj3DRlWMXmQ==" }, "Microsoft.Extensions.Hosting.Abstractions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "5LugpYGHk+mkn0a8IZgcyfBca8PCTAU9RQFoMrTdtOOidq88M2SI5f3px6ugnzgxC+eTkvYYJi8pzlUnG5xdAQ==", + "resolved": "10.0.11", + "contentHash": "pwtpF7iF/NNaOBcX+pvMZ7y2+JAVbH5KkNrH9uMZtuVxVJsFTDiWiCR7Tk3HVptsAijaetipUZVRVK2LLq+nvA==", "dependencies": { - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Diagnostics.Abstractions": "10.0.10", - "Microsoft.Extensions.FileProviders.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10" + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Diagnostics.Abstractions": "10.0.11", + "Microsoft.Extensions.FileProviders.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Logging.Configuration": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "cLrqxkuEfcilZ8SjK+9KAnpLk9lOoMPaOokF+wRUYie+iUEcdX4/p/+gJkt0BYgWLthjpBUCkVTBI6Kxg0nsOw==", + "resolved": "10.0.11", + "contentHash": "S7LvLeVHKNPaY2NMyxW7c2TBGsLgxoSUBCV5Ev5iN8kgC7EPR2UB7eW7vHsElGMcIUDwRmoxLfvGDynCn3q6EA==", "dependencies": { - "Microsoft.Extensions.Configuration": "10.0.10", - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Configuration.Binder": "10.0.10", - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10", - "Microsoft.Extensions.Options.ConfigurationExtensions": "10.0.10" + "Microsoft.Extensions.Configuration": "10.0.11", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Configuration.Binder": "10.0.11", + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11", + "Microsoft.Extensions.Options.ConfigurationExtensions": "10.0.11" } }, "Microsoft.Extensions.Logging.Console": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "VIlNzPwPS0GeQVSmCqqo36ugryX3LpE9ul6gEkks5VLET3weH/XMLeWmclwfoGn4Nxi2mwVibB+OZBVJ9tDqvg==", + "resolved": "10.0.11", + "contentHash": "dFc0yDudyD1iIg6z9XT7ofsT3hVO7Y4ylrxGHIVRR0GaZ4CUk4ujOrMoy7wWEdNHZhvJooySg6hZpOxyS8zEVA==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging.Configuration": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging.Configuration": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11" } }, "Microsoft.Extensions.Logging.Debug": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "8+TZBnV5fgBXoVNJ5ROSErUwYogk4hOgV7c2HWK1u5cqKGmiUTUn7+KqZ35iQu8e/B7Ykccyz5OTjdXcidNZ9g==", + "resolved": "10.0.11", + "contentHash": "wr+j1bjdFXhc8lKTLoq+RbwFM8M+orcMS9xrcqLmDGOxJcXpKizEeE5h6v/GKwCZV02FmhaA7OlNjoq072jZpQ==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11" } }, "Microsoft.Extensions.Logging.EventLog": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "0RE4951AzQ+YD4gVrvbq0BhdsiBgSDo44yM7+QBZ2mrmMJeNjY+teCIYfUjqDPVYnKs0HR6SkkhgrX1YgXZq3Q==", + "resolved": "10.0.11", + "contentHash": "Eck9GpCCpvZ3f6L7IUlN+mPtRVefnf7PsiIG5vi61QawPtLNCEAv2TPD/M3SojcU0PFaef+BxiVGPOShFHtDog==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10", - "System.Diagnostics.EventLog": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11", + "System.Diagnostics.EventLog": "10.0.11" } }, "Microsoft.Extensions.Logging.EventSource": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "85SAPwXhJtdBInzN2k7SChiFiBGh3KOWay5AfoY+GREF6P7oZA98+ST2p7Z9384iLKYjkZSKIZ/FqIO5aojtNw==", + "resolved": "10.0.11", + "contentHash": "hs6QWECLLohi2VKqUvSGRUvrg7eXR1DqKL95Jrtz3cdD2g2nBA+yJPdRQLZ7SLmnTZWycxfMDK2s0ho+rfst5w==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Logging": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Logging": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Options": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "srnhnk7nE8krBiIXp71LvBmKBtraBONWSRzdjJgRv1Ko9Mp8IVNqv4vIS9hGeVteBig8aQkva9ZG+sC+o5sVcA==", + "resolved": "10.0.11", + "contentHash": "eY1GAKcTfD2maP27J84X9IovT3yjHJ2dVDzPmDg6/XqYvt3jMzJhtfQCLjG9pVsZGAd+8DQ2QrjaDcs2+VQLGw==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Options.ConfigurationExtensions": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "tnBmu/LwF25ZQK+HBNCu2xrwnkKoB/XEbJyooGGoYxHrhvxbSKi7eOFiJ4AXBy/QU4vtCvCJfoi8k9Ej72qzOQ==", + "resolved": "10.0.11", + "contentHash": "syEhXQ/sEaSBFaqzlp9gDGHX/nk6gkQkh1sIUpBO1mlBj3Phu1rmb4ML1uCiyPW9N6Kxfxv3y5FGObC+bV01Qw==", "dependencies": { - "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", - "Microsoft.Extensions.Configuration.Binder": "10.0.10", - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10", - "Microsoft.Extensions.Primitives": "10.0.10" + "Microsoft.Extensions.Configuration.Abstractions": "10.0.11", + "Microsoft.Extensions.Configuration.Binder": "10.0.11", + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11", + "Microsoft.Extensions.Primitives": "10.0.11" } }, "Microsoft.Extensions.Primitives": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "5wu/GrYVd8mG2DVUw3vFJzF+O336TyTGg/Kmcgw9bfwYhCoFiV5lR5QeEmKecJyrW4W54nMfD3p3589E8a7czQ==" + "resolved": "10.0.11", + "contentHash": "SXcz+kF+4Oo9b1+55zntpJFYfwb1jw66ioxptyNOOTDc8g2FHnBFWjZpsWfCvZIhzr0x+4e2trVTs4OKwQfBtw==" }, "Microsoft.Identity.Client": { "type": "Transitive", @@ -530,8 +530,8 @@ }, "ModelContextProtocol.Core": { "type": "Transitive", - "resolved": "2.1.0", - "contentHash": "cU/urrhRxE4/iSyBIJI7QOaFqSP1FOEnwEHsct9n6t6/XluCAFD9iqnrPkBAsEYr+f/G4tVQ21U+6wN/6fQvOg==", + "resolved": "2.2.0", + "contentHash": "FeBfXU6T8k+jw4afg4sfxdEX2rL/e5oKOk9ROOGztu9k47+7Bz08sdaToYt2XvMY1opNbwxYQOFMj6wH9TInhA==", "dependencies": { "Microsoft.Extensions.AI.Abstractions": "10.8.3", "Microsoft.Extensions.Logging.Abstractions": "10.0.10" @@ -709,8 +709,8 @@ }, "System.Diagnostics.EventLog": { "type": "Transitive", - "resolved": "10.0.10", - "contentHash": "OvGz3PrzuAI/Sj7LTcXcCe3FClRI1IyRMZjNONcZtFh+Ww7nAtSh4kh08r8KVe/xxkXJPjR0Y1jF7H+N42d4xQ==" + "resolved": "10.0.11", + "contentHash": "QTXEoQBzz00SFWbo7nAg1Ogd4f99lwqcO9uAJ7MYSLEUR28f6As32QktrqG2Fr9cfAfd1GjLyGYspE7Ipj7P6w==" }, "System.IdentityModel.Tokens.Jwt": { "type": "Transitive", @@ -746,14 +746,14 @@ "type": "Project", "dependencies": { "CredentialManagement": "[1.0.2, )", - "Microsoft.Extensions.Logging.Abstractions": "[10.0.10, )", - "ModelContextProtocol": "[2.1.0, )" + "Microsoft.Extensions.Logging.Abstractions": "[10.0.11, )", + "ModelContextProtocol": "[2.2.0, )" } }, "performancemonitor.notifications": { "type": "Project", "dependencies": { - "Microsoft.Extensions.Logging.Abstractions": "[10.0.10, )", + "Microsoft.Extensions.Logging.Abstractions": "[10.0.11, )", "PerformanceMonitor.Analysis": "[1.0.0, )" } }, @@ -761,7 +761,8 @@ "type": "Project", "dependencies": { "Microsoft.Data.SqlClient": "[7.0.2, )", - "PerformanceMonitor.Common": "[1.0.0, )" + "PerformanceMonitor.Common": "[1.0.0, )", + "System.Security.Cryptography.ProtectedData": "[10.0.11, )" } }, "performancemonitor.ui": { @@ -774,23 +775,29 @@ }, "Microsoft.Extensions.Logging": { "type": "CentralTransitive", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "Tf6z5HsL0VDYRTfvsoNrTGHGheCwkTsZBA2FFh5ATJUbkAwug+FFNISJK2gjpUNemlAOoWllAK52HOWCjto3EQ==", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "nUOJwgFkSiLHiVGFpU22pIJtuWYewuSYQ3JVuP/gdK8ASMT807Px+TYQiRWs6uSsOmoyFTaVCwKXTasczV6BpA==", "dependencies": { - "Microsoft.Extensions.DependencyInjection": "10.0.10", - "Microsoft.Extensions.Logging.Abstractions": "10.0.10", - "Microsoft.Extensions.Options": "10.0.10" + "Microsoft.Extensions.DependencyInjection": "10.0.11", + "Microsoft.Extensions.Logging.Abstractions": "10.0.11", + "Microsoft.Extensions.Options": "10.0.11" } }, "Microsoft.Extensions.Logging.Abstractions": { "type": "CentralTransitive", - "requested": "[10.0.10, )", - "resolved": "10.0.10", - "contentHash": "zkFxGYUvdxAvIKTyXHrmW+Sux53D4SezD9dMyZ6hrwwzPQJNuwCRy1f5W7AvYTqacEGhWF2XderRQG1OvbV8og==", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "Ljd0Uxoq5XpScD2Bg0nM/r3mwx7Ao5Uq24eo2ARxbGvqJ7Zht6rt2cJtwVRH4Cv+1ZVMdXz6TB43KbpmsxRrvQ==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.11" } + }, + "System.Security.Cryptography.ProtectedData": { + "type": "CentralTransitive", + "requested": "[10.0.11, )", + "resolved": "10.0.11", + "contentHash": "PNoxCTPb+Tlux+GJyq4c89ddYdpioVSqfGx8pqOF6shCSKwUNNctXhQtRkCICjbJmGrJJsW8NY52kVYO/b8mlQ==" } } } diff --git a/deprecated/Installer.Core/packages.lock.json b/deprecated/Installer.Core/packages.lock.json index f0dfeab10..feb2a1f3a 100644 --- a/deprecated/Installer.Core/packages.lock.json +++ b/deprecated/Installer.Core/packages.lock.json @@ -290,7 +290,7 @@ }, "Microsoft.Extensions.Logging.Abstractions": { "type": "CentralTransitive", - "requested": "[10.0.10, )", + "requested": "[10.0.11, )", "resolved": "10.0.3", "contentHash": "lxl0WLk7ROgBFAsjcOYjQ8/DVK+VMszxGBzUhgtQmAsTNldLL5pk9NG/cWTsXHq0lUhUEAtZkEE7jOGOA8bGKQ==", "dependencies": { @@ -299,7 +299,7 @@ }, "System.Security.Cryptography.ProtectedData": { "type": "CentralTransitive", - "requested": "[10.0.10, )", + "requested": "[10.0.11, )", "resolved": "9.0.13", "contentHash": "t8S9IDpjJKsLpLkeBdW8cWtcPyYqrGu93Dej1RO6WwuL/lkFSqWlan3rMJfortqz1mRIh+sys2AFsSA6jWJ3Jg==" } diff --git a/deprecated/Installer.Tests/Installer.Tests.csproj b/deprecated/Installer.Tests/Installer.Tests.csproj index a8b5c9747..5887d6723 100644 --- a/deprecated/Installer.Tests/Installer.Tests.csproj +++ b/deprecated/Installer.Tests/Installer.Tests.csproj @@ -5,16 +5,12 @@ false enable true + Exe CA1849;CA2007;CA1508;CA1822;CA1805;CA1510;CA1816;CA1861;CA1845;CA2201;CS4014;NU1701;CA1001;CA1848;CA1852;CA1305;CA1860;CA1707;CA1507;CA1806 false - - - all - runtime; build; native; contentfiles; analyzers; buildtransitive - diff --git a/deprecated/Installer.Tests/packages.lock.json b/deprecated/Installer.Tests/packages.lock.json index 5f09b29ec..ee02640ad 100644 --- a/deprecated/Installer.Tests/packages.lock.json +++ b/deprecated/Installer.Tests/packages.lock.json @@ -20,22 +20,6 @@ "System.Security.Cryptography.Pkcs": "9.0.13" } }, - "Microsoft.NET.Test.Sdk": { - "type": "Direct", - "requested": "[18.8.1, )", - "resolved": "18.8.1", - "contentHash": "dknJL3/9Y3t4XuCBqnc0PevPxgLsUMmVhjwup/b1HNovA8zWcj3XsfIf7c6p05363DWcqL7X/YhDL9B+Zymv1w==", - "dependencies": { - "Microsoft.CodeCoverage": "18.8.1", - "Microsoft.TestPlatform.TestHost": "18.8.1" - } - }, - "xunit.runner.visualstudio": { - "type": "Direct", - "requested": "[3.1.5, )", - "resolved": "3.1.5", - "contentHash": "tKi7dSTwP4m5m9eXPM2Ime4Kn7xNf4x4zT9sdLO/G4hZVnQCRiMTWoSZqI/pYTVeI27oPPqHBKYI/DjJ9GsYgA==" - }, "xunit.v3": { "type": "Direct", "requested": "[3.2.2, )", @@ -82,11 +66,6 @@ "resolved": "9.0.13", "contentHash": "5T+bH3Lb1nEe8Hf/ixMxLmhlrx5wRi53wv7OhVwG2F1ZviW1ejFRS1NHur3uqPpJRGtkQwUchtY6zhVK2R+v+w==" }, - "Microsoft.CodeCoverage": { - "type": "Transitive", - "resolved": "18.8.1", - "contentHash": "Eclse/ZZjr4lmWzZFNN9h/OluhKL+SK/QbUyKUewgX139aGeyMEO/DkMPwuFs2MixvanTnz6891rF8UHDg+W4Q==" - }, "Microsoft.Data.SqlClient.Extensions.Abstractions": { "type": "Transitive", "resolved": "7.0.2", @@ -294,19 +273,6 @@ "Microsoft.Testing.Platform": "1.9.1" } }, - "Microsoft.TestPlatform.ObjectModel": { - "type": "Transitive", - "resolved": "18.8.1", - "contentHash": "qLbktNB1+b1XZLNJBTzaWVVJAd6PEzD7cgD406geMb6PcFZhp3EDNa1tctWx1+mtMU6MP/6ozVvFPC9vs2a9rw==" - }, - "Microsoft.TestPlatform.TestHost": { - "type": "Transitive", - "resolved": "18.8.1", - "contentHash": "FaQHPDTUOcE+SFTjssNPfrub2lT9Zyon4J2W/KLHt/efLJACb1TCeWXyOgh0D/4Q1e4n+S3E6mOKud+9nLZlEA==", - "dependencies": { - "Microsoft.TestPlatform.ObjectModel": "18.8.1" - } - }, "Microsoft.Win32.Registry": { "type": "Transitive", "resolved": "5.0.0", @@ -447,7 +413,7 @@ }, "Microsoft.Extensions.Logging.Abstractions": { "type": "CentralTransitive", - "requested": "[10.0.10, )", + "requested": "[10.0.11, )", "resolved": "10.0.3", "contentHash": "lxl0WLk7ROgBFAsjcOYjQ8/DVK+VMszxGBzUhgtQmAsTNldLL5pk9NG/cWTsXHq0lUhUEAtZkEE7jOGOA8bGKQ==", "dependencies": { @@ -456,7 +422,7 @@ }, "System.Security.Cryptography.ProtectedData": { "type": "CentralTransitive", - "requested": "[10.0.10, )", + "requested": "[10.0.11, )", "resolved": "9.0.13", "contentHash": "t8S9IDpjJKsLpLkeBdW8cWtcPyYqrGu93Dej1RO6WwuL/lkFSqWlan3rMJfortqz1mRIh+sys2AFsSA6jWJ3Jg==" } diff --git a/deprecated/Installer/packages.lock.json b/deprecated/Installer/packages.lock.json index 6fcafc034..85f751308 100644 --- a/deprecated/Installer/packages.lock.json +++ b/deprecated/Installer/packages.lock.json @@ -303,7 +303,7 @@ }, "Microsoft.Extensions.Logging.Abstractions": { "type": "CentralTransitive", - "requested": "[10.0.10, )", + "requested": "[10.0.11, )", "resolved": "10.0.3", "contentHash": "lxl0WLk7ROgBFAsjcOYjQ8/DVK+VMszxGBzUhgtQmAsTNldLL5pk9NG/cWTsXHq0lUhUEAtZkEE7jOGOA8bGKQ==", "dependencies": { @@ -312,7 +312,7 @@ }, "System.Security.Cryptography.ProtectedData": { "type": "CentralTransitive", - "requested": "[10.0.10, )", + "requested": "[10.0.11, )", "resolved": "9.0.13", "contentHash": "t8S9IDpjJKsLpLkeBdW8cWtcPyYqrGu93Dej1RO6WwuL/lkFSqWlan3rMJfortqz1mRIh+sys2AFsSA6jWJ3Jg==" }