diff --git a/.gitattributes b/.gitattributes index e71a7d8dd..aa168a0ae 100644 --- a/.gitattributes +++ b/.gitattributes @@ -1,5 +1,8 @@ * text=auto eol=crlf +# Shell scripts must stay LF (this repo uses Git Bash) — CRLF breaks them. +*.sh text eol=lf + # Explicit binary markers (prevent text mis-detection / corruption): *.png binary *.jpg binary diff --git a/.github/PULL_REQUEST_TEMPLATE.md b/.github/PULL_REQUEST_TEMPLATE.md index 6b13b4d16..311320e9d 100644 --- a/.github/PULL_REQUEST_TEMPLATE.md +++ b/.github/PULL_REQUEST_TEMPLATE.md @@ -9,7 +9,6 @@ A clear description of the change and why it's being made. - [ ] Lite Tests - [ ] SQL collection scripts - [ ] CLI Installer -- [ ] GUI Installer - [ ] Documentation ## How was this tested? diff --git a/.github/workflows/build.yml b/.github/workflows/build.yml index 8a23a2e39..82ab912f8 100644 --- a/.github/workflows/build.yml +++ b/.github/workflows/build.yml @@ -230,7 +230,9 @@ jobs: GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} VERSION: ${{ steps.version.outputs.VERSION }} run: | - dotnet tool install -g vpk + # Pin vpk to the Velopack library version (keep in sync with the Velopack + # PackageReference in Dashboard.csproj / PerformanceMonitorLite.csproj). + dotnet tool install -g vpk --version 1.2.0 New-Item -ItemType Directory -Force -Path releases/velopack-dashboard New-Item -ItemType Directory -Force -Path releases/velopack-lite diff --git a/CHANGELOG.md b/CHANGELOG.md index b28b938b9..4663fd62b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,6 +5,60 @@ All notable changes to this project will be documented in this file. The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). +## [Unreleased] + +## [3.1.0] - 2026-06-28 + +### Added + +- **Lite and Dashboard: a shared block-chain viewer reconstructs the full apex → victim blocking tree** ([#1207]) — right-click *View Block Chain* (or double-click a row) on the Dashboard **Locking** / Lite **Blocking** grids opens a focused graph of the clicked session's blocking chain, rooted at its apex (lead blocker) with the full hierarchy of waiters beneath it. Node cards lead with session identity (login/host/app), the contended object, and the SQL text rather than a bare SPID; the clicked node auto-selects and scrolls into view; and the canvas has zoom/pan/fit plus a properties panel. Reconstruction, tree-building (DAG → tree, cycle-safe), and layout are pure and unit-tested in shared `PerformanceMonitor.Common` (`BlockingChainModel` / `BlockingChainTreeBuilder` / `BlockingChainLayout`) with the WPF shell in `PerformanceMonitor.Ui/BlockingChainControl`, so the two apps can't drift, and the fetch/reconstruct/build runs off the UI thread. The chain is picked edge-precisely around the clicked row's own `event_time` (±5 min) so double-clicking any visible row reliably opens that event's chain, and sessions are keyed by `(monitor_loop, spid, ecid)` — mirroring `sp_HumanEventsBlockViewer` — with `(spid, ecid)` and SPID-only fallbacks so historical rows still resolve. Covered by `BlockingChainTreeBuilderTests` and `BlockingChainLayoutTests`, with `BlockingChainReconstructorTests` mirrored in both suites +- **Lite and Dashboard: a shared deadlock graph viewer shows the waits-for cycles** ([#1216]) — right-click *View Deadlock Graph* (or double-click a row) on the deadlock grid renders the deadlock as a **cycle view** of the waits-for graph — not an SSMS process|resource bipartite graph — because a real deadlock is typically a bundle of independent small cycles rather than one large N-cycle. Each process card leads with SPID + statement + lock + wait; the rolled-back victim gets a red border and a VICTIM badge; directional arrows are labeled by the contended resource (resolved to an object name from the deadlock XML, e.g. `PAGE hammerdb_tpcc.dbo.new_order`); and each independent cycle is framed in its own dashed *Cycle N* box with the count noted in the summary. The parser/model/layout live in shared `PerformanceMonitor.Common` (`DeadlockGraphParser` / `DeadlockGraphModel` / `DeadlockGraphLayout` — marks all victims, handles the Azure event wrapper, parallelism self-edges, and a nameless-resource fallback, and never throws) with the control in `PerformanceMonitor.Ui/DeadlockGraphControl`, and parsing runs off the UI thread. Covered by `DeadlockGraphParserTests` and `DeadlockGraphLayoutTests` against real 2/5/8-process golden samples +- **Lite and Dashboard: an always-on DMV blocking-snapshot fallback keeps blocking visible without the blocked-process report** ([#1227]) — a new `dmv_blocking_snapshots` collector captures point-in-time blocking from `sys.dm_os_waiting_tasks` + `sys.dm_exec_*`, independent of the `blocked_process_report` Extended Event (which only fires when `blocked process threshold` is set — off by default, and unsettable via `sp_configure` on AWS RDS). It feeds the same source-agnostic chain reconstructor, so the blocking grid, the chain viewer, the flat top-blocking drill-down, the slicer/trend, and the count/badge surfaces all stay populated even when no blocked-process reports were captured. DMV rows merge **BPR-preferred** through the shared `PerformanceMonitor.Analysis/BlockingPairRowMerge` (synthetic negative `monitor_loop` so DMV episodes never collide with real ones); counts use `COALESCE(NULLIF(bpr,0), dmv)` and bucketed surfaces add DMV rows only where no BPR exists in the window, so a server with both sources never double-counts, and a *Source* badge marks each row's origin. The collector applies layered minimum-wait floors by contention class (LCK 2s, PAGELATCH 0.5s, PAGEIOLATCH 1s, RESOURCE_SEMAPHORE 5s) to cut noise, and the slicer/drill-down probe for the table so a not-yet-upgraded server degrades to BPR-only instead of erroring. Covered by `BlockingPairRowMergeTests` in both suites +- **Lite and Dashboard: analysis findings are grouped into trackable incidents on the Recommendations surface** ([#1214]) — instead of a flat, severity-sorted sea of cards, the recommendation surfaces group findings by a stable `incident_id` into one collapsible report per incident (header = the primary/highest-severity finding + finding count + severity), with each card keeping its own advice / Copy / Apply. The id is a fingerprint of the incident's primary finding's root key + database, scoped to the server, so the **same recurring incident keeps the same id across runs** (trackable, not a per-run GUID); findings are clustered into incidents by connected components over the analysis relationship graph's active edges, so a run's genuinely-distinct problems get distinct incidents while related facets (a plan regression that's really part of a CPU incident, or a THREADPOOL→LCK blocking-driven thread exhaustion) merge into one. The `incident_id` is persisted in both stores (Lite analysis-schema, Dashboard `config.analysis_findings`) and emitted per-finding by the MCP `analyze_server` / `get_analysis_findings`, alongside a "what else fired in this window" cross-reference. Covered by `IncidentIdTests`, `InferenceEngineTests`, and the recommendations view-model tests in both suites +- **Lite and Dashboard: FinOps storage analysis becomes an object-growth → index drill with in-grid heatmaps** ([#1138]) — the standalone *Object Sizes & Growth* and *Index Usage* tabs are replaced by a **Storage Growth → object → index** drill: double-click a database to its object-growth heatmap (rows = top-N objects by reserved-MB growth over the window, X = daily buckets, color = reserved MB on a per-column log scale) with a companion grid, then double-click an object for its per-index size + usage. The *Locking & Contention* tab becomes a heatmap-in-grid — the four wait-time columns are shaded per-column by their wait category's hue (Lock / Buffer Latch / Buffer IO) with the numbers still visible — plus a database selector and a per-index lock/latch drill. The top-N ranking, long → matrix pivot, and color-scale math are pure and shared (`PerformanceMonitor.Common/FinOpsHeatmap`, `PerformanceMonitor.Ui/FinOpsHeatmapRenderer`, `HeatIntensityToBrushConverter`) so both apps render identically, and the read layer ranks the top-N cheaply before series-scanning only those rows (DB-scoped on the covering index's lead column to avoid the [#1135] uncovered scan). The existing `GetObjectSizeGrowthAsync` / `GetIndexUsageAsync` stay for the MCP. Covered by `FinOpsHeatmapBuilderTests` (plus Lite integration tests) in both suites +- **Lite and Dashboard: per-server override for the alert delivery mode** ([#1236]) — the per-event vs. summary delivery mode ([#1141]) can now be overridden per server (e.g. Per-event for one noisy prod box while the global default stays Summary). `ServerConnection` gains a nullable `AlertDeliveryModeOverride` persisted in the existing `servers.json` (no new store; null inherits the global), the Add/Edit Server dialog in both apps gets an *Alert delivery* combo (Use global setting / Summary / Per-event), and a shared `AlertDeliveryModeResolver` centralizes the precedence so the two apps can't drift. Covered by `AlertDeliveryModeOverrideTests` in both suites +- **Lite and Dashboard: MCP tools return a structured status envelope for non-data outcomes** ([#1224]) — `McpHelpers.Status(status, message, hints?)` lets an LLM client distinguish a true negative (`empty`) from a bad input (`not_collected`) from a transient substrate gap (`unavailable`) instead of parsing prose, routed through every data-miss return across both apps' MCP tools (102 sites / 38 files). Lite's `get_perfmon_trend` / `get_wait_trend` return the actually-collected counters / wait types as hints, and `analyze_server`'s no-findings result becomes `empty`. Setup, validation, and resolution messages and the data-returning success paths are unchanged. Covered by `McpStatusEnvelopeTests` +- **Lite and Dashboard: in-app plan navigation on every query-identifying surface** ([#1184]) — *View Plan* (open the collected/cached plan) and *Get Actual Plan* (re-execute for a fresh actual plan) are now available from every view that identifies a single query in both apps — the history/drill-down windows (Query Stats, Query Store, Procedure, Wait drill-down), the comparison grids (whose menu items previously no-opped because the switch lacked the comparison-item cases), and the FinOps High-Impact / Expensive-Queries grids — via a shared `PlanNavigationController` so the confirm/cancel/busy/error handling lives once and the two apps stay in parity +- **Full Dashboard: a "Collection Stopped" alert for when the collector itself goes dark** ([#1246]) — the app tracked collection *health* (did data arrive?) but never collection *state*, so disabling the SQL Agent collector jobs was silent: the Dashboard kept looking healthy — calmer, even, since the live cards read zero rows from the `collect.*` tables — until a collector aged into STALE on the Collection Health tab after 24 hours. A new app-side check that survives the collector being off (it is the collector that fills every other table) now watches two things: a live read of `msdb.dbo.sysjobs.enabled` for the `PerformanceMonitor%` jobs (the immediate, specific cause) and a `config.collection_log` freshness backstop (no run in 30+ minutes, which also catches the Agent service being stopped or collectors silently erroring). It surfaces as a proactive "Collection Stopped" tray/email alert — a new `NotifyOnCollectionStopped` setting (default on), with a "Collection Resumed" clear and the usual cooldown/mute, mirroring the Capture Down pattern — plus a banner on the Collection Health tab, so it shows immediately instead of only after the 24-hour STALE lag. The msdb read is gated off on Azure SQL Database and degrades gracefully where msdb is restricted (e.g. AWS RDS / no `SQLAgentReaderRole`), so it never reports "disabled" when it simply could not look. The decision logic is extracted to `DatabaseService.DecideCollectionStopped` and unit-tested (9 cases). Full edition only — Lite runs its own in-app scheduler, not Agent jobs. + +### Changed + +- **Lite and Dashboard: the recommendation engine's operator advice is rebuilt to be sourced and composed from your server's own facts** ([#1189], [#1196], [#1197], [#1198], [#1203], [#1244]) — the advice on every Recommendation (headline, investigation, remediation) was largely unsourced folklore that hedged ("maybe, if…") and pointed at MCP tool names (`get_*`) instead of stating the measured number. It now **composes from the collected fact set at analysis time**: each of the 56 advice blocks states your server's actual values — current MAXDOP/CTFP, max server memory, the RCSI-off database count, the dominant lock mode, the SOS signal-wait share, the lead blocker's identity, and the named contended object (`dbo.Orders` / `index IX_…`, via a new `Fact.ObjectName` carrier the collectors had been dropping) — and each remediation states the co-fired findings that actually fired rather than a bag of generic guesses. Tool names, internal field names, and the speculative "bag of tricks" are stripped from both the composed path and the static fallbacks, and the composed advice is frozen into each finding's story at analysis time so it reads identically in the apps and the MCP. Covered by `FactAdviceComposeTests` and `FactAdviceCorrectnessTests` in both suites. (A larger prose rewrite is planned for a future release.) +- **Lite and Dashboard: click a chart's legend key (or a series line) to isolate that series** — on any multi-series trend chart, left-clicking a series' built-in legend key — or its line in the plot — now dims every other series and auto-fits the Y axis to the clicked one, so a wait type (or metric) that sits flat against the axis underneath the big lines becomes readable. Clicking it again, clicking a different series, or double-clicking (autoscale) restores the full view. It is a transient view toggle only: it never changes the picker selection and never refetches or deletes data, and any re-render (picker/time-range change or a background poll) resets it. The whole mechanic lives in the shared `PerformanceMonitor.Ui/ChartHoverHelper` that every dynamic-legend chart in both apps already routes its series through — so the two apps can't drift — and the Y-fit transiently clears then restores Dashboard's `LockedVertical` axis rule so the fit actually sticks. The dim uses each series' own identity color (so it works in Dark/Light/CoolBreeze), and clicking a memory-pressure **bar** legend key is intentionally a no-op (stacked bars are deferred). Covered by `ChartClickIsolateTests` in both suites +- **Lite and Dashboard: alert payloads carry a stable dedup fingerprint and involved-object list** ([#1140]) — deadlock and blocking alerts (plus query/job/disk) now attach a deterministic, server-scoped fingerprint and the list of involved objects, computed in the shared `PerformanceMonitor.Notifications/AlertFingerprint.cs` (sorted, literal-stripped, excluding volatile fields) and grouped into `AlertContext.Incidents` via `IncidentGrouping`. This gives every downstream consumer — cooldown ([#1154]), per-event delivery ([#1141]), and restart dedup ([#1145]) — a stable key for "the same incident" instead of re-alerting on cosmetically-different text. Wired into both apps' alert engines; covered by `AlertFingerprintTests`, `IncidentGroupingTests`, and `AlertIncidentRenderTests` +- **Lite and Dashboard: optional per-event notification mode for deadlocks and blocking** ([#1141]) — alerts can now be delivered as one message per distinct incident instead of the default batched per-cycle summary, controlled by a new `AlertDeliveryMode` (Summary | PerEvent) setting plus `AlertPerEventMaxPerCycle` (default 10, with a "+N more" overflow) in both apps' Settings. Per-event cards keep the full forensic detail and a numeric current value. Built on the [#1140] incident grouping; covered by `PerEventNotificationTests`. (A per-server override is tracked as a fast-follow — Dashboard has no per-server settings store yet.) +- **Lite and Dashboard: the low-disk (Volume Free Space) alert is now severity-graded instead of always INFO** ([#1136]) — the metric name was absent from the shared severity map, so every low-disk alert fell through to the default INFO tier (blue, lowest severity) regardless of how full the volume was. A volume nearing full stops data/log file growth — transactions fail and the database can go into recovery/suspect — so this under-prioritized a potentially critical condition, including for downstream severity-based webhook routing (Teams/Slack). The alert now renders **WARNING** for a normal breach and **CRITICAL** when the worst breached volume is critically low (≤ 3% free **or** ≤ 2 GB free — a second, lower tier beneath the user-configured fire threshold). The critical grading lives in the shared `LowDiskAlertGate.IsCriticallyLow`, and the tier rides through to the email badge/accent, Teams card, and Slack sidebar via an `AlertContext.SeverityOverride`, so the metric name (hence mute rules, cooldown, and Alert-History matching) is unchanged. The existing "notify only on a fresh or worsening breach" firing rule is untouched — this is purely the severity classification. Fixed identically in both apps. Covered by `AlertSeverityTests` and `LowDiskAlertGateTests` +- **Lite and Dashboard: the execution-plan viewer is now one shared control** ([#1205]) — Dashboard and Lite each carried a near-identical 6-file `PlanViewerControl`; the plan parsing/layout/analysis already lived in the shared `PerformanceMonitor.PlanAnalysis`, so only the WPF shell was duplicated. That shell is now a single `PerformanceMonitor.Ui.PlanViewerControl` referenced by both apps, so it can no longer drift. Two host-visible effects: (1) **Lite's multi-statement picker becomes a sortable grid** — for a plan with more than one statement, Lite previously used a `Statements:` dropdown and now shows Dashboard's collapsible **Statements** panel (a sortable grid of #, query text, CPU/Elapsed/UDF or estimated cost, and Critical/Warning counts, with right-click → Copy Query Text), so you can sort a batch by cost or runtime to find the heavy statement; single-statement plans are unchanged (no panel). (2) The shared control parses and analyzes **off the UI thread**, so opening a large showplan no longer briefly freezes the window. The theme-change handler also moves to a `Cleanup()`-at-teardown lifecycle in both apps — it is no longer detached on `Unloaded` (which a `TabControl` raises on every tab switch, the previous Dashboard behavior that could leave a backgrounded plan tab no longer re-theming); each app now calls `Cleanup()` at every real teardown (plan window/tab close, server-tab close, and the malformed-XML error path) +- **Lite and Dashboard: "Show Active Queries at This Time" right-click drill-down now works on every resource chart** ([#1208]) — the chart drill-down (right-click a point to jump to Active Queries scoped to that ±30-minute window, originally [#682]) was wired on only a subset of charts and had drifted between the two apps. A full audit of every chart surfaced three problems, all fixed: (1) the Dashboard **Resources/TempDB tab** drill-down was dead plumbing — the control defined the `AddDrillDown` mechanism and a `ChartDrillDownRequested` event that the server tab even subscribed to, but never wired a single chart, so none of its 11 charts could drill down; (2) the TempDB **Allocated MB** chart (both apps) and Lite's **Memory Pressure Events** chart had no right-click menu at all (no save/export *or* drill-down) because they were never registered — they now get the full menu plus a new hover helper so the click can resolve a timestamp; (3) bidirectional parity drift — Lite lacked it on Memory Clerks/Grants/Pressure and Current/Lock Waits while Dashboard had them, and Dashboard lacked it on TempDB Stats while Lite had it. The drill-down is now present on every time-series resource chart in both apps (memory clerks/grants/pressure, tempdb size/stats/file-I/O latency, file-I/O latency + throughput, perfmon, current waits, and — Dashboard only — session/latch/spinlock stats and plan cache; Lite's Lock Wait routes to *Show Blocking* to match Dashboard). The query/procedure/Query Store/execution **duration-trend** charts are intentionally excluded because the grid→slicer overlay already routes there, as are Collector Duration and the system-health event charts. Lite consolidates the new navigation into one shared `OnActiveQueriesDrillDown` handler rather than duplicating the existing CPU/Memory/TempDB drill-down bodies +- **Lite and Dashboard: the Overview correlated-timeline lanes get the same Active Queries drill-down** ([#1210]) — follow-on to [#1208]. The five synchronized lanes (CPU, Wait Stats, Blocking, Memory, File I/O) are a deliberately stripped-chrome view that disables all chart mouse input for its shared crosshair, so they had no right-click at all — and in Lite that control *is* the **Overview / default landing tab** (Dashboard's 2×2 Overview grid already drilled, so this also closes a cross-app gap). Each lane now has a minimal right-click menu with just *Show Active Queries at This Time*; because pan/zoom is disabled the clicked time is read straight from the X axis, and it routes to the same ±30-minute Active Queries navigation as every other chart. The synchronized crosshair is unaffected — it is mouse-move driven, independent of the right-click handler +- **Dashboard: the Queries and Resource tabs load much faster** ([#1181], [#1182], [#1190]) — opening the Queries tab took 3–9s under load. The heavy sub-tabs (heatmap, regressions, patterns) now lazy-load like the big grids; first-load, tab-switch, and time-range changes refresh only the *visible* tab instead of all six at once (which saturated the connection pool); the Query/Procedure/Query Store grids drop from `TOP (500)` to `TOP (50)` (matching Lite, ~10x less query-text decompression); and the large plan / deadlock-graph / blocked-process XML is deferred out of the grid queries and fetched on demand — so Query Stats now opens in well under a second +- **Lite and Dashboard: the Active Queries view refreshes when you open it** ([#1183]) — the sub-tab previously showed the last-loaded snapshot until the next timer tick; it now re-runs the live snapshot fetch when it becomes active (Dashboard had no auto-refresh on it at all), guarded so the wait/chart/heatmap drill-downs that select Active Queries with *filtered* data aren't clobbered by the refresh +- **Lite and Dashboard: resolved/cleared tray toasts are themed to match the condition cards** ([#1186]) — *Blocking Cleared*, *CPU Resolved*, etc. went through a plain unthemed Windows balloon while the triggering alert used the themed card, so a resolution looked nothing like the alert it resolved. All 17 resolved/cleared notifications across both apps now render a themed `StyledBalloon` with a green-check *resolved* accent (the same green as the email/webhook RESOLVED badge); server online/offline, update-available, and analysis-finding toasts stay as OS balloons so they persist in the Action Center + +### Fixed + +- **Lite and Dashboard: Collection Health no longer flags skip-if-unchanged collectors as STALE/NEVER_RUN** ([#1248]) — dedup-snapshot collectors (notably `server_properties` and the config snapshots) log `SKIPPED` when nothing has changed since the last run, which is a successful no-op. But the per-collector health computed "last success" from `SUCCESS` rows only, so a collector that was correctly skipping showed **STALE** — and then **NEVER_RUN** once its last real success aged out of log retention — even though it was firing on schedule every day. `SKIPPED` now counts as a healthy run (matching the semantics the [#1246] Collection-Stopped freshness backstop already uses): the Dashboard `report.collection_health` view counts it toward `last_success_time`/`total_runs` and includes it in the recent-failures window (so a skip-only collector isn't misread as FAILING either), and Lite counts it toward `last_success_time` too. Verified live — `server_properties` flips STALE/NEVER_RUN → HEALTHY. Fixed identically in both apps. + +- **Lite and Dashboard: the Overview's empty Blocking/Deadlocking lane now renders as a live grid instead of a dead black box** ([#1245]) — on a healthy server the Blocking/Deadlocking correlated-timeline lane usually has no events, and its empty state drew as a blank black box: the grid was hidden, both axes used an `EmptyTickGenerator`, and a "No Data" label sat at the origin where `SyncXAxes` immediately shoved it off-screen the moment it set the real time range. The empty lane now renders like the populated ones — grid kept, a 0-1 Y axis with normal numeric ticks, and `DateTimeTicksBottomDateChange` so its vertical gridlines line up with the other lanes (time labels still only on the bottom File I/O lane) — so an idle lane reads as "nothing happening" rather than broken. Mirrored in both apps' sync-paired `CorrelatedTimelineLanesControl`. + +- **Lite and Dashboard: corrected wrong and misleading recommendation advice, plus a MAXDOP code bug** ([#1185], [#1187], [#1192], [#1194]) — an adversarial, source-checked audit of the advice engine found several claims that were outright wrong; the worst are fixed against primary documentation: `SOS_SCHEDULER_YIELD` is no longer described as "demand exceeds supply" (it's quantum-exhaustion CPU work — pressure shows as signal-wait / runnable-queue share); "every UPDATE touches every nonclustered index" is removed (only changed-column indexes are touched); the last-page-contention guidance is corrected to `OPTIMIZE_FOR_SEQUENTIAL_KEY` (the old text recommended *adding* a clustered index, which creates the hotspot); tempdb "autogrowth disabled" is corrected to enable it; range-lock isolation is narrowed to `SERIALIZABLE`; and the dead `PERFMON_PLE` rule (PLE is never collected) is removed. Separately a **code** bug is fixed: `FactRemediation.RecommendedMaxdop` computed MAXDOP from `engine_edition` (8/4/1 by edition — invented; SQL Server's guidance is topology-based, never edition-based) and the engine both recommended *and could apply* it; it now derives from cores-per-socket (`min(cores, 8)`), with edition dropped from the advice. Also fixed: the root-fact value surfaced in the MCP and notifications showed the finding's severity instead of its value. Covered by `FactAdviceCorrectnessTests`. +- **Lite: the Alert History Value/Threshold columns no longer show a raw full-precision float** ([#1134]) — a Volume Free Space alert rendered its Value as `0.9746057751382348` instead of a rounded, unit-aware figure. Lite stores each alert's `current_value`/`threshold_value` as a DuckDB `DOUBLE` and the grid binds to `CurrentValueDisplay`/`ThresholdValueDisplay`, which ran the value through a `FormatValue` helper that special-cased only CPU and TempDB and fell through to `":G"` (full precision) for every other metric — so free-space %, poison-wait ms, long-running-job % of average, and the count metrics all leaked the raw double. `FormatValue` is now keyed on the exact `metric_name` strings Lite's alert engine emits: percent metrics (High CPU, TempDB Space, Volume Free Space, Long-Running Job) render as `F1` + `%`, Poison Wait as whole `ms`, Long-Running Query as whole `m`, and the count metrics (Blocking, Deadlocks, Failed Agent Job) as whole numbers — and the fallback is now `":F2"` instead of `":G"`, so an unmapped metric (e.g. an analysis-finding severity) can never render a raw float again. This was **Lite-only**: the Dashboard's `AlertHistoryDisplayItem.CurrentValue` is a pre-formatted string built at the alert site, so it was structurally immune and is unchanged. Covered by `AlertHistoryValueFormatTests` +- **Dashboard: muted and resolved alerts no longer inflate the sidebar Alert badge** ([#1226], [#1228]) — the left-nav **Alerts** badge counted every non-hidden alert-history row from the last 24h, including muted alerts (known recurring noise, e.g. a muted SQL Sentry trace-reader session) and resolution notices (the "… Cleared/Resolved/Restored" rows logged when a condition recovers). A muted source re-firing every cooldown, or a run of auto-resolved conditions, kept the badge lit so the dashboard looked like it had unhandled alerts when it didn't. The badge now counts only *actionable* alerts: `GetAlertHistory` filters muted and resolved rows out **before** the row limit (so high-volume noise can't crowd real alerts out of the count), while the Alert History grid and the MCP `get_alert_history` tool still show every row for audit. Dashboard-only — Lite's badge tracks live blocking/deadlock health, not history rows. Covered by `JsonAlertHistoryStoreMutedFilterTests` +- **Lite and Dashboard: Alert History rows are classified by one shared metric classifier, fixing the "Restored" row styling** ([#1228]) — both apps independently decided an Alert History row's resolved/critical/warning styling from inline metric-name string checks, and both copies recognized only the "Cleared"/"Resolved" resolution suffixes while **missing "Restored"** — so `Capture Restored` and `Server Restored` rows were styled as warnings even though the UI legend documents Server Restored as a resolved (green) state. The duplicated logic is now a single `PerformanceMonitor.Common.AlertMetricClassifier` (`IsResolution`/`IsCritical`/`IsWarning`) consumed by both grids and the Dashboard badge, so it can no longer drift and the "Restored" rows render as resolved in both apps. Covered by `AlertMetricClassifierTests` +- **Lite and Dashboard: the Capture Down and Failed Agent Job alerts render at their real severity in email and webhook notifications** ([#1229]) — both metric names were absent from the shared `AlertSeverity` map (the same gap class as the low-disk fix [#1136]), so although each is emailed/webhooked they fell through to the default INFO-blue tier. `Capture Down` (blocking/deadlock capture is broken — fired as an Error toast) now renders **CRITICAL**, and `Failed Agent Job` (a Warning toast) renders **WARNING**, matching each alert's tray treatment and feeding severity-based Teams/Slack routing correctly. Fixed in the shared map so both apps' email body, Teams card, and Slack sidebar are corrected; covered by `AlertSeverityTests` +- **Lite and Dashboard: webhook (Teams/Slack) alerts no longer re-fire after an app restart** ([#1145]) — #981's restart-dedup was applied only to the email channel, so a Teams/Slack alert posted shortly before a restart was re-posted afterward. The webhook cooldown now seeds from history on startup via the new `GetLastWebhookSentUtcAsync` (mirroring the email seed), and Lite persists its edge-trigger watermark to a new `config_edge_trigger_watermarks` DuckDB table so it is restored before the first post-restart sweep. Covered by `WebhookCooldownSeedTests` +- **Lite and Dashboard: alert cooldown is keyed on the [#1140] fingerprint, not just (server, metric)** ([#1154]) — distinct concurrent deadlock/blocking incidents on the same server shared a single (server, metric) cooldown, so the second incident was silently dropped from email/webhook (the tray showed it, Teams didn't). A new shared `PerformanceMonitor.Notifications/IncidentCooldown.cs` keys the cooldown per fingerprint — it sends if **any** incident is outside its window, stamps every candidate key, and falls back to the metric key for non-fingerprintable alerts (CPU/memory/poison-wait/tempdb) — and is consumed by both the email and webhook channels in both apps. Covered by `IncidentCooldownTests` +- **Lite: the `index_object_stats` collector no longer times out and returns zero index data on larger estates** ([#1135]) — the v3.0.0 collector ran its entire multi-database sweep as **one** `SqlCommand` under the global 30s `CommandTimeoutSeconds`, cursoring over every online database into a `#temp` and returning a single final `SELECT`. Because nothing streamed back until the end, the 30s was a *cumulative, all-or-nothing* budget across every database — on a server with sizable/many databases the sweep blew past 30s, failed with `Execution Timeout Expired` (SQL `#-2`), and discarded results from **every** database, not just the slow one. Enabled by default and "never-run = due immediately," it failed on first connect right after upgrade and kept retrying the timeout. Now the collector runs **one command per database** (mirroring the Query Store collector): on-prem enumerates databases then sends each through `[db].sys.sp_executesql`, Azure SQL DB connects to each database individually, and each database has its own command, timeout, and `try/catch` — so a slow or inaccessible database fails only itself and the rest still persist. Within each database the three DMVs (`sys.dm_db_partition_stats`, `sys.dm_db_index_usage_stats`, `sys.dm_db_index_operational_stats`) are staged into `#temp` tables with single scans and then joined, giving the optimizer real cardinality and avoiding the bad plans the old single monolithic multi-DMV join produced on large databases (the `sp_IndexCleanup` technique). The collector also gets a dedicated 300s timeout (matching the FinOps `sp_IndexCleanup` path) instead of the 30s meant for lightweight DMV reads. The Dashboard's equivalent SQL collector (`install/55`) — which was not subject to the bug (it runs under SQL Agent and persists per database) — was brought to parity with the same DMV-staging technique for plan quality on large databases +- **Dashboard: the execution-plan node "actual of estimated rows" badge no longer over-reports for multi-execution operators** ([#1205]) — Dashboard compared an operator's *total* actual rows (summed across every execution) against the *per-execution* estimate, so any operator that runs more than once — e.g. the inner side of a nested-loops join — looked far off: a seek executed 1,000 times showed ~1,000x its estimate and turned red even when each execution matched the estimate exactly. It now divides actual rows by execution count before comparing, matching the per-execution estimate — the math Lite's viewer already used and that `PlanAnalyzer` applies elsewhere. Surfaced while consolidating the two apps' plan viewers into the shared control; Lite's badge was already correct and is unchanged +- **Lite: a fresh install no longer floods collection and the wait-stats tab with benign waits** ([#1240]) — a clean Lite install / nightly extract created the per-user `config\` directory empty and never seeded it from the bundled `ignored_wait_types.json`, so `LoadIgnoredWaitTypes()` returned an empty set (cached by the `Lazy`), the wait filter became a no-op, and benign waits (`SOS_WORK_DISPATCHER`, `DISPATCHER_QUEUE_SEMAPHORE`, `CLR_AUTO_EVENT`, …) dominated both collection and the wait-stats tab. A new `ConfigSeeder` seeds the per-user config from the bundled copies on first run (copy-if-absent, never clobbering a user-edited file) with a bundled-copy fallback in the loader so the filter can never silently be empty, and the wait-stats queries (top list, picker distinct types, total trend) now also exclude ignored waits **at display time** via a shared `IgnoredWaitTypes` builder — so a box that collected benign waits before the filter was active is cleaned up in the UI without deleting anything (rows age out via retention). Dashboard seeds its list server-side at install and is unaffected. Covered by `ConfigSeederTests` and `IgnoredWaitTypesTests` +- **Lite: picker-chart trends no longer run an N+1 query loop** ([#1209]) — Wait Stats, Memory Clerks, and Perfmon each fetched one trend query *per* selected picker item in a sequential await loop, so loading several series ran 12–20 serial DuckDB queries (~3.6s for Wait Stats under load) — invisible to the slow-query log because each query was individually under 500ms. Mirroring Dashboard's already-batched pattern, each chart now issues one query returning all selected series (IN-list, grouped by type/counter) with a client-side plot loop (the Wait Stats query partitions its `LAG` window by `wait_type`), cutting Wait Stats chart-build from 3628ms → under 500ms and the Perfmon tab from 3126ms → 534ms and restoring Lite↔Dashboard parity. Also adds a lightweight end-to-end `TabNav` timer to catch future slow tab loads +- **Lite and Dashboard: upgrading the desktop app now replaces the running tray instance instead of surfacing the stale one** ([#1148]) — the apps are single-instance and minimize to the tray, so an old build kept running after the user "closed" it, and launching the new build just handed back the old in-memory version via the mutex — the upgrade appeared not to take effect. A version-aware startup handoff (shared `SingleInstanceCoordinator` / `ProcessInspector` / `SingleInstanceDecision` in `PerformanceMonitor.Ui`) now has a newer build close an older tray-resident one and take over, with an elevated-relaunch path for the run-as-administrator case; same-or-newer instances still just surface the running window. Scoped per exe name, so Lite and Dashboard never close each other +- **Lite: the Blocking tab's ~930ms hitch and interactive UI-thread stalls are removed** ([#1193], [#1202]) — deadlock-graph parsing (`XElement.Parse` plus deep traversal over up to 50 graphs) ran on the UI thread at all five call sites, and 31 further synchronous DuckDB calls on interactive paths (slicer drags, chart/grid drill-downs, View/Download Plan, row-history expands, stat comparisons, the heatmap metric change, alert history + dismiss) blocked the dispatcher because DuckDB.NET is synchronous (a connection-open alone is ~766ms under load). Both now run off the dispatcher; only the grid bind stays on the UI thread +- **Dashboard: "View Plan" was a silent no-op on several surfaces** ([#1181], [#1190]) — the Active Queries drill-down grids returned "no plan found" for every row because `query_snapshots.collection_time` is legacy `datetime(3)` but the lookup bound the parameter as `datetime2`, so the exact match failed on precision; and the blocking-report *View Plan* extracted process nodes as direct children when the stored root is the full XE `` (descendant nodes). Both are fixed and the plan is fetched on demand +- **Dashboard: the FinOps Database Sizes tab no longer shows blank until a manual refresh** ([#1179]) — an init-order race ran the first load before its `DataGridFilterManager` existed, so the loader fetched the rows then discarded them. The manager is now created in the constructor; it was the only FinOps sub-tab whose manager was built in `OnLoaded` +- **Lite and Dashboard: failed-Agent-job alerts no longer coalesce or replay after a restart** ([#1157], [#1173]) — two distinct job failures in the same window shared a single cooldown (the failed-job builder attached no [#1140] incident fingerprint and fell back to the metric key); each failure now carries a per-job fingerprint keyed by job name. Separately the failed-job tray watermark was in-memory only — blocking/deadlock got restart persistence in [#1145] but this path never did — so on reopen every failure still in the lookback window looked new and re-fired toasts the user had already dismissed; the watermark now persists (server-local) and re-seeds on startup +- **Dashboard: anomaly and baseline detection corrected to match Lite's already-fixed behavior** ([#1155]) — a code-sharing drift audit found two Dashboard-only bugs: sparse `(hour, day-of-week)` baseline buckets were written as `HourOnly` because a copy-paste ternary had two identical `Full` arms, and blocking/deadlock spike detection compared a raw count over the whole (default 4h) analysis window against a per-hour baseline mean, so a steady event rate scaled with window length and could trip the spike threshold. Both now match `Lite`'s tested behavior (full buckets labeled `Full`; current counts normalized to per-hour before the ratio) + ## [3.0.0] - 2026-06-15 ### Important @@ -81,6 +135,54 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 [#1116]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1116 [#1121]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1121 [#1122]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1122 +[#1134]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/1134 +[#1135]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/1135 +[#1136]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/1136 +[#1205]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1205 +[#1208]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1208 +[#1210]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1210 +[#1140]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/1140 +[#1141]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/1141 +[#1145]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/1145 +[#1154]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/1154 +[#1226]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1226 +[#1228]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1228 +[#1229]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1229 +[#1138]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/1138 +[#1207]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1207 +[#1209]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1209 +[#1214]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1214 +[#1216]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1216 +[#1224]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1224 +[#1227]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1227 +[#1236]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/1236 +[#1240]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/1240 +[#1185]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1185 +[#1187]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1187 +[#1189]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1189 +[#1192]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1192 +[#1194]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1194 +[#1196]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1196 +[#1197]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1197 +[#1198]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1198 +[#1203]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1203 +[#1244]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1244 +[#1245]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1245 +[#1246]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1246 +[#1248]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1248 +[#1148]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1148 +[#1155]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1155 +[#1157]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1157 +[#1173]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1173 +[#1179]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1179 +[#1181]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1181 +[#1182]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1182 +[#1183]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1183 +[#1184]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1184 +[#1186]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1186 +[#1190]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1190 +[#1193]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1193 +[#1202]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1202 ## [2.11.0] - 2026-05-19 diff --git a/Dashboard.Tests/AlertDeliveryModeOverrideTests.cs b/Dashboard.Tests/AlertDeliveryModeOverrideTests.cs new file mode 100644 index 000000000..98aea0248 --- /dev/null +++ b/Dashboard.Tests/AlertDeliveryModeOverrideTests.cs @@ -0,0 +1,46 @@ +/* + * Performance Monitor Dashboard + * Copyright (c) 2026 Darling Data, LLC + * Licensed under the MIT License - see LICENSE file for details + */ + +using System.Text.Json; +using PerformanceMonitor.Notifications; +using PerformanceMonitorDashboard.Models; +using Xunit; + +namespace PerformanceMonitorDashboard.Tests; + +/// +/// #1236 per-server alert delivery override persists on (servers.json, +/// same WriteIndented options ServerManager uses). A pre-#1236 file with no field inherits the global. +/// The shared precedence rule () is covered in Lite.Tests. +/// +public class AlertDeliveryModeOverrideTests +{ + private static readonly JsonSerializerOptions s_jsonOptions = new() { WriteIndented = true }; + + [Fact] + public void Override_RoundTripsThroughJson() + { + var server = new ServerConnection { ServerName = "S1", AlertDeliveryModeOverride = AlertNotificationMode.PerEvent }; + var back = JsonSerializer.Deserialize(JsonSerializer.Serialize(server, s_jsonOptions), s_jsonOptions); + Assert.Equal(AlertNotificationMode.PerEvent, back!.AlertDeliveryModeOverride); + } + + [Fact] + public void NullOverride_RoundTripsAsNull() + { + var server = new ServerConnection { ServerName = "S1", AlertDeliveryModeOverride = null }; + var back = JsonSerializer.Deserialize(JsonSerializer.Serialize(server, s_jsonOptions), s_jsonOptions); + Assert.Null(back!.AlertDeliveryModeOverride); + } + + [Fact] + public void LegacyServersJson_WithoutField_InheritsGlobal() + { + // A servers.json written before #1236 has no AlertDeliveryModeOverride property -> null -> inherit. + var back = JsonSerializer.Deserialize("{\"ServerName\":\"S1\",\"DisplayName\":\"S1\"}", s_jsonOptions); + Assert.Null(back!.AlertDeliveryModeOverride); + } +} diff --git a/Dashboard.Tests/AlertMetricClassifierTests.cs b/Dashboard.Tests/AlertMetricClassifierTests.cs new file mode 100644 index 000000000..42514f5cc --- /dev/null +++ b/Dashboard.Tests/AlertMetricClassifierTests.cs @@ -0,0 +1,80 @@ +using PerformanceMonitor.Common; +using Xunit; + +namespace PerformanceMonitorDashboard.Tests; + +/// +/// #1225: the shared metric-name classifier that both apps' Alert History grids and the Dashboard +/// sidebar Alert badge rely on. Locks in the resolution suffix set (Cleared/Resolved/Restored) — +/// the "Restored" cases (Capture/Server Restored) are the ones the old duplicated inline copies +/// missed — plus the critical (Deadlock/Poison) and warning buckets, over the metric names the +/// alert engines actually emit. +/// +public class AlertMetricClassifierTests +{ + [Theory] + [InlineData("Blocking Cleared")] + [InlineData("Deadlocks Cleared")] + [InlineData("Poison Waits Cleared")] + [InlineData("Long-Running Queries Cleared")] + [InlineData("Long-Running Jobs Cleared")] + [InlineData("CPU Resolved")] + [InlineData("TempDB Space Resolved")] + [InlineData("Volume Free Space Resolved")] + [InlineData("Capture Restored")] + [InlineData("Server Restored")] + public void IsResolution_True_ForEveryResolutionNotice(string metric) + { + Assert.True(AlertMetricClassifier.IsResolution(metric)); + Assert.False(AlertMetricClassifier.IsWarning(metric)); // a resolution notice is never a warning + } + + [Theory] + [InlineData("Blocking Detected")] + [InlineData("Deadlocks Detected")] + [InlineData("High CPU")] + [InlineData("Poison Wait")] + [InlineData("Long-Running Query")] + [InlineData("TempDB Space")] + [InlineData("Volume Free Space")] + [InlineData("Long-Running Job")] + [InlineData("Failed Agent Job")] + [InlineData("Capture Down")] + [InlineData("Server Unreachable")] + public void IsResolution_False_ForActionableAlerts(string metric) + { + Assert.False(AlertMetricClassifier.IsResolution(metric)); + } + + [Theory] + [InlineData("Deadlocks Detected")] + [InlineData("Poison Wait")] + public void IsCritical_True_ForDeadlockAndPoison(string metric) + { + Assert.True(AlertMetricClassifier.IsCritical(metric)); + } + + [Theory] + [InlineData("Blocking Detected")] + [InlineData("High CPU")] + [InlineData("Long-Running Query")] + public void IsWarning_True_ForOrdinaryActionableAlerts(string metric) + { + Assert.True(AlertMetricClassifier.IsWarning(metric)); + Assert.False(AlertMetricClassifier.IsCritical(metric)); + Assert.False(AlertMetricClassifier.IsResolution(metric)); + } + + [Fact] + public void Classifiers_AreFalse_ForNullOrEmpty() + { + Assert.False(AlertMetricClassifier.IsResolution(null)); + Assert.False(AlertMetricClassifier.IsResolution("")); + Assert.False(AlertMetricClassifier.IsCritical(null)); + Assert.False(AlertMetricClassifier.IsCritical("")); + // IsWarning is the complement of the two signals, so an absent name defaults to warning — + // matching the long-standing display behavior this classifier replaced. + Assert.True(AlertMetricClassifier.IsWarning(null)); + Assert.True(AlertMetricClassifier.IsWarning("")); + } +} diff --git a/Dashboard.Tests/BlockingChainLayoutTests.cs b/Dashboard.Tests/BlockingChainLayoutTests.cs new file mode 100644 index 000000000..c7a232aa8 --- /dev/null +++ b/Dashboard.Tests/BlockingChainLayoutTests.cs @@ -0,0 +1,127 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using PerformanceMonitor.Common; +using Xunit; + +namespace PerformanceMonitorDashboard.Tests; + +/// +/// Pure unit tests for the shared — determinism and no-overlap of the +/// left-to-right tree/forest placement. No WPF (a constant measure callback supplies node heights). +/// +public class BlockingChainLayoutTests +{ + private const double H = 100.0; + private static double Measure(BlockingChainNode _) => H; + + private static BlockingChainNode Node(int spid, params BlockingChainNode[] children) => + new() { Spid = spid, Children = children.ToList() }; + + // A representative forest: a multi-level tree plus a second apex tree. + private static IReadOnlyList SampleForest() => new[] + { + Node(1, + Node(2, Node(4), Node(5)), + Node(3)), + Node(10, + Node(11), + Node(12, Node(13))) + }; + + private static IEnumerable Flatten(IReadOnlyList roots) + { + foreach (var r in roots) + foreach (var n in FlattenOne(r)) + yield return n; + } + + private static IEnumerable FlattenOne(BlockingChainNode n) + { + yield return n; + foreach (var c in n.Children) + foreach (var d in FlattenOne(c)) + yield return d; + } + + [Fact] + public void Layout_IsDeterministic_AcrossIdenticalForests() + { + var a = SampleForest(); + var b = SampleForest(); + + BlockingChainLayout.Layout(a, Measure); + BlockingChainLayout.Layout(b, Measure); + + var fa = Flatten(a).ToList(); + var fb = Flatten(b).ToList(); + Assert.Equal(fa.Count, fb.Count); + for (int i = 0; i < fa.Count; i++) + { + Assert.Equal(fa[i].Spid, fb[i].Spid); + Assert.Equal(fa[i].X, fb[i].X); + Assert.Equal(fa[i].Y, fb[i].Y); + } + } + + [Fact] + public void Layout_PlacesDepthByColumn() + { + var roots = SampleForest(); + BlockingChainLayout.Layout(roots, Measure); + + double depth0 = BlockingChainLayout.Padding; + double depth1 = BlockingChainLayout.Padding + BlockingChainLayout.HorizontalSpacing; + double depth2 = BlockingChainLayout.Padding + 2 * BlockingChainLayout.HorizontalSpacing; + + Assert.Equal(depth0, roots[0].X); // apex + Assert.Equal(depth1, roots[0].Children[0].X); // direct victim + Assert.Equal(depth2, roots[0].Children[0].Children[0].X); // grand-victim + } + + [Fact] + public void Layout_ProducesNoOverlappingNodes() + { + var roots = SampleForest(); + BlockingChainLayout.Layout(roots, Measure); + + var nodes = Flatten(roots).ToList(); + const double w = BlockingChainLayout.NodeWidth; + const double eps = 0.5; + + for (int i = 0; i < nodes.Count; i++) + { + for (int j = i + 1; j < nodes.Count; j++) + { + var a = nodes[i]; + var b = nodes[j]; + bool overlapX = a.X < b.X + w - eps && b.X < a.X + w - eps; + bool overlapY = a.Y < b.Y + H - eps && b.Y < a.Y + H - eps; + Assert.False(overlapX && overlapY, + $"SPID {a.Spid} ({a.X},{a.Y}) overlaps SPID {b.Spid} ({b.X},{b.Y})"); + } + } + } + + [Fact] + public void Layout_ReturnsPositiveExtentCoveringAllNodes() + { + var roots = SampleForest(); + var (width, height) = BlockingChainLayout.Layout(roots, Measure); + + Assert.True(width > 0); + Assert.True(height > 0); + + var nodes = Flatten(roots).ToList(); + Assert.True(width >= nodes.Max(n => n.X) + BlockingChainLayout.NodeWidth); + Assert.True(height >= nodes.Max(n => n.Y) + H); + } + + [Fact] + public void Layout_EmptyForest_ReturnsPaddingExtent() + { + var (width, height) = BlockingChainLayout.Layout(Array.Empty(), Measure); + Assert.True(width > 0); // padding only + Assert.True(height > 0); + } +} diff --git a/Dashboard.Tests/BlockingChainReconstructorTests.cs b/Dashboard.Tests/BlockingChainReconstructorTests.cs index 7fd0fb49f..7333c6787 100644 --- a/Dashboard.Tests/BlockingChainReconstructorTests.cs +++ b/Dashboard.Tests/BlockingChainReconstructorTests.cs @@ -7,9 +7,9 @@ namespace PerformanceMonitorDashboard.Tests; /// -/// Pure unit tests for BlockingChainReconstructor — apex/depth/victim reconstruction, -/// the composite session identity that defeats SPID reuse, the 1900-01-01 sentinel, -/// cycle handling, and the traversal caps. No database. +/// Pure unit tests for BlockingChainReconstructor — apex/depth/victim reconstruction, the (monitor_loop, +/// spid, ecid) session identity (mirroring sp_HumanEventsBlockViewer), cumulative-vs-per-scan scoping, cycle +/// handling, and the traversal caps. No database. /// public class BlockingChainReconstructorTests { @@ -21,7 +21,7 @@ public class BlockingChainReconstructorTests private static BlockingPairRow Pair( int blockedSpid, int blockingSpid, - DateTime? blockedTran = null, DateTime? blockingTran = null, + int? monitorLoop = null, int blockedEcid = 0, int blockingEcid = 0, long waitMs = 1000, string blockingStatus = "running") { return new BlockingPairRow @@ -29,9 +29,12 @@ private static BlockingPairRow Pair( EventTime = new DateTime(2026, 5, 22, 10, 0, 0), DatabaseName = "TestDb", BlockedSpid = blockedSpid, - BlockedTranStarted = blockedTran ?? TranFor(blockedSpid), + BlockedTranStarted = TranFor(blockedSpid), BlockingSpid = blockingSpid, - BlockingTranStarted = blockingTran ?? TranFor(blockingSpid), + BlockingTranStarted = TranFor(blockingSpid), + MonitorLoop = monitorLoop, + BlockedEcid = blockedEcid, + BlockingEcid = blockingEcid, WaitTimeMs = waitMs, LockMode = "X", BlockingStatus = blockingStatus, @@ -40,8 +43,9 @@ private static BlockingPairRow Pair( }; } - private static BlockingReconstruction Run(IEnumerable rows) => - BlockingChainReconstructor.Reconstruct(rows, MaxDepth, MaxPairs, StepBudget); + // Default scope is cumulative (the collector path); the viewer-style tests pass scopeByMonitorLoop: true. + private static BlockingReconstruction Run(IEnumerable rows, bool scopeByMonitorLoop = false) => + BlockingChainReconstructor.Reconstruct(rows, MaxDepth, MaxPairs, StepBudget, scopeByMonitorLoop); [Fact] public void Empty_ProducesNoChains() @@ -79,35 +83,88 @@ public void FanOut_IsDepthOneWithAllVictims() } [Fact] - public void SpidReuse_DifferentTransactionStart_DoesNotSplice() + public void DeepChain_WithNullBlockerTran_StillFormsOneChain() { - // Real chain 200 → 201 → 202, plus SPID 201 reused (different tran) blocking 203. - var reusedTran = TranFor(201).AddHours(3); + // The bug this whole change fixes: blocking_last_tran_started is always NULL, so the old (spid, tran) + // key split a mid-chain node. Keying by spid:ecid within monitor_loop, 116 → 111 → 80 is ONE depth-2 + // chain — 111 unifies as (loop, 111, 0) whether it is the blocker or the blocked party. var result = Run(new[] { - Pair(201, 200), - Pair(202, 201), - Pair(203, 201, blockingTran: reusedTran) // reused 201 — a distinct session - }); + Pair(111, 116, monitorLoop: 620365), + Pair(80, 111, monitorLoop: 620365), + }, scopeByMonitorLoop: true); + + var chain = Assert.Single(result.Chains); + Assert.Equal(116, chain.ApexSpid); + Assert.Equal(2, chain.Depth); + Assert.Equal(2, chain.VictimCount); + } + + [Fact] + public void SeparateEpisodes_SameSpid_DoNotLeak() + { + // The same blocker→blocked pair in two different monitor_loops (episodes) must stay two chains, not + // merge — the cross-episode leakage the monitor_loop scope prevents for the viewer. + var result = Run(new[] + { + Pair(201, 200, monitorLoop: 1), + Pair(201, 200, monitorLoop: 2), + }, scopeByMonitorLoop: true); - // Two distinct chains: apex 200 depth 2, and the reused-201 apex depth 1. Assert.Equal(2, result.Chains.Count); - Assert.Contains(result.Chains, c => c.ApexSpid == 200 && c.Depth == 2); - Assert.Contains(result.Chains, c => c.ApexSpid == 201 && c.Depth == 1); + Assert.All(result.Chains, c => Assert.Equal(200, c.ApexSpid)); + Assert.Contains(result.Chains, c => c.MonitorLoop == 1); + Assert.Contains(result.Chains, c => c.MonitorLoop == 2); } [Fact] - public void Sentinel_TransactionStart_NormalizesToNull() + public void ParallelBlocker_DistinctEcid_AreDistinctNodes() { - // SQL Server's 1900-01-01 "no transaction" sentinel must key the same as NULL. - Assert.Equal( - BlockingChainReconstructor.MakeKey(100, null), - BlockingChainReconstructor.MakeKey(100, new DateTime(1900, 1, 1))); + // Same SPID, different ecid (parallel workers) are distinct sessions: spid 200 ecid 0 and ecid 1 each + // head their own chain. + var result = Run(new[] + { + Pair(201, 200, monitorLoop: 1, blockingEcid: 0), + Pair(202, 200, monitorLoop: 1, blockingEcid: 1), + }, scopeByMonitorLoop: true); + + Assert.Equal(2, result.Chains.Count); + Assert.Contains(result.Chains, c => c.ApexEcid == 0); + Assert.Contains(result.Chains, c => c.ApexEcid == 1); + } - // And a real transaction start must NOT collapse to the sentinel key. - Assert.NotEqual( - BlockingChainReconstructor.MakeKey(100, null), - BlockingChainReconstructor.MakeKey(100, TranFor(100))); + [Fact] + public void Cumulative_MergesAcrossScans() + { + // The collector (scopeByMonitorLoop:false) ignores monitor_loop, so an episode's per-scan re-fires + // merge into ONE chain — preserving window-level depth/victims for the severity fact (per-scan scoping + // would under-count). 200 → 201 (scan 1) and 201 → 202 (scan 2) form one depth-2 chain. + var result = Run(new[] + { + Pair(201, 200, monitorLoop: 1), + Pair(202, 201, monitorLoop: 2), + }, scopeByMonitorLoop: false); + + var chain = Assert.Single(result.Chains); + Assert.Equal(200, chain.ApexSpid); + Assert.Equal(2, chain.Depth); + Assert.Null(chain.MonitorLoop); // cumulative chains carry no episode + } + + [Fact] + public void SpidReuse_DifferentMonitorLoop_DoesNotSplice() + { + // SPID 201 reused across two episodes is two distinct sessions, not one spliced chain. + var result = Run(new[] + { + Pair(201, 200, monitorLoop: 1), + Pair(202, 201, monitorLoop: 1), // 200 → 201 → 202 in episode 1 + Pair(203, 201, monitorLoop: 2), // reused 201 heads its own chain in episode 2 + }, scopeByMonitorLoop: true); + + Assert.Equal(2, result.Chains.Count); + Assert.Contains(result.Chains, c => c.ApexSpid == 200 && c.Depth == 2); + Assert.Contains(result.Chains, c => c.ApexSpid == 201 && c.Depth == 1); } [Fact] @@ -129,7 +186,7 @@ public void DepthCap_IsFlaggedAndBounded() { // A 12-edge line, reconstructed with a maxDepth of 4. var rows = Enumerable.Range(0, 12).Select(i => Pair(501 + i, 500 + i)).ToList(); - var result = BlockingChainReconstructor.Reconstruct(rows, maxDepth: 4, MaxPairs, StepBudget); + var result = BlockingChainReconstructor.Reconstruct(rows, maxDepth: 4, MaxPairs, StepBudget, scopeByMonitorLoop: false); Assert.True(result.DepthCapped); var chain = Assert.Single(result.Chains); @@ -180,4 +237,72 @@ public void Ranking_PutsTheHigherMagnitudeChainFirst() // The 8-victim fan-out out-scores the depth-2 / 2-victim chain. Assert.Equal(800, result.Chains[0].ApexSpid); } + + [Fact] + public void ReconstructedOutput_CarriesTranStartedAndDatabaseName() + { + // Transaction start is display-only now, sourced from the rows (apex from its first outgoing edge). + var result = Run(new[] { Pair(201, 200), Pair(202, 201) }); + + var chain = Assert.Single(result.Chains); + Assert.Equal(200, chain.ApexSpid); + Assert.Equal(TranFor(200), chain.ApexTranStarted); + + Assert.NotEmpty(chain.Levels); + Assert.All(chain.Levels, l => Assert.Equal("TestDb", l.DatabaseName)); + + var top = chain.Levels.Single(l => l.BlockingSpid == 200 && l.BlockedSpid == 201); + Assert.Equal(TranFor(200), top.BlockingTranStarted); + Assert.Equal(TranFor(201), top.BlockedTranStarted); + } + + [Fact] + public void FindChainForSession_AnyMember_ReturnsChainRootedAtApex() + { + // Two separate chains — A: 200 → 201 → 202 ; B: 300 → 301. The viewer scopes to the ONE chain a + // clicked session belongs to, rooted at its apex regardless of where in the chain it sits. + var result = Run(new[] { Pair(201, 200), Pair(202, 201), Pair(301, 300) }); + + // Mid-level 201 -> chain A, rooted at apex 200 (not at the clicked node). monitor_loop null -> matches + // by spid:ecid. + var chainA = BlockingChainReconstructor.FindChainForSession(result, null, 201, 0); + Assert.NotNull(chainA); + Assert.Equal(200, chainA!.ApexSpid); + + Assert.Equal(200, BlockingChainReconstructor.FindChainForSession(result, null, 200, 0)!.ApexSpid); + Assert.Equal(200, BlockingChainReconstructor.FindChainForSession(result, null, 202, 0)!.ApexSpid); + Assert.Equal(300, BlockingChainReconstructor.FindChainForSession(result, null, 301, 0)!.ApexSpid); + Assert.Null(BlockingChainReconstructor.FindChainForSession(result, null, 999, 0)); + } + + [Fact] + public void FindChainForSession_DisambiguatesByMonitorLoop() + { + // Victim 500 was blocked by apex 100 in episode 1 and apex 200 in episode 2 — two scoped chains, both + // containing 500. The clicked row's monitor_loop selects the right one. + var result = Run(new[] + { + Pair(blockedSpid: 500, blockingSpid: 100, monitorLoop: 1), + Pair(blockedSpid: 500, blockingSpid: 200, monitorLoop: 2) + }, scopeByMonitorLoop: true); + + Assert.Equal(100, BlockingChainReconstructor.FindChainForSession(result, 1, 500, 0)!.ApexSpid); + Assert.Equal(200, BlockingChainReconstructor.FindChainForSession(result, 2, 500, 0)!.ApexSpid); + } + + [Fact] + public void FindChainForSession_NullMonitorLoop_FallsBackToSpidEcid() + { + // A lead-blocker grid row may not supply its monitor_loop; the click must still resolve via spid:ecid + // rather than reporting "no reconstructable chain" (supersedes the #1222 SPID-only fallback). + var result = Run(new[] + { + Pair(201, 200, monitorLoop: 7), + Pair(202, 201, monitorLoop: 7), + }, scopeByMonitorLoop: true); + + var chain = BlockingChainReconstructor.FindChainForSession(result, null, 200, 0); + Assert.NotNull(chain); + Assert.Equal(200, chain!.ApexSpid); + } } diff --git a/Dashboard.Tests/BlockingChainTreeBuilderTests.cs b/Dashboard.Tests/BlockingChainTreeBuilderTests.cs new file mode 100644 index 000000000..c7c252d0a --- /dev/null +++ b/Dashboard.Tests/BlockingChainTreeBuilderTests.cs @@ -0,0 +1,256 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using PerformanceMonitor.Common; +using Xunit; + +namespace PerformanceMonitorDashboard.Tests; + +/// +/// Pure unit tests for the shared — the DAG -> tree logic that +/// turns the analysis engine's edge lists into renderable trees. Tested once in Common (not duplicated +/// per app). No database, no WPF. +/// +public class BlockingChainTreeBuilderTests +{ + private static DateTime Tran(int spid) => new DateTime(2026, 5, 22, 9, 0, 0).AddSeconds(spid); + + private static BlockingEdgeInput Edge( + int level, int blocker, int blocked, + long wait = 1000, string lockMode = "X", string db = "TestDb", + DateTime? blockerTran = null, DateTime? blockedTran = null, + int blockerEcid = 0, int blockedEcid = 0) => new() + { + Level = level, + BlockingSpid = blocker, + BlockingEcid = blockerEcid, + BlockingTranStarted = blockerTran ?? Tran(blocker), + BlockedSpid = blocked, + BlockedEcid = blockedEcid, + BlockedTranStarted = blockedTran ?? Tran(blocked), + WaitTimeMs = wait, + LockMode = lockMode, + DatabaseName = db, + BlockingSqlText = $"blocker {blocker}", + BlockedSqlText = $"blocked {blocked}", + // Identity keyed per spid so a node's sourced login/host/app is assertable. + BlockedLoginName = $"login{blocked}", + BlockedHostName = $"host{blocked}", + BlockedClientApp = $"app{blocked}", + BlockingLoginName = $"login{blocker}", + BlockingHostName = $"host{blocker}", + BlockingClientApp = $"app{blocker}" + }; + + private static BlockingChainInput Chain( + int apex, IEnumerable edges, + double magnitude = 1.0, bool sleeping = false, DateTime? apexTran = null) => new() + { + ApexSpid = apex, + ApexTranStarted = apexTran ?? Tran(apex), + ApexSleeping = sleeping, + Magnitude = magnitude, + Edges = edges.ToList() + }; + + private static BlockingChainModel Build(params BlockingChainInput[] chains) => + BlockingChainTreeBuilder.Build(chains, false, false, false); + + private static IEnumerable Flatten(BlockingChainNode n) + { + yield return n; + foreach (var c in n.Children) + foreach (var d in Flatten(c)) + yield return d; + } + + [Fact] + public void SingleChain_BuildsLinearTree_WithSourcedFields() + { + // 200 -> 201 -> 202 -> 203 + var model = Build(Chain(200, new[] + { + Edge(1, 200, 201), Edge(2, 201, 202), Edge(3, 202, 203) + })); + + var root = Assert.Single(model.Roots); + Assert.Equal(200, root.Spid); + Assert.True(root.IsApex); + Assert.Equal(0, root.WaitTimeMs); // apex waits on no one + Assert.Equal(string.Empty, root.LockMode); + Assert.Equal("blocker 200", root.SqlText); // apex SQL sourced from its outgoing edge + Assert.Equal("TestDb", root.DatabaseName); + + var n201 = Assert.Single(root.Children); + Assert.Equal(201, n201.Spid); + Assert.False(n201.IsApex); + Assert.Equal(1000, n201.WaitTimeMs); + Assert.Equal("X", n201.LockMode); + Assert.Equal("blocked 201", n201.SqlText); // victim SQL sourced from its incoming edge + + var n202 = Assert.Single(n201.Children); + var n203 = Assert.Single(n202.Children); + Assert.Equal(203, n203.Spid); + Assert.Empty(n203.Children); + Assert.Equal(4, Flatten(root).Count()); + } + + [Fact] + public void BranchingBlocker_AttachesAllVictimsToApex_OrderedBySpid() + { + var model = Build(Chain(300, new[] + { + Edge(1, 300, 303), Edge(1, 300, 301), Edge(1, 300, 302) + })); + + var root = Assert.Single(model.Roots); + Assert.Equal(3, root.Children.Count); + Assert.Equal(new[] { 301, 302, 303 }, root.Children.Select(c => c.Spid).ToArray()); + Assert.All(root.Children, c => Assert.Empty(c.Children)); + } + + [Fact] + public void DeepLinearChain_NestsRootToGreatGrandchild_WithIdentity() + { + // 1 -> 2 -> 3 -> 4 must render NESTED at every level (root -> child -> grandchild -> great-grandchild), + // not flattened. Also proves identity is sourced from the right edge side. + var model = Build(Chain(1, new[] { Edge(1, 1, 2), Edge(2, 2, 3), Edge(3, 3, 4) })); + + var root = Assert.Single(model.Roots); + Assert.Equal(1, root.Spid); + Assert.True(root.IsApex); + // Apex identity comes from the blocking side of its outgoing edge. + Assert.Equal("login1", root.LoginName); + Assert.Equal("host1", root.HostName); + Assert.Equal("app1", root.ClientApp); + + var c2 = Assert.Single(root.Children); + Assert.Equal(2, c2.Spid); + Assert.Equal("login2", c2.LoginName); // victim identity from the blocked side of its incoming edge + Assert.Equal("host2", c2.HostName); + + var c3 = Assert.Single(c2.Children); + Assert.Equal(3, c3.Spid); + + var c4 = Assert.Single(c3.Children); + Assert.Equal(4, c4.Spid); + Assert.Empty(c4.Children); + + Assert.Equal(4, Flatten(root).Count()); + } + + [Fact] + public void MultiLevelBranching_NestsEachBranchSeparately() + { + // 1 blocks 2 AND 3; 2 blocks 4. Expect 1 -> {2, 3} and 2 -> {4}; no flattening, no duplication. + var model = Build(Chain(1, new[] { Edge(1, 1, 2), Edge(1, 1, 3), Edge(2, 2, 4) })); + + var root = Assert.Single(model.Roots); + Assert.Equal(new[] { 2, 3 }, root.Children.Select(c => c.Spid).OrderBy(x => x).ToArray()); + + var n2 = root.Children.Single(c => c.Spid == 2); + var n3 = root.Children.Single(c => c.Spid == 3); + + var n4 = Assert.Single(n2.Children); + Assert.Equal(4, n4.Spid); + Assert.Empty(n3.Children); + Assert.Empty(n4.Children); + + Assert.Equal(4, Flatten(root).Count()); // 1, 2, 3, 4 — each once + } + + [Fact] + public void SameSpidDifferentEcid_AreKeptAsSeparateNodes() + { + // Apex 200 blocks SPID 201 on two DIFFERENT execution contexts (ecid 0 and 1 — parallel workers). + // The builder keys on spid:ecid, so it must NOT collapse them into one node. + var model = Build(Chain(200, new[] + { + Edge(1, 200, 201, blockedEcid: 0), + Edge(1, 200, 201, blockedEcid: 1) + })); + + var root = Assert.Single(model.Roots); + Assert.Equal(2, root.Children.Count); + Assert.All(root.Children, c => Assert.Equal(201, c.Spid)); + Assert.Equal(new[] { 0, 1 }, root.Children.Select(c => c.Ecid).OrderBy(e => e)); + } + + [Fact] + public void Diamond_AttachesVictimToLowestParentSpid_DroppingExtraInEdge() + { + // 400 -> 401, 400 -> 402, and BOTH 401 and 402 block 403 at the same level. Tie on level breaks + // to the lowest parent SPID (401); the 402 -> 403 in-edge is dropped, and 403 appears once. + var model = Build(Chain(400, new[] + { + Edge(1, 400, 401), Edge(1, 400, 402), + Edge(2, 401, 403), Edge(2, 402, 403) + })); + + var root = Assert.Single(model.Roots); + var n401 = root.Children.Single(c => c.Spid == 401); + var n402 = root.Children.Single(c => c.Spid == 402); + + Assert.Single(n401.Children); + Assert.Equal(403, n401.Children[0].Spid); + Assert.Empty(n402.Children); + Assert.Equal(4, Flatten(root).Count()); // 403 is not duplicated + Assert.Single(Flatten(root), n => n.Spid == 403); + } + + [Fact] + public void Flags_ArePropagatedToModel() + { + var model = BlockingChainTreeBuilder.Build( + new[] { Chain(200, new[] { Edge(1, 200, 201) }) }, + cycleDetected: true, depthCapped: true, traversalTruncated: true); + + Assert.True(model.CycleDetected); + Assert.True(model.DepthCapped); + Assert.True(model.TraversalTruncated); + } + + [Fact] + public void SleepingApex_IsFlaggedOnRoot() + { + var model = Build(Chain(200, new[] { Edge(1, 200, 201) }, sleeping: true)); + var root = Assert.Single(model.Roots); + Assert.True(root.IsApex); + Assert.True(root.IsApexSleeping); + } + + [Fact] + public void Roots_AreRankedByMagnitudeDescending() + { + var weak = Chain(200, new[] { Edge(1, 200, 201) }, magnitude: 0.2); + var strong = Chain(300, new[] { Edge(1, 300, 301) }, magnitude: 0.9); + + var model = BlockingChainTreeBuilder.Build(new[] { weak, strong }, false, false, false); + + Assert.Equal(2, model.Roots.Count); + Assert.Equal(300, model.Roots[0].Spid); // higher magnitude first + Assert.Equal(0.9, model.Roots[0].Magnitude); + Assert.Equal(200, model.Roots[1].Spid); + } + + [Fact] + public void Cycle_DoesNotInfiniteLoop_AndPlacesEachNodeOnce() + { + // 500 -> 501 -> 502, plus a back-edge 502 -> 500. The visited set must stop the cycle and the + // apex must never be re-parented; each session appears once. + var model = BlockingChainTreeBuilder.Build( + new[] + { + Chain(500, new[] + { + Edge(1, 500, 501), Edge(2, 501, 502), Edge(3, 502, 500) + }) + }, + cycleDetected: true, depthCapped: false, traversalTruncated: false); + + var root = Assert.Single(model.Roots); + Assert.Equal(500, root.Spid); + Assert.Equal(3, Flatten(root).Count()); // 500, 501, 502 — no duplicate 500 + Assert.Single(Flatten(root), n => n.Spid == 500); + } +} diff --git a/Dashboard.Tests/BlockingPairRowMergeTests.cs b/Dashboard.Tests/BlockingPairRowMergeTests.cs new file mode 100644 index 000000000..b13882cdc --- /dev/null +++ b/Dashboard.Tests/BlockingPairRowMergeTests.cs @@ -0,0 +1,84 @@ +using System; +using System.Collections.Generic; +using PerformanceMonitor.Analysis; +using Xunit; + +namespace PerformanceMonitorDashboard.Tests; + +/// +/// Unit tests for BlockingPairRowMerge — merging always-on DMV-snapshot pair-rows into blocked-process-report +/// rows: BPR preferred, DMV-only filled in (the AWS RDS / threshold-unset case), overlapping edges deduped. +/// No database. +/// +public class BlockingPairRowMergeTests +{ + private static readonly DateTime BaseTime = new(2026, 5, 22, 10, 0, 0); + + private static BlockingPairRow Pair(int blockedSpid, int blockingSpid, string source, + DateTime? eventTime = null, int blockedEcid = 0, int blockingEcid = 0) => + new() + { + EventTime = eventTime ?? BaseTime, + DatabaseName = "TestDb", + BlockedSpid = blockedSpid, + BlockingSpid = blockingSpid, + BlockedEcid = blockedEcid, + BlockingEcid = blockingEcid, + LockMode = "X", + BlockingStatus = "running", + BlockedSqlText = source, // tag the origin so assertions can tell which row survived + BlockingSqlText = source + }; + + [Fact] + public void EmptySnapshots_LeavesPrimaryUnchanged() + { + var primary = new List { Pair(201, 200, "BPR") }; + BlockingPairRowMerge.MergeInto(primary, Array.Empty()); + Assert.Equal("BPR", Assert.Single(primary).BlockedSqlText); + } + + [Fact] + public void EmptyPrimary_AddsAllSnapshots() + { + // The RDS / threshold-unset case: no blocked-process reports, the DMV snapshot is the only source. + var primary = new List(); + BlockingPairRowMerge.MergeInto(primary, new[] { Pair(201, 200, "DMV"), Pair(203, 202, "DMV") }); + Assert.Equal(2, primary.Count); + Assert.All(primary, r => Assert.Equal("DMV", r.BlockedSqlText)); + } + + [Fact] + public void OverlappingEdge_PrefersBprAndDropsDmv() + { + // Same edge (blocked/blocker spid:ecid, same minute) from both sources -> only the BPR row survives. + var primary = new List { Pair(201, 200, "BPR") }; + BlockingPairRowMerge.MergeInto(primary, new[] { Pair(201, 200, "DMV") }); + Assert.Equal("BPR", Assert.Single(primary).BlockedSqlText); + } + + [Fact] + public void DistinctEdge_AddsDmvRow() + { + var primary = new List { Pair(201, 200, "BPR") }; + BlockingPairRowMerge.MergeInto(primary, new[] { Pair(301, 300, "DMV") }); + Assert.Equal(2, primary.Count); + } + + [Fact] + public void SameSpidsDifferentMinute_AreNotDeduped() + { + var primary = new List { Pair(201, 200, "BPR", BaseTime) }; + BlockingPairRowMerge.MergeInto(primary, new[] { Pair(201, 200, "DMV", BaseTime.AddMinutes(5)) }); + Assert.Equal(2, primary.Count); + } + + [Fact] + public void SameSpidDifferentEcid_AreNotDeduped() + { + // Parallel workers (distinct ecid) are distinct edges, not duplicates. + var primary = new List { Pair(201, 200, "BPR", blockingEcid: 0) }; + BlockingPairRowMerge.MergeInto(primary, new[] { Pair(201, 200, "DMV", blockingEcid: 1) }); + Assert.Equal(2, primary.Count); + } +} diff --git a/Dashboard.Tests/ChartClickIsolateTests.cs b/Dashboard.Tests/ChartClickIsolateTests.cs new file mode 100644 index 000000000..1ffbdc14f --- /dev/null +++ b/Dashboard.Tests/ChartClickIsolateTests.cs @@ -0,0 +1,131 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using PerformanceMonitor.Ui; +using Xunit; + +namespace PerformanceMonitorDashboard.Tests; + +/// +/// Tests the pure, app-agnostic logic behind the shared chart click-to-isolate mechanic +/// (): toggle transitions, the dim-vs-full visual decision, and faithful +/// per-series restore for fill vs line-only charts. Isolate dims the other series and leaves the Y axis +/// untouched (no auto-fit). Mirror of Lite.Tests for parity. +/// +public class ChartClickIsolateTests +{ + // ── NextIsolate: toggle transitions ────────────────────────────────────────────────────── + + [Fact] + public void NextIsolate_FromNothing_IsolatesClicked() + { + Assert.Equal("CXPACKET", ChartHoverHelper.NextIsolate(null, "CXPACKET")); + } + + [Fact] + public void NextIsolate_ClickingIsolatedSeries_TogglesOff() + { + Assert.Null(ChartHoverHelper.NextIsolate("CXPACKET", "CXPACKET")); + } + + [Fact] + public void NextIsolate_ClickingDifferentSeries_SwitchesTarget() + { + Assert.Equal("WRITELOG", ChartHoverHelper.NextIsolate("CXPACKET", "WRITELOG")); + } + + [Fact] + public void NextIsolate_IsCaseSensitive_DifferentCaseIsADifferentSeries() + { + // Labels are exact series identifiers; a case difference is a different series, not a toggle-off. + Assert.Equal("cxpacket", ChartHoverHelper.NextIsolate("CXPACKET", "cxpacket")); + } + + // ── ResolveSeriesVisual: dim vs full decision ──────────────────────────────────────────── + + [Fact] + public void ResolveSeriesVisual_NothingIsolated_EverySeriesIsFull() + { + var v = ChartHoverHelper.ResolveSeriesVisual(null, "AnySeries"); + Assert.False(v.Dim); + Assert.True(v.FillRibbon); + } + + [Fact] + public void ResolveSeriesVisual_TargetSeries_IsFull() + { + var v = ChartHoverHelper.ResolveSeriesVisual("WRITELOG", "WRITELOG"); + Assert.False(v.Dim); + Assert.True(v.FillRibbon); + } + + [Fact] + public void ResolveSeriesVisual_NonTargetSeries_IsDimmedWithNoFill() + { + var v = ChartHoverHelper.ResolveSeriesVisual("WRITELOG", "CXPACKET"); + Assert.True(v.Dim); + Assert.False(v.FillRibbon); // the gradient ribbon is dropped while dimmed + Assert.Equal(ChartHoverHelper.DimAlpha, v.LineAlpha); + } + + [Fact] + public void DimAlpha_IsFaintButVisible() + { + Assert.Equal((byte)40, ChartHoverHelper.DimAlpha); + Assert.Equal((byte)40, ChartHoverHelper.IsolateVisual.Dimmed.LineAlpha); + Assert.True(ChartHoverHelper.IsolateVisual.Full.FillRibbon); + Assert.False(ChartHoverHelper.IsolateVisual.Full.Dim); + } + + // ── RestoreSeriesVisual: faithful restore for line-only AND fill charts (regression) ───────── + + [Fact] + public void RestoreSeriesVisual_LineOnlyChart_StaysLineOnly_NoPhantomMarkersOrFill() + { + // CollectorDuration / trend charts build line-only (MarkerSize 0, no fill) and never call + // StyleScatter. Restore must NOT re-run StyleScatter — that would add density markers + a fill. + var plot = new ScottPlot.Plot(); + var sc = plot.Add.Scatter(new double[] { 1, 2, 3 }, new double[] { 1, 2, 3 }); + var identity = ScottPlot.Color.FromHex("#4E79A7"); + sc.Color = identity; + sc.LineWidth = 1.5f; + sc.MarkerSize = 0; + sc.FillY = false; + var entry = new ChartHoverHelper.SeriesEntry(sc, "Collector", identity, + sc.LineColor, sc.LineWidth, sc.MarkerSize, sc.FillY); + + sc.Color = identity.WithAlpha(ChartHoverHelper.DimAlpha); // simulate a dim + ChartHoverHelper.RestoreSeriesVisual(entry); + + Assert.Equal(0f, sc.MarkerSize); // no phantom density markers + Assert.False(sc.FillY); // no phantom fill ribbon + Assert.Equal(1.5f, sc.LineWidth); // original width preserved + } + + [Fact] + public void RestoreSeriesVisual_FillChart_RebuildsTheStyleScatterLook() + { + // A StyleScatter'd fill chart restores via StyleScatter, which rebuilds the gradient from the + // unchanged data — reproducing the original look (isolate never spans a re-render). + var plot = new ScottPlot.Plot(); + var sc = plot.Add.Scatter(new double[] { 1, 2, 3 }, new double[] { 0, 5, 10 }); + var identity = ScottPlot.Color.FromHex("#4E79A7"); + sc.Color = identity; + ChartStyle.StyleScatter(sc); + var entry = new ChartHoverHelper.SeriesEntry(sc, "Wait", identity, + sc.LineColor, sc.LineWidth, sc.MarkerSize, sc.FillY); + Assert.True(entry.OrigFillY); // StyleScatter set FillY=true (data has a range) + + sc.Color = identity.WithAlpha(ChartHoverHelper.DimAlpha); + sc.FillY = false; // simulate a dim + ChartHoverHelper.RestoreSeriesVisual(entry); + + Assert.True(sc.FillY); // fill ribbon rebuilt + Assert.Equal(2f, sc.LineWidth); // StyleScatter's signature line width + } +} diff --git a/Dashboard.Tests/CoFiredSummaryTests.cs b/Dashboard.Tests/CoFiredSummaryTests.cs new file mode 100644 index 000000000..7f49e3713 --- /dev/null +++ b/Dashboard.Tests/CoFiredSummaryTests.cs @@ -0,0 +1,74 @@ +using System.Collections.Generic; +using PerformanceMonitor.Analysis; +using PerformanceMonitorDashboard.Services.Recommendations; +using Xunit; + +namespace PerformanceMonitorDashboard.Tests; + +/// +/// Tests the shared "what else fired in this window" cross-reference (correlate-and-focus slice 1, +/// review §1d): the pure helper and the Dashboard reader's per-card +/// append. The Lite reader's append is covered in LiteRecommendationsReaderTests. +/// +public class CoFiredSummaryTests +{ + [Fact] + public void OtherTitles_ExcludesSelf_OrdersBySeverityDesc_Dedups() + { + var window = new List<(string, double)> { ("A", 0.5), ("B", 1.9), ("C", 0.8), ("B", 1.0) }; + var others = CoFiredSummary.OtherTitles("A", window); + Assert.Equal(new[] { "B", "C" }, others); // self A excluded; B(1.9) before C(0.8); duplicate B collapsed + } + + [Fact] + public void Line_CapsAndCountsTheRest() + { + var others = new List { "A", "B", "C", "D", "E" }; + Assert.Equal( + "Also surfaced in this analysis window: A; B; C (+2 more).", + CoFiredSummary.Line(others, cap: 3)); + } + + [Fact] + public void Line_NoExtra_OmitsTheCount() + { + Assert.Equal( + "Also surfaced in this analysis window: A; B.", + CoFiredSummary.Line(new List { "A", "B" }, cap: 3)); + } + + [Fact] + public void Line_NullForEmpty() + { + Assert.Null(CoFiredSummary.Line(new List())); + } + + [Fact] + public void DashboardReader_AppendCoFired_AppendsTheOthersToEachCard() + { + var items = new List + { + new() { Title = "High problem", RawSeverity = 1.9, AdviceText = "fix high." }, + new() { Title = "Low problem", RawSeverity = 0.5, AdviceText = "fix low." }, + }; + + RecommendationsReader.AppendCoFired(items); + + Assert.Contains("fix high.", items[0].AdviceText); + Assert.Contains("Also surfaced in this analysis window: Low problem.", items[0].AdviceText); + Assert.Contains("Also surfaced in this analysis window: High problem.", items[1].AdviceText); + } + + [Fact] + public void DashboardReader_AppendCoFired_SingleCard_NoOp() + { + var items = new List + { + new() { Title = "Only problem", RawSeverity = 1.0, AdviceText = "fix it." }, + }; + + RecommendationsReader.AppendCoFired(items); + + Assert.Equal("fix it.", items[0].AdviceText); // nothing else fired -> unchanged + } +} diff --git a/Dashboard.Tests/CollectionStoppedDetectionTests.cs b/Dashboard.Tests/CollectionStoppedDetectionTests.cs new file mode 100644 index 000000000..22ee94bea --- /dev/null +++ b/Dashboard.Tests/CollectionStoppedDetectionTests.cs @@ -0,0 +1,106 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using PerformanceMonitorDashboard.Services; +using Xunit; + +namespace PerformanceMonitorDashboard.Tests; + +/// +/// Verifies the "collection stopped" decision logic (): +/// disabled collector jobs are the immediate, specific signal; a collection-freshness gap is the catch-all +/// for the Agent service being stopped or collectors silently erroring. This is the one health signal the +/// app must compute itself, since the collector that fills every other table is exactly what may be off. +/// +public class CollectionStoppedDetectionTests +{ + private const int Threshold = 30; + + [Fact] + public void AllJobsDisabled_IsStopped_WithAllReason() + { + var r = DatabaseService.DecideCollectionStopped(disabledJobs: 6, totalJobs: 6, minutesSince: 2, thresholdMinutes: Threshold); + + Assert.True(r.Stopped); + Assert.Contains("All 6", r.Reason); + Assert.Equal(6, r.DisabledJobs); + } + + [Fact] + public void SomeJobsDisabled_IsStopped_WithPartialReason() + { + var r = DatabaseService.DecideCollectionStopped(disabledJobs: 2, totalJobs: 6, minutesSince: 1, thresholdMinutes: Threshold); + + Assert.True(r.Stopped); + Assert.Contains("2 of 6", r.Reason); + } + + [Fact] + public void DisabledJobs_WinOverFreshness_EvenWhenDataIsFresh() + { + // Jobs just disabled — collection_log is still fresh, but we must flag it immediately, not wait 30m. + var r = DatabaseService.DecideCollectionStopped(disabledJobs: 1, totalJobs: 6, minutesSince: 0, thresholdMinutes: Threshold); + + Assert.True(r.Stopped); + Assert.Contains("disabled", r.Reason); + } + + [Fact] + public void NoDisabledJobs_FreshData_IsNotStopped() + { + var r = DatabaseService.DecideCollectionStopped(disabledJobs: 0, totalJobs: 6, minutesSince: 3, thresholdMinutes: Threshold); + + Assert.False(r.Stopped); + Assert.Null(r.Reason); + } + + [Fact] + public void NoDisabledJobs_StaleData_IsStopped_WithFreshnessReason() + { + // Jobs enabled but nothing has run for > threshold — Agent stopped or collectors erroring. + var r = DatabaseService.DecideCollectionStopped(disabledJobs: 0, totalJobs: 6, minutesSince: 45, thresholdMinutes: Threshold); + + Assert.True(r.Stopped); + Assert.Contains("45 minutes", r.Reason); + } + + [Fact] + public void FreshnessAtThreshold_IsStopped() + { + var r = DatabaseService.DecideCollectionStopped(disabledJobs: 0, totalJobs: 6, minutesSince: Threshold, thresholdMinutes: Threshold); + + Assert.True(r.Stopped); + } + + [Fact] + public void JobStateUnreadable_FreshData_IsNotStopped() + { + // msdb unreadable (Azure / RDS / no SQLAgentReaderRole): totalJobs = 0 must NOT read as "all disabled". + var r = DatabaseService.DecideCollectionStopped(disabledJobs: 0, totalJobs: 0, minutesSince: 4, thresholdMinutes: Threshold); + + Assert.False(r.Stopped); + } + + [Fact] + public void JobStateUnreadable_StaleData_StillCaughtByFreshness() + { + var r = DatabaseService.DecideCollectionStopped(disabledJobs: 0, totalJobs: 0, minutesSince: 60, thresholdMinutes: Threshold); + + Assert.True(r.Stopped); + Assert.Contains("60 minutes", r.Reason); + } + + [Fact] + public void NeverCollected_NoDisabledJobs_IsNotStopped() + { + // minutesSince null = collection_log empty (fresh install / never ran), not "stopped". + var r = DatabaseService.DecideCollectionStopped(disabledJobs: 0, totalJobs: 6, minutesSince: null, thresholdMinutes: Threshold); + + Assert.False(r.Stopped); + } +} diff --git a/Dashboard.Tests/Dashboard.Tests.csproj b/Dashboard.Tests/Dashboard.Tests.csproj index 97e989265..f77205704 100644 --- a/Dashboard.Tests/Dashboard.Tests.csproj +++ b/Dashboard.Tests/Dashboard.Tests.csproj @@ -9,7 +9,7 @@ - + all runtime; build; native; contentfiles; analyzers; buildtransitive @@ -20,4 +20,9 @@ + + + + + diff --git a/Dashboard.Tests/DeadlockGraphLayoutTests.cs b/Dashboard.Tests/DeadlockGraphLayoutTests.cs new file mode 100644 index 000000000..d052bd846 --- /dev/null +++ b/Dashboard.Tests/DeadlockGraphLayoutTests.cs @@ -0,0 +1,171 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System.Collections.Generic; +using System.Linq; +using PerformanceMonitor.Common; +using Xunit; + +namespace PerformanceMonitorDashboard.Tests; + +/// +/// Pure unit tests for the shared — determinism, no-overlap, and full +/// coverage of the cycle/ring placement at the real 5- and 8-process sizes. No WPF (cards are fixed-size). +/// +public class DeadlockGraphLayoutTests +{ + private static DeadlockGraphModel Parse(string fixture) => + DeadlockGraphParser.Parse(DeadlockGraphParserTests.LoadFixture(fixture)); + + private static void AssertNoOverlap(DeadlockGraphModel m) + { + const double w = DeadlockGraphLayout.NodeWidth; + const double h = DeadlockGraphLayout.NodeHeight; + const double eps = 0.5; + + var nodes = m.Processes; + for (int i = 0; i < nodes.Count; i++) + { + for (int j = i + 1; j < nodes.Count; j++) + { + var a = nodes[i]; + var b = nodes[j]; + bool overlapX = a.X < b.X + w - eps && b.X < a.X + w - eps; + bool overlapY = a.Y < b.Y + h - eps && b.Y < a.Y + h - eps; + Assert.False(overlapX && overlapY, + $"SPID {a.Spid} ({a.X},{a.Y}) overlaps SPID {b.Spid} ({b.X},{b.Y})"); + } + } + } + + [Theory] + [InlineData("deadlock_2proc_real_sql2025.xml")] + [InlineData("deadlock_5proc_real_sql2025.xml")] + [InlineData("deadlock_8proc_multivictim_real_sql2025.xml")] + [InlineData("deadlock_single5cycle_synthetic.xml")] + [InlineData("deadlock_parallel_selfedge_synthetic.xml")] + [InlineData("deadlock_crossproduct_synthetic.xml")] + public void Layout_ProducesNoOverlappingNodes(string fixture) + { + var m = Parse(fixture); + DeadlockGraphLayout.Layout(m); + AssertNoOverlap(m); + } + + [Theory] + [InlineData("deadlock_5proc_real_sql2025.xml")] + [InlineData("deadlock_8proc_multivictim_real_sql2025.xml")] + public void Layout_IsDeterministic(string fixture) + { + var a = Parse(fixture); + var b = Parse(fixture); + + DeadlockGraphLayout.Layout(a); + DeadlockGraphLayout.Layout(b); + + var pa = a.Processes.OrderBy(p => p.Id, System.StringComparer.Ordinal).ToList(); + var pb = b.Processes.OrderBy(p => p.Id, System.StringComparer.Ordinal).ToList(); + Assert.Equal(pa.Count, pb.Count); + for (int i = 0; i < pa.Count; i++) + { + Assert.Equal(pa[i].Id, pb[i].Id); + Assert.Equal(pa[i].X, pb[i].X); + Assert.Equal(pa[i].Y, pb[i].Y); + } + } + + [Theory] + [InlineData("deadlock_2proc_real_sql2025.xml")] + [InlineData("deadlock_5proc_real_sql2025.xml")] + [InlineData("deadlock_8proc_multivictim_real_sql2025.xml")] + public void Layout_ReturnsExtentCoveringAllNodes(string fixture) + { + var m = Parse(fixture); + var (width, height) = DeadlockGraphLayout.Layout(m); + + Assert.True(width > 0); + Assert.True(height > 0); + Assert.True(width >= m.Processes.Max(p => p.X) + DeadlockGraphLayout.NodeWidth); + Assert.True(height >= m.Processes.Max(p => p.Y) + DeadlockGraphLayout.NodeHeight); + } + + [Fact] + public void Layout_AssignsDistinctPositionsToEveryNode() + { + var m = Parse("deadlock_8proc_multivictim_real_sql2025.xml"); + DeadlockGraphLayout.Layout(m); + + var positions = m.Processes.Select(p => (p.X, p.Y)).ToHashSet(); + Assert.Equal(m.Processes.Count, positions.Count); // no two cards share a position + } + + [Fact] + public void Layout_EmptyModel_ReturnsPositiveExtent() + { + var (width, height) = DeadlockGraphLayout.Layout(new DeadlockGraphModel()); + Assert.True(width > 0); + Assert.True(height > 0); + } + + private static bool Within(DeadlockProcessNode n, DeadlockComponentBox b, double eps = 0.5) => + n.X >= b.X - eps && n.X + DeadlockGraphLayout.NodeWidth <= b.X + b.Width + eps && + n.Y >= b.Y - eps && n.Y + DeadlockGraphLayout.NodeHeight <= b.Y + b.Height + eps; + + [Theory] + [InlineData("deadlock_2proc_real_sql2025.xml", 1)] + [InlineData("deadlock_5proc_real_sql2025.xml", 2)] // 3-cycle + 2-cycle + [InlineData("deadlock_8proc_multivictim_real_sql2025.xml", 4)] // four 2-cycles + [InlineData("deadlock_single5cycle_synthetic.xml", 1)] + [InlineData("deadlock_crossproduct_synthetic.xml", 1)] + [InlineData("deadlock_parallel_selfedge_synthetic.xml", 1)] // a<->b via exchange edges = one component + public void Layout_PopulatesOneBoxPerComponent(string fixture, int expectedComponents) + { + var m = Parse(fixture); + DeadlockGraphLayout.Layout(m); + + Assert.Equal(expectedComponents, m.ComponentBoxes.Count); + Assert.Equal(m.Processes.Count, m.ComponentBoxes.Sum(b => b.NodeCount)); // every process is in one box + Assert.Equal(Enumerable.Range(1, expectedComponents), m.ComponentBoxes.Select(b => b.Index)); + + // Every card lands inside exactly one component box (so a drawn frame contains exactly its cycle). + foreach (var n in m.Processes) + Assert.Equal(1, m.ComponentBoxes.Count(b => Within(n, b))); + } + + [Fact] + public void ComponentBoxes_LoneNode_IsNotCountedAsACycle() + { + // The viewer frames/counts only components with >= 2 processes. A node connected solely by a self-edge + // is its own 1-node component and must NOT read as an independent cycle. No real fixture produces one, + // so build it: a<->b is a real 2-cycle and c carries only a self-edge. + var model = new DeadlockGraphModel + { + Processes = new[] + { + new DeadlockProcessNode { Id = "a", Spid = 60 }, + new DeadlockProcessNode { Id = "b", Spid = 61 }, + new DeadlockProcessNode { Id = "c", Spid = 62 }, + }, + Edges = new[] + { + new DeadlockWaitEdge { WaiterProcessId = "a", OwnerProcessId = "b", ResourceKind = "pagelock", ResourceLabel = "PAGE db.dbo.t" }, + new DeadlockWaitEdge { WaiterProcessId = "b", OwnerProcessId = "a", ResourceKind = "pagelock", ResourceLabel = "PAGE db.dbo.t" }, + new DeadlockWaitEdge { WaiterProcessId = "c", OwnerProcessId = "c", ResourceKind = "exchangeEvent", ResourceLabel = "Parallelism" }, + } + }; + + DeadlockGraphLayout.Layout(model); + + Assert.Equal(2, model.ComponentBoxes.Count); // a<->b, and lone c + Assert.Equal(1, model.ComponentBoxes.Count(b => b.NodeCount >= 2)); // only a<->b is a cycle + + var multi = Parse("deadlock_8proc_multivictim_real_sql2025.xml"); + DeadlockGraphLayout.Layout(multi); + Assert.Equal(4, multi.ComponentBoxes.Count(b => b.NodeCount >= 2)); // four real cycles + } +} diff --git a/Dashboard.Tests/DeadlockGraphParserTests.cs b/Dashboard.Tests/DeadlockGraphParserTests.cs new file mode 100644 index 000000000..1ae1dcb45 --- /dev/null +++ b/Dashboard.Tests/DeadlockGraphParserTests.cs @@ -0,0 +1,267 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.IO; +using System.Linq; +using PerformanceMonitor.Common; +using Xunit; + +namespace PerformanceMonitorDashboard.Tests; + +/// +/// Tests for the shared , driven primarily by REAL deadlock graphs +/// captured from sql2025 (HammerDB TPC-C) plus synthetic fixtures for the cases that workload never +/// produced (parallel self-edge, Azure wrapper, nameless resource, cross-product). See +/// Fixtures/Deadlocks/README.md for provenance and the asserted cycle structure of each sample. +/// Tested once in Common (not duplicated per app). No database, no WPF. +/// +public class DeadlockGraphParserTests +{ + // ── fixtures ── + + internal static string LoadFixture(string name) => + File.ReadAllText(Path.Combine(AppContext.BaseDirectory, "Fixtures", "Deadlocks", name)); + + /// The edge set expressed as (waiter spid -> owner spid), for cycle-structure assertions. + private static HashSet<(int Waiter, int Owner)> SpidEdges(DeadlockGraphModel m) + { + var bySpid = m.Processes.ToDictionary(p => p.Id, p => p.Spid, StringComparer.Ordinal); + return m.Edges.Select(e => (bySpid[e.WaiterProcessId], bySpid[e.OwnerProcessId])).ToHashSet(); + } + + private static int[] Spids(DeadlockGraphModel m) => m.Processes.Select(p => p.Spid).OrderBy(s => s).ToArray(); + private static int[] VictimSpids(DeadlockGraphModel m) => + m.Processes.Where(p => p.IsVictim).Select(p => p.Spid).OrderBy(s => s).ToArray(); + + // ── real: 2-process ── + + [Fact] + public void Real2Proc_ParsesProcessesVictimAndTwoCycle() + { + var m = DeadlockGraphParser.Parse(LoadFixture("deadlock_2proc_real_sql2025.xml")); + + Assert.Equal(new[] { 85, 103 }, Spids(m)); + Assert.Equal(new[] { 103 }, VictimSpids(m)); + Assert.False(m.IsParallel); + + // 103 -> 85 (waits on new_order), 85 -> 103 (waits on orders): a 2-cycle. + Assert.Equal(new HashSet<(int, int)> { (103, 85), (85, 103) }, SpidEdges(m)); + + // Resource label leads with the friendly kind + the real object name. + var bySpid = m.Processes.ToDictionary(p => p.Id, p => p.Spid, StringComparer.Ordinal); + var edge103 = m.Edges.Single(e => bySpid[e.WaiterProcessId] == 103); + Assert.Equal("PAGE hammerdb_tpcc.dbo.new_order", edge103.ResourceLabel); + Assert.Equal("pagelock", edge103.ResourceKind); + Assert.Equal("X", edge103.RequestMode); + Assert.Equal("X", edge103.OwnerMode); + } + + [Fact] + public void Real2Proc_LeadsWithStatementAndContext_NotNumericDbId() + { + var m = DeadlockGraphParser.Parse(LoadFixture("deadlock_2proc_real_sql2025.xml")); + var victim = m.Processes.Single(p => p.Spid == 103); + + Assert.False(string.IsNullOrWhiteSpace(victim.SqlText)); // the differentiator + Assert.Equal("hammerdb_tpcc", victim.DatabaseName); // currentdbname, NOT the numeric "7" + Assert.Equal("hammerdb_tpcc.dbo.neword", victim.ProcName); // first real exec-stack frame + Assert.Equal("sa", victim.LoginName); + Assert.Equal("X", victim.LockMode); + Assert.True(victim.WaitTimeMs > 0); + } + + // ── real: 5-process (the common case = a bundle of cycles) ── + + [Fact] + public void Real5Proc_IsThreeCyclePlusTwoCycle_WithOneVictimEach() + { + var m = DeadlockGraphParser.Parse(LoadFixture("deadlock_5proc_real_sql2025.xml")); + + Assert.Equal(new[] { 80, 88, 97, 103, 108 }, Spids(m)); + Assert.Equal(new[] { 88, 108 }, VictimSpids(m)); // multi-victim: one per cycle + Assert.False(m.IsParallel); + + // 3-cycle {80->103->108->80} + 2-cycle {88->97->88}. + Assert.Equal(new HashSet<(int, int)> + { + (80, 103), (103, 108), (108, 80), + (88, 97), (97, 88) + }, SpidEdges(m)); + } + + // ── real: 8-process, multi-victim ── + + [Fact] + public void Real8Proc_IsFourTwoCycles_WithFourVictims() + { + var m = DeadlockGraphParser.Parse(LoadFixture("deadlock_8proc_multivictim_real_sql2025.xml")); + + Assert.Equal(new[] { 105, 107, 114, 122, 123, 125, 135, 139 }, Spids(m)); + Assert.Equal(new[] { 107, 114, 122, 135 }, VictimSpids(m)); // four victims, one per 2-cycle + Assert.False(m.IsParallel); + + Assert.Equal(new HashSet<(int, int)> + { + (105, 122), (122, 105), // cycle 1 + (107, 123), (123, 107), // cycle 2 + (114, 125), (125, 114), // cycle 3 + (135, 139), (139, 135), // cycle 4 + }, SpidEdges(m)); + Assert.Equal(8, m.Edges.Count); + } + + // ── parallel: exchangeEvent self-edge ── + + [Fact] + public void ParallelDeadlock_IsFlagged_AndKeepsSelfEdge() + { + var m = DeadlockGraphParser.Parse(LoadFixture("deadlock_parallel_selfedge_synthetic.xml")); + + Assert.True(m.IsParallel); + Assert.Contains(m.Edges, e => e.IsSelfEdge); // owner == waiter + Assert.Contains(m.Edges, e => e.ResourceKind == "exchangeEvent"); + Assert.All(m.Edges.Where(e => e.ResourceKind == "exchangeEvent"), + e => Assert.Equal("Parallelism", e.ResourceLabel)); // no objectname -> friendly fallback + + // Two ECIDs of the same spid stay distinct nodes (keyed by process id, not spid). + Assert.Equal(2, m.Processes.Count); + Assert.All(m.Processes, p => Assert.Equal(55, p.Spid)); + Assert.Equal(new[] { 1, 2 }, m.Processes.Select(p => p.Ecid).OrderBy(e => e).ToArray()); + } + + // ── cloud: Azure database_xml_deadlock_report wrapper + nameless resource ── + + [Fact] + public void AzureWrappedReport_IsParsed_WrapperIsTransparent() + { + var m = DeadlockGraphParser.Parse(LoadFixture("deadlock_azure_database_report_synthetic.xml")); + + // The wrapper is stripped transparently. + Assert.Equal(2, m.Processes.Count); + Assert.Equal(new[] { 121 }, VictimSpids(m)); + Assert.Equal("salesdb", m.Processes.First().DatabaseName); + Assert.NotNull(m.EventTime); // timestamp read off the event wrapper + } + + [Fact] + public void NamelessResource_FallsBackToKindPlusId() + { + var m = DeadlockGraphParser.Parse(LoadFixture("deadlock_azure_database_report_synthetic.xml")); + + // keylocks carry no objectname (Azure) -> "KEY hobt ". + Assert.All(m.Edges, e => Assert.StartsWith("KEY hobt ", e.ResourceLabel)); + } + + // ── cross-product: one resource, many owners/waiters ── + + [Fact] + public void CrossProductResource_EmitsOwnerByWaiterEdges() + { + var m = DeadlockGraphParser.Parse(LoadFixture("deadlock_crossproduct_synthetic.xml")); + + // One objectlock with 2 owners (201,202) and 2 waiters (203,204) => 2x2 = 4 edges. + Assert.Equal(4, m.Edges.Count); + Assert.Equal(new HashSet<(int, int)> + { + (203, 201), (203, 202), (204, 201), (204, 202) + }, SpidEdges(m)); + Assert.All(m.Edges, e => Assert.Equal("OBJECT StackOverflow.dbo.Votes", e.ResourceLabel)); + Assert.Equal(new[] { 203, 204 }, VictimSpids(m)); + } + + // ── single true N-cycle (synthetic; this workload only bundles small cycles) ── + + [Fact] + public void SingleFiveCycle_FormsOneRing() + { + var m = DeadlockGraphParser.Parse(LoadFixture("deadlock_single5cycle_synthetic.xml")); + + Assert.Equal(new[] { 301, 302, 303, 304, 305 }, Spids(m)); + Assert.Equal(new[] { 305 }, VictimSpids(m)); + Assert.Equal(new HashSet<(int, int)> + { + (301, 302), (302, 303), (303, 304), (304, 305), (305, 301) + }, SpidEdges(m)); + } + + // ── contended object (resolved straight off the deadlock resource-list) ── + + [Fact] + public void ContentiousObject_MatchesEachNodesOutgoingEdgeLabel() + { + var m = DeadlockGraphParser.Parse(LoadFixture("deadlock_2proc_real_sql2025.xml")); + + // A process's contended object is the resolved label of the resource it waits on (its outgoing edge). + foreach (var node in m.Processes) + { + var waitsOn = m.Edges.First(e => e.WaiterProcessId == node.Id && !e.IsSelfEdge); + Assert.Equal(waitsOn.ResourceLabel, node.ContentiousObject); + Assert.False(string.IsNullOrEmpty(node.ContentiousObject)); + } + + // Concretely: the victim contends for the page it was blocked on — no DMV lookup, no raw "7:1:..." id. + var bySpid = m.Processes.ToDictionary(p => p.Spid); + Assert.Equal("PAGE hammerdb_tpcc.dbo.new_order", bySpid[103].ContentiousObject); + } + + [Fact] + public void ContentiousObject_CrossProduct_IsTheSharedObject() + { + var m = DeadlockGraphParser.Parse(LoadFixture("deadlock_crossproduct_synthetic.xml")); + var bySpid = m.Processes.ToDictionary(p => p.Spid); + + // Both waiters on the single objectlock surface the same resolved object. + Assert.Equal("OBJECT StackOverflow.dbo.Votes", bySpid[203].ContentiousObject); + Assert.Equal("OBJECT StackOverflow.dbo.Votes", bySpid[204].ContentiousObject); + } + + [Fact] + public void ContentiousObject_Empty_WhenProcessWaitsOnlyOnAParallelSelfEdge() + { + var m = DeadlockGraphParser.Parse(LoadFixture("deadlock_parallel_selfedge_synthetic.xml")); + + // Parallel threads wait on the exchange itself (a self-edge) — there is no contended OBJECT to show. + Assert.All(m.Processes, p => Assert.Equal(string.Empty, p.ContentiousObject)); + } + + // ── robustness ── + + [Theory] + [InlineData(null)] + [InlineData("")] + [InlineData(" ")] + [InlineData("")] // truncated / not well-formed + [InlineData("")] // well-formed but no deadlock element + [InlineData("complete garbage not xml at all")] + public void MalformedOrEmptyInput_ReturnsEmptyModel(string? xml) + { + var m = DeadlockGraphParser.Parse(xml); + Assert.True(m.IsEmpty); + Assert.Empty(m.Processes); + Assert.Empty(m.Edges); + } + + [Fact] + public void EventWrappedOnPremXml_IsAlsoAccepted() + { + // The raw collect.deadlock_xml form is the full wrapper; the + // parser must handle it as well as the bare the apps store. Build one inline. + var bare = LoadFixture("deadlock_2proc_real_sql2025.xml"); + var wrapped = + "" + + "" + bare + + ""; + + var m = DeadlockGraphParser.Parse(wrapped); + Assert.Equal(new[] { 85, 103 }, Spids(m)); + Assert.Equal(new[] { 103 }, VictimSpids(m)); + Assert.NotNull(m.EventTime); + } +} diff --git a/Dashboard.Tests/FactAdviceComposeTests.cs b/Dashboard.Tests/FactAdviceComposeTests.cs new file mode 100644 index 000000000..c6e597480 --- /dev/null +++ b/Dashboard.Tests/FactAdviceComposeTests.cs @@ -0,0 +1,367 @@ +using System.Collections.Generic; +using PerformanceMonitor.Analysis; +using Xunit; + +namespace PerformanceMonitorDashboard.Tests; + +/// +/// Tests the value-stated advice substrate (FactAdvice.Compose / PopulateStoryText / +/// GetComposedForFinding / Serialize-round-trip). The composer reads the server's ACTUAL settings +/// from the full fact set and states them — current MAXDOP, CTFP, cores — instead of generic +/// folklore, and the result is frozen into StoryText so read-back cards show the same numbers. +/// +public class FactAdviceComposeTests +{ + private static Dictionary Facts(params Fact[] facts) + { + var d = new Dictionary(); + foreach (var f in facts) + d[f.Key] = f; + return d; + } + + private static Fact F(string key, double value) => + new() { Key = key, Source = "config", Value = value, Severity = 0.0 }; + + private static Fact Hardware(int coresPerSocket) => + new() + { + Key = "SERVER_HARDWARE", + Source = "config", + Value = 1, + Severity = 0, + Metadata = new Dictionary { ["cores_per_socket"] = coresPerSocket } + }; + + private static Fact FM(string key, double value, Dictionary meta, string? obj = null) => + new() { Key = key, Source = "x", Value = value, Severity = 1, Metadata = meta, ObjectName = obj }; + + // The audience invariant: composed advice must never contain MCP tool names or internal field names. + private static readonly string[] ToolNameMarkers = + { "get_", "audit_config", "parallel_only", "schema_table", "reconstructed_blocking_chains", "top_cpu_queries", "lock_mode_breakdown", "queries_at_spike", "best_plan_id" }; + + private static void AssertComposedAndClean(AdviceBlock? a) + { + Assert.NotNull(a); + var text = a!.Headline + " " + a.Investigation + " " + a.Remediation; + foreach (var m in ToolNameMarkers) + Assert.DoesNotContain(m, text); + } + + // Regression guard for the routing bug: range-lock findings surface under the real LCK_M_R* keys + // (kept individual by the collector), not "LCK_RANGE" — they must reach the composer (state wait + // totals, no tool names), not fall through to the tool-named static block. + [Fact] + public void RangeLock_RealKey_IsComposed_StatesWaitTotals_NoToolNames() + { + var wait = new Dictionary + { + ["wait_time_ms"] = 75000, ["waiting_tasks_count"] = 300, ["avg_ms_per_wait"] = 250, + ["signal_wait_time_ms"] = 0, ["resource_wait_time_ms"] = 75000, ["period_duration_ms"] = 3600000, + }; + var a = FactAdvice.Compose("LCK_M_RS_S", Facts(FM("LCK_M_RS_S", 0.05, wait))); + AssertComposedAndClean(a); + Assert.Contains("Key-range lock waits", a!.Investigation); // wait-totals prefix = composed, not static + Assert.Contains("SERIALIZABLE", a.Investigation); + } + + [Fact] + public void BlockingChain_StatesDepthAndVictims_NoToolNames() + { + var meta = new Dictionary + { + ["worst_chain_depth"] = 7, ["worst_chain_victim_count"] = 12, ["worst_apex_sleeping"] = 1, + ["worst_chain_max_wait_ms"] = 48000, ["total_reconstructed_chains"] = 23, + }; + var a = FactAdvice.Compose("BLOCKING_CHAIN", Facts(FM("BLOCKING_CHAIN", 7, meta))); + AssertComposedAndClean(a); + Assert.Contains("7", a!.Investigation); + Assert.Contains("12", a.Investigation); + } + + [Fact] + public void AnomalyObjectGrowth_NamesTheObject_NoToolNames() + { + var meta = new Dictionary { ["growth_mb"] = 6200, ["growth_pct"] = 44, ["current_mb"] = 20200 }; + var a = FactAdvice.Compose("ANOMALY_OBJECT_GROWTH", Facts(FM("ANOMALY_OBJECT_GROWTH", 6200, meta, obj: "dbo.Orders"))); + AssertComposedAndClean(a); + Assert.Contains("dbo.Orders", a!.Headline); + } + + [Fact] + public void Sos_StatesSignalWaitShare_NoToolNames() + { + var wait = new Dictionary + { + ["wait_time_ms"] = 600000, ["waiting_tasks_count"] = 5000, ["avg_ms_per_wait"] = 120, + ["signal_wait_time_ms"] = 180000, ["resource_wait_time_ms"] = 420000, ["period_duration_ms"] = 3600000, + }; + var a = FactAdvice.Compose("SOS_SCHEDULER_YIELD", Facts(FM("SOS_SCHEDULER_YIELD", 0.2, wait))); + AssertComposedAndClean(a); + Assert.Contains("signal wait", a!.Investigation); + } + + [Fact] + public void Cxpacket_Composed_NoToolNames() + { + var a = FactAdvice.Compose("CXPACKET", Facts(F("CONFIG_MAXDOP", 16), F("CONFIG_CTFP", 5), Hardware(8))); + AssertComposedAndClean(a); + } + + // ── THREADPOOL_PARALLEL: states the actual MAXDOP/CTFP, recommends the topology cap ── + + [Fact] + public void ThreadpoolParallel_StatesCurrentMaxdopAndCtfp_AndRecommendsTopologyCap() + { + var advice = FactAdvice.Compose( + "THREADPOOL_PARALLEL", Facts(F("CONFIG_MAXDOP", 16), F("CONFIG_CTFP", 5), Hardware(8))); + + Assert.NotNull(advice); + var r = advice!.Remediation; + Assert.Contains("MAXDOP is 16", r); + Assert.Contains("cost threshold for parallelism is 5", r); + Assert.Contains("raise cost threshold for parallelism to 50", r); + Assert.Contains("lower MAXDOP from 16 to 8", r); + // The whole point: no hedging about settings the engine collected. + Assert.DoesNotContain("if those are already set", r); + } + + [Fact] + public void ThreadpoolParallel_AlreadyWithinGuidance_GuardsHarder() + { + // MAXDOP 8 / CTFP 50 are within topology guidance yet the pool still exhausted, so the + // driver is concurrency volume — the workload-aware override guards harder. + var advice = FactAdvice.Compose( + "THREADPOOL_PARALLEL", Facts(F("CONFIG_MAXDOP", 8), F("CONFIG_CTFP", 50), Hardware(8))); + + var r = advice!.Remediation; + Assert.Contains("MAXDOP is 8", r); + Assert.Contains("already within topology guidance", r); + Assert.Contains("MAXDOP from 8 to 4", r); // harder = rec/2 + } + + [Fact] + public void ThreadpoolParallel_NoConfigFacts_FallsBackToStaticBlock() + { + var advice = FactAdvice.Compose("THREADPOOL_PARALLEL", Facts()); + Assert.Equal(FactAdvice.GetForFactKey("THREADPOOL_PARALLEL"), advice); + } + + // ── CONFIG_MAXDOP: headline states the real value (the static block hard-coded "0") ── + + [Theory] + [InlineData(0, "MAXDOP is 0")] + [InlineData(1, "MAXDOP is 1")] + [InlineData(16, "MAXDOP is 16")] + public void ConfigMaxdop_HeadlineStatesActualValue(int maxdop, string expected) + { + var advice = FactAdvice.Compose("CONFIG_MAXDOP", Facts(F("CONFIG_MAXDOP", maxdop), Hardware(8))); + Assert.Contains(expected, advice!.Headline); + } + + [Fact] + public void ConfigMaxdop_AboveGuidance_RecommendsLoweringToCores() + { + var advice = FactAdvice.Compose("CONFIG_MAXDOP", Facts(F("CONFIG_MAXDOP", 32), Hardware(8))); + Assert.Contains("Lower MAXDOP from 32 to 8", advice!.Remediation); + } + + // ── CONFIG_CTFP: states the real value ── + + [Fact] + public void ConfigCtfp_StatesActualValue() + { + var advice = FactAdvice.Compose("CONFIG_CTFP", Facts(F("CONFIG_CTFP", 5))); + Assert.Contains("Cost Threshold for Parallelism is 5", advice!.Headline); + Assert.Contains("Raise it to 50", advice.Remediation); + } + + // ── non-value keys pass through to the static block ── + + [Fact] + public void Compose_NonValueKey_ReturnsStaticBlock() + { + Assert.Equal(FactAdvice.GetForFactKey("DEADLOCKS"), FactAdvice.Compose("DEADLOCKS", Facts())); + } + + // ── serialize / read-back round-trip ── + + [Fact] + public void SerializeForStoryText_RoundTrips() + { + var composed = FactAdvice.Compose( + "THREADPOOL_PARALLEL", Facts(F("CONFIG_MAXDOP", 16), F("CONFIG_CTFP", 5), Hardware(8))); + var json = FactAdvice.SerializeForStoryText(composed); + + Assert.StartsWith("{", json); + var back = FactAdvice.TryReadStoryText(json); + Assert.NotNull(back); + Assert.Equal(composed!.Headline, back!.Headline); + Assert.Equal(composed.Investigation, back.Investigation); + Assert.Equal(composed.Remediation, back.Remediation); + } + + [Fact] + public void TryReadStoryText_LegacyOrEmpty_ReturnsNull() + { + Assert.Null(FactAdvice.TryReadStoryText("")); + Assert.Null(FactAdvice.TryReadStoryText(null)); + Assert.Null(FactAdvice.TryReadStoryText("plain legacy text")); + } + + // ── render entry point: prefer frozen StoryText, fall back to static ── + + [Fact] + public void GetComposedForFinding_PrefersFrozenStoryText() + { + var custom = new AdviceBlock("Frozen headline", "Frozen investigation", "Frozen remediation"); + var finding = new AnalysisFinding + { + RootFactKey = "THREADPOOL_PARALLEL", + StoryText = FactAdvice.SerializeForStoryText(custom) + }; + + var advice = FactAdvice.GetComposedForFinding(finding); + Assert.Equal("Frozen headline", advice!.Headline); + Assert.Equal("Frozen remediation", advice.Remediation); + } + + [Fact] + public void GetComposedForFinding_EmptyStoryText_FallsBackToStatic() + { + var finding = new AnalysisFinding { RootFactKey = "CONFIG_CTFP", StoryText = "" }; + Assert.Equal(FactAdvice.GetForFactKey("CONFIG_CTFP"), FactAdvice.GetComposedForFinding(finding)); + } + + // ── PopulateStoryText: freezes value-stated advice, skips absolution ── + + [Fact] + public void PopulateStoryText_FreezesValueStatedAdvice_OnValueBearingStory() + { + var story = new AnalysisStory { RootFactKey = "THREADPOOL_PARALLEL", IsAbsolution = false }; + var facts = new List { F("CONFIG_MAXDOP", 16), F("CONFIG_CTFP", 5), Hardware(8) }; + + FactAdvice.PopulateStoryText(new[] { story }, facts); + + Assert.StartsWith("{", story.StoryText); + var back = FactAdvice.TryReadStoryText(story.StoryText); + Assert.Contains("MAXDOP is 16", back!.Remediation); + } + + [Fact] + public void PopulateStoryText_SkipsAbsolution() + { + var story = new AnalysisStory { RootFactKey = "server_health", IsAbsolution = true, StoryText = "" }; + FactAdvice.PopulateStoryText(new[] { story }, new List()); + Assert.Equal("", story.StoryText); + } + + // ── B1: parallelism value-gap blocks state the server's actual MAXDOP/CTFP ── + + [Theory] + [InlineData("CXPACKET")] + [InlineData("QUERY_HIGH_DOP")] + [InlineData("THREADPOOL")] + public void ParallelismBlocks_StateCurrentMaxdopCtfp(string key) + { + var facts = Facts(F("CONFIG_MAXDOP", 16), F("CONFIG_CTFP", 5), Hardware(8)); + var advice = FactAdvice.Compose(key, facts); + Assert.Contains("MAXDOP is 16", advice!.Remediation); + Assert.Contains("cost threshold for parallelism is 5", advice.Remediation); + Assert.Contains("lower MAXDOP from 16 to 8", advice.Remediation); + } + + [Fact] + public void ParallelismBlocks_NoConfigFacts_FallBackToStatic() + { + Assert.Equal(FactAdvice.GetForFactKey("CXPACKET"), FactAdvice.Compose("CXPACKET", Facts())); + } + + [Fact] + public void Cxpacket_Investigation_DropsAuditConfigDeferral() + { + Assert.DoesNotContain("call `audit_config` to check CTFP and MAXDOP", + FactAdvice.GetForFactKey("CXPACKET")!.Investigation); + } + + // ── B2: memory value blocks state the actual cap / physical RAM instead of deferring ── + + [Fact] + public void ConfigMaxMemory_StatesConcreteSuggestedCap() + { + var facts = Facts(F("CONFIG_MAX_MEMORY_MB", 2147483647), F("MEMORY_TOTAL_PHYSICAL_MB", 65536)); + var advice = FactAdvice.Compose("CONFIG_MAX_MEMORY_MB", facts); + Assert.Contains("65,536 MB", advice!.Investigation); // total physical RAM stated + Assert.Contains("58,983 MB", advice.Remediation); // suggested cap = total - max(4096, 10%) + Assert.DoesNotContain("audit_config", advice.Investigation); + } + + [Fact] + public void ConfigMinMaxNarrow_StatesConfiguredMinAndMax() + { + var f = new Fact + { + Key = "CONFIG_MIN_MAX_MEMORY_NARROW", Source = "config", Value = 24000, Severity = 0.4, + Metadata = new Dictionary { ["min_memory_mb"] = 24000, ["max_memory_mb"] = 28672 } + }; + var advice = FactAdvice.Compose("CONFIG_MIN_MAX_MEMORY_NARROW", Facts(f)); + Assert.Contains("24,000 MB", advice!.Investigation); + Assert.Contains("28,672 MB", advice.Investigation); + Assert.DoesNotContain("audit_config", advice.Investigation); + } + + [Theory] + [InlineData("PAGEIOLATCH_SH")] + [InlineData("QUERY_SPILLS")] + [InlineData("ANOMALY_MEMORY_PRESSURE")] + public void MemoryWaitBlocks_StateCurrentCap_NoAuditConfigDeferral(string key) + { + var facts = Facts(F("CONFIG_MAX_MEMORY_MB", 28672), F("MEMORY_TOTAL_PHYSICAL_MB", 65536)); + var advice = FactAdvice.Compose(key, facts); + Assert.Contains("max server memory is set to 28,672 MB", advice!.Remediation); + Assert.DoesNotContain("audit_config", advice.Remediation); + } + + [Fact] + public void MemoryWaitBlocks_NoCapFact_FallBackToStatic() + { + Assert.Equal(FactAdvice.GetForFactKey("QUERY_SPILLS"), FactAdvice.Compose("QUERY_SPILLS", Facts())); + } + + // ── B3: RCSI-deferral blocks state the RCSI-off count instead of deferring to audit_config ── + + private static Fact DbConfig(int rcsiOff) => new() + { + Key = "DB_CONFIG", Source = "config", Value = 1, Severity = 0.3, + Metadata = new Dictionary { ["rcsi_off_count"] = rcsiOff } + }; + + [Theory] + [InlineData("BLOCKING_EVENTS")] + [InlineData("DEADLOCKS")] + public void RcsiBlocks_StateRcsiOffCount_WhenDbConfigCoFired(string key) + { + var advice = FactAdvice.Compose(key, Facts(DbConfig(9))); + Assert.Contains("9 databases on this server currently have RCSI off", advice!.Remediation); + } + + [Fact] + public void RcsiBlocks_SingularPhrasing_ForOneDatabase() + { + var advice = FactAdvice.Compose("BLOCKING_EVENTS", Facts(DbConfig(1))); + Assert.Contains("One database on this server currently has RCSI off", advice!.Remediation); + } + + [Fact] + public void RcsiBlocks_NoDbConfig_FallBackToStatic() + { + Assert.Equal(FactAdvice.GetForFactKey("DEADLOCKS"), FactAdvice.Compose("DEADLOCKS", Facts())); + } + + [Fact] + public void BlockingAndDeadlock_DropAuditConfigRcsiDeferral() + { + Assert.DoesNotContain("audit_config", FactAdvice.GetForFactKey("BLOCKING_EVENTS")!.Remediation); + Assert.DoesNotContain("audit_config", FactAdvice.GetForFactKey("DEADLOCKS")!.Investigation); + } +} diff --git a/Dashboard.Tests/FactAdviceCorrectnessTests.cs b/Dashboard.Tests/FactAdviceCorrectnessTests.cs new file mode 100644 index 000000000..826fb315e --- /dev/null +++ b/Dashboard.Tests/FactAdviceCorrectnessTests.cs @@ -0,0 +1,101 @@ +using PerformanceMonitor.Analysis; +using Xunit; + +namespace PerformanceMonitorDashboard.Tests; + +/// +/// Regression guards for the sourced-wrong claims and the retrospective "kill it" survivors the +/// post-merge broad review flagged in FactAdvice's static blocks. Each was either contradicted by +/// MS Learn (range locks are SERIALIZABLE-only; an UPDATE maintains only changed-column indexes; +/// last-page contention is fixed by OPTIMIZE_FOR_SEQUENTIAL_KEY, not by adding a clustered index; +/// tempdb autogrowth should be ENABLED) or gave live-firefighting advice for a tool that reports on +/// windows that have already passed. +/// +public class FactAdviceCorrectnessTests +{ + private static string Text(string key) + { + var a = FactAdvice.GetForFactKey(key); + Assert.NotNull(a); + return $"{a!.Headline}\n{a.Investigation}\n{a.Remediation}"; + } + + [Fact] + public void Sos_DoesNotClaimDemandExceedsSupply_NorCircularDeferral() + { + var t = Text("SOS_SCHEDULER_YIELD"); + Assert.DoesNotContain("demand than capacity", t); + Assert.DoesNotContain("demand exceeds supply", t); + Assert.DoesNotContain("recommendation engine will tell you", t); + Assert.Contains("quantum", t); // states the real cooperative-scheduling mechanism + } + + [Fact] + public void IoWriteLatency_DoesNotClaimEveryUpdateTouchesEveryIndex() + { + Assert.DoesNotContain("every UPDATE touches every nonclustered index", Text("IO_WRITE_LATENCY_MS")); + } + + [Fact] + public void MissingIndex_UpdateMaintenanceClaimIsPrecise() + { + Assert.DoesNotContain("every INSERT/UPDATE/DELETE pays", Text("MISSING_INDEX")); + } + + [Fact] + public void LatchEx_LastPageFix_IsOptimizeForSequentialKey_NotAddClusteredIndex() + { + var t = Text("LATCH_EX"); + Assert.DoesNotContain("add a clustered index", t); + Assert.Contains("OPTIMIZE_FOR_SEQUENTIAL_KEY", t); + } + + [Fact] + public void Tempdb_AutogrowthIsEnabled_NotDisabled_AndNoKill() + { + var t = Text("TEMPDB_USAGE"); + Assert.DoesNotContain("autogrowth disabled", t); + Assert.DoesNotContain("kill it", t); + } + + [Fact] + public void RangeLocks_AreSerializableOnly_NotRepeatableRead() + { + Assert.DoesNotContain("SERIALIZABLE or REPEATABLE READ", Text("LCK_RANGE")); + } + + [Fact] + public void AnomalyBlockingSpike_NoKillIt() + { + Assert.DoesNotContain("kill it", Text("ANOMALY_BLOCKING_SPIKE")); + } + + [Theory] + [InlineData("ANOMALY_READ_LATENCY")] + [InlineData("ANOMALY_WRITE_LATENCY")] + public void LatencyAnomalies_HeadlineNotRightNow(string key) + { + Assert.DoesNotContain("right now", FactAdvice.GetForFactKey(key)!.Headline); + } + + // Review note (§2): SOS rewrite over-swung to "never CPU pressure". The amount + a deep runnable + // queue IS demand-exceeds-capacity; the advice must point at the runnable-queue discriminator. + [Fact] + public void Sos_UsesRunnableQueueDiscriminator_NotAbsoluteNoPressure() + { + var t = Text("SOS_SCHEDULER_YIELD"); + Assert.Contains("runnable queue", t); + Assert.DoesNotContain("not as SOS_SCHEDULER_YIELD", t); // the over-correction is gone + Assert.DoesNotContain("get_cpu_scheduler_pressure", t); // MCP tool names removed from human prose + } + + // Review note (§1): MAXDOP can be set per database; the advice must acknowledge the DB-scoped + // override and point at get_database_scoped_config (the engine itself scores only the server default). + [Fact] + public void ConfigMaxdop_NotesPerDatabaseScopedOverride() + { + var advice = FactAdvice.GetForFactKey("CONFIG_MAXDOP"); + Assert.Contains("DATABASE SCOPED CONFIGURATION", advice!.Investigation); + Assert.DoesNotContain("get_database_scoped_config", advice.Investigation); // MCP tool name removed + } +} diff --git a/Dashboard.Tests/FactScorerTests.cs b/Dashboard.Tests/FactScorerTests.cs index 657a8a5f2..1afd50583 100644 --- a/Dashboard.Tests/FactScorerTests.cs +++ b/Dashboard.Tests/FactScorerTests.cs @@ -360,4 +360,22 @@ public void PlanAdvisoryKeys_HaveAdviceBlocks(string key) Assert.False(string.IsNullOrWhiteSpace(advice.Remediation)); Assert.Null(advice.RemediationTsql); } + + // C (workload-aware MAXDOP/CTFP): every THREADPOOL attribution variant InferenceEngine can root + // on (parallel/blocking/mixed) plus the generic fallback has an advice block — a relabeled root + // with no advice would render an empty card (the dead-fact bug class). + [Theory] + [InlineData("THREADPOOL")] + [InlineData("THREADPOOL_PARALLEL")] + [InlineData("THREADPOOL_BLOCKING")] + [InlineData("THREADPOOL_MIXED")] + public void ThreadpoolVariants_HaveAdviceBlocks(string key) + { + var advice = FactAdvice.GetForFactKey(key); + + Assert.NotNull(advice); + Assert.False(string.IsNullOrWhiteSpace(advice!.Headline)); + Assert.False(string.IsNullOrWhiteSpace(advice.Investigation)); + Assert.False(string.IsNullOrWhiteSpace(advice.Remediation)); + } } diff --git a/Dashboard.Tests/FailedJobWatermarkPreferenceTests.cs b/Dashboard.Tests/FailedJobWatermarkPreferenceTests.cs new file mode 100644 index 000000000..c354fe09e --- /dev/null +++ b/Dashboard.Tests/FailedJobWatermarkPreferenceTests.cs @@ -0,0 +1,77 @@ +/* + * Performance Monitor Dashboard + * Copyright (c) 2026 Darling Data, LLC + * Licensed under the MIT License - see LICENSE file for details + */ + +using System; +using System.Collections.Generic; +using System.Text.Json; +using PerformanceMonitorDashboard.Models; +using Xunit; + +namespace PerformanceMonitorDashboard.Tests; + +/// +/// Failed-job tray watermark persistence (the Dashboard half of the #1145 parity fix): a reopen +/// must not re-fire tray toasts for failures still inside the lookback window that the user already +/// saw and dismissed. The watermark is the newest already-alerted failure's server-local run time, +/// stored as in . +/// These pin the JSON persistence contract UserPreferencesService relies on (it serializes/ +/// deserializes the whole UserPreferences object with the same options). +/// +public class FailedJobWatermarkPreferenceTests +{ + /// Serializer options matching UserPreferencesService (WriteIndented = true). + private static readonly JsonSerializerOptions s_jsonOptions = new() { WriteIndented = true }; + + [Fact] + public void FailedJobAlertWatermarkTicks_DefaultsToEmpty() + { + var prefs = new UserPreferences(); + Assert.NotNull(prefs.FailedJobAlertWatermarkTicks); + Assert.Empty(prefs.FailedJobAlertWatermarkTicks); + } + + [Fact] + public void FailedJobAlertWatermarkTicks_RoundTripsExactValue_ThroughJson() + { + /* Ticks keep the server-local run time basis-exact across the JSON round-trip — the value is + compared directly against FailedJobInfo.RunDateTime, so DateTimeKind drift would corrupt it. */ + var serverLocal = new DateTime(2026, 6, 19, 14, 30, 15); + var prefs = new UserPreferences + { + FailedJobAlertWatermarkTicks = new Dictionary + { + ["server-a"] = serverLocal.Ticks, + ["server-b"] = new DateTime(2026, 6, 19, 9, 5, 0).Ticks + } + }; + + var json = JsonSerializer.Serialize(prefs, s_jsonOptions); + var loaded = JsonSerializer.Deserialize(json); + + Assert.NotNull(loaded); + Assert.Equal(2, loaded!.FailedJobAlertWatermarkTicks.Count); + Assert.Equal(serverLocal, new DateTime(loaded.FailedJobAlertWatermarkTicks["server-a"])); + Assert.Equal(new DateTime(2026, 6, 19, 9, 5, 0), new DateTime(loaded.FailedJobAlertWatermarkTicks["server-b"])); + } + + [Fact] + public void FailedJobAlertWatermarkTicks_AbsentFromExistingJson_DeserializesToEmpty() + { + // A preferences.json written before this PR has no "FailedJobAlertWatermarkTicks" key; + // the default initializer (= empty dict) must apply so upgraded installs don't NRE on seed. + const string legacyJson = """ + { + "AnalysisIntervalMinutes": 30 + } + """; + + var loaded = JsonSerializer.Deserialize(legacyJson); + + Assert.NotNull(loaded); + Assert.NotNull(loaded!.FailedJobAlertWatermarkTicks); + Assert.Empty(loaded.FailedJobAlertWatermarkTicks); + } +} diff --git a/Dashboard.Tests/FinOpsHeatmapBuilderTests.cs b/Dashboard.Tests/FinOpsHeatmapBuilderTests.cs new file mode 100644 index 000000000..380c8ca10 --- /dev/null +++ b/Dashboard.Tests/FinOpsHeatmapBuilderTests.cs @@ -0,0 +1,173 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using PerformanceMonitor.Common; +using Xunit; + +namespace PerformanceMonitorDashboard.Tests; + +/// +/// Tests the shared, app-agnostic FinOps heatmap shaping (#1138): the top-N-by-growth ranking and the +/// long→matrix pivot that both Dashboard and Lite depend on. (Phase-2 color-scale math is exercised by the +/// ColumnLogIntensities tests below.) Pure functions, no database. +/// +public class FinOpsHeatmapBuilderTests +{ + private static readonly DateTime D1 = new(2026, 6, 1); + private static readonly DateTime D2 = new(2026, 6, 2); + private static readonly DateTime D3 = new(2026, 6, 3); + + // ── RankTopGrowers ── + + [Fact] + public void RankTopGrowers_OrdersByGrowthDescending_TakesTopN() + { + var ranked = FinOpsHeatmapBuilder.RankTopGrowers( + new[] { ("a", 10.0), ("b", 50.0), ("c", 30.0), ("d", 5.0) }, topN: 2); + + Assert.Equal(new[] { "b", "c" }, ranked); + } + + [Fact] + public void RankTopGrowers_TieBreaksOnKeyOrdinal_ForDeterminism() + { + var ranked = FinOpsHeatmapBuilder.RankTopGrowers( + new[] { ("z", 10.0), ("a", 10.0), ("m", 10.0) }, topN: 2); + + Assert.Equal(new[] { "a", "m" }, ranked); + } + + [Fact] + public void RankTopGrowers_TopNExceedsCount_ReturnsAll() + { + var ranked = FinOpsHeatmapBuilder.RankTopGrowers( + new[] { ("a", 1.0), ("b", 2.0) }, topN: 10); + + Assert.Equal(new[] { "b", "a" }, ranked); + } + + // ── BuildMatrix (long → dense pivot) ── + + [Fact] + public void BuildMatrix_PivotsToRowsByDays_WithMissingCellsZero() + { + var samples = new[] + { + new FinOpsObjectDaySample("big", D1, 100), + new FinOpsObjectDaySample("big", D2, 200), + new FinOpsObjectDaySample("small", D1, 10), + // "small" has no D2 sample — that cell must stay 0. + }; + + // Rows bottom-to-top: index 0 = "small" (bottom), index 1 = "big" (top). + var matrix = FinOpsHeatmapBuilder.BuildMatrix(new[] { "small", "big" }, samples); + + Assert.Equal(new[] { "small", "big" }, matrix.RowLabels); + Assert.Equal(new[] { D1, D2 }, matrix.Days); // ascending + Assert.Equal(10, matrix.Intensities[0, 0]); // small / D1 + Assert.Equal(0, matrix.Intensities[0, 1]); // small / D2 (missing) + Assert.Equal(100, matrix.Intensities[1, 0]); // big / D1 + Assert.Equal(200, matrix.Intensities[1, 1]); // big / D2 + } + + [Fact] + public void BuildMatrix_IgnoresSamplesNotInRowKeys() + { + var samples = new[] + { + new FinOpsObjectDaySample("a", D1, 5), + new FinOpsObjectDaySample("ghost", D1, 999), + }; + + var matrix = FinOpsHeatmapBuilder.BuildMatrix(new[] { "a" }, samples); + + Assert.Single(matrix.RowLabels); + Assert.Equal(5, matrix.Intensities[0, 0]); + // "ghost" introduced no extra row; only its day survives as a column. + Assert.Equal(1, matrix.Intensities.GetLength(0)); + } + + [Fact] + public void BuildMatrix_SumsDuplicateKeyDayPairs() + { + var samples = new[] + { + new FinOpsObjectDaySample("a", D1, 5), + new FinOpsObjectDaySample("a", D1, 7), + }; + + var matrix = FinOpsHeatmapBuilder.BuildMatrix(new[] { "a" }, samples); + + Assert.Equal(12, matrix.Intensities[0, 0]); + } + + [Fact] + public void BuildMatrix_OrdersDaysAscendingRegardlessOfInputOrder() + { + var samples = new[] + { + new FinOpsObjectDaySample("a", D3, 3), + new FinOpsObjectDaySample("a", D1, 1), + new FinOpsObjectDaySample("a", D2, 2), + }; + + var matrix = FinOpsHeatmapBuilder.BuildMatrix(new[] { "a" }, samples); + + Assert.Equal(new[] { D1, D2, D3 }, matrix.Days); + Assert.Equal(1, matrix.Intensities[0, 0]); + Assert.Equal(2, matrix.Intensities[0, 1]); + Assert.Equal(3, matrix.Intensities[0, 2]); + } + + [Fact] + public void BuildMatrix_NoSamples_IsEmpty() + { + var matrix = FinOpsHeatmapBuilder.BuildMatrix(new[] { "a" }, Array.Empty()); + Assert.True(matrix.IsEmpty); + } + + // ── ColumnLogIntensities (Phase 2 color-scale binding) ── + + [Fact] + public void ColumnLogIntensities_MaxIsOne_ZeroIsZero_OthersBetween() + { + var intensities = FinOpsHeatmapBuilder.ColumnLogIntensities(new long[] { 0, 10, 100 }); + + Assert.Equal(0.0, intensities[0]); // zero -> no shade + Assert.Equal(1.0, intensities[2], 6); // column max -> full shade + Assert.InRange(intensities[1], 0.0001, 0.9999); // mid -> partial + Assert.True(intensities[1] < intensities[2]); + } + + [Fact] + public void ColumnLogIntensities_AllZero_ReturnsAllZero() + { + var intensities = FinOpsHeatmapBuilder.ColumnLogIntensities(new long[] { 0, 0, 0 }); + Assert.All(intensities, v => Assert.Equal(0.0, v)); // max=0 divide-by-zero guard (§3B) + } + + [Fact] + public void ColumnLogIntensities_NegativeClampsToZero() + { + var intensities = FinOpsHeatmapBuilder.ColumnLogIntensities(new long[] { -5, 100 }); + Assert.Equal(0.0, intensities[0]); + Assert.Equal(1.0, intensities[1], 6); + } + + [Fact] + public void ColumnLogIntensities_LogScale_CompressesLargeRange() + { + // A single hot value among many quiet ones: the quiet values still get a visible (non-tiny) + // shade because of the log scale, not a linear one. + var intensities = FinOpsHeatmapBuilder.ColumnLogIntensities(new long[] { 10, 10000 }); + Assert.True(intensities[0] > 0.2, $"log scale should keep the small value visible, got {intensities[0]}"); + Assert.Equal(1.0, intensities[1], 6); + } +} diff --git a/Dashboard.Tests/Fixtures/Deadlocks/README.md b/Dashboard.Tests/Fixtures/Deadlocks/README.md new file mode 100644 index 000000000..c21da286f --- /dev/null +++ b/Dashboard.Tests/Fixtures/Deadlocks/README.md @@ -0,0 +1,43 @@ +# Deadlock graph parser/layout test fixtures + +Inputs for `DeadlockGraphParserTests` / `DeadlockGraphLayoutTests` (both exercise the shared +`PerformanceMonitor.Common` deadlock types). Each file is a bare `` element — exactly what +both apps hand the parser at runtime (Lite `deadlock_graph_xml` and Dashboard +`collect.deadlocks.deadlock_graph` are both `evt.query('.../value/deadlock')`) — except the Azure +fixture, which keeps its `` wrapper on purpose (see below). + +## Real samples — captured from sql2025 `PerformanceMonitor.collect.deadlock_xml` (2026-06-21 HammerDB TPC-C) + +The `` symbol-address dumps were stripped (the parser never reads them); everything the +parser does read — processes, `inputbuf`, `executionStack/frame@procname`, `resource-list`, +`victim-list` — is byte-for-byte the original. + +- `deadlock_2proc_real_sql2025.xml` — id 46574. 2 processes (spid 85, 103), 1 victim (103). One + 2-cycle: 103 →(new_order) 85, 85 →(orders) 103. +- `deadlock_5proc_real_sql2025.xml` — id 46559. 5 processes (80,88,97,103,108), 2 victims (88,108). + **Two disjoint cycles**: a 3-cycle {80→103→108→80} and a 2-cycle {88→97→88}. One victim per cycle. +- `deadlock_8proc_multivictim_real_sql2025.xml` — id 75970. 8 processes + (105,107,114,122,123,125,135,139), 4 victims (107,114,122,135). **Four disjoint 2-cycles**: + {105,122}, {107,123}, {114,125}, {135,139}. One victim per cycle. + +Key real-world finding that shaped the layout: in this workload a single "N-process deadlock" is a +**bundle of independent small cycles** (mostly 2- and 3-cycles), NOT one big N-cycle. Every 5-proc +deadlock sampled was a 3-cycle + 2-cycle; the 8-proc was four 2-cycles. So the layout detects +connected components and lays each out as its own ring, tiled — a single global ring would render the +disjoint cycles as meaningless chords. + +## Synthetic samples — for cases this workload never produced + +- `deadlock_parallel_selfedge_synthetic.xml` — intra-query parallel deadlock (two ECIDs of spid 55). + `exchangeEvent` resources with no `objectname`/`mode`; includes one exchangeEvent whose owner and + waiter are the **same** process (a self-edge). Exercises `IsParallel`, the "Parallelism" label + fallback, empty modes, and the layout's self-loop tolerance. +- `deadlock_azure_database_report_synthetic.xml` — Azure SQL DB `database_xml_deadlock_report` event + wrapper (the only cloud difference is the XE event *name*; the inner `` is identical). + Proves the parser is wrapper-agnostic (anchors on `Descendants("deadlock")`). Its `keylock`s carry + **no `objectname`** → exercises the `KEY ` label fallback. +- `deadlock_crossproduct_synthetic.xml` — one `objectlock` with 2 owners (201,202) and 2 waiters + (203,204) → the parser emits the 2×2 = 4 cross-product waiter→owner edges. +- `deadlock_single5cycle_synthetic.xml` — a true single 5-process simple cycle + (301→302→303→304→305→301). Exercises the single-ring layout path that the bundled real samples + don't. diff --git a/Dashboard.Tests/Fixtures/Deadlocks/deadlock_2proc_real_sql2025.xml b/Dashboard.Tests/Fixtures/Deadlocks/deadlock_2proc_real_sql2025.xml new file mode 100644 index 000000000..169512bf8 --- /dev/null +++ b/Dashboard.Tests/Fixtures/Deadlocks/deadlock_2proc_real_sql2025.xml @@ -0,0 +1,13 @@ + +INSERT dbo.new_order(no_o_id, no_d_id, no_w_id) + VALUES (@o_id, @no_d_id, @no_w_id +EXEC neword @no_w_id = @P1, @no_max_w_id = @P2, @no_d_id = @P3, @no_c_id = @P4, @no_o_ol_cnt = @P5, @TIMESTAMP = @P +(@P1 int,@P2 int,@P3 int,@P4 int,@P5 int,@P6 datetime2)EXEC neword @no_w_id = @P1, @no_max_w_id = @P2, @no_d_id = @P3, @no_c_id = @P4, @no_o_ol_cnt = @P5, @TIMESTAMP = @P6 +UPDATE dbo.orders + SET o_carrier_id = @d_o_carrier_id + , @d_c_id = orders.o_c_id + WHERE orders.o_id = @d_no_o_id + AND orders.o_d_id = @d_d_id + AND orders.o_w_id = @d_w_i +EXEC delivery @d_w_id = @P1, @d_o_carrier_id = @P2, @timestamp = @P +(@P1 int,@P2 int,@P3 datetime2)EXEC delivery @d_w_id = @P1, @d_o_carrier_id = @P2, @timestamp = @P3 \ No newline at end of file diff --git a/Dashboard.Tests/Fixtures/Deadlocks/deadlock_5proc_real_sql2025.xml b/Dashboard.Tests/Fixtures/Deadlocks/deadlock_5proc_real_sql2025.xml new file mode 100644 index 000000000..7995c68f0 --- /dev/null +++ b/Dashboard.Tests/Fixtures/Deadlocks/deadlock_5proc_real_sql2025.xml @@ -0,0 +1,29 @@ + +INSERT dbo.new_order(no_o_id, no_d_id, no_w_id) + VALUES (@o_id, @no_d_id, @no_w_id +EXEC neword @no_w_id = @P1, @no_max_w_id = @P2, @no_d_id = @P3, @no_c_id = @P4, @no_o_ol_cnt = @P5, @TIMESTAMP = @P +(@P1 int,@P2 int,@P3 int,@P4 int,@P5 int,@P6 datetime2)EXEC neword @no_w_id = @P1, @no_max_w_id = @P2, @no_d_id = @P3, @no_c_id = @P4, @no_o_ol_cnt = @P5, @TIMESTAMP = @P6 +INSERT dbo.new_order(no_o_id, no_d_id, no_w_id) + VALUES (@o_id, @no_d_id, @no_w_id +EXEC neword @no_w_id = @P1, @no_max_w_id = @P2, @no_d_id = @P3, @no_c_id = @P4, @no_o_ol_cnt = @P5, @TIMESTAMP = @P +(@P1 int,@P2 int,@P3 int,@P4 int,@P5 int,@P6 datetime2)EXEC neword @no_w_id = @P1, @no_max_w_id = @P2, @no_d_id = @P3, @no_c_id = @P4, @no_o_ol_cnt = @P5, @TIMESTAMP = @P6 +INSERT dbo.orders( o_id, o_d_id, o_w_id, o_c_id, o_entry_d, o_ol_cnt, o_all_local) + VALUES ( @o_id, @no_d_id, @no_w_id, @no_c_id, @TIMESTAMP, @no_o_ol_cnt, @no_o_all_local +EXEC neword @no_w_id = @P1, @no_max_w_id = @P2, @no_d_id = @P3, @no_c_id = @P4, @no_o_ol_cnt = @P5, @TIMESTAMP = @P +(@P1 int,@P2 int,@P3 int,@P4 int,@P5 int,@P6 datetime2)EXEC neword @no_w_id = @P1, @no_max_w_id = @P2, @no_d_id = @P3, @no_c_id = @P4, @no_o_ol_cnt = @P5, @TIMESTAMP = @P6 +UPDATE dbo.order_line + SET ol_delivery_d = @timestamp + , @d_ol_total = @d_ol_total + ol_amount + WHERE order_line.ol_o_id = @d_no_o_id + AND order_line.ol_d_id = @d_d_id + AND order_line.ol_w_id = @d_w_i +EXEC delivery @d_w_id = @P1, @d_o_carrier_id = @P2, @timestamp = @P +(@P1 int,@P2 int,@P3 datetime2)EXEC delivery @d_w_id = @P1, @d_o_carrier_id = @P2, @timestamp = @P3 +UPDATE dbo.orders + SET o_carrier_id = @d_o_carrier_id + , @d_c_id = orders.o_c_id + WHERE orders.o_id = @d_no_o_id + AND orders.o_d_id = @d_d_id + AND orders.o_w_id = @d_w_i +EXEC delivery @d_w_id = @P1, @d_o_carrier_id = @P2, @timestamp = @P +(@P1 int,@P2 int,@P3 datetime2)EXEC delivery @d_w_id = @P1, @d_o_carrier_id = @P2, @timestamp = @P3 \ No newline at end of file diff --git a/Dashboard.Tests/Fixtures/Deadlocks/deadlock_8proc_multivictim_real_sql2025.xml b/Dashboard.Tests/Fixtures/Deadlocks/deadlock_8proc_multivictim_real_sql2025.xml new file mode 100644 index 000000000..928f312d1 --- /dev/null +++ b/Dashboard.Tests/Fixtures/Deadlocks/deadlock_8proc_multivictim_real_sql2025.xml @@ -0,0 +1,52 @@ + +SELECT order_line.ol_i_id + , order_line.ol_supply_w_id + , order_line.ol_quantity + , order_line.ol_amount + , order_line.ol_delivery_d + FROM dbo.order_line WITH (repeatableread) + WHERE order_line.ol_o_id = @os_o_id + AND order_line.ol_d_id = @os_d_id + AND order_line.ol_w_id = @os_w_i +EXEC ostat @os_w_id = @P1, @os_d_id = @P2, @os_c_id = @P3, @byname = @P4, @os_c_last = @P +(@P1 int,@P2 int,@P3 int,@P4 int,@P5 char(20))EXEC ostat @os_w_id = @P1, @os_d_id = @P2, @os_c_id = @P3, @byname = @P4, @os_c_last = @P5 +INSERT dbo.orders( o_id, o_d_id, o_w_id, o_c_id, o_entry_d, o_ol_cnt, o_all_local) + VALUES ( @o_id, @no_d_id, @no_w_id, @no_c_id, @TIMESTAMP, @no_o_ol_cnt, @no_o_all_local +EXEC neword @no_w_id = @P1, @no_max_w_id = @P2, @no_d_id = @P3, @no_c_id = @P4, @no_o_ol_cnt = @P5, @TIMESTAMP = @P +(@P1 int,@P2 int,@P3 int,@P4 int,@P5 int,@P6 datetime2)EXEC neword @no_w_id = @P1, @no_max_w_id = @P2, @no_d_id = @P3, @no_c_id = @P4, @no_o_ol_cnt = @P5, @TIMESTAMP = @P6 +INSERT dbo.new_order(no_o_id, no_d_id, no_w_id) + VALUES (@o_id, @no_d_id, @no_w_id +EXEC neword @no_w_id = @P1, @no_max_w_id = @P2, @no_d_id = @P3, @no_c_id = @P4, @no_o_ol_cnt = @P5, @TIMESTAMP = @P +(@P1 int,@P2 int,@P3 int,@P4 int,@P5 int,@P6 datetime2)EXEC neword @no_w_id = @P1, @no_max_w_id = @P2, @no_d_id = @P3, @no_c_id = @P4, @no_o_ol_cnt = @P5, @TIMESTAMP = @P6 +INSERT dbo.new_order(no_o_id, no_d_id, no_w_id) + VALUES (@o_id, @no_d_id, @no_w_id +EXEC neword @no_w_id = @P1, @no_max_w_id = @P2, @no_d_id = @P3, @no_c_id = @P4, @no_o_ol_cnt = @P5, @TIMESTAMP = @P +(@P1 int,@P2 int,@P3 int,@P4 int,@P5 int,@P6 datetime2)EXEC neword @no_w_id = @P1, @no_max_w_id = @P2, @no_d_id = @P3, @no_c_id = @P4, @no_o_ol_cnt = @P5, @TIMESTAMP = @P6 +INSERT dbo.new_order(no_o_id, no_d_id, no_w_id) + VALUES (@o_id, @no_d_id, @no_w_id +EXEC neword @no_w_id = @P1, @no_max_w_id = @P2, @no_d_id = @P3, @no_c_id = @P4, @no_o_ol_cnt = @P5, @TIMESTAMP = @P +(@P1 int,@P2 int,@P3 int,@P4 int,@P5 int,@P6 datetime2)EXEC neword @no_w_id = @P1, @no_max_w_id = @P2, @no_d_id = @P3, @no_c_id = @P4, @no_o_ol_cnt = @P5, @TIMESTAMP = @P6 +UPDATE dbo.orders + SET o_carrier_id = @d_o_carrier_id + , @d_c_id = orders.o_c_id + WHERE orders.o_id = @d_no_o_id + AND orders.o_d_id = @d_d_id + AND orders.o_w_id = @d_w_i +EXEC delivery @d_w_id = @P1, @d_o_carrier_id = @P2, @timestamp = @P +(@P1 int,@P2 int,@P3 datetime2)EXEC delivery @d_w_id = @P1, @d_o_carrier_id = @P2, @timestamp = @P3 +UPDATE dbo.orders + SET o_carrier_id = @d_o_carrier_id + , @d_c_id = orders.o_c_id + WHERE orders.o_id = @d_no_o_id + AND orders.o_d_id = @d_d_id + AND orders.o_w_id = @d_w_i +EXEC delivery @d_w_id = @P1, @d_o_carrier_id = @P2, @timestamp = @P +(@P1 int,@P2 int,@P3 datetime2)EXEC delivery @d_w_id = @P1, @d_o_carrier_id = @P2, @timestamp = @P3 +UPDATE dbo.orders + SET o_carrier_id = @d_o_carrier_id + , @d_c_id = orders.o_c_id + WHERE orders.o_id = @d_no_o_id + AND orders.o_d_id = @d_d_id + AND orders.o_w_id = @d_w_i +EXEC delivery @d_w_id = @P1, @d_o_carrier_id = @P2, @timestamp = @P +(@P1 int,@P2 int,@P3 datetime2)EXEC delivery @d_w_id = @P1, @d_o_carrier_id = @P2, @timestamp = @P3 \ No newline at end of file diff --git a/Dashboard.Tests/Fixtures/Deadlocks/deadlock_azure_database_report_synthetic.xml b/Dashboard.Tests/Fixtures/Deadlocks/deadlock_azure_database_report_synthetic.xml new file mode 100644 index 000000000..a230a78dd --- /dev/null +++ b/Dashboard.Tests/Fixtures/Deadlocks/deadlock_azure_database_report_synthetic.xml @@ -0,0 +1,5 @@ + +UPDATE dbo.Inventory SET Qty = Qty - @n WHERE ProductId = @p +EXEC dbo.PlaceOrder @ProductId = 42, @Qty = 3 +UPDATE dbo.Inventory SET Qty = Qty - @n WHERE ProductId = @p +EXEC dbo.PlaceOrder @ProductId = 99, @Qty = 1 diff --git a/Dashboard.Tests/Fixtures/Deadlocks/deadlock_crossproduct_synthetic.xml b/Dashboard.Tests/Fixtures/Deadlocks/deadlock_crossproduct_synthetic.xml new file mode 100644 index 000000000..3c70d56c6 --- /dev/null +++ b/Dashboard.Tests/Fixtures/Deadlocks/deadlock_crossproduct_synthetic.xml @@ -0,0 +1 @@ +UPDATE dbo.Votes SET BountyAmount = 50 WHERE Id = 1UPDATE dbo.Votes SET BountyAmount = 50 WHERE Id = 1UPDATE dbo.Votes SET BountyAmount = 75 WHERE Id = 2UPDATE dbo.Votes SET BountyAmount = 75 WHERE Id = 2ALTER TABLE dbo.Votes ADD Flagged bit NULLALTER TABLE dbo.Votes ADD Flagged bit NULLCREATE INDEX IX_Votes ON dbo.Votes(PostId)CREATE INDEX IX_Votes ON dbo.Votes(PostId) diff --git a/Dashboard.Tests/Fixtures/Deadlocks/deadlock_parallel_selfedge_synthetic.xml b/Dashboard.Tests/Fixtures/Deadlocks/deadlock_parallel_selfedge_synthetic.xml new file mode 100644 index 000000000..109dec3bb --- /dev/null +++ b/Dashboard.Tests/Fixtures/Deadlocks/deadlock_parallel_selfedge_synthetic.xml @@ -0,0 +1,5 @@ + +SELECT TOP 100 u.DisplayName, COUNT_BIG(*) FROM dbo.Users AS u JOIN dbo.Posts AS p ON p.OwnerUserId = u.Id GROUP BY u.DisplayName ORDER BY COUNT_BIG(*) DESC OPTION (MAXDOP 4) +SELECT TOP 100 u.DisplayName, COUNT_BIG(*) FROM dbo.Users AS u JOIN dbo.Posts AS p ON p.OwnerUserId = u.Id GROUP BY u.DisplayName ORDER BY COUNT_BIG(*) DESC OPTION (MAXDOP 4) +SELECT TOP 100 u.DisplayName, COUNT_BIG(*) FROM dbo.Users AS u JOIN dbo.Posts AS p ON p.OwnerUserId = u.Id GROUP BY u.DisplayName ORDER BY COUNT_BIG(*) DESC OPTION (MAXDOP 4) +SELECT TOP 100 u.DisplayName, COUNT_BIG(*) FROM dbo.Users AS u JOIN dbo.Posts AS p ON p.OwnerUserId = u.Id GROUP BY u.DisplayName ORDER BY COUNT_BIG(*) DESC OPTION (MAXDOP 4) diff --git a/Dashboard.Tests/Fixtures/Deadlocks/deadlock_single5cycle_synthetic.xml b/Dashboard.Tests/Fixtures/Deadlocks/deadlock_single5cycle_synthetic.xml new file mode 100644 index 000000000..ffe04d6a1 --- /dev/null +++ b/Dashboard.Tests/Fixtures/Deadlocks/deadlock_single5cycle_synthetic.xml @@ -0,0 +1 @@ +UPDATE dbo.A SET x=1UPDATE dbo.A SET x=1UPDATE dbo.B SET x=1UPDATE dbo.B SET x=1UPDATE dbo.C SET x=1UPDATE dbo.C SET x=1UPDATE dbo.D SET x=1UPDATE dbo.D SET x=1UPDATE dbo.E SET x=1UPDATE dbo.E SET x=1 diff --git a/Dashboard.Tests/IncidentIdTests.cs b/Dashboard.Tests/IncidentIdTests.cs new file mode 100644 index 000000000..9ee003599 --- /dev/null +++ b/Dashboard.Tests/IncidentIdTests.cs @@ -0,0 +1,83 @@ +using System.Collections.Generic; +using PerformanceMonitor.Analysis; +using Xunit; + +namespace PerformanceMonitorDashboard.Tests; + +/// +/// Tests the incident id (correlate-and-focus slice 2): one analysis run = one incident, so every +/// non-absolution story shares the id; the id fingerprints the primary (highest-severity) finding + +/// database so the same recurring incident is trackable across runs. +/// +public class IncidentIdTests +{ + private static AnalysisStory Story(string key, double sev, string? db = null, bool absolution = false) => + new() { RootFactKey = key, Severity = sev, DatabaseName = db, IsAbsolution = absolution }; + + [Fact] + public void StampStories_StampsAllNonAbsolution_WithTheSameId() + { + var stories = new List { Story("CPU_SQL_PERCENT", 1.6), Story("BLOCKING_EVENTS", 1.0) }; + IncidentId.StampStories("SQL1", stories); + + Assert.NotEmpty(stories[0].IncidentId); + Assert.Equal(stories[0].IncidentId, stories[1].IncidentId); // one incident per run + } + + [Fact] + public void StampStories_SamePrimaryProblem_SameIdAcrossRuns() + { + // Different co-fired members, SAME primary (CPU on Sales) -> trackable as one recurring incident. + var run1 = new List { Story("CPU_SQL_PERCENT", 1.6, "Sales"), Story("CXPACKET", 0.9) }; + var run2 = new List { Story("CPU_SQL_PERCENT", 1.7, "Sales"), Story("WRITELOG", 0.8) }; + IncidentId.StampStories("SQL1", run1); + IncidentId.StampStories("SQL1", run2); + + Assert.Equal(run1[0].IncidentId, run2[0].IncidentId); + } + + [Fact] + public void StampStories_DifferentPrimaryServerOrDb_DifferentId() + { + var a = new List { Story("CPU_SQL_PERCENT", 1.6, "Sales") }; + var b = new List { Story("CPU_SQL_PERCENT", 1.6, "Orders") }; // different database + var c = new List { Story("BLOCKING_EVENTS", 1.6, "Sales") }; // different primary key + IncidentId.StampStories("SQL1", a); + IncidentId.StampStories("SQL1", b); + IncidentId.StampStories("SQL2", c); + + Assert.NotEqual(a[0].IncidentId, b[0].IncidentId); + Assert.NotEqual(a[0].IncidentId, c[0].IncidentId); + } + + [Fact] + public void StampStories_AbsolutionOnly_LeavesIdsEmpty() + { + var stories = new List { Story("server_health", 0, absolution: true) }; + IncidentId.StampStories("SQL1", stories); + Assert.Equal(string.Empty, stories[0].IncidentId); + } + + [Fact] + public void Compute_IsDeterministic_OnSeverityTies() + { + var s1 = new List { Story("AAA", 1.0), Story("BBB", 1.0) }; + var s2 = new List { Story("BBB", 1.0), Story("AAA", 1.0) }; // same set, reversed order + Assert.Equal(IncidentId.Compute("SQL1", s1), IncidentId.Compute("SQL1", s2)); // tiebroken by root key + } + + // ── StampClusters: per-component ids (the clustering refinement) ── + + [Fact] + public void StampClusters_GivesEachComponentItsOwnId() + { + var component1 = new List { Story("CPU_SQL_PERCENT", 1.6), Story("CXPACKET", 0.9) }; + var component2 = new List { Story("DISK_SPACE", 1.5) }; + + IncidentId.StampClusters("SQL1", new[] { component1, component2 }); + + Assert.NotEmpty(component1[0].IncidentId); + Assert.Equal(component1[0].IncidentId, component1[1].IncidentId); // same incident -> same id + Assert.NotEqual(component1[0].IncidentId, component2[0].IncidentId); // different incidents -> different ids + } +} diff --git a/Dashboard.Tests/InferenceEngineTests.cs b/Dashboard.Tests/InferenceEngineTests.cs index 134bfa50d..f817de6a8 100644 --- a/Dashboard.Tests/InferenceEngineTests.cs +++ b/Dashboard.Tests/InferenceEngineTests.cs @@ -1,4 +1,5 @@ using System.Collections.Generic; +using System.Linq; using PerformanceMonitor.Analysis; using Xunit; @@ -218,4 +219,129 @@ public void BuildStories_WithDuplicateFactKeys_DoesNotThrow() Assert.Null(ex); } + + // C (workload-aware MAXDOP/CTFP): a THREADPOOL root is relabeled by its co-elevated cause so the + // persisted root key carries parallel-vs-blocking — only the parallel flavor is a MAXDOP/CTFP + // problem. CXPACKET/high-DOP co-fired -> _PARALLEL; blocking co-fired -> _BLOCKING; both -> + // _MIXED; neither -> generic THREADPOOL. Co-causes are seeded below 0.5 so only THREADPOOL roots. + [Fact] + public void Threadpool_RootIsAttributedByCoElevatedCause() + { + var engine = new InferenceEngine(new RelationshipGraph()); + + string ThreadpoolRoot(params Fact[] facts) => + engine.BuildStories(facts.ToList()) + .First(s => s.RootFactKey.StartsWith("THREADPOOL")) + .RootFactKey; + + Fact Tp() => new() { Key = "THREADPOOL", Source = "waits", Value = 0.5, Severity = 0.9 }; + Fact Cx() => new() { Key = "CXPACKET", Source = "waits", Value = 0.3, Severity = 0.3 }; + Fact Dop() => new() { Key = "QUERY_HIGH_DOP", Source = "queries", Value = 6, Severity = 0.3 }; + Fact Blk() => new() { Key = "BLOCKING_EVENTS", Source = "blocking", Value = 20, Severity = 0.3 }; + + Assert.Equal("THREADPOOL_PARALLEL", ThreadpoolRoot(Tp(), Cx())); + Assert.Equal("THREADPOOL_PARALLEL", ThreadpoolRoot(Tp(), Dop())); + Assert.Equal("THREADPOOL_BLOCKING", ThreadpoolRoot(Tp(), Blk())); + Assert.Equal("THREADPOOL_MIXED", ThreadpoolRoot(Tp(), Cx(), Blk())); + Assert.Equal("THREADPOOL", ThreadpoolRoot(Tp())); + } + + // Regression: RootFactValue/LeafFactValue must carry the fact's RAW collected VALUE, not its + // severity. They were both set to .Severity, so a CONFIG_CTFP finding (value 5, severity 0.4) + // reported RootFactValue 0.4 — which the MCP root_fact.value contract and the notification + // headline surface as "the value", contradicting the value-stated advice ("CTFP is 5"). + [Fact] + public void BuildStory_RootFactValue_IsFactValue_NotSeverity() + { + var engine = new InferenceEngine(new RelationshipGraph()); + var facts = new List + { + new() { Key = "CONFIG_CTFP", Source = "config", Value = 5, Severity = 0.4 } + }; + + var story = engine.BuildStories(facts).First(s => s.RootFactKey == "CONFIG_CTFP"); + + Assert.Equal(5, story.RootFactValue); // the collected value, NOT 0.4 + Assert.Equal(0.4, story.Severity); // severity still on its own field + } + + [Fact] + public void BuildStory_LeafFactValue_IsFactValue_NotSeverity() + { + var engine = new InferenceEngine(new RelationshipGraph()); + // CXPACKET roots (severity >= 0.5) and traverses to SOS_SCHEDULER_YIELD (edge fires when SOS + // severity is high), so the story has a leaf whose value must be SOS's value, not its severity. + var facts = new List + { + new() { Key = "CXPACKET", Source = "waits", Value = 0.6, Severity = 0.9 }, + new() { Key = "SOS_SCHEDULER_YIELD", Source = "waits", Value = 0.5, Severity = 0.67 } + }; + + var story = engine.BuildStories(facts).First(s => s.RootFactKey == "CXPACKET"); + + Assert.Equal(0.6, story.RootFactValue); // CXPACKET value, not 0.9 + Assert.Equal("SOS_SCHEDULER_YIELD", story.LeafFactKey); + Assert.Equal(0.5, story.LeafFactValue); // SOS value, not 0.67 + } + + // ── correlate-and-focus: graph-connectivity incident clustering ── + + [Fact] + public void ClusterIntoIncidents_MergesGraphConnectedStories() + { + var engine = new InferenceEngine(new RelationshipGraph()); + var stories = new List + { + new() { RootFactKey = "CPU_SQL_PERCENT", Severity = 1.6, Path = ["CPU_SQL_PERCENT"] }, + new() { RootFactKey = "PLAN_REGRESSION", Severity = 0.6, Path = ["PLAN_REGRESSION"] }, + }; + var facts = new List + { + new() { Key = "CPU_SQL_PERCENT", Source = "cpu", Value = 90, Severity = 1.6 }, + new() { Key = "PLAN_REGRESSION", Source = "queries", Value = 1, Severity = 0.6, BaseSeverity = 0.6 }, + }; + + // CPU_SQL_PERCENT -> PLAN_REGRESSION is an active edge (PLAN_REGRESSION.BaseSeverity > 0), + // so the two stories are one incident even though the greedy traversal left them separate. + Assert.Single(engine.ClusterIntoIncidents(stories, facts)); + } + + [Fact] + public void ClusterIntoIncidents_KeepsIndependentStoriesSeparate() + { + var engine = new InferenceEngine(new RelationshipGraph()); + var stories = new List + { + new() { RootFactKey = "CPU_SQL_PERCENT", Severity = 1.6, Path = ["CPU_SQL_PERCENT"] }, + new() { RootFactKey = "DISK_SPACE", Severity = 1.6, Path = ["DISK_SPACE"] }, + }; + var facts = new List + { + new() { Key = "CPU_SQL_PERCENT", Source = "cpu", Value = 90, Severity = 1.6 }, + new() { Key = "DISK_SPACE", Source = "disk", Value = 5, Severity = 1.6 }, + }; + + // No graph edge between CPU and disk space -> two separate incidents. + Assert.Equal(2, engine.ClusterIntoIncidents(stories, facts).Count); + } + + [Fact] + public void ClusterIntoIncidents_NormalizesThreadpoolRelabel_AndBridgesToBlocking() + { + var engine = new InferenceEngine(new RelationshipGraph()); + var stories = new List + { + new() { RootFactKey = "THREADPOOL_BLOCKING", Severity = 0.9, Path = ["THREADPOOL_BLOCKING"] }, + new() { RootFactKey = "LCK", Severity = 0.9, Path = ["LCK"] }, + }; + var facts = new List + { + new() { Key = "THREADPOOL", Source = "waits", Value = 0.5, Severity = 0.9 }, + new() { Key = "LCK", Source = "waits", Value = 0.5, Severity = 0.9 }, + }; + + // The relabeled THREADPOOL_BLOCKING path key normalizes to THREADPOOL, whose active + // THREADPOOL->LCK bridge (LCK severity >= 0.5) merges it with the blocking story. + Assert.Single(engine.ClusterIntoIncidents(stories, facts)); + } } diff --git a/Dashboard.Tests/JsonAlertHistoryStoreDedupKeyTests.cs b/Dashboard.Tests/JsonAlertHistoryStoreDedupKeyTests.cs new file mode 100644 index 000000000..daa2be9fe --- /dev/null +++ b/Dashboard.Tests/JsonAlertHistoryStoreDedupKeyTests.cs @@ -0,0 +1,82 @@ +using System; +using System.Collections.Generic; +using System.Threading.Tasks; +using PerformanceMonitor.Notifications; +using PerformanceMonitorDashboard.Interfaces; +using PerformanceMonitorDashboard.Models; +using PerformanceMonitorDashboard.Services; +using Xunit; + +namespace PerformanceMonitorDashboard.Tests; + +/// +/// #1154: the Dashboard JSON store's per-fingerprint cooldown seed. Proves the in-memory scan +/// restricts to rows whose ContextJson carries the #1140 dedup key, isolates per-fingerprint, and +/// — critically — does NOT NRE on the many null-ContextJson rows (tray/muted) the scan also visits +/// (the round-1 MAJOR). A unique serverId keeps the test isolated from any on-disk alert_history.json +/// the store loads in its constructor. +/// +public class JsonAlertHistoryStoreDedupKeyTests +{ + /// Minimal IUserPreferencesService over one UserPreferences instance (mirrors the other Dashboard tests). + private sealed class FakePreferencesService : IUserPreferencesService + { + public UserPreferences Preferences { get; } = new(); + public UserPreferences GetPreferences() => Preferences; + public void SavePreferences(UserPreferences preferences) { } + public void UpdateWaitStatsRange(int hoursBack, DateTime? fromDate = null, DateTime? toDate = null) { } + public void UpdateCpuRange(int hoursBack, DateTime? fromDate = null, DateTime? toDate = null) { } + public void UpdateMemoryRange(int hoursBack, DateTime? fromDate = null, DateTime? toDate = null) { } + public void UpdateFileIoRange(int hoursBack, DateTime? fromDate = null, DateTime? toDate = null) { } + public void UpdateExpensiveQueriesRange(int hoursBack, DateTime? fromDate = null, DateTime? toDate = null) { } + public void UpdateBlockingRange(int hoursBack, DateTime? fromDate = null, DateTime? toDate = null) { } + public void UpdateCollectionHealthRange(int hoursBack, DateTime? fromDate = null, DateTime? toDate = null) { } + } + + private static string JsonWith(string dedupKey) + { + var ctx = new AlertContext + { + Incidents = new List { new(dedupKey, new[] { "db.dbo.T" }) } + }; + return AlertContextSerializer.Serialize(ctx); + } + + private static Task RecordAsync(JsonAlertHistoryStore store, string serverId, string metric, string type, string? contextJson) + => store.RecordAlertAsync(new AlertHistoryRecord( + serverId, "Srv", metric, "4", "1", null, null, + AlertSent: true, NotificationType: type, SendError: null, + Muted: false, DetailText: null, ContextJson: contextJson)); + + [Fact] + public async Task GetLastSentUtc_WithDedupKey_FiltersToFingerprint_AndIgnoresNullContextRows() + { + var store = new JsonAlertHistoryStore(new FakePreferencesService()); + var srv = "test-" + Guid.NewGuid().ToString("N"); // unique -> isolated from on-disk history + + // Email rows: AAA earlier, BBB later, then a null-context "email" row (must not NRE, must be + // excluded by a dedupKey filter, but still counts for the metric-level seed). + await RecordAsync(store, srv, "Deadlocks Detected", "email", JsonWith("aaaa1111")); + await Task.Delay(10, TestContext.Current.CancellationToken); + await RecordAsync(store, srv, "Deadlocks Detected", "email", JsonWith("bbbb2222")); + await Task.Delay(10, TestContext.Current.CancellationToken); + await RecordAsync(store, srv, "Deadlocks Detected", "email", null); + + var lastAaa = await store.GetLastEmailSentUtcAsync(srv, "Deadlocks Detected", "aaaa1111"); + var lastBbb = await store.GetLastEmailSentUtcAsync(srv, "Deadlocks Detected", "bbbb2222"); + var lastCcc = await store.GetLastEmailSentUtcAsync(srv, "Deadlocks Detected", "cccc3333"); + var lastMetric = await store.GetLastEmailSentUtcAsync(srv, "Deadlocks Detected"); // metric-level (null key) + + Assert.NotNull(lastAaa); + Assert.NotNull(lastBbb); + Assert.Null(lastCcc); // no such fingerprint + Assert.True(lastAaa!.Value < lastBbb!.Value); // the filter isolates per-fingerprint + Assert.NotNull(lastMetric); + Assert.True(lastMetric!.Value >= lastBbb.Value); // metric-level still sees the later null-context row + + // Webhook channel parity. + await RecordAsync(store, srv, "Blocking Detected", "webhook", JsonWith("dddd4444")); + Assert.NotNull(await store.GetLastWebhookSentUtcAsync(srv, "Blocking Detected", "dddd4444")); + Assert.Null(await store.GetLastWebhookSentUtcAsync(srv, "Blocking Detected", "eeee5555")); + } +} diff --git a/Dashboard.Tests/JsonAlertHistoryStoreMutedFilterTests.cs b/Dashboard.Tests/JsonAlertHistoryStoreMutedFilterTests.cs new file mode 100644 index 000000000..fe5c8f285 --- /dev/null +++ b/Dashboard.Tests/JsonAlertHistoryStoreMutedFilterTests.cs @@ -0,0 +1,114 @@ +using System; +using System.Linq; +using System.Threading.Tasks; +using PerformanceMonitor.Common; +using PerformanceMonitor.Notifications; +using PerformanceMonitorDashboard.Interfaces; +using PerformanceMonitorDashboard.Models; +using PerformanceMonitorDashboard.Services; +using Xunit; + +namespace PerformanceMonitorDashboard.Tests; + +/// +/// #1225: muted alerts and resolution notices ("... Cleared/Resolved/Restored") are written to +/// history for audit but must not count toward the sidebar Alert badge. Proves GetAlertHistory's +/// includeMuted/includeResolved filters drop those rows, that the defaults still return them for +/// the history grid / MCP tool, and — the regression that bit the reporter — that both filters run +/// BEFORE the row limit, so noise can't evict a real alert from the counted window. A unique +/// serverId keeps the test isolated from any on-disk alert_history.json the store loads in its +/// constructor. +/// +public class JsonAlertHistoryStoreMutedFilterTests +{ + /// Minimal IUserPreferencesService over one UserPreferences instance (mirrors the other Dashboard tests). + private sealed class FakePreferencesService : IUserPreferencesService + { + public UserPreferences Preferences { get; } = new(); + public UserPreferences GetPreferences() => Preferences; + public void SavePreferences(UserPreferences preferences) { } + public void UpdateWaitStatsRange(int hoursBack, DateTime? fromDate = null, DateTime? toDate = null) { } + public void UpdateCpuRange(int hoursBack, DateTime? fromDate = null, DateTime? toDate = null) { } + public void UpdateMemoryRange(int hoursBack, DateTime? fromDate = null, DateTime? toDate = null) { } + public void UpdateFileIoRange(int hoursBack, DateTime? fromDate = null, DateTime? toDate = null) { } + public void UpdateExpensiveQueriesRange(int hoursBack, DateTime? fromDate = null, DateTime? toDate = null) { } + public void UpdateBlockingRange(int hoursBack, DateTime? fromDate = null, DateTime? toDate = null) { } + public void UpdateCollectionHealthRange(int hoursBack, DateTime? fromDate = null, DateTime? toDate = null) { } + } + + private static Task RecordAsync(JsonAlertHistoryStore store, string serverId, string metric, bool muted) + => store.RecordAlertAsync(new AlertHistoryRecord( + serverId, "Srv", metric, "4", "1", null, null, + AlertSent: !muted, NotificationType: muted ? "muted" : "tray", SendError: null, + Muted: muted, DetailText: null, ContextJson: null)); + + [Fact] + public async Task GetAlertHistory_ExcludeMuted_DropsMutedRows_ButDefaultKeepsThem() + { + var store = new JsonAlertHistoryStore(new FakePreferencesService()); + var srv = "test-" + Guid.NewGuid().ToString("N"); // unique -> isolated from on-disk history + + await RecordAsync(store, srv, "Long Running Query", muted: false); + await RecordAsync(store, srv, "Long Running Query", muted: true); + await RecordAsync(store, srv, "Deadlocks Detected", muted: false); + await RecordAsync(store, srv, "Long Running Query", muted: true); + + var all = store.GetAlertHistory(hoursBack: 24, limit: 100); // history grid / MCP + var actionable = store.GetAlertHistory(hoursBack: 24, limit: 100, includeMuted: false); // sidebar badge + + Assert.Equal(4, all.Count(a => a.ServerId == srv)); // default: muted rows still shown for audit + var mine = actionable.Where(a => a.ServerId == srv).ToList(); + Assert.Equal(2, mine.Count); // only the two unmuted alerts count + Assert.All(mine, a => Assert.False(a.Muted)); + } + + [Fact] + public async Task GetAlertHistory_ExcludeMuted_FiltersBeforeLimit_SoMutedNoiseCannotEvictRealAlert() + { + var store = new JsonAlertHistoryStore(new FakePreferencesService()); + var srv = "test-" + Guid.NewGuid().ToString("N"); + + // One real alert first (oldest), then a flood of newer muted noise (the reporter's scenario: + // a muted trace-reader session re-firing every cooldown). + await RecordAsync(store, srv, "Deadlocks Detected", muted: false); + await Task.Delay(10, TestContext.Current.CancellationToken); + for (int i = 0; i < 5; i++) + await RecordAsync(store, srv, "Long Running Query", muted: true); + + // Newest-first ordering + a small limit: the muted flood crowds the real alert out of the + // window when muted rows are included (this is what made the old badge under/over-count). + var withMuted = store.GetAlertHistory(hoursBack: 24, limit: 3); + Assert.DoesNotContain(withMuted, a => a.ServerId == srv && !a.Muted); + + // Excluding muted BEFORE the limit keeps the real alert inside the counted window (#1225). + var actionable = store.GetAlertHistory(hoursBack: 24, limit: 3, includeMuted: false); + var mine = actionable.Where(a => a.ServerId == srv).ToList(); + Assert.Single(mine); + Assert.False(mine[0].Muted); + } + + [Fact] + public async Task GetAlertHistory_ExcludeResolved_DropsResolutionRows_FiltersBeforeLimit() + { + var store = new JsonAlertHistoryStore(new FakePreferencesService()); + var srv = "test-" + Guid.NewGuid().ToString("N"); + + // One real alert first (oldest), then a flood of newer resolution notices. "Capture Restored" + // and "Server Restored" are included deliberately: the pre-#1225 classifier matched only + // Cleared/Resolved, so those "Restored" rows would have slipped into the badge count. + await RecordAsync(store, srv, "Deadlocks Detected", muted: false); + await Task.Delay(10, TestContext.Current.CancellationToken); + foreach (var resolved in new[] { "Blocking Cleared", "CPU Resolved", "Capture Restored", "TempDB Space Resolved", "Server Restored" }) + await RecordAsync(store, srv, resolved, muted: false); + + // Default includes resolution rows for audit (history grid / MCP); a small limit fills with them. + var all = store.GetAlertHistory(hoursBack: 24, limit: 3); + Assert.DoesNotContain(all, a => a.ServerId == srv && !AlertMetricClassifier.IsResolution(a.MetricName)); + + // Excluding resolution rows before the limit keeps the one real alert in the counted window. + var actionable = store.GetAlertHistory(hoursBack: 24, limit: 3, includeResolved: false); + var mine = actionable.Where(a => a.ServerId == srv).ToList(); + Assert.Single(mine); + Assert.Equal("Deadlocks Detected", mine[0].MetricName); + } +} diff --git a/Dashboard.Tests/RecommendationsViewModelTests.cs b/Dashboard.Tests/RecommendationsViewModelTests.cs index 09dee1c89..0911d4baf 100644 --- a/Dashboard.Tests/RecommendationsViewModelTests.cs +++ b/Dashboard.Tests/RecommendationsViewModelTests.cs @@ -50,7 +50,8 @@ private static RecommendationItem Item( DateTime? windowStartUtc = null, DateTime? windowEndUtc = null, string? storyHash = "hash", - string? storyPath = "root>leaf") + string? storyPath = "root>leaf", + string incidentId = "") { return new RecommendationItem { @@ -66,7 +67,8 @@ private static RecommendationItem Item( WindowStartUtc = windowStartUtc, WindowEndUtc = windowEndUtc, StoryPathHash = source == RecommendationSource.Engine ? storyHash : null, - StoryPath = source == RecommendationSource.Engine ? storyPath : null + StoryPath = source == RecommendationSource.Engine ? storyPath : null, + IncidentId = incidentId }; } @@ -129,33 +131,29 @@ public void InsufficientData_WithBlankMessage_FallsBackToDefault(string? engineM // ---- grouping --------------------------------------------------------- [Fact] - public void FromItems_GroupsBySeverity_IntoThreeSectionsInFixedOrder() + public void FromItems_GroupsByIncidentId() { + // Reader-sorted (severity desc); two findings share incident "inc1", a third is its own. var items = new[] { - Item(CanonicalSeverity.Info, title: "i1"), - Item(CanonicalSeverity.Critical, title: "c1"), - Item(CanonicalSeverity.Warning, title: "w1"), - Item(CanonicalSeverity.Critical, title: "c2"), + Item(CanonicalSeverity.Critical, title: "c1", rawSeverity: 1.9, incidentId: "inc1"), + Item(CanonicalSeverity.Warning, title: "w1", rawSeverity: 1.0, incidentId: "inc1"), + Item(CanonicalSeverity.Warning, title: "w2", rawSeverity: 0.9, incidentId: "inc2"), }; var vm = RecommendationsViewModel.FromItems(items); - // Fixed display order Critical -> Warning -> Info, regardless of input order. - Assert.Equal(3, vm.Sections.Count); - Assert.Equal(CanonicalSeverity.Critical, vm.Sections[0].Severity); - Assert.Equal(CanonicalSeverity.Warning, vm.Sections[1].Severity); - Assert.Equal(CanonicalSeverity.Info, vm.Sections[2].Severity); - - Assert.Equal(2, vm.Sections[0].Count); - Assert.Equal(1, vm.Sections[1].Count); - Assert.Equal(1, vm.Sections[2].Count); - Assert.Equal(4, vm.TotalCount); + Assert.Equal(2, vm.Sections.Count); // two incidents + Assert.Equal(2, vm.Sections[0].Count); // inc1: c1 + w1 + Assert.Equal(CanonicalSeverity.Critical, vm.Sections[0].Severity); // primary = c1 + Assert.Single(vm.Sections[1].Cards); // inc2: w2 only + Assert.Equal(3, vm.TotalCount); } [Fact] - public void FromItems_OmitsEmptySeveritySections() + public void FromItems_NoIncidentId_EachItemIsItsOwnGroup() { + // Legacy / pre-incident_id rows carry no id, so each is a standalone single-card group. var items = new[] { Item(CanonicalSeverity.Warning, title: "w1"), @@ -164,20 +162,19 @@ public void FromItems_OmitsEmptySeveritySections() var vm = RecommendationsViewModel.FromItems(items); - Assert.Single(vm.Sections); - Assert.Equal(CanonicalSeverity.Warning, vm.Sections[0].Severity); - Assert.Equal(2, vm.Sections[0].Count); + Assert.Equal(2, vm.Sections.Count); + Assert.All(vm.Sections, s => Assert.Single(s.Cards)); } [Fact] - public void FromItems_PreservesReaderOrderWithinASection() + public void FromItems_PreservesReaderOrderWithinAnIncident() { - // The reader returns severity-desc; within a band the order must be preserved as-is. + // The reader returns severity-desc; within an incident the order must be preserved as-is. var items = new[] { - Item(CanonicalSeverity.Critical, title: "first", rawSeverity: 1.9), - Item(CanonicalSeverity.Critical, title: "second", rawSeverity: 1.6), - Item(CanonicalSeverity.Critical, title: "third", rawSeverity: 1.51), + Item(CanonicalSeverity.Critical, title: "first", rawSeverity: 1.9, incidentId: "inc1"), + Item(CanonicalSeverity.Critical, title: "second", rawSeverity: 1.6, incidentId: "inc1"), + Item(CanonicalSeverity.Critical, title: "third", rawSeverity: 1.51, incidentId: "inc1"), }; var vm = RecommendationsViewModel.FromItems(items); @@ -204,18 +201,27 @@ public void FromItems_CriticalAndWarningExpanded_InfoCollapsed_ByDefault() } [Fact] - public void SectionHeader_IncludesCount() + public void IncidentHeader_NamesPrimaryFindingCountAndSeverity() { var items = new[] { - Item(CanonicalSeverity.Critical, title: "c1"), - Item(CanonicalSeverity.Critical, title: "c2"), - Item(CanonicalSeverity.Critical, title: "c3"), + Item(CanonicalSeverity.Critical, title: "SQL CPU pegged", rawSeverity: 1.9, incidentId: "inc1"), + Item(CanonicalSeverity.Warning, title: "Plan regression", rawSeverity: 1.0, incidentId: "inc1"), }; var vm = RecommendationsViewModel.FromItems(items); - Assert.Equal("Critical (3)", vm.Sections[0].Header); + Assert.Equal("SQL CPU pegged · 2 findings · CRITICAL", vm.Sections[0].Header); + } + + [Fact] + public void IncidentHeader_SingleFinding_OmitsCount() + { + var items = new[] { Item(CanonicalSeverity.Warning, title: "RCSI is OFF — Sales", incidentId: "inc1") }; + + var vm = RecommendationsViewModel.FromItems(items); + + Assert.Equal("RCSI is OFF — Sales · WARNING", vm.Sections[0].Header); } [Fact] diff --git a/Dashboard.Tests/RemediationTests.cs b/Dashboard.Tests/RemediationTests.cs index e80dfad17..f93adb321 100644 --- a/Dashboard.Tests/RemediationTests.cs +++ b/Dashboard.Tests/RemediationTests.cs @@ -1646,7 +1646,12 @@ public void Detector_AnomalousButTrivialCpu_NoTarget() public void Detector_CpuPercent_IsRealWindowShare_NotHardcodedZero() { var dir = FindDashboardSourceDir(); - var src = File.ReadAllText(Path.Combine(dir, "Analysis", "SqlServerDrillDownCollector.cs")); + // CollectAbnormalCpuPlans lives in a per-domain partial (SqlServerDrillDownCollector.Queries.cs) + // after the partial-class split; read every partial so the assertions hold wherever it sits. + var src = string.Concat( + Directory.GetFiles(Path.Combine(dir, "Analysis"), "SqlServerDrillDownCollector*.cs") + .OrderBy(p => p, StringComparer.Ordinal) + .Select(File.ReadAllText)); // The CollectAbnormalCpuPlans detector must NO LONGER hardcode cpu_percent = 0. Assert.DoesNotContain("cpu_percent = 0,", src); @@ -1943,7 +1948,6 @@ private sealed class FakeExecutor : IRemediationExecutor public bool AuditWriteResult = true; public Func? PreflightFunc; public Func? ForceFunc; - public Func? UnforceFunc; public int ForceCalls; public int UnforceCalls; @@ -1974,7 +1978,7 @@ public Task ForcePlanAsync(string database, long queryId, long public Task UnforcePlanAsync(string database, long queryId, long planId, RemediationIdentity identity, CancellationToken ct) { UnforceCalls++; - return Task.FromResult(UnforceFunc?.Invoke(database, queryId, planId) ?? new ForcePlanOutcome + return Task.FromResult(new ForcePlanOutcome { Database = database, QueryId = queryId, PlanId = planId, Status = RemediationStatus.Success, Forced = true, ExecutingLogin = "sa", GateSpid = 55, ExecSpid = 55 diff --git a/Dashboard.Tests/ServerConfigRecommendationTests.cs b/Dashboard.Tests/ServerConfigRecommendationTests.cs index bd0daf8c0..e4bcb22cd 100644 --- a/Dashboard.Tests/ServerConfigRecommendationTests.cs +++ b/Dashboard.Tests/ServerConfigRecommendationTests.cs @@ -18,7 +18,7 @@ namespace PerformanceMonitorDashboard.Tests; /// -/// WS3 server-level config: the FactRemediation builder (edition-aware capped MAXDOP / flat CTFP / +/// WS3 server-level config: the FactRemediation builder (topology-based MAXDOP / flat CTFP / /// advise-only memory), the persisted-action DTO round-trip, the Recommendations reader fan-out /// (one card per ServerConfigTarget; MAXDOP/CTFP carry Apply, memory cards are copy-paste only), and /// the card affordance model (Copy + Apply / Copy-only, NOT incidents). @@ -42,20 +42,18 @@ public class ServerConfigRecommendationTests }; [Theory] - // Enterprise (3) -> 8, capped at cores-per-socket when smaller. - [InlineData(3, 16, 8)] // 8 <= 16 cores -> 8 - [InlineData(3, 4, 4)] // 8 capped to 4 cores - [InlineData(3, 0, 8)] // cores unknown -> edition value stands - // Standard (2) / unknown -> 4. - [InlineData(2, 16, 4)] - [InlineData(0, 16, 4)] - [InlineData(2, 2, 2)] // 4 capped to 2 cores - // Express (4) -> 1. - [InlineData(4, 16, 1)] - public void BuildServerConfig_Maxdop_EditionAware_CappedAtCores(int edition, int cores, long expectedRecommended) + // MAXDOP is topology-based: min(cores-per-socket, 8). Edition is NOT consulted. + [InlineData(16, 8)] // > 8 cores per socket -> capped at 8 + [InlineData(8, 8)] // exactly 8 -> 8 + [InlineData(4, 4)] // <= 8 -> the core count + [InlineData(2, 2)] // small box -> 2 + [InlineData(1, 1)] // single core -> 1 + [InlineData(0, 8)] // cores unknown -> the safe general cap of 8 + public void BuildServerConfig_Maxdop_TopologyBased_CappedAt8(int cores, long expectedRecommended) { + // edition is supplied to mirror the production row shape but is no longer consulted. var finding = ServerConfigFinding("CONFIG_MAXDOP", - new { setting = "maxdop", current_value = 0, edition, cores_per_socket = cores }); + new { setting = "maxdop", current_value = 0, edition = 3, cores_per_socket = cores }); var action = FactRemediation.BuildServerConfigAction(finding); diff --git a/Dashboard/AddServerDialog.xaml b/Dashboard/AddServerDialog.xaml index 22e64dd00..3f4a8c358 100644 --- a/Dashboard/AddServerDialog.xaml +++ b/Dashboard/AddServerDialog.xaml @@ -129,6 +129,18 @@ ToolTip="What this server costs per month (license, compute, storage combined). Used for FinOps cost attribution. Leave 0 to hide cost columns."/> + + + + + + + + + + diff --git a/Dashboard/AddServerDialog.xaml.cs b/Dashboard/AddServerDialog.xaml.cs index 552a45e5c..0f17a0b4a 100644 --- a/Dashboard/AddServerDialog.xaml.cs +++ b/Dashboard/AddServerDialog.xaml.cs @@ -21,6 +21,7 @@ using PerformanceMonitorDashboard.Models; using PerformanceMonitorDashboard.Services; using PerformanceMonitor.Common; +using PerformanceMonitor.Notifications; namespace PerformanceMonitorDashboard { @@ -83,6 +84,12 @@ public AddServerDialog(ServerConnection existingServer) TrustServerCertificateCheckBox.IsChecked = existingServer.TrustServerCertificate; ReadOnlyIntentCheckBox.IsChecked = existingServer.ReadOnlyIntent; MultiSubnetFailoverCheckBox.IsChecked = existingServer.MultiSubnetFailover; + AlertDeliveryOverrideComboBox.SelectedIndex = existingServer.AlertDeliveryModeOverride switch + { + AlertNotificationMode.Summary => 1, + AlertNotificationMode.PerEvent => 2, + _ => 0 + }; if (existingServer.AuthenticationType == AuthenticationTypes.EntraMFA) { @@ -135,6 +142,13 @@ private void AuthType_Changed(object sender, RoutedEventArgs e) } } + private AlertNotificationMode? GetSelectedDeliveryOverride() => AlertDeliveryOverrideComboBox.SelectedIndex switch + { + 1 => AlertNotificationMode.Summary, + 2 => AlertNotificationMode.PerEvent, + _ => null + }; + private string GetSelectedEncryptMode() { return EncryptModeComboBox.SelectedIndex switch @@ -847,6 +861,7 @@ private void SetFormEnabled(bool enabled) MultiSubnetFailoverCheckBox.IsEnabled = enabled; IsFavoriteCheckBox.IsEnabled = enabled; MonthlyCostTextBox.IsEnabled = enabled; + AlertDeliveryOverrideComboBox.IsEnabled = enabled; DescriptionTextBox.IsEnabled = enabled; TestConnectionButton.IsEnabled = enabled; SaveButton.IsEnabled = enabled; @@ -942,6 +957,7 @@ private async void Save_Click(object sender, RoutedEventArgs e) ServerConnection.MultiSubnetFailover = MultiSubnetFailoverCheckBox.IsChecked == true; if (decimal.TryParse(MonthlyCostTextBox.Text, System.Globalization.NumberStyles.Any, System.Globalization.CultureInfo.InvariantCulture, out var editCost) && editCost >= 0) ServerConnection.MonthlyCostUsd = editCost; + ServerConnection.AlertDeliveryModeOverride = GetSelectedDeliveryOverride(); } else { @@ -962,7 +978,8 @@ private async void Save_Click(object sender, RoutedEventArgs e) TrustServerCertificate = TrustServerCertificateCheckBox.IsChecked == true, ReadOnlyIntent = ReadOnlyIntentCheckBox.IsChecked == true, MultiSubnetFailover = MultiSubnetFailoverCheckBox.IsChecked == true, - MonthlyCostUsd = monthlyCost + MonthlyCostUsd = monthlyCost, + AlertDeliveryModeOverride = GetSelectedDeliveryOverride() }; } diff --git a/Dashboard/Analysis/AnalysisService.cs b/Dashboard/Analysis/AnalysisService.cs index 87a7eb3a0..2558ffca9 100644 --- a/Dashboard/Analysis/AnalysisService.cs +++ b/Dashboard/Analysis/AnalysisService.cs @@ -184,6 +184,19 @@ public async Task> AnalyzeAsync(AnalysisContext context) // 3. Build stories via graph traversal var stories = _engine.BuildStories(facts); + // 3.5. Freeze value-stated advice (current MAXDOP/CTFP/etc.) into each story's StoryText + // from the FULL fact set, BEFORE the store copies StoryText onto the finding. This is the + // only place the raw fact VALUES are in scope; read-back cards then state the numbers + // (FactAdvice.GetComposedForFinding) instead of generic folklore. No schema change. + FactAdvice.PopulateStoryText(stories, facts); + + // 3.6. Cluster the run's stories into causally-related incidents (graph-connectivity) and + // stamp each with its own trackable id, BEFORE the store copies it onto the finding. The + // grouped surface renders one report per incident; the id fingerprints the incident's + // primary so the same recurring incident is trackable across runs. + var incidents = _engine.ClusterIntoIncidents(stories, facts); + IncidentId.StampClusters(context.ServerName, incidents); + // 4. Mute-filter the stories into the surviving findings (P2 reorder) — WITHOUT // inserting yet, so enrichment + action-build happen on the survivors first // and the BUILT RemediationAction is persisted on each row (D2). Muted/ diff --git a/Dashboard/Analysis/BlockingChainViewerProjection.cs b/Dashboard/Analysis/BlockingChainViewerProjection.cs new file mode 100644 index 000000000..c3907b841 --- /dev/null +++ b/Dashboard/Analysis/BlockingChainViewerProjection.cs @@ -0,0 +1,74 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using PerformanceMonitor.Analysis; +using PerformanceMonitor.Common; + +namespace PerformanceMonitorDashboard.Analysis; + +/// +/// The trivial, mechanical projection from the analysis engine's (internal) reconstruction into the shared +/// public for the viewer. This is the only per-app block-chain code — it +/// must bridge an internal analysis type to a Common type, and no shared assembly can see both (Ui sees +/// Common but not Analysis; Analysis does not reference Common by design). Kept byte-identical to Lite's copy. +/// +internal static class BlockingChainViewerProjection +{ + public const string EmptyStateDetail = + "No blocking was captured for this session in the selected time range — from either the " + + "blocked-process-report Extended Event or the always-on DMV blocking snapshot. Its wait may not " + + "have overlapped a collection in this window."; + + /// + /// Builds the model for the ONE chain that contains the clicked session (matched by SessionKey — the + /// clicked session may be the apex, a mid-level blocker, or a leaf victim). The chain is rooted at its + /// apex with the full descendant hierarchy. Returns null when the session isn't in any reconstructed + /// chain (its wait may not have met the blocked-process threshold in the selected range). + /// + public static BlockingChainModel? BuildModelForSession( + BlockingReconstruction reconstruction, + int? monitorLoop, int spid, int ecid) + { + var chain = BlockingChainReconstructor.FindChainForSession( + reconstruction, monitorLoop, spid, ecid); + if (chain == null) + return null; + + return BlockingChainTreeBuilder.Build( + new[] { ToInput(chain) }, + reconstruction.CycleDetected, + reconstruction.DepthCapped, + reconstruction.TraversalTruncated); + } + + private static BlockingChainInput ToInput(ReconstructedChain c) => new() + { + ApexSpid = c.ApexSpid, + ApexEcid = c.ApexEcid, + ApexTranStarted = c.ApexTranStarted, + ApexSleeping = c.ApexSleeping, + Magnitude = c.Magnitude, + Edges = c.Levels.Select(static l => new BlockingEdgeInput + { + Level = l.Level, + BlockingSpid = l.BlockingSpid, + BlockingEcid = l.BlockingEcid, + BlockingTranStarted = l.BlockingTranStarted, + BlockedSpid = l.BlockedSpid, + BlockedEcid = l.BlockedEcid, + BlockedTranStarted = l.BlockedTranStarted, + LockMode = l.LockMode, + WaitTimeMs = l.WaitTimeMs, + DatabaseName = l.DatabaseName, + BlockingSqlText = l.BlockingSqlText, + BlockedSqlText = l.BlockedSqlText, + BlockedLoginName = l.BlockedLoginName, + BlockedHostName = l.BlockedHostName, + BlockedClientApp = l.BlockedClientApp, + BlockingLoginName = l.BlockingLoginName, + BlockingHostName = l.BlockingHostName, + BlockingClientApp = l.BlockingClientApp, + ContentiousObject = l.ContentiousObject + }).ToList() + }; +} diff --git a/Dashboard/Analysis/BlockingPairRowQuery.cs b/Dashboard/Analysis/BlockingPairRowQuery.cs new file mode 100644 index 000000000..800a9f69b --- /dev/null +++ b/Dashboard/Analysis/BlockingPairRowQuery.cs @@ -0,0 +1,237 @@ +using System; +using System.Collections.Generic; +using System.Data.Common; +using System.Threading.Tasks; +using Microsoft.Data.SqlClient; +using PerformanceMonitor.Analysis; + +namespace PerformanceMonitorDashboard.Analysis; + +/// +/// The single source of the blocked-process-report pair-row query used to reconstruct blocking chains. +/// Shared by the three Dashboard consumers — the drill-down collector, the BLOCKING_CHAIN fact collector, +/// and the viewer's data-service fetch — so the apex-determining blocking_spid IS NOT NULL filter +/// (and the SELECT/ordinals) stay in lockstep. activity = 'blocked' picks the canonical per-event +/// side; blocking_spid IS NOT NULL drops rows whose source XML had an empty +/// <blocking-process><process/></blocking-process> (system task / torn-down session), +/// which cannot contribute to a chain. +/// +internal static class BlockingPairRowQuery +{ + private const string ReadUncommitted = "SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED;\n"; + + /// + /// The core pair SELECT (no SET preamble; ordinals 0-14). Kept separate from so the + /// viewer variant can reuse it verbatim as a derived table — the + /// column list and the apex filter live in exactly one place and can't drift between the two queries. + /// Ordinals 11-13 are the BLOCKED row's own identity (login/host/app); ordinal 14 is the contended + /// object (resolved by the collector / sp_HumanEventsBlockViewer). The blocker's identity is not stored + /// on the blocked row, so a pure apex has no identity from this query alone (see the variant). + /// + public const string SelectBody = @" +SELECT TOP (5000) + event_time, + database_name, + spid, + last_transaction_started, + blocking_spid, + blocking_last_tran_started, + wait_time_ms, + lock_mode, + blocking_status, + blocked_sql_text, + blocking_sql_text, + login_name, + host_name, + client_app, + contentious_object, + ecid, + blocking_ecid, + monitor_loop +FROM collect.blocking_BlockedProcessReport +WHERE collection_time >= @collectionWindow +AND event_time >= @startTime +AND event_time <= @endTime +AND activity = 'blocked' +AND blocking_spid IS NOT NULL +ORDER BY collection_time DESC"; + + /// The collector/fact query: the core SELECT under READ UNCOMMITTED. Maps via . + public const string Sql = ReadUncommitted + SelectBody; + + /// + /// Viewer-only variant that also resolves the BLOCKER's identity (login / host / client app) per pair, + /// so the apex node — which only ever appears as a blocker, never as a blocked row — shows WHO it is + /// instead of a bare SPID. The Dashboard table stores identity per process row, so the blocker's lives + /// on its activity='blocking' row for the same event; we correlate to it by (event_time, spid) + /// via OUTER APPLY. TOP (1) collapses the one-to-many 'blocking' rows a lead blocker emits when it + /// blocks several victims in one report (identity is identical across them). The apply is bounded by the + /// same collection_time window as the core query, so the clustered PK (collection_time, + /// blocking_id) range-limits the lookup. The background collectors keep the lighter + /// (they score chains and never display identity), so this extra correlation is paid only on the + /// infrequent, small-window viewer open. Maps via (adds 15-17). + /// Result row order is intentionally unspecified: the inner TOP (5000) / ORDER BY selects the + /// most-recent pairs, reconstruction is order-independent, and the sole caller's maxPairs equals the + /// cap — so no outer ORDER BY is added (and collection_time is deliberately not projected, so don't + /// "fix" this with ORDER BY c.collection_time — it isn't a column of the derived table). + /// + public const string SqlWithBlockerIdentity = ReadUncommitted + @" +SELECT + c.*, + blocking_login_name = bk.login_name, + blocking_host_name = bk.host_name, + blocking_client_app = bk.client_app +FROM +(" + SelectBody + @" +) AS c +OUTER APPLY +( + SELECT TOP (1) + k.login_name, + k.host_name, + k.client_app + FROM collect.blocking_BlockedProcessReport AS k + WHERE k.activity = 'blocking' + AND k.spid = c.blocking_spid + AND k.event_time = c.event_time + AND k.collection_time >= @collectionWindow + ORDER BY k.collection_time DESC +) AS bk"; + + /// + /// Adds the three parameters. The collection-time floor is a generous bound (window start minus an + /// hour) so rows whose event_time is inside the window but whose collection_time lags slightly are + /// still caught. Shared by both and (the apply + /// reuses @collectionWindow to bound its own lookup). + /// + public static void AddParameters(SqlCommand cmd, DateTime start, DateTime end) + { + cmd.Parameters.Add(new SqlParameter("@collectionWindow", start.AddHours(-1))); + cmd.Parameters.Add(new SqlParameter("@startTime", start)); + cmd.Parameters.Add(new SqlParameter("@endTime", end)); + } + + /// + /// Maps ordinals 0-13 of the core query: per-event fields plus the BLOCKED row's own identity. The + /// blocking-side identity stays empty here — the Dashboard table parses only the blocker's + /// status/tran/SQL from the report XML, not its login/host/app — so a pure apex read via the collector + /// path shows SQL + db but no identity. The viewer fills it via . + /// + public static BlockingPairRow Read(DbDataReader reader) => new() + { + EventTime = reader.IsDBNull(0) ? default : reader.GetDateTime(0), + DatabaseName = reader.IsDBNull(1) ? string.Empty : reader.GetString(1), + BlockedSpid = reader.IsDBNull(2) ? 0 : Convert.ToInt32(reader.GetValue(2)), + BlockedTranStarted = reader.IsDBNull(3) ? (DateTime?)null : reader.GetDateTime(3), + BlockingSpid = reader.IsDBNull(4) ? 0 : Convert.ToInt32(reader.GetValue(4)), + BlockingTranStarted = reader.IsDBNull(5) ? (DateTime?)null : reader.GetDateTime(5), + WaitTimeMs = reader.IsDBNull(6) ? 0L : Convert.ToInt64(reader.GetValue(6)), + LockMode = reader.IsDBNull(7) ? string.Empty : reader.GetString(7), + BlockingStatus = reader.IsDBNull(8) ? string.Empty : reader.GetString(8), + BlockedSqlText = reader.IsDBNull(9) ? string.Empty : reader.GetString(9), + BlockingSqlText = reader.IsDBNull(10) ? string.Empty : reader.GetString(10), + BlockedLoginName = reader.IsDBNull(11) ? string.Empty : reader.GetString(11), + BlockedHostName = reader.IsDBNull(12) ? string.Empty : reader.GetString(12), + BlockedClientApp = reader.IsDBNull(13) ? string.Empty : reader.GetString(13), + ContentiousObject = reader.IsDBNull(14) ? string.Empty : reader.GetString(14), + // 15-17: spid:ecid + monitor_loop typed columns (populated at collect time by install/23). The + // collector reconstructs cumulatively (scopeByMonitorLoop:false) so MonitorLoop is read but ignored; + // the viewer scopes by it. + BlockedEcid = reader.IsDBNull(15) ? 0 : Convert.ToInt32(reader.GetValue(15)), + BlockingEcid = reader.IsDBNull(16) ? 0 : Convert.ToInt32(reader.GetValue(16)), + MonitorLoop = reader.IsDBNull(17) ? (int?)null : Convert.ToInt32(reader.GetValue(17)) + }; + + /// + /// Maps the viewer's result: the core (0-17) + /// plus the correlated blocker identity at ordinals 18-20, so the apex node carries login/host/app. + /// + public static BlockingPairRow ReadWithBlockerIdentity(DbDataReader reader) + { + var row = Read(reader); + row.BlockingLoginName = reader.IsDBNull(18) ? string.Empty : reader.GetString(18); + row.BlockingHostName = reader.IsDBNull(19) ? string.Empty : reader.GetString(19); + row.BlockingClientApp = reader.IsDBNull(20) ? string.Empty : reader.GetString(20); + return row; + } + + /// + /// The DMV-snapshot pair-row source (collect.dmv_blocking_snapshots, the always-on blocking fallback). + /// Same projection and ordinals (0-20) as so the shared + /// maps it unchanged — but the blocker identity is stored inline + /// (the snapshot collector captures both sides natively), so no OUTER APPLY correlation is needed. + /// + public const string DmvSql = ReadUncommitted + @" +SELECT TOP (5000) + event_time, + database_name, + spid, + last_transaction_started, + blocking_spid, + blocking_last_tran_started, + wait_time_ms, + lock_mode, + blocking_status, + blocked_sql_text, + blocking_sql_text, + login_name, + host_name, + client_app, + contentious_object, + ecid, + blocking_ecid, + monitor_loop, + blocking_login_name, + blocking_host_name, + blocking_client_app +FROM collect.dmv_blocking_snapshots +WHERE collection_time >= @collectionWindow +AND event_time >= @startTime +AND event_time <= @endTime +ORDER BY collection_time DESC"; + + /// + /// Fetches DMV-snapshot pair-rows for the window and merges them into + /// (BPR-preferred — see ). Called by all three pair-row consumers + /// after they build their BPR rows, so the DMV fallback feeds the reconstructor everywhere the + /// blocked-process-report does. Tolerates a not-yet-upgraded database (missing table -> no-op). + /// + internal static async Task AppendDmvSnapshotRowsAsync(SqlConnection connection, List rows, DateTime start, DateTime end) + { + var dmv = new List(); + try + { + using var cmd = new SqlCommand(DmvSql, connection); + cmd.CommandTimeout = 120; + AddParameters(cmd, start, end); + using var reader = await cmd.ExecuteReaderAsync(); + while (await reader.ReadAsync()) + dmv.Add(ReadWithBlockerIdentity(reader)); + } + catch (SqlException ex) when (ex.Number == 208) + { + /* Invalid object name: collect.dmv_blocking_snapshots not present yet (pre-upgrade DB). + Degrade to BPR-only rather than failing the whole block-chain fetch. */ + return; + } + + BlockingPairRowMerge.MergeInto(rows, dmv); + } + + /// + /// Cheap existence probe for collect.dmv_blocking_snapshots. The slicer and the flat top-blocking + /// drill-down inline this table in a single combined CTE / UNION batch, which fails to COMPILE (Msg 208) + /// on a not-yet-upgraded database — a runtime try/catch can't rescue one combined batch the way the + /// separate fetch can. Those callers use this to drop the DMV + /// branch from the query text entirely on such servers, degrading to BPR-only instead of throwing. + /// + internal static async Task DmvSnapshotsTableExistsAsync(SqlConnection connection) + { + using var cmd = new SqlCommand( + "SELECT CASE WHEN OBJECT_ID(N'collect.dmv_blocking_snapshots', N'U') IS NOT NULL THEN 1 ELSE 0 END;", + connection); + cmd.CommandTimeout = 30; + var result = await cmd.ExecuteScalarAsync(); + return result is not null && result != DBNull.Value && Convert.ToInt32(result) == 1; + } +} diff --git a/Dashboard/Analysis/SqlServerAnomalyDetector.cs b/Dashboard/Analysis/SqlServerAnomalyDetector.cs index a95ebfdc0..78d5493dd 100644 --- a/Dashboard/Analysis/SqlServerAnomalyDetector.cs +++ b/Dashboard/Analysis/SqlServerAnomalyDetector.cs @@ -215,6 +215,8 @@ ORDER BY if (await reader.ReadAsync()) { var db = reader.GetString(0); + var gSchema = reader.IsDBNull(1) ? null : reader.GetValue(1)?.ToString(); + var gTable = reader.IsDBNull(2) ? null : reader.GetValue(2)?.ToString(); var growthMb = Convert.ToDouble(reader.GetValue(5)); var growthPct = Convert.ToDouble(reader.GetValue(6)); anomalies.Add(new Fact @@ -224,6 +226,7 @@ ORDER BY Value = growthMb, ServerId = context.ServerId, DatabaseName = db, + ObjectName = string.IsNullOrEmpty(gTable) ? null : string.IsNullOrEmpty(gSchema) ? gTable : $"{gSchema}.{gTable}", Metadata = new Dictionary { ["prior_mb"] = Convert.ToDouble(reader.GetValue(3)), @@ -321,7 +324,17 @@ ORDER BY if (await reader.ReadAsync()) { var db = reader.GetString(0); + var cSchema = reader.IsDBNull(1) ? null : reader.GetValue(1)?.ToString(); + var cTable = reader.IsDBNull(2) ? null : reader.GetValue(2)?.ToString(); + var cIndex = reader.IsDBNull(3) ? null : reader.GetValue(3)?.ToString(); var msDelta = Convert.ToDouble(reader.GetValue(4)); + string? contendedObject = null; + if (!string.IsNullOrEmpty(cTable)) + { + contendedObject = string.IsNullOrEmpty(cSchema) ? cTable : $"{cSchema}.{cTable}"; + if (!string.IsNullOrEmpty(cIndex)) + contendedObject += $", index {cIndex}"; + } anomalies.Add(new Fact { Source = "anomaly", @@ -329,6 +342,7 @@ ORDER BY Value = msDelta, ServerId = context.ServerId, DatabaseName = db, + ObjectName = contendedObject, Metadata = new Dictionary { ["lock_wait_ms_delta"] = msDelta, @@ -572,17 +586,27 @@ private async Task DetectBlockingAnomalies(AnalysisContext context, List a var currentBlocking = Convert.ToInt64(reader.GetValue(0)); var currentDeadlocks = Convert.ToInt64(reader.GetValue(1)); + /* Baseline mean is events per hour-of-day/dow bucket (≈ events per hour at this time of + day). current_* are raw counts over the whole analysis window (hoursBack, default 4), + so normalize them to per-hour before the ratio — otherwise the ratio scales with the + window length, not the workload, and a steady event rate trips the spike threshold. */ + var windowHours = (context.TimeRangeEnd - context.TimeRangeStart).TotalHours; + if (windowHours <= 0) windowHours = 1; + var currentBlockingPerHour = currentBlocking / windowHours; + var currentDeadlocksPerHour = currentDeadlocks / windowHours; + + // Baseline mean = events per hour for this hour+dow bucket var baselineBlockingRate = blockingBaseline.SampleCount > 0 ? blockingBaseline.Mean : 0; var baselineDeadlockRate = deadlockBaseline.SampleCount > 0 ? deadlockBaseline.Mean : 0; - // Blocking spike: at least 5 events AND 3x baseline rate (or no baseline) - if (currentBlocking >= 5 && (baselineBlockingRate <= 0 || currentBlocking / Math.Max(baselineBlockingRate, 1) >= DefaultEventRatioThreshold)) + // Blocking spike: at least 5 events in the window AND per-hour rate >= 3x baseline (or no baseline) + if (currentBlocking >= 5 && (baselineBlockingRate <= 0 || currentBlockingPerHour / Math.Max(baselineBlockingRate, 1) >= DefaultEventRatioThreshold)) { var metadata = new Dictionary { ["current_count"] = currentBlocking, ["baseline_rate"] = baselineBlockingRate, - ["ratio"] = baselineBlockingRate > 0 ? currentBlocking / baselineBlockingRate : 100.0 + ["ratio"] = baselineBlockingRate > 0 ? currentBlockingPerHour / baselineBlockingRate : 100.0 }; AddBaselineContext(metadata, blockingBaseline); @@ -596,14 +620,14 @@ private async Task DetectBlockingAnomalies(AnalysisContext context, List a }); } - // Deadlock spike: at least 3 events AND 3x baseline rate (or no baseline) - if (currentDeadlocks >= 3 && (baselineDeadlockRate <= 0 || currentDeadlocks / Math.Max(baselineDeadlockRate, 1) >= DefaultEventRatioThreshold)) + // Deadlock spike: at least 3 events in the window AND per-hour rate >= 3x baseline (or no baseline) + if (currentDeadlocks >= 3 && (baselineDeadlockRate <= 0 || currentDeadlocksPerHour / Math.Max(baselineDeadlockRate, 1) >= DefaultEventRatioThreshold)) { var metadata = new Dictionary { ["current_count"] = currentDeadlocks, ["baseline_rate"] = baselineDeadlockRate, - ["ratio"] = baselineDeadlockRate > 0 ? currentDeadlocks / baselineDeadlockRate : 100.0 + ["ratio"] = baselineDeadlockRate > 0 ? currentDeadlocksPerHour / baselineDeadlockRate : 100.0 }; AddBaselineContext(metadata, deadlockBaseline); diff --git a/Dashboard/Analysis/SqlServerBaselineProvider.cs b/Dashboard/Analysis/SqlServerBaselineProvider.cs index 6ef0fef3e..dcecdad35 100644 --- a/Dashboard/Analysis/SqlServerBaselineProvider.cs +++ b/Dashboard/Analysis/SqlServerBaselineProvider.cs @@ -169,9 +169,11 @@ public async Task GetBaselineAsync(string metricName, DateTime a Mean = mean, StdDev = stddev, SampleCount = count, - Tier = count >= RestoreThreshold ? BaselineTier.Full - : count >= CollapseThreshold ? BaselineTier.Full - : BaselineTier.HourOnly + // Every bucket here is a full (hour, day-of-week) bucket; the HourOnly/Flat + // tiers are assigned only on the collapse/flat paths below. A sparse full + // bucket is still Full, just low-sample. (Was a copy-paste of two identical + // Full branches that mislabeled sparse buckets HourOnly in baseline_tier.) + Tier = BaselineTier.Full }; } diff --git a/Dashboard/Analysis/SqlServerDrillDownCollector.Blocking.cs b/Dashboard/Analysis/SqlServerDrillDownCollector.Blocking.cs new file mode 100644 index 000000000..3b0df4c08 --- /dev/null +++ b/Dashboard/Analysis/SqlServerDrillDownCollector.Blocking.cs @@ -0,0 +1,243 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading.Tasks; +using Microsoft.Data.SqlClient; +using PerformanceMonitor.Analysis; +using PerformanceMonitor.PlanAnalysis; +using PerformanceMonitorDashboard.Helpers; +using PerformanceMonitorDashboard.Mcp; +using PerformanceMonitorDashboard.Models; +using PerformanceMonitorDashboard.Services; +using PerformanceMonitor.Common; +using PerformanceMonitor.Notifications; + +namespace PerformanceMonitorDashboard.Analysis; + +public partial class SqlServerDrillDownCollector +{ + private async Task CollectTopDeadlocks(AnalysisFinding finding, AnalysisContext context) + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT TOP 3 + collection_time, + event_date, + spid, + LEFT(CAST(query AS NVARCHAR(MAX)), 500) AS victim_sql, + CAST(deadlock_graph AS NVARCHAR(MAX)) AS deadlock_graph +FROM collect.deadlocks +WHERE collection_time >= @startTime AND collection_time <= @endTime +ORDER BY collection_time DESC;"; + + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + var items = new List(); + using var reader = await cmd.ExecuteReaderAsync(); + while (await reader.ReadAsync()) + { + /* #1140: parse the involved objects from the graph for the dedup fingerprint + a readable + Objects field. The raw graph XML is NOT surfaced (it would bloat the alert detail). */ + var objects = DeadlockObjectExtractor.FromGraphXml(reader.IsDBNull(4) ? null : reader.GetString(4)); + items.Add(new + { + time = reader.IsDBNull(0) ? "" : reader.GetDateTime(0).ToString("o"), + deadlock_time = reader.IsDBNull(1) ? "" : reader.GetDateTime(1).ToString("o"), + victim = reader.IsDBNull(2) ? "" : reader.GetValue(2).ToString(), + victim_sql = reader.IsDBNull(3) ? "" : reader.GetString(3), + objects = string.Join(", ", objects) + }); + } + + if (items.Count > 0) + finding.DrillDown!["top_deadlocks"] = items; + } + + private async Task CollectTopBlockingChains(AnalysisFinding finding, AnalysisContext context) + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + /* BPR + always-on DMV blocking snapshot, so the flat top-blocking list isn't empty when the + blocked-process-report XE captured nothing (AWS RDS). Worst-by-wait surfaces regardless of + source; on a box with both, BPR and DMV may each contribute (this is a top-5 list, not a count). + The DMV UNION branch is dropped on a not-yet-upgraded server (no dmv_blocking_snapshots table) -- + inlining a missing table here would fail the whole combined batch at compile (Msg 208). */ + bool dmvExists = await BlockingPairRowQuery.DmvSnapshotsTableExistsAsync(connection); + string dmvUnion = dmvExists ? @" + + UNION ALL + + SELECT + collection_time, + database_name, + blocked_spid = spid, + blocking_spid, + wait_time_ms, + lock_mode, + blocked_sql = LEFT(CAST(blocked_sql_text AS NVARCHAR(MAX)), 500), + blocking_sql = LEFT(CAST(blocking_sql_text AS NVARCHAR(MAX)), 500), + contentious_object + FROM collect.dmv_blocking_snapshots + WHERE collection_time >= @startTime AND collection_time <= @endTime" : ""; + cmd.CommandText = $@" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT TOP 5 + collection_time, + database_name, + blocked_spid, + blocking_spid, + wait_time_ms, + lock_mode, + blocked_sql, + blocking_sql, + contentious_object +FROM +( + SELECT + collection_time, + database_name, + blocked_spid = spid, + blocking_spid = 0, + wait_time_ms, + lock_mode, + blocked_sql = LEFT(CAST(query_text AS NVARCHAR(MAX)), 500), + blocking_sql = LEFT(blocking_tree, 500), + contentious_object + FROM collect.blocking_BlockedProcessReport + WHERE collection_time >= @startTime AND collection_time <= @endTime{dmvUnion} +) AS combined +ORDER BY wait_time_ms DESC;"; + + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + var items = new List(); + using var reader = await cmd.ExecuteReaderAsync(); + while (await reader.ReadAsync()) + { + items.Add(new + { + time = reader.IsDBNull(0) ? "" : reader.GetDateTime(0).ToString("o"), + database = reader.IsDBNull(1) ? "" : reader.GetString(1), + blocked_spid = reader.IsDBNull(2) ? 0 : Convert.ToInt32(reader.GetValue(2)), + blocking_spid = reader.IsDBNull(3) ? 0 : Convert.ToInt32(reader.GetValue(3)), + wait_time_ms = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)), + lock_mode = reader.IsDBNull(5) ? "" : reader.GetString(5), + blocked_sql = reader.IsDBNull(6) ? "" : reader.GetString(6), + blocking_sql = reader.IsDBNull(7) ? "" : reader.GetString(7), + contentious_object = reader.IsDBNull(8) ? "" : reader.GetString(8) + }); + } + + if (items.Count > 0) + finding.DrillDown!["top_blocking_chains"] = items; + } + + /// + /// Reconstructs blocking chains (same logic as the collector) and surfaces the top 3 + /// by magnitude — apex, depth, victim count, and the level-by-level structure that + /// the flat top_blocking_chains list cannot show. + /// + private async Task CollectReconstructedBlockingChains(AnalysisFinding finding, AnalysisContext context) + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + // Shared query/filter — see BlockingPairRowQuery (keeps the drill-down, the BLOCKING_CHAIN fact + // collector, and the viewer fetch in lockstep on the apex-determining blocking_spid filter). + cmd.CommandText = BlockingPairRowQuery.Sql; + BlockingPairRowQuery.AddParameters(cmd, context.TimeRangeStart, context.TimeRangeEnd); + + var rows = new List(); + using (var reader = await cmd.ExecuteReaderAsync()) + { + while (await reader.ReadAsync()) + rows.Add(BlockingPairRowQuery.Read(reader)); + } + + // Always-on DMV blocking snapshot fallback. Merge BEFORE the empty check so DMV-only blocking + // (blocked-process-report unavailable, e.g. AWS RDS) still reconstructs. + await BlockingPairRowQuery.AppendDmvSnapshotRowsAsync(connection, rows, context.TimeRangeStart, context.TimeRangeEnd); + + if (rows.Count == 0) return; + + var reconstruction = BlockingChainReconstructor.Reconstruct( + rows, maxDepth: 50, maxPairs: 5000, stepBudget: 100_000, scopeByMonitorLoop: false); + + var items = new List(); + foreach (var chain in reconstruction.Chains.Take(3)) + { + items.Add(new + { + apex_spid = chain.ApexSpid, + apex_sleeping = chain.ApexSleeping, + depth = chain.Depth, + // Distinct sessions blocked under this apex over the window — cumulative, not peak-concurrent. + victim_count = chain.VictimCount, + max_wait_ms = chain.MaxWaitMs, + levels = chain.Levels.Select(l => new + { + level = l.Level, + blocking_spid = l.BlockingSpid, + blocked_spid = l.BlockedSpid, + lock_mode = l.LockMode, + wait_time_ms = l.WaitTimeMs, + blocking_sql = l.BlockingSqlText, + blocked_sql = l.BlockedSqlText + }).ToList() + }); + } + + if (items.Count > 0) + finding.DrillDown!["reconstructed_blocking_chains"] = items; + } + + private async Task CollectLockModeBreakdown(AnalysisFinding finding, AnalysisContext context) + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT TOP 10 + wait_type, + CAST(SUM(wait_time_ms_delta) AS BIGINT) AS total_wait_ms, + CAST(SUM(waiting_tasks_count_delta) AS BIGINT) AS total_count +FROM collect.wait_stats +WHERE collection_time >= @startTime AND collection_time <= @endTime +AND wait_type LIKE 'LCK%' +AND wait_time_ms_delta > 0 +GROUP BY wait_type +ORDER BY CAST(SUM(wait_time_ms_delta) AS BIGINT) DESC;"; + + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + var items = new List(); + using var reader = await cmd.ExecuteReaderAsync(); + while (await reader.ReadAsync()) + { + items.Add(new + { + lock_type = reader.IsDBNull(0) ? "" : reader.GetString(0), + total_wait_ms = reader.IsDBNull(1) ? 0.0 : Convert.ToDouble(reader.GetValue(1)), + waiting_tasks = reader.IsDBNull(2) ? 0.0 : Convert.ToDouble(reader.GetValue(2)) + }); + } + + if (items.Count > 0) + finding.DrillDown!["lock_mode_breakdown"] = items; + } +} diff --git a/Dashboard/Analysis/SqlServerDrillDownCollector.Config.cs b/Dashboard/Analysis/SqlServerDrillDownCollector.Config.cs new file mode 100644 index 000000000..95e6b64b6 --- /dev/null +++ b/Dashboard/Analysis/SqlServerDrillDownCollector.Config.cs @@ -0,0 +1,241 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading.Tasks; +using Microsoft.Data.SqlClient; +using PerformanceMonitor.Analysis; +using PerformanceMonitor.PlanAnalysis; +using PerformanceMonitorDashboard.Helpers; +using PerformanceMonitorDashboard.Mcp; +using PerformanceMonitorDashboard.Models; +using PerformanceMonitorDashboard.Services; +using PerformanceMonitor.Common; +using PerformanceMonitor.Notifications; + +namespace PerformanceMonitorDashboard.Analysis; + +public partial class SqlServerDrillDownCollector +{ + private async Task CollectConfigIssues(AnalysisFinding finding, AnalysisContext context) + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + // The Dashboard uses config.database_configuration_history which stores + // settings as rows (setting_type, setting_name, setting_value) not columns. + // Pivot the latest snapshot into the format we need. + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +;WITH latest AS ( + SELECT database_name, setting_name, + CAST(setting_value AS NVARCHAR(256)) AS setting_value, + ROW_NUMBER() OVER (PARTITION BY database_name, setting_name ORDER BY collection_time DESC) AS rn + FROM config.database_configuration_history + WHERE setting_name IN ( + 'recovery_model_desc', 'is_auto_shrink_on', 'is_auto_close_on', + 'is_read_committed_snapshot_on', 'page_verify_option_desc', 'is_query_store_on' + ) +), +pivoted AS ( + SELECT + database_name, + MAX(CASE WHEN setting_name = 'recovery_model_desc' THEN setting_value END) AS recovery_model, + MAX(CASE WHEN setting_name = 'is_auto_shrink_on' THEN setting_value END) AS is_auto_shrink_on, + MAX(CASE WHEN setting_name = 'is_auto_close_on' THEN setting_value END) AS is_auto_close_on, + MAX(CASE WHEN setting_name = 'is_read_committed_snapshot_on' THEN setting_value END) AS is_rcsi_on, + MAX(CASE WHEN setting_name = 'page_verify_option_desc' THEN setting_value END) AS page_verify_option, + MAX(CASE WHEN setting_name = 'is_query_store_on' THEN setting_value END) AS is_query_store_on + FROM latest + WHERE rn = 1 + GROUP BY database_name +) +SELECT database_name, recovery_model, + is_auto_shrink_on, is_auto_close_on, + is_rcsi_on, page_verify_option, is_query_store_on +FROM pivoted +WHERE is_auto_shrink_on = '1' OR is_auto_close_on = '1' + OR is_rcsi_on = '0' OR page_verify_option != 'CHECKSUM' +ORDER BY database_name;"; + + // Read the pivoted rows first into a typed buffer so we can determine the + // RCSI-OFF set BEFORE building the emitted items — the §3.3 enrichment is + // computed only for RCSI-off databases and injected into those rows. + var rows = new List(); + using (var reader = await cmd.ExecuteReaderAsync()) + { + while (await reader.ReadAsync()) + { + rows.Add(new ConfigIssueRow + { + Database = reader.IsDBNull(0) ? "" : reader.GetString(0), + RecoveryModel = reader.IsDBNull(1) ? "" : reader.GetString(1), + AutoShrink = (reader.IsDBNull(2) ? "" : reader.GetString(2)) == "1", + AutoClose = (reader.IsDBNull(3) ? "" : reader.GetString(3)) == "1", + Rcsi = (reader.IsDBNull(4) ? "" : reader.GetString(4)) == "1", + PageVerify = reader.IsDBNull(5) ? "" : reader.GetString(5), + QueryStore = (reader.IsDBNull(6) ? "" : reader.GetString(6)) == "1" + }); + } + } + + // §3.3 enrichment: per-DB blocking/deadlock counts + reader/writer split, ONLY + // for RCSI-off databases, over the analysis window, from already-collected + // monitoring tables (no fresh probe of the target server). + var rcsiOff = rows.Where(r => !r.Rcsi && !string.IsNullOrEmpty(r.Database)) + .Select(r => r.Database) + .ToList(); + var enrichment = rcsiOff.Count > 0 + ? await CollectRcsiInactionFigures(connection, rcsiOff, context) + : new Dictionary(StringComparer.Ordinal); + + var items = new List(); + foreach (var r in rows) + { + var issues = new List(); + if (r.AutoShrink) issues.Add("auto_shrink ON"); + if (r.AutoClose) issues.Add("auto_close ON"); + if (!r.Rcsi) issues.Add("RCSI OFF"); + if (!string.IsNullOrEmpty(r.PageVerify) && r.PageVerify != "CHECKSUM") issues.Add($"page_verify={r.PageVerify}"); + + // RCSI-off rows carry the three structured inaction-risk fields (M-2: + // identical JSON names + types to Lite, which emits them null/0). RCSI-on + // rows omit them entirely (the affordance never applies there). + if (!r.Rcsi) + { + enrichment.TryGetValue(r.Database, out var fig); + items.Add(new + { + database = r.Database, + recovery_model = r.RecoveryModel, + rcsi = r.Rcsi, + query_store = r.QueryStore, + issues, + auto_shrink = r.AutoShrink, + auto_close = r.AutoClose, + page_verify = r.PageVerify, + // §3.3: int / int / nullable-int. Counts from blocking_deadlock_stats + // (O-P3-F — NOT a blocking_BlockedProcessReport row count); the split + // is a separate pass over blocking_BlockedProcessReport. + rcsi_blocking_events = fig?.BlockingEvents ?? 0, + rcsi_deadlocks = fig?.Deadlocks ?? 0, + rcsi_reader_writer_pct = fig?.ReaderWriterPct + }); + } + else + { + items.Add(new + { + database = r.Database, + recovery_model = r.RecoveryModel, + rcsi = r.Rcsi, + query_store = r.QueryStore, + issues, + // §4.1: structured, wording-independent fields the shared extractor + // (FactRemediation.ExtractDbConfigTargets) reads. Identical JSON names + // and types to the Lite collector (bool / bool / string). + auto_shrink = r.AutoShrink, + auto_close = r.AutoClose, + page_verify = r.PageVerify + }); + } + } + + if (items.Count > 0) + finding.DrillDown!["config_issues"] = items; + } + + /// + /// Emits the server_config drill-down for a CONFIG_* finding (WS3): the single bad + /// server-level setting that rooted this finding, with its latest value_in_use plus the + /// server's engine_edition and cores_per_socket (needed to compute the + /// edition-aware, core-capped MAXDOP recommendation in + /// ). The structured fields + /// (setting, current_value, edition, cores_per_socket) are what the + /// shared extractor reads; nothing here is executed. The four CONFIG_* keys root SEPARATE + /// findings, so each finding emits exactly the one row for its own setting. + /// + private async Task CollectServerConfig(AnalysisFinding finding, AnalysisContext context, HashSet pathKeys) + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + // Latest value per requested config (ROW_NUMBER, not TOP N — same dedup correctness as the + // fact collector), plus the latest server_properties row for edition + cores-per-socket. + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +;WITH latest_config AS ( + SELECT + configuration_name, + CAST(value_in_use AS BIGINT) AS value_in_use, + ROW_NUMBER() OVER (PARTITION BY configuration_name ORDER BY collection_time DESC) AS rn + FROM config.server_configuration_history + WHERE configuration_name IN ( + 'cost threshold for parallelism', + 'max degree of parallelism', + 'max server memory (MB)', + 'min server memory (MB)' + ) +) +SELECT + configuration_name, + value_in_use +FROM latest_config +WHERE rn = 1; + +SELECT TOP (1) + engine_edition, + cores_per_socket +FROM collect.server_properties +ORDER BY collection_time DESC;"; + + long? ctfp = null, maxdop = null, maxMem = null, minMem = null; + int edition = 0, coresPerSocket = 0; + + using (var reader = await cmd.ExecuteReaderAsync()) + { + while (await reader.ReadAsync()) + { + var name = reader.IsDBNull(0) ? "" : reader.GetString(0); + var value = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); + switch (name) + { + case "cost threshold for parallelism": ctfp = value; break; + case "max degree of parallelism": maxdop = value; break; + case "max server memory (MB)": maxMem = value; break; + case "min server memory (MB)": minMem = value; break; + } + } + + if (await reader.NextResultAsync() && await reader.ReadAsync()) + { + edition = reader.IsDBNull(0) ? 0 : Convert.ToInt32(reader.GetValue(0)); + coresPerSocket = reader.IsDBNull(1) ? 0 : Convert.ToInt32(reader.GetValue(1)); + } + } + + var items = new List(); + + // Emit ONLY the setting that rooted this finding (the engine roots one CONFIG_* per finding). + // edition / cores_per_socket ride every row (the extractor uses them only for MAXDOP). + if (pathKeys.Contains("CONFIG_MAXDOP") && maxdop is { } md) + items.Add(new { setting = "maxdop", current_value = md, edition, cores_per_socket = coresPerSocket }); + + if (pathKeys.Contains("CONFIG_CTFP") && ctfp is { } ct) + items.Add(new { setting = "ctfp", current_value = ct, edition, cores_per_socket = coresPerSocket }); + + if (pathKeys.Contains("CONFIG_MAX_MEMORY_MB") && maxMem is { } mx) + items.Add(new { setting = "max_memory", current_value = mx, edition, cores_per_socket = coresPerSocket }); + + // For the narrow-memory finding the bad value the operator acts on is MIN server memory + // (lower it); carry it as the min_memory target. + if (pathKeys.Contains("CONFIG_MIN_MAX_MEMORY_NARROW") && minMem is { } mn) + items.Add(new { setting = "min_memory", current_value = mn, edition, cores_per_socket = coresPerSocket }); + + if (items.Count > 0) + finding.DrillDown!["server_config"] = items; + } +} diff --git a/Dashboard/Analysis/SqlServerDrillDownCollector.Plans.cs b/Dashboard/Analysis/SqlServerDrillDownCollector.Plans.cs new file mode 100644 index 000000000..24995d121 --- /dev/null +++ b/Dashboard/Analysis/SqlServerDrillDownCollector.Plans.cs @@ -0,0 +1,194 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading.Tasks; +using Microsoft.Data.SqlClient; +using PerformanceMonitor.Analysis; +using PerformanceMonitor.PlanAnalysis; +using PerformanceMonitorDashboard.Helpers; +using PerformanceMonitorDashboard.Mcp; +using PerformanceMonitorDashboard.Models; +using PerformanceMonitorDashboard.Services; +using PerformanceMonitor.Common; +using PerformanceMonitor.Notifications; + +namespace PerformanceMonitorDashboard.Analysis; + +public partial class SqlServerDrillDownCollector +{ + /// + /// For findings that have query hashes (bad actors), fetch the execution plan + /// live from SQL Server via IPlanFetcher, then run PlanAnalyzer to surface + /// warnings and missing indexes. No plan storage needed -- fetch on demand + /// only for queries that make it into high-impact findings. + /// + private async Task CollectPlanAnalysis(AnalysisFinding finding, AnalysisContext context) + { + if (finding.DrillDown == null || _planFetcher == null) return; + + // Only analyze plans for bad actor findings (1 plan each). + // Skip top_cpu_queries (5 plans would be too heavy). + if (!finding.RootFactKey.StartsWith("BAD_ACTOR_", StringComparison.OrdinalIgnoreCase)) return; + + var queryHash = finding.RootFactKey.Replace("BAD_ACTOR_", ""); + if (string.IsNullOrEmpty(queryHash)) return; + + // Look up plan_handle from collect.query_stats for this query_hash + string? planHandle = null; + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT TOP 1 CONVERT(VARCHAR(130), plan_handle, 1) AS plan_handle +FROM collect.query_stats +WHERE query_hash = CONVERT(BINARY(8), @queryHash, 1) +AND plan_handle IS NOT NULL +ORDER BY collection_time DESC;"; + + cmd.Parameters.Add(new SqlParameter("@queryHash", queryHash)); + + using var reader = await cmd.ExecuteReaderAsync(); + if (await reader.ReadAsync() && !reader.IsDBNull(0)) + planHandle = reader.GetString(0); + } + catch { return; } + + if (string.IsNullOrEmpty(planHandle)) return; + + // Fetch plan XML live from SQL Server + var planXml = await _planFetcher.FetchPlanXmlAsync(context.ServerId, planHandle); + if (string.IsNullOrEmpty(planXml)) return; + + try + { + var plan = ShowPlanParser.Parse(planXml); + PlanAnalyzer.Analyze(plan); + + var allWarnings = plan.Batches + .SelectMany(b => b.Statements) + .Where(s => s.RootNode != null) + .SelectMany(s => + { + var nodeWarnings = new List(); + CollectPlanNodes(s.RootNode!, nodeWarnings); + return s.PlanWarnings + .Concat(nodeWarnings.SelectMany(n => n.Warnings)); + }) + .ToList(); + + var missingIndexes = plan.AllMissingIndexes; + + if (allWarnings.Count == 0 && missingIndexes.Count == 0) return; + + finding.DrillDown["plan_analysis"] = new + { + query_hash = queryHash, + warning_count = allWarnings.Count, + critical_count = allWarnings.Count(w => w.Severity == PlanWarningSeverity.Critical), + warnings = allWarnings + .OrderByDescending(w => w.Severity) + .Take(10) + .Select(w => new + { + severity = w.Severity.ToString(), + type = w.WarningType, + message = McpHelpers.Truncate(w.Message, 300) + }), + missing_indexes = missingIndexes.Take(5).Select(idx => new + { + table = $"{idx.Schema}.{idx.Table}", + impact = idx.Impact, + create_statement = idx.CreateStatement + }) + }; + } + catch + { + // Plan parsing can fail on malformed XML -- skip silently + } + } + + /// + /// WS4: re-parses the top collected query plans (same top-10-by-cost set the fact collector + /// summarized) and attaches the specific missing indexes / plan warnings to a MISSING_INDEX or + /// PLAN_WARNING finding's drill-down. The fact carries only counts (Fact.Metadata is numeric), + /// so the strings — CREATE statements, warning messages — are rendered here. Best-effort: a + /// read/parse failure leaves the finding with no plan-advisory detail rather than aborting. + /// + private async Task CollectPlanAdvisoryDetail(AnalysisFinding finding, AnalysisContext context, HashSet pathKeys) + { + try + { + var planXmls = new List(); + + using (var connection = new SqlConnection(_connectionString)) + { + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT TOP (10) + plan_xml = CAST(DECOMPRESS(qs.query_plan_text) AS nvarchar(max)) +FROM collect.query_stats AS qs +WHERE qs.collection_time >= @startTime +AND qs.collection_time <= @endTime +AND qs.query_plan_text IS NOT NULL +ORDER BY + qs.total_worker_time DESC;"; + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + using var reader = await cmd.ExecuteReaderAsync(); + while (await reader.ReadAsync()) + { + if (!reader.IsDBNull(0)) + planXmls.Add(reader.GetString(0)); + } + } + + if (planXmls.Count == 0) + return; + + var details = PlanAdvisoryAggregator.Extract(planXmls); + + if (pathKeys.Contains("MISSING_INDEX") && details.MissingIndexes.Count > 0) + { + finding.DrillDown!["missing_indexes"] = details.MissingIndexes + .OrderByDescending(i => i.Impact) + .Take(5) + .Select(i => new + { + table = $"{i.Schema}.{i.Table}", + impact = Math.Round(i.Impact, 1), + create_statement = i.CreateStatement + }) + .ToList(); + } + + if (pathKeys.Contains("PLAN_WARNING") && details.Warnings.Count > 0) + { + finding.DrillDown!["plan_warnings"] = details.Warnings + .OrderByDescending(w => w.Severity) + .Take(5) + .Select(w => new + { + type = w.WarningType, + severity = w.Severity.ToString(), + message = McpHelpers.Truncate(w.Message, 300) + }) + .ToList(); + } + } + catch + { + // Plan read/parse can fail on malformed XML -- skip, the detail is best-effort. + } + } +} diff --git a/Dashboard/Analysis/SqlServerDrillDownCollector.Queries.cs b/Dashboard/Analysis/SqlServerDrillDownCollector.Queries.cs new file mode 100644 index 000000000..1c0b62b1e --- /dev/null +++ b/Dashboard/Analysis/SqlServerDrillDownCollector.Queries.cs @@ -0,0 +1,753 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading.Tasks; +using Microsoft.Data.SqlClient; +using PerformanceMonitor.Analysis; +using PerformanceMonitor.PlanAnalysis; +using PerformanceMonitorDashboard.Helpers; +using PerformanceMonitorDashboard.Mcp; +using PerformanceMonitorDashboard.Models; +using PerformanceMonitorDashboard.Services; +using PerformanceMonitor.Common; +using PerformanceMonitor.Notifications; + +namespace PerformanceMonitorDashboard.Analysis; + +public partial class SqlServerDrillDownCollector +{ + private async Task CollectQueriesAtSpike(AnalysisFinding finding, AnalysisContext context) + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + // Check if query_snapshots table exists (created dynamically by sp_WhoIsActive) + using var checkCmd = connection.CreateCommand(); + checkCmd.CommandText = "SELECT OBJECT_ID(N'collect.query_snapshots', N'U')"; + var tableExists = await checkCmd.ExecuteScalarAsync(); + if (tableExists == null || tableExists == DBNull.Value) return; + + // Step 1: Find when the spike occurred + using var peakCmd = connection.CreateCommand(); + peakCmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT TOP 1 collection_time, sqlserver_cpu_utilization +FROM collect.cpu_utilization_stats +WHERE collection_time >= @startTime AND collection_time <= @endTime +ORDER BY sqlserver_cpu_utilization DESC;"; + + peakCmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + peakCmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + DateTime? peakTime = null; + int peakCpu = 0; + using (var peakReader = await peakCmd.ExecuteReaderAsync()) + { + if (await peakReader.ReadAsync()) + { + peakTime = peakReader.GetDateTime(0); + peakCpu = peakReader.GetInt32(1); + } + } + + if (peakTime == null) return; + + // Step 2: Get queries active within 2 minutes of peak + using var queryCmd = connection.CreateCommand(); + queryCmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT TOP 5 + collection_time, + [session_id], + [database_name], + [status], + DATEDIFF(MILLISECOND, 0, [CPU]) AS cpu_time_ms, + DATEDIFF(MILLISECOND, 0, [elapsed_time]) AS total_elapsed_time_ms, + [reads] AS logical_reads, + [wait_info] AS wait_type, + 0 AS dop, + 0 AS parallel_worker_count, + LEFT(CAST([sql_text] AS NVARCHAR(MAX)), 500) AS query_text +FROM collect.query_snapshots +WHERE collection_time >= @spikeStart +AND collection_time <= @spikeEnd +AND CAST([sql_text] AS NVARCHAR(MAX)) NOT LIKE 'WAITFOR%' +ORDER BY DATEDIFF(MILLISECOND, 0, [CPU]) DESC;"; + + queryCmd.Parameters.Add(new SqlParameter("@spikeStart", peakTime.Value.AddMinutes(-2))); + queryCmd.Parameters.Add(new SqlParameter("@spikeEnd", peakTime.Value.AddMinutes(2))); + + var items = new List(); + using (var reader = await queryCmd.ExecuteReaderAsync()) + { + while (await reader.ReadAsync()) + { + items.Add(new + { + time = reader.IsDBNull(0) ? "" : reader.GetDateTime(0).ToString("o"), + session_id = reader.IsDBNull(1) ? 0 : Convert.ToInt32(reader.GetValue(1)), + database = reader.IsDBNull(2) ? "" : reader.GetString(2), + status = reader.IsDBNull(3) ? "" : reader.GetString(3), + cpu_time_ms = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)), + elapsed_time_ms = reader.IsDBNull(5) ? 0L : Convert.ToInt64(reader.GetValue(5)), + logical_reads = reader.IsDBNull(6) ? 0L : Convert.ToInt64(reader.GetValue(6)), + wait_type = reader.IsDBNull(7) ? "" : reader.GetString(7), + dop = reader.IsDBNull(8) ? 0 : Convert.ToInt32(reader.GetValue(8)), + parallel_workers = reader.IsDBNull(9) ? 0 : Convert.ToInt32(reader.GetValue(9)), + query_text = reader.IsDBNull(10) ? "" : reader.GetString(10) + }); + } + } + + if (items.Count > 0) + { + finding.DrillDown!["spike_peak"] = new + { + time = peakTime.Value.ToString("o"), + cpu_percent = peakCpu + }; + finding.DrillDown!["queries_at_spike"] = items; + } + } + + private async Task CollectTopCpuQueries(AnalysisFinding finding, AnalysisContext context) + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT TOP 5 + database_name, + CONVERT(VARCHAR(18), query_hash, 1) AS query_hash, + CAST(SUM(total_worker_time_delta) AS BIGINT) AS total_cpu_us, + CAST(SUM(execution_count_delta) AS BIGINT) AS exec_count, + MAX(max_dop) AS max_dop, + CAST(SUM(total_spills) AS BIGINT) AS spills, + LEFT(CAST(DECOMPRESS(MAX(query_text)) AS NVARCHAR(MAX)), 500) AS query_text +FROM collect.query_stats +WHERE collection_time >= @startTime AND collection_time <= @endTime +AND total_worker_time_delta > 0 +GROUP BY database_name, query_hash +ORDER BY CAST(SUM(total_worker_time_delta) AS BIGINT) DESC;"; + + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + var items = new List(); + using var reader = await cmd.ExecuteReaderAsync(); + while (await reader.ReadAsync()) + { + items.Add(new + { + database = reader.IsDBNull(0) ? "" : reader.GetString(0), + query_hash = reader.IsDBNull(1) ? "" : reader.GetString(1), + total_cpu_ms = reader.IsDBNull(2) ? 0.0 : Convert.ToDouble(reader.GetValue(2)) / 1000.0, + execution_count = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)), + max_dop = reader.IsDBNull(4) ? 0 : Convert.ToInt32(reader.GetValue(4)), + spills = reader.IsDBNull(5) ? 0L : Convert.ToInt64(reader.GetValue(5)), + query_text = reader.IsDBNull(6) ? "" : reader.GetString(6) + }); + } + + if (items.Count > 0 && !finding.DrillDown!.ContainsKey("top_cpu_queries")) + finding.DrillDown!["top_cpu_queries"] = items; + } + + /// + /// The plan-cache anomaly detector (§2): one row per offending query_hash whose + /// CURRENT per-exec CPU has jumped to an abnormal multiple of its OWN trailing baseline, + /// and which is a material CPU contributor in the window. This is the enrichment the + /// (PR-B) "Clear cached plan (advanced)" affordance reads. PR-A emits the drill-down but + /// does NOT register the handler / emit the affordance — dead-code-safe display only. + /// + /// + /// §2a ROW-LEVEL exclusion (the round-2 correctness fix): the delta framework + /// (install/05_delta_framework.sql:218-307) assigns delta = full cumulative raw + /// total on TWO arms — first-collection-of-a-plan_handle (the pc.collection_id + /// IS NULL arm, :233-235) and the first-post-restart row (the + /// server_start_time >= pc.collection_time arm, :235-236). Both inject false + /// anomalies on exactly this feature's target population. We exclude BOTH, per row, in + /// BOTH the current and baseline windows: a row counts as a REAL prior-delta row only + /// when an earlier collection exists for the same (sql_handle, offsets, plan_handle) + /// AND that earlier collection_time is > this row's server_start_time (i.e. the + /// delta interval started at a real prior collection, not at compile/restart). The + /// per-exec math (M-3) and the materiality CPU sum use ONLY these real-delta rows. + /// + /// + private async Task CollectAbnormalCpuPlans(AnalysisFinding finding, AnalysisContext context, HashSet pathKeys) + { + // The anomaly threshold (sibling of PLAN_REGRESSION's regression_factor) and the + // materiality floor (a query must contribute at least this much CPU in the window + // for clearing to be worth offering). Conservative defaults — start ~3x. + const double AnomalyThreshold = 3.0; + const double MaterialCpuMsFloor = 1000.0; // 1s of CPU in the window + + // Co-fired-fact awareness (drives the §5 disclosure steer): whether this CPU + // finding's story crossed PLAN_REGRESSION / PARAMETER_SENSITIVITY. The analysis + // already holds the story path — no extra SQL. + var planRegressionCoFired = pathKeys.Contains("PLAN_REGRESSION"); + var parameterSensitivityCoFired = pathKeys.Contains("PARAMETER_SENSITIVITY"); + + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + // The baseline window is the @baselineDays preceding the current window (NOT + // overlapping it). The current window is [@startTime, @endTime]. + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +DECLARE @baselineStart datetime2(7) = DATEADD(DAY, -@baselineDays, @startTime); + +/* +Real-delta rows only (§2a row-level exclusion): a row is a genuine inter-collection +delta when an EARLIER collection exists for the same (sql_handle, offsets, plan_handle) +whose collection_time is BOTH < this row's collection_time (it is a prior) AND +> this row's server_start_time (so the delta interval did not start at compile/restart). +This drops the first-collection-of-a-plan_handle row and the first-post-restart row in +one predicate, by row, without using sample_interval_seconds (R2-MIN-A). +*/ +WITH + real_delta AS +( + SELECT + qs.query_hash, + qs.database_name, + qs.collection_time, + qs.total_worker_time_delta, + qs.execution_count_delta, + qs.plan_handle, + qs.query_text + FROM collect.query_stats AS qs + WHERE qs.query_hash IS NOT NULL + AND qs.collection_time >= @baselineStart + AND qs.collection_time <= @endTime + AND EXISTS + ( + SELECT 1 + FROM collect.query_stats AS prior + WHERE prior.sql_handle = qs.sql_handle + AND prior.statement_start_offset = qs.statement_start_offset + AND prior.statement_end_offset = qs.statement_end_offset + AND prior.plan_handle = qs.plan_handle + AND prior.collection_time < qs.collection_time + AND prior.collection_time > qs.server_start_time + ) +), + windowed AS +( + SELECT + rd.query_hash, + /* current-window per-exec CPU (ms): SUM(worker)/SUM(execs) on real-delta rows */ + current_worker_ms = + CAST(SUM(CASE WHEN rd.collection_time >= @startTime THEN rd.total_worker_time_delta ELSE 0 END) AS float) / 1000.0, + current_execs = + SUM(CASE WHEN rd.collection_time >= @startTime THEN rd.execution_count_delta ELSE 0 END), + /* baseline-window per-exec CPU (ms): the preceding window, same exclusion */ + baseline_worker_ms = + CAST(SUM(CASE WHEN rd.collection_time < @startTime THEN rd.total_worker_time_delta ELSE 0 END) AS float) / 1000.0, + baseline_execs = + SUM(CASE WHEN rd.collection_time < @startTime THEN rd.execution_count_delta ELSE 0 END), + /* window CPU contribution (ms), §2a exclusion applied (R2-MIN-B) */ + current_total_cpu_ms = + CAST(SUM(CASE WHEN rd.collection_time >= @startTime THEN rd.total_worker_time_delta ELSE 0 END) AS float) / 1000.0 + FROM real_delta AS rd + GROUP BY rd.query_hash +), + /* + LOW-1 fix: the window's TOTAL query CPU (ms) over the SAME §2a real-delta rows in the + current window, so each query's cpu_percent is its real share of query CPU over the + window (NOT the hardcoded 0 that understated risk in PR-A). Using the same exclusion + keeps the numerator and denominator consistent (a query can't show a share computed on + contaminated raw-total CPU it won't actually clear, R2-MIN-B). + */ + window_total AS +( + SELECT + total_cpu_ms = + CAST(SUM(CASE WHEN rd.collection_time >= @startTime THEN rd.total_worker_time_delta ELSE 0 END) AS float) / 1000.0 + FROM real_delta AS rd +) +SELECT TOP 5 + query_hash = CONVERT(VARCHAR(18), w.query_hash, 1), + database_name = + ( + SELECT TOP (1) rd2.database_name + FROM real_delta AS rd2 + WHERE rd2.query_hash = w.query_hash + ORDER BY rd2.collection_time DESC + ), + current_cpu_per_exec_ms = w.current_worker_ms / NULLIF(w.current_execs, 0), + baseline_cpu_per_exec_ms = w.baseline_worker_ms / NULLIF(w.baseline_execs, 0), + anomaly_ratio = + (w.current_worker_ms / NULLIF(w.current_execs, 0)) / + NULLIF(w.baseline_worker_ms / NULLIF(w.baseline_execs, 0), 0), + execution_count = w.current_execs, + total_cpu_ms = w.current_total_cpu_ms, + /* LOW-1: real share of the window's total query CPU (rounded int %), 0 when the window + total is non-positive (degenerate). Display-only — carried into the disclosure. */ + cpu_percent = + CONVERT(int, ROUND(100.0 * w.current_total_cpu_ms / NULLIF(wt.total_cpu_ms, 0), 0)), + latest_plan_handle = + ( + SELECT TOP (1) CONVERT(VARCHAR(130), rd3.plan_handle, 1) + FROM real_delta AS rd3 + WHERE rd3.query_hash = w.query_hash + AND rd3.plan_handle IS NOT NULL + ORDER BY rd3.collection_time DESC + ), + query_text = + ( + SELECT TOP (1) LEFT(CAST(DECOMPRESS(rd4.query_text) AS NVARCHAR(MAX)), 500) + FROM real_delta AS rd4 + WHERE rd4.query_hash = w.query_hash + ORDER BY rd4.collection_time DESC + ) +FROM windowed AS w +CROSS JOIN window_total AS wt +WHERE w.current_execs > 0 +AND w.baseline_execs > 0 +AND w.baseline_worker_ms > 0 +AND w.current_total_cpu_ms >= @materialFloor +/* the anomaly gate: current per-exec >= T x baseline per-exec */ +AND (w.current_worker_ms / NULLIF(w.current_execs, 0)) >= + @threshold * (w.baseline_worker_ms / NULLIF(w.baseline_execs, 0)) +ORDER BY w.current_total_cpu_ms DESC;"; + + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + cmd.Parameters.Add(new SqlParameter("@baselineDays", 7)); + cmd.Parameters.Add(new SqlParameter("@threshold", AnomalyThreshold)); + cmd.Parameters.Add(new SqlParameter("@materialFloor", MaterialCpuMsFloor)); + + var items = new List(); + using var reader = await cmd.ExecuteReaderAsync(); + while (await reader.ReadAsync()) + { + items.Add(new + { + query_hash = reader.IsDBNull(0) ? "" : reader.GetString(0), + database = reader.IsDBNull(1) ? "" : reader.GetString(1), + current_cpu_per_exec_ms = reader.IsDBNull(2) ? 0.0 : Convert.ToDouble(reader.GetValue(2)), + baseline_cpu_per_exec_ms = reader.IsDBNull(3) ? 0.0 : Convert.ToDouble(reader.GetValue(3)), + anomaly_ratio = reader.IsDBNull(4) ? 0.0 : Convert.ToDouble(reader.GetValue(4)), + execution_count = reader.IsDBNull(5) ? 0L : Convert.ToInt64(reader.GetValue(5)), + total_cpu_ms = reader.IsDBNull(6) ? 0.0 : Convert.ToDouble(reader.GetValue(6)), + // LOW-1 fix: the REAL window CPU share (was hardcoded 0 in PR-A, which made + // the disclosure render "responsible for 0% of CPU" and understate the risk). + cpu_percent = reader.IsDBNull(7) ? 0 : Convert.ToInt32(reader.GetValue(7)), + latest_plan_handle = reader.IsDBNull(8) ? "" : reader.GetString(8), + query_text = reader.IsDBNull(9) ? "" : reader.GetString(9), + // §2b co-fired flags (drive the §5 disclosure steer); display-only. + plan_regression_cofired = planRegressionCoFired, + parameter_sensitivity_cofired = parameterSensitivityCoFired + }); + } + + if (items.Count > 0) + finding.DrillDown!["abnormal_cpu_plans"] = items; + } + + private async Task CollectTopSpillingQueries(AnalysisFinding finding, AnalysisContext context) + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT TOP 5 + database_name, + CONVERT(VARCHAR(18), query_hash, 1) AS query_hash, + CAST(SUM(total_spills) AS BIGINT) AS total_spills, + CAST(SUM(execution_count_delta) AS BIGINT) AS exec_count, + LEFT(CAST(DECOMPRESS(MAX(query_text)) AS NVARCHAR(MAX)), 500) AS query_text +FROM collect.query_stats +WHERE collection_time >= @startTime AND collection_time <= @endTime +AND total_spills > 0 +GROUP BY database_name, query_hash +ORDER BY CAST(SUM(total_spills) AS BIGINT) DESC;"; + + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + var items = new List(); + using var reader = await cmd.ExecuteReaderAsync(); + while (await reader.ReadAsync()) + { + items.Add(new + { + database = reader.IsDBNull(0) ? "" : reader.GetString(0), + query_hash = reader.IsDBNull(1) ? "" : reader.GetString(1), + total_spills = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)), + execution_count = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)), + query_text = reader.IsDBNull(4) ? "" : reader.GetString(4) + }); + } + + if (items.Count > 0) + finding.DrillDown!["top_spilling_queries"] = items; + } + + /// + /// Top parameter-sensitive plans behind a PARAMETER_SENSITIVITY finding. + /// Re-runs the detector for the top 5 offenders. + /// + private async Task CollectParameterSensitiveQueries(AnalysisFinding finding, AnalysisContext context) + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +WITH latest AS +( + SELECT + database_name, + query_hash, + query_plan_hash, + execution_count, + creation_time, + min_worker_time, + max_worker_time, + min_grant_kb, + max_grant_kb, + min_spills, + max_spills, + query_text, + ROW_NUMBER() OVER + ( + PARTITION BY database_name, query_hash, query_plan_hash + ORDER BY collection_time DESC + ) AS rn + FROM collect.query_stats + WHERE collection_time >= @startTime + AND collection_time <= @endTime + AND execution_count_delta > 0 +) +SELECT + database_name, + CONVERT(varchar(18), query_hash, 1) AS query_hash, + CONVERT(varchar(18), query_plan_hash, 1) AS query_plan_hash, + execution_count, + min_worker_time, + max_worker_time, + CAST(max_worker_time AS float) / NULLIF(min_worker_time, 0) AS worker_ratio, + CAST(max_grant_kb AS float) / NULLIF(min_grant_kb, 0) AS grant_ratio, + CASE WHEN max_spills > 0 AND min_spills = 0 THEN 1 ELSE 0 END AS spill_divergence, + LEFT(CAST(DECOMPRESS(query_text) AS NVARCHAR(MAX)), 500) AS query_text +FROM latest +WHERE rn = 1 +AND min_worker_time >= 10000 +AND max_worker_time >= 250000 +AND execution_count >= 20 +AND creation_time <= @startTime +AND CAST(max_worker_time AS float) / NULLIF(min_worker_time, 0) >= 10 +ORDER BY worker_ratio DESC +OFFSET 0 ROWS FETCH NEXT 5 ROWS ONLY"; + + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + var items = new List(); + using var reader = await cmd.ExecuteReaderAsync(); + while (await reader.ReadAsync()) + { + items.Add(new + { + database = reader.IsDBNull(0) ? "" : reader.GetString(0), + query_hash = reader.IsDBNull(1) ? "" : reader.GetString(1), + query_plan_hash = reader.IsDBNull(2) ? "" : reader.GetString(2), + execution_count = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)), + min_worker_time_us = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)), + max_worker_time_us = reader.IsDBNull(5) ? 0L : Convert.ToInt64(reader.GetValue(5)), + worker_ratio = reader.IsDBNull(6) ? 0.0 : Convert.ToDouble(reader.GetValue(6)), + grant_ratio = reader.IsDBNull(7) ? 0.0 : Convert.ToDouble(reader.GetValue(7)), + spills_on_some_inputs = !reader.IsDBNull(8) && Convert.ToInt32(reader.GetValue(8)) == 1, + query_text = reader.IsDBNull(9) ? "" : reader.GetString(9) + }); + } + + if (items.Count > 0) + finding.DrillDown!["parameter_sensitive_queries"] = items; + } + + /// + /// Top regressed queries behind a PLAN_REGRESSION finding. + /// Uses the same 14-day server_last_execution_time comparison window as the detector + /// (NOT the standard analysis window) so the days-old "best plan" baseline is present. + /// + private async Task CollectRegressedQueries(AnalysisFinding finding, AnalysisContext context) + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +WITH deduped AS +( + SELECT + database_name, + query_id, + plan_id, + query_plan_hash, + count_executions, + avg_cpu_time, + avg_duration, + server_last_execution_time, + ROW_NUMBER() OVER + ( + PARTITION BY database_name, query_id, plan_id, server_first_execution_time + ORDER BY collection_time DESC + ) AS rn + FROM collect.query_store_data + WHERE execution_type_desc = N'Regular' + AND server_last_execution_time >= @windowStart +), +plan_agg AS +( + -- query_plan_hash is invariant within a plan_id, so include it in the GROUP BY + -- (MS Learn's MAX page does not list binary/varbinary in the accepted types). + SELECT + database_name, + query_id, + plan_id, + query_plan_hash, + SUM(count_executions) AS execs, + CASE WHEN SUM(count_executions) > 0 + THEN SUM(avg_cpu_time * count_executions) / NULLIF(SUM(count_executions), 0) + ELSE 0 END AS cpu_per_exec, + CASE WHEN SUM(count_executions) > 0 + THEN SUM(avg_duration * count_executions) / NULLIF(SUM(count_executions), 0) + ELSE 0 END AS dur_per_exec, + MAX(server_last_execution_time) AS last_exec + FROM deduped + WHERE rn = 1 + GROUP BY database_name, query_id, plan_id, query_plan_hash +), +plan_dedup AS +( + -- MAX(plan_id) carries the most recently observed plan_id in the hash partition + -- forward — newer plans are less likely evicted by Query Store retention. + -- Functionally any plan_id sharing the hash forces the same execution shape. + SELECT + database_name, + query_id, + query_plan_hash, + MAX(plan_id) AS plan_id, + SUM(execs) AS execs, + CASE WHEN SUM(execs) > 0 + THEN SUM(cpu_per_exec * execs) / NULLIF(SUM(execs), 0) + ELSE 0 END AS cpu_per_exec, + CASE WHEN SUM(execs) > 0 + THEN SUM(dur_per_exec * execs) / NULLIF(SUM(execs), 0) + ELSE 0 END AS dur_per_exec, + MAX(last_exec) AS last_exec + FROM plan_agg + GROUP BY database_name, query_id, query_plan_hash + HAVING SUM(execs) >= 25 +), +ranked AS +( + SELECT + *, + ROW_NUMBER() OVER (PARTITION BY database_name, query_id ORDER BY last_exec DESC) AS recency, + ROW_NUMBER() OVER (PARTITION BY database_name, query_id ORDER BY cpu_per_exec ASC) AS cheapness + FROM plan_dedup +), +compared AS +( + SELECT + l.database_name, + l.query_id, + l.query_plan_hash AS latest_plan_hash, + l.cpu_per_exec AS latest_cpu, + l.dur_per_exec AS latest_dur, + b.query_plan_hash AS best_plan_hash, + b.plan_id AS best_plan_id, + b.cpu_per_exec AS best_cpu, + b.dur_per_exec AS best_dur, + (SELECT MAX(v) + FROM (VALUES + (CAST(l.cpu_per_exec AS float) / NULLIF(b.cpu_per_exec, 0)), + (CAST(l.dur_per_exec AS float) / NULLIF(b.dur_per_exec, 0)) + ) AS x(v)) AS regression_factor + FROM ranked AS l + JOIN ranked AS b + ON b.database_name = l.database_name + AND b.query_id = l.query_id + AND b.cheapness = 1 + WHERE l.recency = 1 + AND l.query_plan_hash <> b.query_plan_hash +) +SELECT + c.database_name, + c.query_id, + CONVERT(varchar(18), c.latest_plan_hash, 1) AS latest_plan_hash, + c.latest_cpu, + c.latest_dur, + CONVERT(varchar(18), c.best_plan_hash, 1) AS best_plan_hash, + c.best_plan_id, + c.best_cpu, + c.best_dur, + c.regression_factor, + -- query_sql_text is varbinary(max); fetch it via APPLY (MAX() on varbinary(max) is invalid). + LEFT(CAST(DECOMPRESS(qt.query_sql_text) AS NVARCHAR(MAX)), 500) AS query_text +FROM compared AS c +OUTER APPLY +( + SELECT TOP (1) qs.query_sql_text + FROM collect.query_store_data AS qs + WHERE qs.database_name = c.database_name + AND qs.query_id = c.query_id + AND qs.query_plan_hash = c.latest_plan_hash + AND qs.server_last_execution_time >= @windowStart + ORDER BY qs.server_last_execution_time DESC +) AS qt +WHERE c.regression_factor >= 2 +ORDER BY c.regression_factor DESC +OFFSET 0 ROWS FETCH NEXT 5 ROWS ONLY"; + + cmd.Parameters.Add(new SqlParameter("@windowStart", context.TimeRangeStart.AddDays(-14))); + + var items = new List(); + using var reader = await cmd.ExecuteReaderAsync(); + while (await reader.ReadAsync()) + { + items.Add(new + { + database = reader.IsDBNull(0) ? "" : reader.GetString(0), + query_id = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)), + latest_plan_hash = reader.IsDBNull(2) ? "" : reader.GetString(2), + latest_cpu_per_exec_us = reader.IsDBNull(3) ? 0.0 : Convert.ToDouble(reader.GetValue(3)), + latest_duration_per_exec_us = reader.IsDBNull(4) ? 0.0 : Convert.ToDouble(reader.GetValue(4)), + best_plan_hash = reader.IsDBNull(5) ? "" : reader.GetString(5), + best_plan_id = reader.IsDBNull(6) ? 0L : Convert.ToInt64(reader.GetValue(6)), + best_cpu_per_exec_us = reader.IsDBNull(7) ? 0.0 : Convert.ToDouble(reader.GetValue(7)), + best_duration_per_exec_us = reader.IsDBNull(8) ? 0.0 : Convert.ToDouble(reader.GetValue(8)), + regression_factor = reader.IsDBNull(9) ? 0.0 : Convert.ToDouble(reader.GetValue(9)), + query_text = reader.IsDBNull(10) ? "" : reader.GetString(10) + }); + } + + if (items.Count > 0) + finding.DrillDown!["regressed_queries"] = items; + } + + private async Task CollectBadActorDetail(AnalysisFinding finding, AnalysisContext context) + { + // Extract query_hash from the fact key (BAD_ACTOR_0x...) + var queryHash = finding.RootFactKey.Replace("BAD_ACTOR_", ""); + if (string.IsNullOrEmpty(queryHash)) return; + + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT + database_name, + CONVERT(VARCHAR(18), query_hash, 1) AS query_hash, + LEFT(CAST(DECOMPRESS(MAX(query_text)) AS NVARCHAR(MAX)), 500) AS query_text, + CAST(SUM(execution_count_delta) AS BIGINT) AS exec_count, + CASE WHEN SUM(execution_count_delta) > 0 + THEN CAST(SUM(total_worker_time_delta) AS FLOAT) / SUM(execution_count_delta) / 1000.0 + ELSE 0 END AS avg_cpu_ms, + CASE WHEN SUM(execution_count_delta) > 0 + THEN CAST(SUM(total_elapsed_time_delta) AS FLOAT) / SUM(execution_count_delta) / 1000.0 + ELSE 0 END AS avg_elapsed_ms, + CASE WHEN SUM(execution_count_delta) > 0 + THEN CAST(SUM(total_logical_reads_delta) AS FLOAT) / SUM(execution_count_delta) + ELSE 0 END AS avg_reads, + CAST(SUM(total_worker_time_delta) AS BIGINT) AS total_cpu_us, + CAST(SUM(total_logical_reads_delta) AS BIGINT) AS total_reads, + CAST(SUM(total_spills) AS BIGINT) AS total_spills, + MAX(max_dop) AS max_dop +FROM collect.query_stats +WHERE collection_time >= @startTime +AND collection_time <= @endTime +AND query_hash = CONVERT(BINARY(8), @queryHash, 1) +GROUP BY database_name, query_hash;"; + + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + cmd.Parameters.Add(new SqlParameter("@queryHash", queryHash)); + + using var reader = await cmd.ExecuteReaderAsync(); + if (await reader.ReadAsync()) + { + finding.DrillDown!["bad_actor_query"] = new + { + database = reader.IsDBNull(0) ? "" : reader.GetString(0), + query_hash = reader.IsDBNull(1) ? "" : reader.GetString(1), + query_text = reader.IsDBNull(2) ? "" : reader.GetString(2), + execution_count = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)), + avg_cpu_ms = reader.IsDBNull(4) ? 0.0 : Math.Round(Convert.ToDouble(reader.GetValue(4)), 2), + avg_elapsed_ms = reader.IsDBNull(5) ? 0.0 : Math.Round(Convert.ToDouble(reader.GetValue(5)), 2), + avg_reads = reader.IsDBNull(6) ? 0.0 : Math.Round(Convert.ToDouble(reader.GetValue(6)), 0), + total_cpu_ms = reader.IsDBNull(7) ? 0.0 : Convert.ToDouble(reader.GetValue(7)) / 1000.0, + total_reads = reader.IsDBNull(8) ? 0L : Convert.ToInt64(reader.GetValue(8)), + total_spills = reader.IsDBNull(9) ? 0L : Convert.ToInt64(reader.GetValue(9)), + max_dop = reader.IsDBNull(10) ? 0 : Convert.ToInt32(reader.GetValue(10)) + }; + } + } + + private async Task CollectPendingGrants(AnalysisFinding finding, AnalysisContext context) + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT TOP 5 + collection_time, + target_memory_mb, total_memory_mb, available_memory_mb, + granted_memory_mb, used_memory_mb, + grantee_count, waiter_count, + timeout_error_count_delta, forced_grant_count_delta +FROM collect.memory_grant_stats +WHERE collection_time >= @startTime AND collection_time <= @endTime +AND waiter_count > 0 +ORDER BY waiter_count DESC;"; + + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + var items = new List(); + using var reader = await cmd.ExecuteReaderAsync(); + while (await reader.ReadAsync()) + { + items.Add(new + { + time = reader.IsDBNull(0) ? "" : reader.GetDateTime(0).ToString("o"), + target_memory_mb = reader.IsDBNull(1) ? 0.0 : Convert.ToDouble(reader.GetValue(1)), + total_memory_mb = reader.IsDBNull(2) ? 0.0 : Convert.ToDouble(reader.GetValue(2)), + available_memory_mb = reader.IsDBNull(3) ? 0.0 : Convert.ToDouble(reader.GetValue(3)), + granted_memory_mb = reader.IsDBNull(4) ? 0.0 : Convert.ToDouble(reader.GetValue(4)), + used_memory_mb = reader.IsDBNull(5) ? 0.0 : Convert.ToDouble(reader.GetValue(5)), + grantee_count = reader.IsDBNull(6) ? 0 : reader.GetInt32(6), + waiter_count = reader.IsDBNull(7) ? 0 : reader.GetInt32(7), + timeout_errors = reader.IsDBNull(8) ? 0L : Convert.ToInt64(reader.GetValue(8)), + forced_grants = reader.IsDBNull(9) ? 0L : Convert.ToInt64(reader.GetValue(9)) + }); + } + + if (items.Count > 0) + finding.DrillDown!["pending_grants"] = items; + } +} diff --git a/Dashboard/Analysis/SqlServerDrillDownCollector.Storage.cs b/Dashboard/Analysis/SqlServerDrillDownCollector.Storage.cs new file mode 100644 index 000000000..70b9da181 --- /dev/null +++ b/Dashboard/Analysis/SqlServerDrillDownCollector.Storage.cs @@ -0,0 +1,176 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading.Tasks; +using Microsoft.Data.SqlClient; +using PerformanceMonitor.Analysis; +using PerformanceMonitor.PlanAnalysis; +using PerformanceMonitorDashboard.Helpers; +using PerformanceMonitorDashboard.Mcp; +using PerformanceMonitorDashboard.Models; +using PerformanceMonitorDashboard.Services; +using PerformanceMonitor.Common; +using PerformanceMonitor.Notifications; + +namespace PerformanceMonitorDashboard.Analysis; + +public partial class SqlServerDrillDownCollector +{ + private async Task CollectFileLatencyBreakdown(AnalysisFinding finding, AnalysisContext context) + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT TOP 10 + database_name, + file_type_desc AS file_type, + AVG(io_stall_read_ms_delta * 1.0 / NULLIF(num_of_reads_delta, 0)) AS avg_read_ms, + AVG(io_stall_write_ms_delta * 1.0 / NULLIF(num_of_writes_delta, 0)) AS avg_write_ms, + CAST(SUM(num_of_reads_delta) AS BIGINT) AS total_reads, + CAST(SUM(num_of_writes_delta) AS BIGINT) AS total_writes +FROM collect.file_io_stats +WHERE collection_time >= @startTime AND collection_time <= @endTime +AND (num_of_reads_delta > 0 OR num_of_writes_delta > 0) +GROUP BY database_name, file_type_desc +ORDER BY AVG(io_stall_read_ms_delta * 1.0 / NULLIF(num_of_reads_delta, 0)) DESC;"; + + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + var items = new List(); + using var reader = await cmd.ExecuteReaderAsync(); + while (await reader.ReadAsync()) + { + items.Add(new + { + database = reader.IsDBNull(0) ? "" : reader.GetString(0), + file_type = reader.IsDBNull(1) ? "" : reader.GetString(1), + avg_read_latency_ms = reader.IsDBNull(2) ? 0.0 : Math.Round(Convert.ToDouble(reader.GetValue(2)), 2), + avg_write_latency_ms = reader.IsDBNull(3) ? 0.0 : Math.Round(Convert.ToDouble(reader.GetValue(3)), 2), + total_reads = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)), + total_writes = reader.IsDBNull(5) ? 0L : Convert.ToInt64(reader.GetValue(5)) + }); + } + + if (items.Count > 0) + finding.DrillDown!["file_latency_breakdown"] = items; + } + + /// + /// Lists the large (>= 10 GB) data/log files on PERCENTAGE autogrowth (WS3), latest + /// snapshot per file, excluding system databases — and attaches a copy-paste + /// ALTER DATABASE ... MODIFY FILE fix per file (FILEGROWTH set to a size-tiered + /// fixed MB). The structured fields (database, logical_file_name, + /// total_size_mb, growth_pct) are what the shared extractor + /// (FactRemediation.ExtractFileGrowthTargets) reads; the rendered alter_statement + /// uses the SHARED renderer so it is byte-identical to the reader's copy-paste rebuild. + /// Advisory only — no Apply. + /// + private async Task CollectAutogrowthPercentFiles(AnalysisFinding finding, AnalysisContext context) + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +;WITH latest AS ( + SELECT + database_name, + file_id, + file_type_desc, + file_name, + total_size_mb, + is_percent_growth, + growth_pct, + ROW_NUMBER() OVER (PARTITION BY database_name, file_id ORDER BY collection_time DESC) AS rn + FROM collect.database_size_stats + WHERE database_name NOT IN ('master', 'msdb', 'model', 'tempdb') +) +SELECT TOP (50) + database_name, + file_type_desc, + file_name, + total_size_mb, + growth_pct +FROM latest +WHERE rn = 1 +AND is_percent_growth = 1 +AND total_size_mb >= @minSizeMb +ORDER BY total_size_mb DESC;"; + + cmd.Parameters.Add(new SqlParameter("@minSizeMb", 10240.0)); /* 10 GB */ + + var items = new List(); + using var reader = await cmd.ExecuteReaderAsync(); + while (await reader.ReadAsync()) + { + var database = reader.IsDBNull(0) ? "" : reader.GetString(0); + var fileType = reader.IsDBNull(1) ? "" : reader.GetString(1); + var logical = reader.IsDBNull(2) ? "" : reader.GetString(2); + var sizeMb = reader.IsDBNull(3) ? 0.0 : Convert.ToDouble(reader.GetValue(3)); + var growthPct = reader.IsDBNull(4) ? 0 : Convert.ToInt32(reader.GetValue(4)); + if (string.IsNullOrEmpty(database) || string.IsNullOrEmpty(logical)) continue; + + var growthMb = FactRemediation.RecommendedGrowthMbFor(fileType); + items.Add(new + { + database, + logical_file_name = logical, + file_type = fileType, + total_size_mb = sizeMb, + growth_pct = growthPct, + issue = $"{growthPct}% autogrowth on {sizeMb / 1024.0:N1} GB {fileType} file", + alter_statement = FactRemediation.BuildModifyFileStatement(database, logical, growthMb) + }); + } + + if (items.Count > 0) + finding.DrillDown!["autogrowth_percent_files"] = items; + } + + private async Task CollectTempDbBreakdown(AnalysisFinding finding, AnalysisContext context) + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT TOP 5 + collection_time, + user_object_reserved_mb, + internal_object_reserved_mb, + version_store_reserved_mb, + unallocated_mb +FROM collect.tempdb_stats +WHERE collection_time >= @startTime AND collection_time <= @endTime +ORDER BY (user_object_reserved_mb + internal_object_reserved_mb + version_store_reserved_mb) DESC;"; + + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + var items = new List(); + using var reader = await cmd.ExecuteReaderAsync(); + while (await reader.ReadAsync()) + { + items.Add(new + { + time = reader.GetDateTime(0).ToString("o"), + user_objects_mb = reader.IsDBNull(1) ? 0.0 : Convert.ToDouble(reader.GetValue(1)), + internal_objects_mb = reader.IsDBNull(2) ? 0.0 : Convert.ToDouble(reader.GetValue(2)), + version_store_mb = reader.IsDBNull(3) ? 0.0 : Convert.ToDouble(reader.GetValue(3)), + unallocated_mb = reader.IsDBNull(4) ? 0.0 : Convert.ToDouble(reader.GetValue(4)) + }); + } + + if (items.Count > 0) + finding.DrillDown!["tempdb_breakdown"] = items; + } +} diff --git a/Dashboard/Analysis/SqlServerDrillDownCollector.cs b/Dashboard/Analysis/SqlServerDrillDownCollector.cs index f3a3ba408..3e9a87a06 100644 --- a/Dashboard/Analysis/SqlServerDrillDownCollector.cs +++ b/Dashboard/Analysis/SqlServerDrillDownCollector.cs @@ -10,6 +10,7 @@ using PerformanceMonitorDashboard.Models; using PerformanceMonitorDashboard.Services; using PerformanceMonitor.Common; +using PerformanceMonitor.Notifications; namespace PerformanceMonitorDashboard.Analysis; @@ -24,7 +25,7 @@ namespace PerformanceMonitorDashboard.Analysis; /// Port of Lite's DrillDownCollector -- uses SQL Server collect.* tables instead of DuckDB views. /// No server_id filtering -- Dashboard monitors one server per database. /// -public class SqlServerDrillDownCollector +public partial class SqlServerDrillDownCollector { private readonly string _connectionString; private readonly IPlanFetcher? _planFetcher; @@ -150,849 +151,6 @@ THIS finding is emitted (the engine roots one CONFIG_* fact per finding). */ } } - private async Task CollectTopDeadlocks(AnalysisFinding finding, AnalysisContext context) - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT TOP 3 - collection_time, - event_date, - spid, - LEFT(CAST(query AS NVARCHAR(MAX)), 500) AS victim_sql -FROM collect.deadlocks -WHERE collection_time >= @startTime AND collection_time <= @endTime -ORDER BY collection_time DESC;"; - - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - var items = new List(); - using var reader = await cmd.ExecuteReaderAsync(); - while (await reader.ReadAsync()) - { - items.Add(new - { - time = reader.IsDBNull(0) ? "" : reader.GetDateTime(0).ToString("o"), - deadlock_time = reader.IsDBNull(1) ? "" : reader.GetDateTime(1).ToString("o"), - victim = reader.IsDBNull(2) ? "" : reader.GetValue(2).ToString(), - victim_sql = reader.IsDBNull(3) ? "" : reader.GetString(3) - }); - } - - if (items.Count > 0) - finding.DrillDown!["top_deadlocks"] = items; - } - - private async Task CollectTopBlockingChains(AnalysisFinding finding, AnalysisContext context) - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT TOP 5 - collection_time, - database_name, - spid AS blocked_spid, - 0 AS blocking_spid, - wait_time_ms, - lock_mode, - LEFT(CAST(query_text AS NVARCHAR(MAX)), 500) AS blocked_sql, - LEFT(blocking_tree, 500) AS blocking_sql -FROM collect.blocking_BlockedProcessReport -WHERE collection_time >= @startTime AND collection_time <= @endTime -ORDER BY wait_time_ms DESC;"; - - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - var items = new List(); - using var reader = await cmd.ExecuteReaderAsync(); - while (await reader.ReadAsync()) - { - items.Add(new - { - time = reader.IsDBNull(0) ? "" : reader.GetDateTime(0).ToString("o"), - database = reader.IsDBNull(1) ? "" : reader.GetString(1), - blocked_spid = reader.IsDBNull(2) ? 0 : Convert.ToInt32(reader.GetValue(2)), - blocking_spid = reader.IsDBNull(3) ? 0 : Convert.ToInt32(reader.GetValue(3)), - wait_time_ms = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)), - lock_mode = reader.IsDBNull(5) ? "" : reader.GetString(5), - blocked_sql = reader.IsDBNull(6) ? "" : reader.GetString(6), - blocking_sql = reader.IsDBNull(7) ? "" : reader.GetString(7) - }); - } - - if (items.Count > 0) - finding.DrillDown!["top_blocking_chains"] = items; - } - - private async Task CollectQueriesAtSpike(AnalysisFinding finding, AnalysisContext context) - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - // Check if query_snapshots table exists (created dynamically by sp_WhoIsActive) - using var checkCmd = connection.CreateCommand(); - checkCmd.CommandText = "SELECT OBJECT_ID(N'collect.query_snapshots', N'U')"; - var tableExists = await checkCmd.ExecuteScalarAsync(); - if (tableExists == null || tableExists == DBNull.Value) return; - - // Step 1: Find when the spike occurred - using var peakCmd = connection.CreateCommand(); - peakCmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT TOP 1 collection_time, sqlserver_cpu_utilization -FROM collect.cpu_utilization_stats -WHERE collection_time >= @startTime AND collection_time <= @endTime -ORDER BY sqlserver_cpu_utilization DESC;"; - - peakCmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - peakCmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - DateTime? peakTime = null; - int peakCpu = 0; - using (var peakReader = await peakCmd.ExecuteReaderAsync()) - { - if (await peakReader.ReadAsync()) - { - peakTime = peakReader.GetDateTime(0); - peakCpu = peakReader.GetInt32(1); - } - } - - if (peakTime == null) return; - - // Step 2: Get queries active within 2 minutes of peak - using var queryCmd = connection.CreateCommand(); - queryCmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT TOP 5 - collection_time, - [session_id], - [database_name], - [status], - DATEDIFF(MILLISECOND, 0, [CPU]) AS cpu_time_ms, - DATEDIFF(MILLISECOND, 0, [elapsed_time]) AS total_elapsed_time_ms, - [reads] AS logical_reads, - [wait_info] AS wait_type, - 0 AS dop, - 0 AS parallel_worker_count, - LEFT(CAST([sql_text] AS NVARCHAR(MAX)), 500) AS query_text -FROM collect.query_snapshots -WHERE collection_time >= @spikeStart -AND collection_time <= @spikeEnd -AND CAST([sql_text] AS NVARCHAR(MAX)) NOT LIKE 'WAITFOR%' -ORDER BY DATEDIFF(MILLISECOND, 0, [CPU]) DESC;"; - - queryCmd.Parameters.Add(new SqlParameter("@spikeStart", peakTime.Value.AddMinutes(-2))); - queryCmd.Parameters.Add(new SqlParameter("@spikeEnd", peakTime.Value.AddMinutes(2))); - - var items = new List(); - using (var reader = await queryCmd.ExecuteReaderAsync()) - { - while (await reader.ReadAsync()) - { - items.Add(new - { - time = reader.IsDBNull(0) ? "" : reader.GetDateTime(0).ToString("o"), - session_id = reader.IsDBNull(1) ? 0 : Convert.ToInt32(reader.GetValue(1)), - database = reader.IsDBNull(2) ? "" : reader.GetString(2), - status = reader.IsDBNull(3) ? "" : reader.GetString(3), - cpu_time_ms = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)), - elapsed_time_ms = reader.IsDBNull(5) ? 0L : Convert.ToInt64(reader.GetValue(5)), - logical_reads = reader.IsDBNull(6) ? 0L : Convert.ToInt64(reader.GetValue(6)), - wait_type = reader.IsDBNull(7) ? "" : reader.GetString(7), - dop = reader.IsDBNull(8) ? 0 : Convert.ToInt32(reader.GetValue(8)), - parallel_workers = reader.IsDBNull(9) ? 0 : Convert.ToInt32(reader.GetValue(9)), - query_text = reader.IsDBNull(10) ? "" : reader.GetString(10) - }); - } - } - - if (items.Count > 0) - { - finding.DrillDown!["spike_peak"] = new - { - time = peakTime.Value.ToString("o"), - cpu_percent = peakCpu - }; - finding.DrillDown!["queries_at_spike"] = items; - } - } - - private async Task CollectTopCpuQueries(AnalysisFinding finding, AnalysisContext context) - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT TOP 5 - database_name, - CONVERT(VARCHAR(18), query_hash, 1) AS query_hash, - CAST(SUM(total_worker_time_delta) AS BIGINT) AS total_cpu_us, - CAST(SUM(execution_count_delta) AS BIGINT) AS exec_count, - MAX(max_dop) AS max_dop, - CAST(SUM(total_spills) AS BIGINT) AS spills, - LEFT(CAST(DECOMPRESS(MAX(query_text)) AS NVARCHAR(MAX)), 500) AS query_text -FROM collect.query_stats -WHERE collection_time >= @startTime AND collection_time <= @endTime -AND total_worker_time_delta > 0 -GROUP BY database_name, query_hash -ORDER BY CAST(SUM(total_worker_time_delta) AS BIGINT) DESC;"; - - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - var items = new List(); - using var reader = await cmd.ExecuteReaderAsync(); - while (await reader.ReadAsync()) - { - items.Add(new - { - database = reader.IsDBNull(0) ? "" : reader.GetString(0), - query_hash = reader.IsDBNull(1) ? "" : reader.GetString(1), - total_cpu_ms = reader.IsDBNull(2) ? 0.0 : Convert.ToDouble(reader.GetValue(2)) / 1000.0, - execution_count = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)), - max_dop = reader.IsDBNull(4) ? 0 : Convert.ToInt32(reader.GetValue(4)), - spills = reader.IsDBNull(5) ? 0L : Convert.ToInt64(reader.GetValue(5)), - query_text = reader.IsDBNull(6) ? "" : reader.GetString(6) - }); - } - - if (items.Count > 0 && !finding.DrillDown!.ContainsKey("top_cpu_queries")) - finding.DrillDown!["top_cpu_queries"] = items; - } - - /// - /// The plan-cache anomaly detector (§2): one row per offending query_hash whose - /// CURRENT per-exec CPU has jumped to an abnormal multiple of its OWN trailing baseline, - /// and which is a material CPU contributor in the window. This is the enrichment the - /// (PR-B) "Clear cached plan (advanced)" affordance reads. PR-A emits the drill-down but - /// does NOT register the handler / emit the affordance — dead-code-safe display only. - /// - /// - /// §2a ROW-LEVEL exclusion (the round-2 correctness fix): the delta framework - /// (install/05_delta_framework.sql:218-307) assigns delta = full cumulative raw - /// total on TWO arms — first-collection-of-a-plan_handle (the pc.collection_id - /// IS NULL arm, :233-235) and the first-post-restart row (the - /// server_start_time >= pc.collection_time arm, :235-236). Both inject false - /// anomalies on exactly this feature's target population. We exclude BOTH, per row, in - /// BOTH the current and baseline windows: a row counts as a REAL prior-delta row only - /// when an earlier collection exists for the same (sql_handle, offsets, plan_handle) - /// AND that earlier collection_time is > this row's server_start_time (i.e. the - /// delta interval started at a real prior collection, not at compile/restart). The - /// per-exec math (M-3) and the materiality CPU sum use ONLY these real-delta rows. - /// - /// - private async Task CollectAbnormalCpuPlans(AnalysisFinding finding, AnalysisContext context, HashSet pathKeys) - { - // The anomaly threshold (sibling of PLAN_REGRESSION's regression_factor) and the - // materiality floor (a query must contribute at least this much CPU in the window - // for clearing to be worth offering). Conservative defaults — start ~3x. - const double AnomalyThreshold = 3.0; - const double MaterialCpuMsFloor = 1000.0; // 1s of CPU in the window - - // Co-fired-fact awareness (drives the §5 disclosure steer): whether this CPU - // finding's story crossed PLAN_REGRESSION / PARAMETER_SENSITIVITY. The analysis - // already holds the story path — no extra SQL. - var planRegressionCoFired = pathKeys.Contains("PLAN_REGRESSION"); - var parameterSensitivityCoFired = pathKeys.Contains("PARAMETER_SENSITIVITY"); - - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - // The baseline window is the @baselineDays preceding the current window (NOT - // overlapping it). The current window is [@startTime, @endTime]. - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -DECLARE @baselineStart datetime2(7) = DATEADD(DAY, -@baselineDays, @startTime); - -/* -Real-delta rows only (§2a row-level exclusion): a row is a genuine inter-collection -delta when an EARLIER collection exists for the same (sql_handle, offsets, plan_handle) -whose collection_time is BOTH < this row's collection_time (it is a prior) AND -> this row's server_start_time (so the delta interval did not start at compile/restart). -This drops the first-collection-of-a-plan_handle row and the first-post-restart row in -one predicate, by row, without using sample_interval_seconds (R2-MIN-A). -*/ -WITH - real_delta AS -( - SELECT - qs.query_hash, - qs.database_name, - qs.collection_time, - qs.total_worker_time_delta, - qs.execution_count_delta, - qs.plan_handle, - qs.query_text - FROM collect.query_stats AS qs - WHERE qs.query_hash IS NOT NULL - AND qs.collection_time >= @baselineStart - AND qs.collection_time <= @endTime - AND EXISTS - ( - SELECT 1 - FROM collect.query_stats AS prior - WHERE prior.sql_handle = qs.sql_handle - AND prior.statement_start_offset = qs.statement_start_offset - AND prior.statement_end_offset = qs.statement_end_offset - AND prior.plan_handle = qs.plan_handle - AND prior.collection_time < qs.collection_time - AND prior.collection_time > qs.server_start_time - ) -), - windowed AS -( - SELECT - rd.query_hash, - /* current-window per-exec CPU (ms): SUM(worker)/SUM(execs) on real-delta rows */ - current_worker_ms = - CAST(SUM(CASE WHEN rd.collection_time >= @startTime THEN rd.total_worker_time_delta ELSE 0 END) AS float) / 1000.0, - current_execs = - SUM(CASE WHEN rd.collection_time >= @startTime THEN rd.execution_count_delta ELSE 0 END), - /* baseline-window per-exec CPU (ms): the preceding window, same exclusion */ - baseline_worker_ms = - CAST(SUM(CASE WHEN rd.collection_time < @startTime THEN rd.total_worker_time_delta ELSE 0 END) AS float) / 1000.0, - baseline_execs = - SUM(CASE WHEN rd.collection_time < @startTime THEN rd.execution_count_delta ELSE 0 END), - /* window CPU contribution (ms), §2a exclusion applied (R2-MIN-B) */ - current_total_cpu_ms = - CAST(SUM(CASE WHEN rd.collection_time >= @startTime THEN rd.total_worker_time_delta ELSE 0 END) AS float) / 1000.0 - FROM real_delta AS rd - GROUP BY rd.query_hash -), - /* - LOW-1 fix: the window's TOTAL query CPU (ms) over the SAME §2a real-delta rows in the - current window, so each query's cpu_percent is its real share of query CPU over the - window (NOT the hardcoded 0 that understated risk in PR-A). Using the same exclusion - keeps the numerator and denominator consistent (a query can't show a share computed on - contaminated raw-total CPU it won't actually clear, R2-MIN-B). - */ - window_total AS -( - SELECT - total_cpu_ms = - CAST(SUM(CASE WHEN rd.collection_time >= @startTime THEN rd.total_worker_time_delta ELSE 0 END) AS float) / 1000.0 - FROM real_delta AS rd -) -SELECT TOP 5 - query_hash = CONVERT(VARCHAR(18), w.query_hash, 1), - database_name = - ( - SELECT TOP (1) rd2.database_name - FROM real_delta AS rd2 - WHERE rd2.query_hash = w.query_hash - ORDER BY rd2.collection_time DESC - ), - current_cpu_per_exec_ms = w.current_worker_ms / NULLIF(w.current_execs, 0), - baseline_cpu_per_exec_ms = w.baseline_worker_ms / NULLIF(w.baseline_execs, 0), - anomaly_ratio = - (w.current_worker_ms / NULLIF(w.current_execs, 0)) / - NULLIF(w.baseline_worker_ms / NULLIF(w.baseline_execs, 0), 0), - execution_count = w.current_execs, - total_cpu_ms = w.current_total_cpu_ms, - /* LOW-1: real share of the window's total query CPU (rounded int %), 0 when the window - total is non-positive (degenerate). Display-only — carried into the disclosure. */ - cpu_percent = - CONVERT(int, ROUND(100.0 * w.current_total_cpu_ms / NULLIF(wt.total_cpu_ms, 0), 0)), - latest_plan_handle = - ( - SELECT TOP (1) CONVERT(VARCHAR(130), rd3.plan_handle, 1) - FROM real_delta AS rd3 - WHERE rd3.query_hash = w.query_hash - AND rd3.plan_handle IS NOT NULL - ORDER BY rd3.collection_time DESC - ), - query_text = - ( - SELECT TOP (1) LEFT(CAST(DECOMPRESS(rd4.query_text) AS NVARCHAR(MAX)), 500) - FROM real_delta AS rd4 - WHERE rd4.query_hash = w.query_hash - ORDER BY rd4.collection_time DESC - ) -FROM windowed AS w -CROSS JOIN window_total AS wt -WHERE w.current_execs > 0 -AND w.baseline_execs > 0 -AND w.baseline_worker_ms > 0 -AND w.current_total_cpu_ms >= @materialFloor -/* the anomaly gate: current per-exec >= T x baseline per-exec */ -AND (w.current_worker_ms / NULLIF(w.current_execs, 0)) >= - @threshold * (w.baseline_worker_ms / NULLIF(w.baseline_execs, 0)) -ORDER BY w.current_total_cpu_ms DESC;"; - - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - cmd.Parameters.Add(new SqlParameter("@baselineDays", 7)); - cmd.Parameters.Add(new SqlParameter("@threshold", AnomalyThreshold)); - cmd.Parameters.Add(new SqlParameter("@materialFloor", MaterialCpuMsFloor)); - - var items = new List(); - using var reader = await cmd.ExecuteReaderAsync(); - while (await reader.ReadAsync()) - { - items.Add(new - { - query_hash = reader.IsDBNull(0) ? "" : reader.GetString(0), - database = reader.IsDBNull(1) ? "" : reader.GetString(1), - current_cpu_per_exec_ms = reader.IsDBNull(2) ? 0.0 : Convert.ToDouble(reader.GetValue(2)), - baseline_cpu_per_exec_ms = reader.IsDBNull(3) ? 0.0 : Convert.ToDouble(reader.GetValue(3)), - anomaly_ratio = reader.IsDBNull(4) ? 0.0 : Convert.ToDouble(reader.GetValue(4)), - execution_count = reader.IsDBNull(5) ? 0L : Convert.ToInt64(reader.GetValue(5)), - total_cpu_ms = reader.IsDBNull(6) ? 0.0 : Convert.ToDouble(reader.GetValue(6)), - // LOW-1 fix: the REAL window CPU share (was hardcoded 0 in PR-A, which made - // the disclosure render "responsible for 0% of CPU" and understate the risk). - cpu_percent = reader.IsDBNull(7) ? 0 : Convert.ToInt32(reader.GetValue(7)), - latest_plan_handle = reader.IsDBNull(8) ? "" : reader.GetString(8), - query_text = reader.IsDBNull(9) ? "" : reader.GetString(9), - // §2b co-fired flags (drive the §5 disclosure steer); display-only. - plan_regression_cofired = planRegressionCoFired, - parameter_sensitivity_cofired = parameterSensitivityCoFired - }); - } - - if (items.Count > 0) - finding.DrillDown!["abnormal_cpu_plans"] = items; - } - - private async Task CollectTopSpillingQueries(AnalysisFinding finding, AnalysisContext context) - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT TOP 5 - database_name, - CONVERT(VARCHAR(18), query_hash, 1) AS query_hash, - CAST(SUM(total_spills) AS BIGINT) AS total_spills, - CAST(SUM(execution_count_delta) AS BIGINT) AS exec_count, - LEFT(CAST(DECOMPRESS(MAX(query_text)) AS NVARCHAR(MAX)), 500) AS query_text -FROM collect.query_stats -WHERE collection_time >= @startTime AND collection_time <= @endTime -AND total_spills > 0 -GROUP BY database_name, query_hash -ORDER BY CAST(SUM(total_spills) AS BIGINT) DESC;"; - - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - var items = new List(); - using var reader = await cmd.ExecuteReaderAsync(); - while (await reader.ReadAsync()) - { - items.Add(new - { - database = reader.IsDBNull(0) ? "" : reader.GetString(0), - query_hash = reader.IsDBNull(1) ? "" : reader.GetString(1), - total_spills = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)), - execution_count = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)), - query_text = reader.IsDBNull(4) ? "" : reader.GetString(4) - }); - } - - if (items.Count > 0) - finding.DrillDown!["top_spilling_queries"] = items; - } - - private async Task CollectFileLatencyBreakdown(AnalysisFinding finding, AnalysisContext context) - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT TOP 10 - database_name, - file_type_desc AS file_type, - AVG(io_stall_read_ms_delta * 1.0 / NULLIF(num_of_reads_delta, 0)) AS avg_read_ms, - AVG(io_stall_write_ms_delta * 1.0 / NULLIF(num_of_writes_delta, 0)) AS avg_write_ms, - CAST(SUM(num_of_reads_delta) AS BIGINT) AS total_reads, - CAST(SUM(num_of_writes_delta) AS BIGINT) AS total_writes -FROM collect.file_io_stats -WHERE collection_time >= @startTime AND collection_time <= @endTime -AND (num_of_reads_delta > 0 OR num_of_writes_delta > 0) -GROUP BY database_name, file_type_desc -ORDER BY AVG(io_stall_read_ms_delta * 1.0 / NULLIF(num_of_reads_delta, 0)) DESC;"; - - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - var items = new List(); - using var reader = await cmd.ExecuteReaderAsync(); - while (await reader.ReadAsync()) - { - items.Add(new - { - database = reader.IsDBNull(0) ? "" : reader.GetString(0), - file_type = reader.IsDBNull(1) ? "" : reader.GetString(1), - avg_read_latency_ms = reader.IsDBNull(2) ? 0.0 : Math.Round(Convert.ToDouble(reader.GetValue(2)), 2), - avg_write_latency_ms = reader.IsDBNull(3) ? 0.0 : Math.Round(Convert.ToDouble(reader.GetValue(3)), 2), - total_reads = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)), - total_writes = reader.IsDBNull(5) ? 0L : Convert.ToInt64(reader.GetValue(5)) - }); - } - - if (items.Count > 0) - finding.DrillDown!["file_latency_breakdown"] = items; - } - - private async Task CollectLockModeBreakdown(AnalysisFinding finding, AnalysisContext context) - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT TOP 10 - wait_type, - CAST(SUM(wait_time_ms_delta) AS BIGINT) AS total_wait_ms, - CAST(SUM(waiting_tasks_count_delta) AS BIGINT) AS total_count -FROM collect.wait_stats -WHERE collection_time >= @startTime AND collection_time <= @endTime -AND wait_type LIKE 'LCK%' -AND wait_time_ms_delta > 0 -GROUP BY wait_type -ORDER BY CAST(SUM(wait_time_ms_delta) AS BIGINT) DESC;"; - - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - var items = new List(); - using var reader = await cmd.ExecuteReaderAsync(); - while (await reader.ReadAsync()) - { - items.Add(new - { - lock_type = reader.IsDBNull(0) ? "" : reader.GetString(0), - total_wait_ms = reader.IsDBNull(1) ? 0.0 : Convert.ToDouble(reader.GetValue(1)), - waiting_tasks = reader.IsDBNull(2) ? 0.0 : Convert.ToDouble(reader.GetValue(2)) - }); - } - - if (items.Count > 0) - finding.DrillDown!["lock_mode_breakdown"] = items; - } - - private async Task CollectConfigIssues(AnalysisFinding finding, AnalysisContext context) - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - // The Dashboard uses config.database_configuration_history which stores - // settings as rows (setting_type, setting_name, setting_value) not columns. - // Pivot the latest snapshot into the format we need. - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -;WITH latest AS ( - SELECT database_name, setting_name, - CAST(setting_value AS NVARCHAR(256)) AS setting_value, - ROW_NUMBER() OVER (PARTITION BY database_name, setting_name ORDER BY collection_time DESC) AS rn - FROM config.database_configuration_history - WHERE setting_name IN ( - 'recovery_model_desc', 'is_auto_shrink_on', 'is_auto_close_on', - 'is_read_committed_snapshot_on', 'page_verify_option_desc', 'is_query_store_on' - ) -), -pivoted AS ( - SELECT - database_name, - MAX(CASE WHEN setting_name = 'recovery_model_desc' THEN setting_value END) AS recovery_model, - MAX(CASE WHEN setting_name = 'is_auto_shrink_on' THEN setting_value END) AS is_auto_shrink_on, - MAX(CASE WHEN setting_name = 'is_auto_close_on' THEN setting_value END) AS is_auto_close_on, - MAX(CASE WHEN setting_name = 'is_read_committed_snapshot_on' THEN setting_value END) AS is_rcsi_on, - MAX(CASE WHEN setting_name = 'page_verify_option_desc' THEN setting_value END) AS page_verify_option, - MAX(CASE WHEN setting_name = 'is_query_store_on' THEN setting_value END) AS is_query_store_on - FROM latest - WHERE rn = 1 - GROUP BY database_name -) -SELECT database_name, recovery_model, - is_auto_shrink_on, is_auto_close_on, - is_rcsi_on, page_verify_option, is_query_store_on -FROM pivoted -WHERE is_auto_shrink_on = '1' OR is_auto_close_on = '1' - OR is_rcsi_on = '0' OR page_verify_option != 'CHECKSUM' -ORDER BY database_name;"; - - // Read the pivoted rows first into a typed buffer so we can determine the - // RCSI-OFF set BEFORE building the emitted items — the §3.3 enrichment is - // computed only for RCSI-off databases and injected into those rows. - var rows = new List(); - using (var reader = await cmd.ExecuteReaderAsync()) - { - while (await reader.ReadAsync()) - { - rows.Add(new ConfigIssueRow - { - Database = reader.IsDBNull(0) ? "" : reader.GetString(0), - RecoveryModel = reader.IsDBNull(1) ? "" : reader.GetString(1), - AutoShrink = (reader.IsDBNull(2) ? "" : reader.GetString(2)) == "1", - AutoClose = (reader.IsDBNull(3) ? "" : reader.GetString(3)) == "1", - Rcsi = (reader.IsDBNull(4) ? "" : reader.GetString(4)) == "1", - PageVerify = reader.IsDBNull(5) ? "" : reader.GetString(5), - QueryStore = (reader.IsDBNull(6) ? "" : reader.GetString(6)) == "1" - }); - } - } - - // §3.3 enrichment: per-DB blocking/deadlock counts + reader/writer split, ONLY - // for RCSI-off databases, over the analysis window, from already-collected - // monitoring tables (no fresh probe of the target server). - var rcsiOff = rows.Where(r => !r.Rcsi && !string.IsNullOrEmpty(r.Database)) - .Select(r => r.Database) - .ToList(); - var enrichment = rcsiOff.Count > 0 - ? await CollectRcsiInactionFigures(connection, rcsiOff, context) - : new Dictionary(StringComparer.Ordinal); - - var items = new List(); - foreach (var r in rows) - { - var issues = new List(); - if (r.AutoShrink) issues.Add("auto_shrink ON"); - if (r.AutoClose) issues.Add("auto_close ON"); - if (!r.Rcsi) issues.Add("RCSI OFF"); - if (!string.IsNullOrEmpty(r.PageVerify) && r.PageVerify != "CHECKSUM") issues.Add($"page_verify={r.PageVerify}"); - - // RCSI-off rows carry the three structured inaction-risk fields (M-2: - // identical JSON names + types to Lite, which emits them null/0). RCSI-on - // rows omit them entirely (the affordance never applies there). - if (!r.Rcsi) - { - enrichment.TryGetValue(r.Database, out var fig); - items.Add(new - { - database = r.Database, - recovery_model = r.RecoveryModel, - rcsi = r.Rcsi, - query_store = r.QueryStore, - issues, - auto_shrink = r.AutoShrink, - auto_close = r.AutoClose, - page_verify = r.PageVerify, - // §3.3: int / int / nullable-int. Counts from blocking_deadlock_stats - // (O-P3-F — NOT a blocking_BlockedProcessReport row count); the split - // is a separate pass over blocking_BlockedProcessReport. - rcsi_blocking_events = fig?.BlockingEvents ?? 0, - rcsi_deadlocks = fig?.Deadlocks ?? 0, - rcsi_reader_writer_pct = fig?.ReaderWriterPct - }); - } - else - { - items.Add(new - { - database = r.Database, - recovery_model = r.RecoveryModel, - rcsi = r.Rcsi, - query_store = r.QueryStore, - issues, - // §4.1: structured, wording-independent fields the shared extractor - // (FactRemediation.ExtractDbConfigTargets) reads. Identical JSON names - // and types to the Lite collector (bool / bool / string). - auto_shrink = r.AutoShrink, - auto_close = r.AutoClose, - page_verify = r.PageVerify - }); - } - } - - if (items.Count > 0) - finding.DrillDown!["config_issues"] = items; - } - - /// - /// Lists the large (>= 10 GB) data/log files on PERCENTAGE autogrowth (WS3), latest - /// snapshot per file, excluding system databases — and attaches a copy-paste - /// ALTER DATABASE ... MODIFY FILE fix per file (FILEGROWTH set to a size-tiered - /// fixed MB). The structured fields (database, logical_file_name, - /// total_size_mb, growth_pct) are what the shared extractor - /// (FactRemediation.ExtractFileGrowthTargets) reads; the rendered alter_statement - /// uses the SHARED renderer so it is byte-identical to the reader's copy-paste rebuild. - /// Advisory only — no Apply. - /// - private async Task CollectAutogrowthPercentFiles(AnalysisFinding finding, AnalysisContext context) - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -;WITH latest AS ( - SELECT - database_name, - file_id, - file_type_desc, - file_name, - total_size_mb, - is_percent_growth, - growth_pct, - ROW_NUMBER() OVER (PARTITION BY database_name, file_id ORDER BY collection_time DESC) AS rn - FROM collect.database_size_stats - WHERE database_name NOT IN ('master', 'msdb', 'model', 'tempdb') -) -SELECT TOP (50) - database_name, - file_type_desc, - file_name, - total_size_mb, - growth_pct -FROM latest -WHERE rn = 1 -AND is_percent_growth = 1 -AND total_size_mb >= @minSizeMb -ORDER BY total_size_mb DESC;"; - - cmd.Parameters.Add(new SqlParameter("@minSizeMb", 10240.0)); /* 10 GB */ - - var items = new List(); - using var reader = await cmd.ExecuteReaderAsync(); - while (await reader.ReadAsync()) - { - var database = reader.IsDBNull(0) ? "" : reader.GetString(0); - var fileType = reader.IsDBNull(1) ? "" : reader.GetString(1); - var logical = reader.IsDBNull(2) ? "" : reader.GetString(2); - var sizeMb = reader.IsDBNull(3) ? 0.0 : Convert.ToDouble(reader.GetValue(3)); - var growthPct = reader.IsDBNull(4) ? 0 : Convert.ToInt32(reader.GetValue(4)); - if (string.IsNullOrEmpty(database) || string.IsNullOrEmpty(logical)) continue; - - var growthMb = FactRemediation.RecommendedGrowthMbFor(fileType); - items.Add(new - { - database, - logical_file_name = logical, - file_type = fileType, - total_size_mb = sizeMb, - growth_pct = growthPct, - issue = $"{growthPct}% autogrowth on {sizeMb / 1024.0:N1} GB {fileType} file", - alter_statement = FactRemediation.BuildModifyFileStatement(database, logical, growthMb) - }); - } - - if (items.Count > 0) - finding.DrillDown!["autogrowth_percent_files"] = items; - } - - /// - /// Emits the server_config drill-down for a CONFIG_* finding (WS3): the single bad - /// server-level setting that rooted this finding, with its latest value_in_use plus the - /// server's engine_edition and cores_per_socket (needed to compute the - /// edition-aware, core-capped MAXDOP recommendation in - /// ). The structured fields - /// (setting, current_value, edition, cores_per_socket) are what the - /// shared extractor reads; nothing here is executed. The four CONFIG_* keys root SEPARATE - /// findings, so each finding emits exactly the one row for its own setting. - /// - private async Task CollectServerConfig(AnalysisFinding finding, AnalysisContext context, HashSet pathKeys) - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - // Latest value per requested config (ROW_NUMBER, not TOP N — same dedup correctness as the - // fact collector), plus the latest server_properties row for edition + cores-per-socket. - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -;WITH latest_config AS ( - SELECT - configuration_name, - CAST(value_in_use AS BIGINT) AS value_in_use, - ROW_NUMBER() OVER (PARTITION BY configuration_name ORDER BY collection_time DESC) AS rn - FROM config.server_configuration_history - WHERE configuration_name IN ( - 'cost threshold for parallelism', - 'max degree of parallelism', - 'max server memory (MB)', - 'min server memory (MB)' - ) -) -SELECT - configuration_name, - value_in_use -FROM latest_config -WHERE rn = 1; - -SELECT TOP (1) - engine_edition, - cores_per_socket -FROM collect.server_properties -ORDER BY collection_time DESC;"; - - long? ctfp = null, maxdop = null, maxMem = null, minMem = null; - int edition = 0, coresPerSocket = 0; - - using (var reader = await cmd.ExecuteReaderAsync()) - { - while (await reader.ReadAsync()) - { - var name = reader.IsDBNull(0) ? "" : reader.GetString(0); - var value = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); - switch (name) - { - case "cost threshold for parallelism": ctfp = value; break; - case "max degree of parallelism": maxdop = value; break; - case "max server memory (MB)": maxMem = value; break; - case "min server memory (MB)": minMem = value; break; - } - } - - if (await reader.NextResultAsync() && await reader.ReadAsync()) - { - edition = reader.IsDBNull(0) ? 0 : Convert.ToInt32(reader.GetValue(0)); - coresPerSocket = reader.IsDBNull(1) ? 0 : Convert.ToInt32(reader.GetValue(1)); - } - } - - var items = new List(); - - // Emit ONLY the setting that rooted this finding (the engine roots one CONFIG_* per finding). - // edition / cores_per_socket ride every row (the extractor uses them only for MAXDOP). - if (pathKeys.Contains("CONFIG_MAXDOP") && maxdop is { } md) - items.Add(new { setting = "maxdop", current_value = md, edition, cores_per_socket = coresPerSocket }); - - if (pathKeys.Contains("CONFIG_CTFP") && ctfp is { } ct) - items.Add(new { setting = "ctfp", current_value = ct, edition, cores_per_socket = coresPerSocket }); - - if (pathKeys.Contains("CONFIG_MAX_MEMORY_MB") && maxMem is { } mx) - items.Add(new { setting = "max_memory", current_value = mx, edition, cores_per_socket = coresPerSocket }); - - // For the narrow-memory finding the bad value the operator acts on is MIN server memory - // (lower it); carry it as the min_memory target. - if (pathKeys.Contains("CONFIG_MIN_MAX_MEMORY_NARROW") && minMem is { } mn) - items.Add(new { setting = "min_memory", current_value = mn, edition, cores_per_socket = coresPerSocket }); - - if (items.Count > 0) - finding.DrillDown!["server_config"] = items; - } - private sealed class ConfigIssueRow { public string Database = ""; @@ -1112,189 +270,6 @@ FROM classified AS c return result; } - private async Task CollectTempDbBreakdown(AnalysisFinding finding, AnalysisContext context) - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT TOP 5 - collection_time, - user_object_reserved_mb, - internal_object_reserved_mb, - version_store_reserved_mb, - unallocated_mb -FROM collect.tempdb_stats -WHERE collection_time >= @startTime AND collection_time <= @endTime -ORDER BY (user_object_reserved_mb + internal_object_reserved_mb + version_store_reserved_mb) DESC;"; - - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - var items = new List(); - using var reader = await cmd.ExecuteReaderAsync(); - while (await reader.ReadAsync()) - { - items.Add(new - { - time = reader.GetDateTime(0).ToString("o"), - user_objects_mb = reader.IsDBNull(1) ? 0.0 : Convert.ToDouble(reader.GetValue(1)), - internal_objects_mb = reader.IsDBNull(2) ? 0.0 : Convert.ToDouble(reader.GetValue(2)), - version_store_mb = reader.IsDBNull(3) ? 0.0 : Convert.ToDouble(reader.GetValue(3)), - unallocated_mb = reader.IsDBNull(4) ? 0.0 : Convert.ToDouble(reader.GetValue(4)) - }); - } - - if (items.Count > 0) - finding.DrillDown!["tempdb_breakdown"] = items; - } - - private async Task CollectPendingGrants(AnalysisFinding finding, AnalysisContext context) - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT TOP 5 - collection_time, - target_memory_mb, total_memory_mb, available_memory_mb, - granted_memory_mb, used_memory_mb, - grantee_count, waiter_count, - timeout_error_count_delta, forced_grant_count_delta -FROM collect.memory_grant_stats -WHERE collection_time >= @startTime AND collection_time <= @endTime -AND waiter_count > 0 -ORDER BY waiter_count DESC;"; - - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - var items = new List(); - using var reader = await cmd.ExecuteReaderAsync(); - while (await reader.ReadAsync()) - { - items.Add(new - { - time = reader.IsDBNull(0) ? "" : reader.GetDateTime(0).ToString("o"), - target_memory_mb = reader.IsDBNull(1) ? 0.0 : Convert.ToDouble(reader.GetValue(1)), - total_memory_mb = reader.IsDBNull(2) ? 0.0 : Convert.ToDouble(reader.GetValue(2)), - available_memory_mb = reader.IsDBNull(3) ? 0.0 : Convert.ToDouble(reader.GetValue(3)), - granted_memory_mb = reader.IsDBNull(4) ? 0.0 : Convert.ToDouble(reader.GetValue(4)), - used_memory_mb = reader.IsDBNull(5) ? 0.0 : Convert.ToDouble(reader.GetValue(5)), - grantee_count = reader.IsDBNull(6) ? 0 : reader.GetInt32(6), - waiter_count = reader.IsDBNull(7) ? 0 : reader.GetInt32(7), - timeout_errors = reader.IsDBNull(8) ? 0L : Convert.ToInt64(reader.GetValue(8)), - forced_grants = reader.IsDBNull(9) ? 0L : Convert.ToInt64(reader.GetValue(9)) - }); - } - - if (items.Count > 0) - finding.DrillDown!["pending_grants"] = items; - } - - /// - /// For findings that have query hashes (bad actors), fetch the execution plan - /// live from SQL Server via IPlanFetcher, then run PlanAnalyzer to surface - /// warnings and missing indexes. No plan storage needed -- fetch on demand - /// only for queries that make it into high-impact findings. - /// - private async Task CollectPlanAnalysis(AnalysisFinding finding, AnalysisContext context) - { - if (finding.DrillDown == null || _planFetcher == null) return; - - // Only analyze plans for bad actor findings (1 plan each). - // Skip top_cpu_queries (5 plans would be too heavy). - if (!finding.RootFactKey.StartsWith("BAD_ACTOR_", StringComparison.OrdinalIgnoreCase)) return; - - var queryHash = finding.RootFactKey.Replace("BAD_ACTOR_", ""); - if (string.IsNullOrEmpty(queryHash)) return; - - // Look up plan_handle from collect.query_stats for this query_hash - string? planHandle = null; - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT TOP 1 CONVERT(VARCHAR(130), plan_handle, 1) AS plan_handle -FROM collect.query_stats -WHERE query_hash = CONVERT(BINARY(8), @queryHash, 1) -AND plan_handle IS NOT NULL -ORDER BY collection_time DESC;"; - - cmd.Parameters.Add(new SqlParameter("@queryHash", queryHash)); - - using var reader = await cmd.ExecuteReaderAsync(); - if (await reader.ReadAsync() && !reader.IsDBNull(0)) - planHandle = reader.GetString(0); - } - catch { return; } - - if (string.IsNullOrEmpty(planHandle)) return; - - // Fetch plan XML live from SQL Server - var planXml = await _planFetcher.FetchPlanXmlAsync(context.ServerId, planHandle); - if (string.IsNullOrEmpty(planXml)) return; - - try - { - var plan = ShowPlanParser.Parse(planXml); - PlanAnalyzer.Analyze(plan); - - var allWarnings = plan.Batches - .SelectMany(b => b.Statements) - .Where(s => s.RootNode != null) - .SelectMany(s => - { - var nodeWarnings = new List(); - CollectPlanNodes(s.RootNode!, nodeWarnings); - return s.PlanWarnings - .Concat(nodeWarnings.SelectMany(n => n.Warnings)); - }) - .ToList(); - - var missingIndexes = plan.AllMissingIndexes; - - if (allWarnings.Count == 0 && missingIndexes.Count == 0) return; - - finding.DrillDown["plan_analysis"] = new - { - query_hash = queryHash, - warning_count = allWarnings.Count, - critical_count = allWarnings.Count(w => w.Severity == PlanWarningSeverity.Critical), - warnings = allWarnings - .OrderByDescending(w => w.Severity) - .Take(10) - .Select(w => new - { - severity = w.Severity.ToString(), - type = w.WarningType, - message = McpHelpers.Truncate(w.Message, 300) - }), - missing_indexes = missingIndexes.Take(5).Select(idx => new - { - table = $"{idx.Schema}.{idx.Table}", - impact = idx.Impact, - create_statement = idx.CreateStatement - }) - }; - } - catch - { - // Plan parsing can fail on malformed XML -- skip silently - } - } - private static void CollectPlanNodes(PlanNode node, List nodes) { nodes.Add(node); @@ -1302,488 +277,4 @@ private static void CollectPlanNodes(PlanNode node, List nodes) CollectPlanNodes(child, nodes); } - /// - /// WS4: re-parses the top collected query plans (same top-10-by-cost set the fact collector - /// summarized) and attaches the specific missing indexes / plan warnings to a MISSING_INDEX or - /// PLAN_WARNING finding's drill-down. The fact carries only counts (Fact.Metadata is numeric), - /// so the strings — CREATE statements, warning messages — are rendered here. Best-effort: a - /// read/parse failure leaves the finding with no plan-advisory detail rather than aborting. - /// - private async Task CollectPlanAdvisoryDetail(AnalysisFinding finding, AnalysisContext context, HashSet pathKeys) - { - try - { - var planXmls = new List(); - - using (var connection = new SqlConnection(_connectionString)) - { - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT TOP (10) - plan_xml = CAST(DECOMPRESS(qs.query_plan_text) AS nvarchar(max)) -FROM collect.query_stats AS qs -WHERE qs.collection_time >= @startTime -AND qs.collection_time <= @endTime -AND qs.query_plan_text IS NOT NULL -ORDER BY - qs.total_worker_time DESC;"; - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - using var reader = await cmd.ExecuteReaderAsync(); - while (await reader.ReadAsync()) - { - if (!reader.IsDBNull(0)) - planXmls.Add(reader.GetString(0)); - } - } - - if (planXmls.Count == 0) - return; - - var details = PlanAdvisoryAggregator.Extract(planXmls); - - if (pathKeys.Contains("MISSING_INDEX") && details.MissingIndexes.Count > 0) - { - finding.DrillDown!["missing_indexes"] = details.MissingIndexes - .OrderByDescending(i => i.Impact) - .Take(5) - .Select(i => new - { - table = $"{i.Schema}.{i.Table}", - impact = Math.Round(i.Impact, 1), - create_statement = i.CreateStatement - }) - .ToList(); - } - - if (pathKeys.Contains("PLAN_WARNING") && details.Warnings.Count > 0) - { - finding.DrillDown!["plan_warnings"] = details.Warnings - .OrderByDescending(w => w.Severity) - .Take(5) - .Select(w => new - { - type = w.WarningType, - severity = w.Severity.ToString(), - message = McpHelpers.Truncate(w.Message, 300) - }) - .ToList(); - } - } - catch - { - // Plan read/parse can fail on malformed XML -- skip, the detail is best-effort. - } - } - - private async Task CollectBadActorDetail(AnalysisFinding finding, AnalysisContext context) - { - // Extract query_hash from the fact key (BAD_ACTOR_0x...) - var queryHash = finding.RootFactKey.Replace("BAD_ACTOR_", ""); - if (string.IsNullOrEmpty(queryHash)) return; - - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT - database_name, - CONVERT(VARCHAR(18), query_hash, 1) AS query_hash, - LEFT(CAST(DECOMPRESS(MAX(query_text)) AS NVARCHAR(MAX)), 500) AS query_text, - CAST(SUM(execution_count_delta) AS BIGINT) AS exec_count, - CASE WHEN SUM(execution_count_delta) > 0 - THEN CAST(SUM(total_worker_time_delta) AS FLOAT) / SUM(execution_count_delta) / 1000.0 - ELSE 0 END AS avg_cpu_ms, - CASE WHEN SUM(execution_count_delta) > 0 - THEN CAST(SUM(total_elapsed_time_delta) AS FLOAT) / SUM(execution_count_delta) / 1000.0 - ELSE 0 END AS avg_elapsed_ms, - CASE WHEN SUM(execution_count_delta) > 0 - THEN CAST(SUM(total_logical_reads_delta) AS FLOAT) / SUM(execution_count_delta) - ELSE 0 END AS avg_reads, - CAST(SUM(total_worker_time_delta) AS BIGINT) AS total_cpu_us, - CAST(SUM(total_logical_reads_delta) AS BIGINT) AS total_reads, - CAST(SUM(total_spills) AS BIGINT) AS total_spills, - MAX(max_dop) AS max_dop -FROM collect.query_stats -WHERE collection_time >= @startTime -AND collection_time <= @endTime -AND query_hash = CONVERT(BINARY(8), @queryHash, 1) -GROUP BY database_name, query_hash;"; - - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - cmd.Parameters.Add(new SqlParameter("@queryHash", queryHash)); - - using var reader = await cmd.ExecuteReaderAsync(); - if (await reader.ReadAsync()) - { - finding.DrillDown!["bad_actor_query"] = new - { - database = reader.IsDBNull(0) ? "" : reader.GetString(0), - query_hash = reader.IsDBNull(1) ? "" : reader.GetString(1), - query_text = reader.IsDBNull(2) ? "" : reader.GetString(2), - execution_count = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)), - avg_cpu_ms = reader.IsDBNull(4) ? 0.0 : Math.Round(Convert.ToDouble(reader.GetValue(4)), 2), - avg_elapsed_ms = reader.IsDBNull(5) ? 0.0 : Math.Round(Convert.ToDouble(reader.GetValue(5)), 2), - avg_reads = reader.IsDBNull(6) ? 0.0 : Math.Round(Convert.ToDouble(reader.GetValue(6)), 0), - total_cpu_ms = reader.IsDBNull(7) ? 0.0 : Convert.ToDouble(reader.GetValue(7)) / 1000.0, - total_reads = reader.IsDBNull(8) ? 0L : Convert.ToInt64(reader.GetValue(8)), - total_spills = reader.IsDBNull(9) ? 0L : Convert.ToInt64(reader.GetValue(9)), - max_dop = reader.IsDBNull(10) ? 0 : Convert.ToInt32(reader.GetValue(10)) - }; - } - } - - /// - /// Top parameter-sensitive plans behind a PARAMETER_SENSITIVITY finding. - /// Re-runs the detector for the top 5 offenders. - /// - private async Task CollectParameterSensitiveQueries(AnalysisFinding finding, AnalysisContext context) - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -WITH latest AS -( - SELECT - database_name, - query_hash, - query_plan_hash, - execution_count, - creation_time, - min_worker_time, - max_worker_time, - min_grant_kb, - max_grant_kb, - min_spills, - max_spills, - query_text, - ROW_NUMBER() OVER - ( - PARTITION BY database_name, query_hash, query_plan_hash - ORDER BY collection_time DESC - ) AS rn - FROM collect.query_stats - WHERE collection_time >= @startTime - AND collection_time <= @endTime - AND execution_count_delta > 0 -) -SELECT - database_name, - CONVERT(varchar(18), query_hash, 1) AS query_hash, - CONVERT(varchar(18), query_plan_hash, 1) AS query_plan_hash, - execution_count, - min_worker_time, - max_worker_time, - CAST(max_worker_time AS float) / NULLIF(min_worker_time, 0) AS worker_ratio, - CAST(max_grant_kb AS float) / NULLIF(min_grant_kb, 0) AS grant_ratio, - CASE WHEN max_spills > 0 AND min_spills = 0 THEN 1 ELSE 0 END AS spill_divergence, - LEFT(CAST(DECOMPRESS(query_text) AS NVARCHAR(MAX)), 500) AS query_text -FROM latest -WHERE rn = 1 -AND min_worker_time >= 10000 -AND max_worker_time >= 250000 -AND execution_count >= 20 -AND creation_time <= @startTime -AND CAST(max_worker_time AS float) / NULLIF(min_worker_time, 0) >= 10 -ORDER BY worker_ratio DESC -OFFSET 0 ROWS FETCH NEXT 5 ROWS ONLY"; - - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - var items = new List(); - using var reader = await cmd.ExecuteReaderAsync(); - while (await reader.ReadAsync()) - { - items.Add(new - { - database = reader.IsDBNull(0) ? "" : reader.GetString(0), - query_hash = reader.IsDBNull(1) ? "" : reader.GetString(1), - query_plan_hash = reader.IsDBNull(2) ? "" : reader.GetString(2), - execution_count = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)), - min_worker_time_us = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)), - max_worker_time_us = reader.IsDBNull(5) ? 0L : Convert.ToInt64(reader.GetValue(5)), - worker_ratio = reader.IsDBNull(6) ? 0.0 : Convert.ToDouble(reader.GetValue(6)), - grant_ratio = reader.IsDBNull(7) ? 0.0 : Convert.ToDouble(reader.GetValue(7)), - spills_on_some_inputs = !reader.IsDBNull(8) && Convert.ToInt32(reader.GetValue(8)) == 1, - query_text = reader.IsDBNull(9) ? "" : reader.GetString(9) - }); - } - - if (items.Count > 0) - finding.DrillDown!["parameter_sensitive_queries"] = items; - } - - /// - /// Top regressed queries behind a PLAN_REGRESSION finding. - /// Uses the same 14-day server_last_execution_time comparison window as the detector - /// (NOT the standard analysis window) so the days-old "best plan" baseline is present. - /// - private async Task CollectRegressedQueries(AnalysisFinding finding, AnalysisContext context) - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -WITH deduped AS -( - SELECT - database_name, - query_id, - plan_id, - query_plan_hash, - count_executions, - avg_cpu_time, - avg_duration, - server_last_execution_time, - ROW_NUMBER() OVER - ( - PARTITION BY database_name, query_id, plan_id, server_first_execution_time - ORDER BY collection_time DESC - ) AS rn - FROM collect.query_store_data - WHERE execution_type_desc = N'Regular' - AND server_last_execution_time >= @windowStart -), -plan_agg AS -( - -- query_plan_hash is invariant within a plan_id, so include it in the GROUP BY - -- (MS Learn's MAX page does not list binary/varbinary in the accepted types). - SELECT - database_name, - query_id, - plan_id, - query_plan_hash, - SUM(count_executions) AS execs, - CASE WHEN SUM(count_executions) > 0 - THEN SUM(avg_cpu_time * count_executions) / NULLIF(SUM(count_executions), 0) - ELSE 0 END AS cpu_per_exec, - CASE WHEN SUM(count_executions) > 0 - THEN SUM(avg_duration * count_executions) / NULLIF(SUM(count_executions), 0) - ELSE 0 END AS dur_per_exec, - MAX(server_last_execution_time) AS last_exec - FROM deduped - WHERE rn = 1 - GROUP BY database_name, query_id, plan_id, query_plan_hash -), -plan_dedup AS -( - -- MAX(plan_id) carries the most recently observed plan_id in the hash partition - -- forward — newer plans are less likely evicted by Query Store retention. - -- Functionally any plan_id sharing the hash forces the same execution shape. - SELECT - database_name, - query_id, - query_plan_hash, - MAX(plan_id) AS plan_id, - SUM(execs) AS execs, - CASE WHEN SUM(execs) > 0 - THEN SUM(cpu_per_exec * execs) / NULLIF(SUM(execs), 0) - ELSE 0 END AS cpu_per_exec, - CASE WHEN SUM(execs) > 0 - THEN SUM(dur_per_exec * execs) / NULLIF(SUM(execs), 0) - ELSE 0 END AS dur_per_exec, - MAX(last_exec) AS last_exec - FROM plan_agg - GROUP BY database_name, query_id, query_plan_hash - HAVING SUM(execs) >= 25 -), -ranked AS -( - SELECT - *, - ROW_NUMBER() OVER (PARTITION BY database_name, query_id ORDER BY last_exec DESC) AS recency, - ROW_NUMBER() OVER (PARTITION BY database_name, query_id ORDER BY cpu_per_exec ASC) AS cheapness - FROM plan_dedup -), -compared AS -( - SELECT - l.database_name, - l.query_id, - l.query_plan_hash AS latest_plan_hash, - l.cpu_per_exec AS latest_cpu, - l.dur_per_exec AS latest_dur, - b.query_plan_hash AS best_plan_hash, - b.plan_id AS best_plan_id, - b.cpu_per_exec AS best_cpu, - b.dur_per_exec AS best_dur, - (SELECT MAX(v) - FROM (VALUES - (CAST(l.cpu_per_exec AS float) / NULLIF(b.cpu_per_exec, 0)), - (CAST(l.dur_per_exec AS float) / NULLIF(b.dur_per_exec, 0)) - ) AS x(v)) AS regression_factor - FROM ranked AS l - JOIN ranked AS b - ON b.database_name = l.database_name - AND b.query_id = l.query_id - AND b.cheapness = 1 - WHERE l.recency = 1 - AND l.query_plan_hash <> b.query_plan_hash -) -SELECT - c.database_name, - c.query_id, - CONVERT(varchar(18), c.latest_plan_hash, 1) AS latest_plan_hash, - c.latest_cpu, - c.latest_dur, - CONVERT(varchar(18), c.best_plan_hash, 1) AS best_plan_hash, - c.best_plan_id, - c.best_cpu, - c.best_dur, - c.regression_factor, - -- query_sql_text is varbinary(max); fetch it via APPLY (MAX() on varbinary(max) is invalid). - LEFT(CAST(DECOMPRESS(qt.query_sql_text) AS NVARCHAR(MAX)), 500) AS query_text -FROM compared AS c -OUTER APPLY -( - SELECT TOP (1) qs.query_sql_text - FROM collect.query_store_data AS qs - WHERE qs.database_name = c.database_name - AND qs.query_id = c.query_id - AND qs.query_plan_hash = c.latest_plan_hash - AND qs.server_last_execution_time >= @windowStart - ORDER BY qs.server_last_execution_time DESC -) AS qt -WHERE c.regression_factor >= 2 -ORDER BY c.regression_factor DESC -OFFSET 0 ROWS FETCH NEXT 5 ROWS ONLY"; - - cmd.Parameters.Add(new SqlParameter("@windowStart", context.TimeRangeStart.AddDays(-14))); - - var items = new List(); - using var reader = await cmd.ExecuteReaderAsync(); - while (await reader.ReadAsync()) - { - items.Add(new - { - database = reader.IsDBNull(0) ? "" : reader.GetString(0), - query_id = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)), - latest_plan_hash = reader.IsDBNull(2) ? "" : reader.GetString(2), - latest_cpu_per_exec_us = reader.IsDBNull(3) ? 0.0 : Convert.ToDouble(reader.GetValue(3)), - latest_duration_per_exec_us = reader.IsDBNull(4) ? 0.0 : Convert.ToDouble(reader.GetValue(4)), - best_plan_hash = reader.IsDBNull(5) ? "" : reader.GetString(5), - best_plan_id = reader.IsDBNull(6) ? 0L : Convert.ToInt64(reader.GetValue(6)), - best_cpu_per_exec_us = reader.IsDBNull(7) ? 0.0 : Convert.ToDouble(reader.GetValue(7)), - best_duration_per_exec_us = reader.IsDBNull(8) ? 0.0 : Convert.ToDouble(reader.GetValue(8)), - regression_factor = reader.IsDBNull(9) ? 0.0 : Convert.ToDouble(reader.GetValue(9)), - query_text = reader.IsDBNull(10) ? "" : reader.GetString(10) - }); - } - - if (items.Count > 0) - finding.DrillDown!["regressed_queries"] = items; - } - - /// - /// Reconstructs blocking chains (same logic as the collector) and surfaces the top 3 - /// by magnitude — apex, depth, victim count, and the level-by-level structure that - /// the flat top_blocking_chains list cannot show. - /// - private async Task CollectReconstructedBlockingChains(AnalysisFinding finding, AnalysisContext context) - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - // blocking_spid IS NOT NULL filters out rows whose source XML had an empty - // (system task / torn-down session) — - // those can't contribute to a reconstructed chain. - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT TOP (5000) - event_time, - database_name, - spid, - last_transaction_started, - blocking_spid, - blocking_last_tran_started, - wait_time_ms, - lock_mode, - blocking_status, - blocked_sql_text, - blocking_sql_text -FROM collect.blocking_BlockedProcessReport -WHERE collection_time >= @collectionWindow -AND event_time >= @startTime -AND event_time <= @endTime -AND activity = 'blocked' -AND blocking_spid IS NOT NULL -ORDER BY collection_time DESC"; - - cmd.Parameters.Add(new SqlParameter("@collectionWindow", context.TimeRangeStart.AddHours(-1))); - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - var rows = new List(); - using (var reader = await cmd.ExecuteReaderAsync()) - { - while (await reader.ReadAsync()) - { - rows.Add(new BlockingPairRow - { - EventTime = reader.IsDBNull(0) ? default : reader.GetDateTime(0), - DatabaseName = reader.IsDBNull(1) ? string.Empty : reader.GetString(1), - BlockedSpid = reader.IsDBNull(2) ? 0 : Convert.ToInt32(reader.GetValue(2)), - BlockedTranStarted = reader.IsDBNull(3) ? (DateTime?)null : reader.GetDateTime(3), - BlockingSpid = reader.IsDBNull(4) ? 0 : Convert.ToInt32(reader.GetValue(4)), - BlockingTranStarted = reader.IsDBNull(5) ? (DateTime?)null : reader.GetDateTime(5), - WaitTimeMs = reader.IsDBNull(6) ? 0L : Convert.ToInt64(reader.GetValue(6)), - LockMode = reader.IsDBNull(7) ? string.Empty : reader.GetString(7), - BlockingStatus = reader.IsDBNull(8) ? string.Empty : reader.GetString(8), - BlockedSqlText = reader.IsDBNull(9) ? string.Empty : reader.GetString(9), - BlockingSqlText = reader.IsDBNull(10) ? string.Empty : reader.GetString(10) - }); - } - } - - if (rows.Count == 0) return; - - var reconstruction = BlockingChainReconstructor.Reconstruct( - rows, maxDepth: 50, maxPairs: 5000, stepBudget: 100_000); - - var items = new List(); - foreach (var chain in reconstruction.Chains.Take(3)) - { - items.Add(new - { - apex_spid = chain.ApexSpid, - apex_sleeping = chain.ApexSleeping, - depth = chain.Depth, - // Distinct sessions blocked under this apex over the window — cumulative, not peak-concurrent. - victim_count = chain.VictimCount, - max_wait_ms = chain.MaxWaitMs, - levels = chain.Levels.Select(l => new - { - level = l.Level, - blocking_spid = l.BlockingSpid, - blocked_spid = l.BlockedSpid, - lock_mode = l.LockMode, - wait_time_ms = l.WaitTimeMs, - blocking_sql = l.BlockingSqlText, - blocked_sql = l.BlockedSqlText - }).ToList() - }); - } - - if (items.Count > 0) - finding.DrillDown!["reconstructed_blocking_chains"] = items; - } } diff --git a/Dashboard/Analysis/SqlServerFactCollector.Activity.cs b/Dashboard/Analysis/SqlServerFactCollector.Activity.cs new file mode 100644 index 000000000..fc829448c --- /dev/null +++ b/Dashboard/Analysis/SqlServerFactCollector.Activity.cs @@ -0,0 +1,306 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading.Tasks; +using Microsoft.Data.SqlClient; +using PerformanceMonitor.Analysis; +using PerformanceMonitor.PlanAnalysis; +using PerformanceMonitorDashboard.Helpers; + +namespace PerformanceMonitorDashboard.Analysis; + +public partial class SqlServerFactCollector +{ + /// + /// Identifies individual queries that are consistently terrible ("bad actors"). + /// These queries don't necessarily cause server-level symptoms but waste resources + /// on every execution. Detection uses execution count tiers x per-execution impact. + /// Top 5 worst offenders become individual BAD_ACTOR facts. + /// Dashboard query_hash is binary(8) — convert to hex string for fact key. + /// + private async Task CollectBadActorFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT TOP 5 + database_name, + CONVERT(VARCHAR(18), query_hash, 1) AS query_hash, + CAST(SUM(execution_count_delta) AS BIGINT) AS exec_count, + CASE WHEN SUM(execution_count_delta) > 0 + THEN CAST(SUM(total_worker_time_delta) AS FLOAT) / SUM(execution_count_delta) / 1000.0 + ELSE 0 END AS avg_cpu_ms, + CASE WHEN SUM(execution_count_delta) > 0 + THEN CAST(SUM(total_elapsed_time_delta) AS FLOAT) / SUM(execution_count_delta) / 1000.0 + ELSE 0 END AS avg_elapsed_ms, + CASE WHEN SUM(execution_count_delta) > 0 + THEN CAST(SUM(total_logical_reads_delta) AS FLOAT) / SUM(execution_count_delta) + ELSE 0 END AS avg_reads, + CAST(SUM(total_worker_time_delta) AS BIGINT) AS total_cpu_us, + CAST(SUM(total_logical_reads_delta) AS BIGINT) AS total_reads, + CAST(SUM(total_spills) AS BIGINT) AS total_spills, + MAX(max_dop) AS max_dop, + LEFT(CAST(DECOMPRESS(MAX(query_text)) AS NVARCHAR(MAX)), 200) AS query_text +FROM collect.query_stats +WHERE collection_time >= @startTime +AND collection_time <= @endTime +AND execution_count_delta > 0 +GROUP BY database_name, query_hash +HAVING SUM(execution_count_delta) >= 100 +ORDER BY CAST(SUM(total_worker_time_delta) AS FLOAT) / NULLIF(SUM(execution_count_delta), 0) * + LOG(NULLIF(SUM(execution_count_delta), 0)) DESC"; + + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + using var reader = await cmd.ExecuteReaderAsync(); + while (await reader.ReadAsync()) + { + var dbName = reader.IsDBNull(0) ? "" : reader.GetString(0); + var queryHash = reader.IsDBNull(1) ? "" : reader.GetString(1); + var execCount = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); + var avgCpuMs = reader.IsDBNull(3) ? 0.0 : Convert.ToDouble(reader.GetValue(3)); + var avgElapsedMs = reader.IsDBNull(4) ? 0.0 : Convert.ToDouble(reader.GetValue(4)); + var avgReads = reader.IsDBNull(5) ? 0.0 : Convert.ToDouble(reader.GetValue(5)); + var totalCpuUs = reader.IsDBNull(6) ? 0L : Convert.ToInt64(reader.GetValue(6)); + var totalReads = reader.IsDBNull(7) ? 0L : Convert.ToInt64(reader.GetValue(7)); + var totalSpills = reader.IsDBNull(8) ? 0L : Convert.ToInt64(reader.GetValue(8)); + var maxDop = reader.IsDBNull(9) ? 0 : Convert.ToInt32(reader.GetValue(9)); + var queryText = reader.IsDBNull(10) ? "" : reader.GetString(10); + + // Skip low-impact queries — need meaningful per-execution cost + if (avgCpuMs < 10 && avgReads < 1000) continue; + + facts.Add(new Fact + { + Source = "bad_actor", + Key = $"BAD_ACTOR_{queryHash}", + Value = avgCpuMs, // Primary scoring dimension + ServerId = context.ServerId, + DatabaseName = dbName, + Metadata = new Dictionary + { + ["execution_count"] = execCount, + ["avg_cpu_ms"] = avgCpuMs, + ["avg_elapsed_ms"] = avgElapsedMs, + ["avg_reads"] = avgReads, + ["total_cpu_us"] = totalCpuUs, + ["total_reads"] = totalReads, + ["total_spills"] = totalSpills, + ["max_dop"] = maxDop + } + }); + } + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectBadActorFactsAsync failed", ex); + } + } + + /// + /// Collects active query snapshot facts: long-running queries, blocked sessions, high DOP. + /// Dashboard query_snapshots table is created by sp_WhoIsActive dynamically. + /// We query it if it exists. + /// + private async Task CollectActiveQueryFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + // Check if the table exists first (created dynamically by sp_WhoIsActive) + using var checkCmd = connection.CreateCommand(); + checkCmd.CommandText = "SELECT OBJECT_ID(N'collect.query_snapshots', N'U')"; + var tableExists = await checkCmd.ExecuteScalarAsync(); + if (tableExists == null || tableExists == DBNull.Value) return; + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT + COUNT(*) AS total_snapshots, + COUNT(CASE WHEN DATEDIFF(MILLISECOND, 0, [elapsed_time]) > 30000 THEN 1 END) AS long_running_count, + COUNT(CASE WHEN [blocking_session_id] IS NOT NULL AND [blocking_session_id] != '' THEN 1 END) AS blocked_count, + MAX(DATEDIFF(MILLISECOND, 0, [elapsed_time])) AS max_elapsed_ms, + COUNT(DISTINCT [session_id]) AS distinct_sessions +FROM collect.query_snapshots +WHERE collection_time >= @startTime +AND collection_time <= @endTime"; + + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + using var reader = await cmd.ExecuteReaderAsync(); + if (!await reader.ReadAsync()) return; + + var totalSnapshots = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); + if (totalSnapshots == 0) return; + + var longRunning = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); + var blocked = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); + var maxElapsed = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)); + var distinctSessions = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)); + + facts.Add(new Fact + { + Source = "queries", + Key = "ACTIVE_QUERIES", + Value = longRunning, + ServerId = context.ServerId, + Metadata = new Dictionary + { + ["total_snapshots"] = totalSnapshots, + ["long_running_count"] = longRunning, + ["blocked_count"] = blocked, + ["max_elapsed_ms"] = maxElapsed, + ["distinct_sessions"] = distinctSessions + } + }); + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectActiveQueryFactsAsync failed", ex); + } + } + + /// + /// Collects running job facts: jobs currently running long vs historical averages. + /// + private async Task CollectRunningJobFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT + COUNT(*) AS running_count, + COUNT(CASE WHEN is_running_long = 1 THEN 1 END) AS running_long_count, + MAX(percent_of_average) AS max_percent_of_avg, + MAX(current_duration_seconds) AS max_duration_seconds +FROM collect.running_jobs +WHERE collection_time >= @startTime +AND collection_time <= @endTime"; + + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + using var reader = await cmd.ExecuteReaderAsync(); + if (!await reader.ReadAsync()) return; + + var runningCount = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); + if (runningCount == 0) return; + + var runningLong = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); + var maxPctAvg = reader.IsDBNull(2) ? 0.0 : Convert.ToDouble(reader.GetValue(2)); + var maxDuration = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)); + + facts.Add(new Fact + { + Source = "jobs", + Key = "RUNNING_JOBS", + Value = runningLong, + ServerId = context.ServerId, + Metadata = new Dictionary + { + ["running_count"] = runningCount, + ["running_long_count"] = runningLong, + ["max_percent_of_average"] = maxPctAvg, + ["max_duration_seconds"] = maxDuration + } + }); + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectRunningJobFactsAsync failed", ex); + } + } + + /// + /// Collects session stats: connection counts, total connections. + /// Dashboard session_stats is a flat table (not per-program_name), so we adapt. + /// + private async Task CollectSessionFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +;WITH latest AS ( + SELECT + total_sessions, + running_sessions, + sleeping_sessions, + dormant_sessions, + databases_with_connections, + top_application_connections, + ROW_NUMBER() OVER (ORDER BY collection_time DESC) AS rn + FROM collect.session_stats + WHERE collection_time >= @startTime + AND collection_time <= @endTime +) +SELECT + total_sessions AS total_connections, + running_sessions AS total_running, + sleeping_sessions AS total_sleeping, + dormant_sessions AS total_dormant, + databases_with_connections AS distinct_apps, + top_application_connections AS max_app_connections +FROM latest WHERE rn = 1"; + + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + using var reader = await cmd.ExecuteReaderAsync(); + if (!await reader.ReadAsync()) return; + + var totalConns = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); + if (totalConns == 0) return; + + var totalRunning = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); + var totalSleeping = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); + var totalDormant = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)); + var distinctApps = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)); + var maxAppConns = reader.IsDBNull(5) ? 0L : Convert.ToInt64(reader.GetValue(5)); + + facts.Add(new Fact + { + Source = "sessions", + Key = "SESSION_STATS", + Value = totalConns, + ServerId = context.ServerId, + Metadata = new Dictionary + { + ["total_connections"] = totalConns, + ["total_running"] = totalRunning, + ["total_sleeping"] = totalSleeping, + ["total_dormant"] = totalDormant, + ["distinct_applications"] = distinctApps, + ["max_app_connections"] = maxAppConns + } + }); + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectSessionFactsAsync failed", ex); + } + } +} diff --git a/Dashboard/Analysis/SqlServerFactCollector.Config.cs b/Dashboard/Analysis/SqlServerFactCollector.Config.cs new file mode 100644 index 000000000..3eaada178 --- /dev/null +++ b/Dashboard/Analysis/SqlServerFactCollector.Config.cs @@ -0,0 +1,390 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading.Tasks; +using Microsoft.Data.SqlClient; +using PerformanceMonitor.Analysis; +using PerformanceMonitor.PlanAnalysis; +using PerformanceMonitorDashboard.Helpers; + +namespace PerformanceMonitorDashboard.Analysis; + +public partial class SqlServerFactCollector +{ + /// + /// Collects server configuration settings relevant to analysis. + /// These become facts that amplifiers and the config audit tool can reference + /// to make recommendations specific (e.g., "your CTFP is 50" vs "check CTFP"). + /// + private async Task CollectServerConfigFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + // Latest value PER configuration_name (ROW_NUMBER, not TOP N): server_configuration_history + // accumulates a row per collection, so a naive TOP-N-ORDER-BY-time returns the newest N + // ROWS — which collapses to one config when collections are frequent, silently dropping + // settings. Partition by name and take rn = 1 so each requested setting is its latest value. + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +;WITH latest AS ( + SELECT + configuration_name, + CAST(value_in_use AS BIGINT) AS value_in_use, + ROW_NUMBER() OVER (PARTITION BY configuration_name ORDER BY collection_time DESC) AS rn + FROM config.server_configuration_history + WHERE configuration_name IN ( + 'cost threshold for parallelism', + 'max degree of parallelism', + 'max server memory (MB)', + 'min server memory (MB)', + 'max worker threads' + ) +) +SELECT + configuration_name, + value_in_use +FROM latest +WHERE rn = 1"; + + // max/min server memory are read alongside the rooted CONFIG_* facts so the + // narrow-memory derivation below can compare them without a second query. + double? maxMemoryMb = null; + double? minMemoryMb = null; + + using (var reader = await cmd.ExecuteReaderAsync()) + { + while (await reader.ReadAsync()) + { + var configName = reader.GetString(0); + var value = Convert.ToDouble(reader.GetValue(1)); + + if (configName == "max server memory (MB)") maxMemoryMb = value; + if (configName == "min server memory (MB)") minMemoryMb = value; + + var factKey = configName switch + { + "cost threshold for parallelism" => "CONFIG_CTFP", + "max degree of parallelism" => "CONFIG_MAXDOP", + "max server memory (MB)" => "CONFIG_MAX_MEMORY_MB", + "min server memory (MB)" => "CONFIG_MIN_MEMORY_MB", + "max worker threads" => "CONFIG_MAX_WORKER_THREADS", + _ => null + }; + + if (factKey == null) continue; + + facts.Add(new Fact + { + Source = "config", + Key = factKey, + Value = value, + ServerId = context.ServerId, + Metadata = new Dictionary + { + ["value_in_use"] = value + } + }); + } + } + + // CONFIG_MIN_MAX_MEMORY_NARROW: emitted only when max is configured AND min is pinned + // near it (shared rule so Dashboard/Lite agree). + var narrow = FactRemediation.BuildNarrowMemoryFact(context.ServerId, maxMemoryMb, minMemoryMb); + if (narrow is not null) + facts.Add(narrow); + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectServerConfigFactsAsync failed", ex); + } + } + + /// + /// Collects SQL Server edition and major version from the server_properties table. + /// + private async Task CollectServerMetadataFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT TOP 1 + engine_edition, + CAST(LEFT(product_version, CHARINDEX('.', product_version) - 1) AS INT) AS major_version +FROM collect.server_properties +ORDER BY collection_time DESC"; + + using var reader = await cmd.ExecuteReaderAsync(); + if (!await reader.ReadAsync()) return; + + var edition = reader.IsDBNull(0) ? 0 : Convert.ToInt32(reader.GetValue(0)); + var majorVersion = reader.IsDBNull(1) ? 0 : Convert.ToInt32(reader.GetValue(1)); + + if (edition > 0) + facts.Add(new Fact { Source = "config", Key = "SERVER_EDITION", Value = edition, ServerId = context.ServerId }); + if (majorVersion > 0) + facts.Add(new Fact { Source = "config", Key = "SERVER_MAJOR_VERSION", Value = majorVersion, ServerId = context.ServerId }); + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectServerMetadataFactsAsync failed", ex); + } + } + + /// + /// Collects database configuration facts: RCSI status, auto_shrink, auto_close, + /// recovery model. Aggregates counts across databases. + /// Dashboard stores config as individual setting rows in config.database_configuration_history. + /// We pivot from the per-setting rows into aggregated counts. + /// + private async Task CollectDatabaseConfigFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +;WITH latest AS ( + SELECT + database_name, + setting_name, + setting_value, + ROW_NUMBER() OVER (PARTITION BY database_name, setting_name ORDER BY collection_time DESC) AS rn + FROM config.database_configuration_history + WHERE setting_type = 'DATABASE_PROPERTY' /* collector writes 'DATABASE_PROPERTY' (install/39); 'database_option' matched 0 rows → DB_CONFIG fact never fired */ + AND database_name NOT IN ('master', 'msdb', 'model', 'tempdb') +), +pivoted AS ( + SELECT + database_name, + MAX(CASE WHEN setting_name = 'recovery_model_desc' THEN CAST(setting_value AS NVARCHAR(128)) END) AS recovery_model, + MAX(CASE WHEN setting_name = 'is_auto_shrink_on' THEN CAST(setting_value AS NVARCHAR(10)) END) AS is_auto_shrink_on, + MAX(CASE WHEN setting_name = 'is_auto_close_on' THEN CAST(setting_value AS NVARCHAR(10)) END) AS is_auto_close_on, + MAX(CASE WHEN setting_name = 'is_read_committed_snapshot_on' THEN CAST(setting_value AS NVARCHAR(10)) END) AS is_read_committed_snapshot_on, + MAX(CASE WHEN setting_name = 'is_auto_create_stats_on' THEN CAST(setting_value AS NVARCHAR(10)) END) AS is_auto_create_stats_on, + MAX(CASE WHEN setting_name = 'is_auto_update_stats_on' THEN CAST(setting_value AS NVARCHAR(10)) END) AS is_auto_update_stats_on, + MAX(CASE WHEN setting_name = 'page_verify_option_desc' THEN CAST(setting_value AS NVARCHAR(128)) END) AS page_verify_option, + MAX(CASE WHEN setting_name = 'is_query_store_on' THEN CAST(setting_value AS NVARCHAR(10)) END) AS is_query_store_on + FROM latest + WHERE rn = 1 + GROUP BY database_name +) +SELECT + COUNT(*) AS database_count, + COUNT(CASE WHEN is_auto_shrink_on = '1' OR is_auto_shrink_on = 'True' THEN 1 END) AS auto_shrink_count, + COUNT(CASE WHEN is_auto_close_on = '1' OR is_auto_close_on = 'True' THEN 1 END) AS auto_close_count, + COUNT(CASE WHEN is_read_committed_snapshot_on = '0' OR is_read_committed_snapshot_on = 'False' THEN 1 END) AS rcsi_off_count, + COUNT(CASE WHEN is_auto_create_stats_on = '0' OR is_auto_create_stats_on = 'False' THEN 1 END) AS auto_create_stats_off_count, + COUNT(CASE WHEN is_auto_update_stats_on = '0' OR is_auto_update_stats_on = 'False' THEN 1 END) AS auto_update_stats_off_count, + COUNT(CASE WHEN page_verify_option IS NOT NULL AND page_verify_option != 'CHECKSUM' THEN 1 END) AS page_verify_not_checksum_count, + COUNT(CASE WHEN recovery_model = 'FULL' THEN 1 END) AS full_recovery_count, + COUNT(CASE WHEN recovery_model = 'SIMPLE' THEN 1 END) AS simple_recovery_count, + COUNT(CASE WHEN is_query_store_on = '1' OR is_query_store_on = 'True' THEN 1 END) AS query_store_on_count +FROM pivoted"; + + using var reader = await cmd.ExecuteReaderAsync(); + if (!await reader.ReadAsync()) return; + + var dbCount = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); + if (dbCount == 0) return; + + var autoShrink = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); + var autoClose = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); + var rcsiOff = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)); + var autoCreateOff = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)); + var autoUpdateOff = reader.IsDBNull(5) ? 0L : Convert.ToInt64(reader.GetValue(5)); + var pageVerifyBad = reader.IsDBNull(6) ? 0L : Convert.ToInt64(reader.GetValue(6)); + var fullRecovery = reader.IsDBNull(7) ? 0L : Convert.ToInt64(reader.GetValue(7)); + var simpleRecovery = reader.IsDBNull(8) ? 0L : Convert.ToInt64(reader.GetValue(8)); + var queryStoreOn = reader.IsDBNull(9) ? 0L : Convert.ToInt64(reader.GetValue(9)); + + facts.Add(new Fact + { + Source = "database_config", + Key = "DB_CONFIG", + Value = dbCount, + ServerId = context.ServerId, + Metadata = new Dictionary + { + ["database_count"] = dbCount, + ["auto_shrink_on_count"] = autoShrink, + ["auto_close_on_count"] = autoClose, + ["rcsi_off_count"] = rcsiOff, + ["auto_create_stats_off_count"] = autoCreateOff, + ["auto_update_stats_off_count"] = autoUpdateOff, + ["page_verify_not_checksum_count"] = pageVerifyBad, + ["full_recovery_count"] = fullRecovery, + ["simple_recovery_count"] = simpleRecovery, + ["query_store_on_count"] = queryStoreOn + } + }); + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectDatabaseConfigFactsAsync failed", ex); + } + } + + /// + /// Collects active global trace flags. Context for the AI to factor into recommendations. + /// + private async Task CollectTraceFlagFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +;WITH latest AS ( + SELECT + trace_flag, + status, + ROW_NUMBER() OVER (PARTITION BY trace_flag ORDER BY collection_time DESC) AS rn + FROM config.trace_flags_history + WHERE is_global = 1 +) +SELECT trace_flag +FROM latest WHERE rn = 1 AND status = 1 +ORDER BY trace_flag"; + + using var reader = await cmd.ExecuteReaderAsync(); + var metadata = new Dictionary(); + var flagCount = 0; + + while (await reader.ReadAsync()) + { + var flag = Convert.ToInt32(reader.GetValue(0)); + metadata[$"TF_{flag}"] = 1; + flagCount++; + } + + if (flagCount == 0) return; + + metadata["flag_count"] = flagCount; + + facts.Add(new Fact + { + Source = "config", + Key = "TRACE_FLAGS", + Value = flagCount, + ServerId = context.ServerId, + Metadata = metadata + }); + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectTraceFlagFactsAsync failed", ex); + } + } + + private async Task CollectServerPropertiesFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + // Version-skew resilience: a server whose PerformanceMonitor DB has not yet had the WS5 + // upgrade lacks some or all of the three server-health columns. Probe EACH independently + // (COL_LENGTH returns NULL for an absent column) and reference only the present ones, so + // the core SERVER_HARDWARE read never fails — and keeps flowing — regardless of which + // columns a partially-upgraded or out-of-order schema happens to have. + bool hasLpim, hasIfi, hasDumps; + using (var probe = connection.CreateCommand()) + { + probe.CommandText = $@" +SELECT + COL_LENGTH('collect.server_properties', '{LpimColumn}'), + COL_LENGTH('collect.server_properties', '{IfiColumn}'), + COL_LENGTH('collect.server_properties', '{DumpsColumn}');"; + using var probeReader = await probe.ExecuteReaderAsync(); + await probeReader.ReadAsync(); + hasLpim = !probeReader.IsDBNull(0); + hasIfi = !probeReader.IsDBNull(1); + hasDumps = !probeReader.IsDBNull(2); + } + + using var cmd = connection.CreateCommand(); + cmd.CommandText = BuildServerPropertiesQuery(hasLpim, hasIfi, hasDumps); + + using var reader = await cmd.ExecuteReaderAsync(); + if (!await reader.ReadAsync()) return; + + var cpuCount = reader.IsDBNull(0) ? 0 : Convert.ToInt32(reader.GetValue(0)); + var htRatio = reader.IsDBNull(1) ? 0 : Convert.ToInt32(reader.GetValue(1)); + var physicalMemMb = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); + var socketCount = reader.IsDBNull(3) ? 0 : Convert.ToInt32(reader.GetValue(3)); + var coresPerSocket = reader.IsDBNull(4) ? 0 : Convert.ToInt32(reader.GetValue(4)); + var hadrEnabled = !reader.IsDBNull(5) && Convert.ToBoolean(reader.GetValue(5)); + var edition = reader.IsDBNull(6) ? string.Empty : reader.GetString(6); + + // Read each PRESENT health column by name — the SELECT includes only the columns that + // exist, so their ordinals shift with the subset; GetOrdinal resolves each regardless. + // An absent column stays null, so EmitServerHealthFacts emits nothing for it. + bool? lpim = null; + bool? ifi = null; + int? dumpCount = null; + if (hasLpim) + { + var ord = reader.GetOrdinal(LpimColumn); + lpim = reader.IsDBNull(ord) ? (bool?)null : Convert.ToBoolean(reader.GetValue(ord)); + } + if (hasIfi) + { + var ord = reader.GetOrdinal(IfiColumn); + ifi = reader.IsDBNull(ord) ? (bool?)null : Convert.ToBoolean(reader.GetValue(ord)); + } + if (hasDumps) + { + var ord = reader.GetOrdinal(DumpsColumn); + dumpCount = reader.IsDBNull(ord) ? (int?)null : Convert.ToInt32(reader.GetValue(ord)); + } + + if (cpuCount == 0) return; + + facts.Add(new Fact + { + Source = "config", + Key = "SERVER_HARDWARE", + Value = cpuCount, + ServerId = context.ServerId, + Metadata = new Dictionary + { + ["cpu_count"] = cpuCount, + ["hyperthread_ratio"] = htRatio, + ["physical_memory_mb"] = physicalMemMb, + ["socket_count"] = socketCount, + ["cores_per_socket"] = coresPerSocket, + ["hadr_enabled"] = hadrEnabled ? 1 : 0 + } + }); + + // WS5 server-health advisories (advise-only). Gating lives here so a fact that would + // score 0 is simply never emitted (noise control); the scorer then scores the emitted + // fact's Value. Shared with Lite — keep the rules identical (see DuckDbFactCollector). + FactCollectorHelpers.EmitServerHealthFacts(context, facts, edition, physicalMemMb, lpim, ifi, dumpCount); + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectServerPropertiesFactsAsync failed", ex); + } + } +} diff --git a/Dashboard/Analysis/SqlServerFactCollector.QueryPerf.cs b/Dashboard/Analysis/SqlServerFactCollector.QueryPerf.cs new file mode 100644 index 000000000..424de0cfb --- /dev/null +++ b/Dashboard/Analysis/SqlServerFactCollector.QueryPerf.cs @@ -0,0 +1,542 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading.Tasks; +using Microsoft.Data.SqlClient; +using PerformanceMonitor.Analysis; +using PerformanceMonitor.PlanAnalysis; +using PerformanceMonitorDashboard.Helpers; + +namespace PerformanceMonitorDashboard.Analysis; + +public partial class SqlServerFactCollector +{ + /// + /// Collects query-level aggregate facts from query_stats. + /// Focuses on spills (memory grant misestimates) and high-parallelism queries. + /// + private async Task CollectQueryStatsFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT + SUM(total_spills) AS total_spills, + COUNT(CASE WHEN max_dop > 8 THEN 1 END) AS high_dop_queries, + COUNT(CASE WHEN total_spills > 0 THEN 1 END) AS spilling_queries, + SUM(execution_count_delta) AS total_executions, + SUM(total_worker_time_delta) AS total_cpu_time_us +FROM collect.query_stats +WHERE collection_time >= @startTime +AND collection_time <= @endTime +AND execution_count_delta > 0"; + + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + using var reader = await cmd.ExecuteReaderAsync(); + if (!await reader.ReadAsync()) return; + + var totalSpills = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); + var highDopQueries = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); + var spillingQueries = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); + var totalExecutions = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)); + var totalCpuTimeUs = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)); + + if (totalSpills > 0) + { + facts.Add(new Fact + { + Source = "queries", + Key = "QUERY_SPILLS", + Value = totalSpills, + ServerId = context.ServerId, + Metadata = new Dictionary + { + ["total_spills"] = totalSpills, + ["spilling_query_count"] = spillingQueries, + ["total_executions"] = totalExecutions + } + }); + } + + if (highDopQueries > 0) + { + facts.Add(new Fact + { + Source = "queries", + Key = "QUERY_HIGH_DOP", + Value = highDopQueries, + ServerId = context.ServerId, + Metadata = new Dictionary + { + ["high_dop_query_count"] = highDopQueries, + ["total_cpu_time_us"] = totalCpuTimeUs, + ["total_executions"] = totalExecutions + } + }); + } + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectQueryStatsFactsAsync failed", ex); + } + } + + /// + /// Detects parameter-sensitive cached plans: a single query_plan_hash whose + /// per-execution worker time varies wildly — one plan serving very different + /// parameter values. Emits one aggregate PARAMETER_SENSITIVITY fact. + /// Note min_*/max_* are cumulative over the plan's cached lifetime, so the + /// finding means "this plan, active now, has a history of widely varying cost". + /// + private async Task CollectParameterSensitivityFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +WITH latest AS +( + SELECT + query_hash, + query_plan_hash, + database_name, + execution_count, + creation_time, + min_worker_time, + max_worker_time, + min_grant_kb, + max_grant_kb, + min_spills, + max_spills, + ROW_NUMBER() OVER + ( + PARTITION BY database_name, query_hash, query_plan_hash + ORDER BY collection_time DESC + ) AS rn + FROM collect.query_stats + WHERE collection_time >= @startTime + AND collection_time <= @endTime + AND execution_count_delta > 0 +) +SELECT + min_worker_time, + max_worker_time, + CAST(max_worker_time AS float) / NULLIF(min_worker_time, 0) AS worker_ratio, + CAST(max_grant_kb AS float) / NULLIF(min_grant_kb, 0) AS grant_ratio, + CASE WHEN max_spills > 0 AND min_spills = 0 THEN 1 ELSE 0 END AS spill_divergence +FROM latest +WHERE rn = 1 +AND min_worker_time >= 10000 +AND max_worker_time >= 250000 +AND execution_count >= 20 +AND creation_time <= @startTime +AND CAST(max_worker_time AS float) / NULLIF(min_worker_time, 0) >= 10 +ORDER BY worker_ratio DESC +OFFSET 0 ROWS FETCH NEXT 20 ROWS ONLY"; + + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + var offenderCount = 0; + var worstRatio = 0.0; + var worstMinWorker = 0L; + var worstMaxWorker = 0L; + var worstGrantRatio = 0.0; + var worstSpillDivergence = 0; + + using var reader = await cmd.ExecuteReaderAsync(); + while (await reader.ReadAsync()) + { + // Rows arrive ordered by worker_ratio DESC — the first row is the worst offender. + if (offenderCount == 0) + { + worstMinWorker = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); + worstMaxWorker = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); + worstRatio = reader.IsDBNull(2) ? 0.0 : Convert.ToDouble(reader.GetValue(2)); + worstGrantRatio = reader.IsDBNull(3) ? 0.0 : Convert.ToDouble(reader.GetValue(3)); + worstSpillDivergence = reader.IsDBNull(4) ? 0 : Convert.ToInt32(reader.GetValue(4)); + } + offenderCount++; + } + + if (offenderCount == 0) return; + + facts.Add(new Fact + { + Source = "queries", + Key = "PARAMETER_SENSITIVITY", + Value = worstRatio, + ServerId = context.ServerId, + Metadata = new Dictionary + { + ["offender_count"] = offenderCount, + ["worst_ratio"] = worstRatio, + ["worst_min_worker_us"] = worstMinWorker, + ["worst_max_worker_us"] = worstMaxWorker, + ["worst_grant_ratio"] = worstGrantRatio, + ["grant_divergence"] = worstGrantRatio >= 5 ? 1 : 0, + ["spill_divergence"] = worstSpillDivergence + } + }); + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectParameterSensitivityFactsAsync failed", ex); + } + } + + /// + /// Detects plan regressions: a query whose currently-active plan has per-execution + /// cost >= 2x the best plan that query is known to perform well with. Emits one + /// aggregate PLAN_REGRESSION fact. Sourced from Query Store (collect.query_store_data); + /// no fact when Query Store is not enabled on the monitored databases. + /// Unlike other collectors this windows on server_last_execution_time (14-day + /// comparison window), NOT collection_time. + /// + private async Task CollectPlanRegressionFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +WITH deduped AS +( + -- Collapse incremental re-collections of the same open runtime-stats interval: + -- keep only the latest collection_time row per logical interval. + SELECT + database_name, + query_id, + plan_id, + query_plan_hash, + count_executions, + avg_cpu_time, + avg_duration, + server_last_execution_time, + is_forced_plan, + force_failure_count, + ROW_NUMBER() OVER + ( + PARTITION BY database_name, query_id, plan_id, server_first_execution_time + ORDER BY collection_time DESC + ) AS rn + FROM collect.query_store_data + WHERE execution_type_desc = N'Regular' + AND server_last_execution_time >= @windowStart +), +plan_agg AS +( + -- Execution-weighted per-exec cost per plan_id. query_plan_hash is invariant + -- within a plan_id, so include it in the GROUP BY rather than aggregating it + -- (MS Learn's MAX page does not list binary/varbinary in the accepted types). + SELECT + database_name, + query_id, + plan_id, + query_plan_hash, + SUM(count_executions) AS execs, + CASE WHEN SUM(count_executions) > 0 + THEN SUM(avg_cpu_time * count_executions) / NULLIF(SUM(count_executions), 0) + ELSE 0 END AS cpu_per_exec, + CASE WHEN SUM(count_executions) > 0 + THEN SUM(avg_duration * count_executions) / NULLIF(SUM(count_executions), 0) + ELSE 0 END AS dur_per_exec, + MAX(server_last_execution_time) AS last_exec, + MAX(CAST(is_forced_plan AS tinyint)) AS is_forced_plan, + MAX(force_failure_count) AS force_failure_count + FROM deduped + WHERE rn = 1 + GROUP BY database_name, query_id, plan_id, query_plan_hash +), +plan_dedup AS +( + -- Collapse plan_ids that share a query_plan_hash (a recompile can produce an + -- identical plan under a new plan_id); keep only plans with enough executions. + SELECT + database_name, + query_id, + query_plan_hash, + SUM(execs) AS execs, + CASE WHEN SUM(execs) > 0 + THEN SUM(cpu_per_exec * execs) / NULLIF(SUM(execs), 0) + ELSE 0 END AS cpu_per_exec, + CASE WHEN SUM(execs) > 0 + THEN SUM(dur_per_exec * execs) / NULLIF(SUM(execs), 0) + ELSE 0 END AS dur_per_exec, + MAX(last_exec) AS last_exec, + MAX(is_forced_plan) AS is_forced_plan, + MAX(force_failure_count) AS force_failure_count + FROM plan_agg + GROUP BY database_name, query_id, query_plan_hash + HAVING SUM(execs) >= 25 +), +ranked AS +( + SELECT + *, + ROW_NUMBER() OVER (PARTITION BY database_name, query_id ORDER BY last_exec DESC) AS recency, + ROW_NUMBER() OVER (PARTITION BY database_name, query_id ORDER BY cpu_per_exec ASC) AS cheapness + FROM plan_dedup +), +compared AS +( + -- Latest active plan vs the best-performing plan for the same query. + SELECT + l.query_id, + l.cpu_per_exec AS latest_cpu, + l.dur_per_exec AS latest_dur, + l.is_forced_plan AS latest_is_forced, + l.force_failure_count AS force_failure_count, + b.cpu_per_exec AS best_cpu, + b.dur_per_exec AS best_dur, + (SELECT MAX(v) + FROM (VALUES + (CAST(l.cpu_per_exec AS float) / NULLIF(b.cpu_per_exec, 0)), + (CAST(l.dur_per_exec AS float) / NULLIF(b.dur_per_exec, 0)) + ) AS x(v)) AS regression_factor + FROM ranked AS l + JOIN ranked AS b + ON b.database_name = l.database_name + AND b.query_id = l.query_id + AND b.cheapness = 1 + WHERE l.recency = 1 + AND l.query_plan_hash <> b.query_plan_hash +) +SELECT + query_id, + latest_cpu, + latest_dur, + latest_is_forced, + force_failure_count, + best_cpu, + best_dur, + regression_factor +FROM compared +WHERE regression_factor >= 2 +ORDER BY regression_factor DESC +OFFSET 0 ROWS FETCH NEXT 20 ROWS ONLY"; + + cmd.Parameters.Add(new SqlParameter("@windowStart", context.TimeRangeStart.AddDays(-14))); + + var offenderCount = 0; + var worstFactor = 0.0; + var worstQueryId = 0L; + var worstLatestCpu = 0.0; + var worstBestCpu = 0.0; + var worstDimension = 1; + var worstLatestForced = 0; + var worstForceFailures = 0L; + + using var reader = await cmd.ExecuteReaderAsync(); + while (await reader.ReadAsync()) + { + // Rows arrive ordered by regression_factor DESC — the first row is the worst offender. + if (offenderCount == 0) + { + worstQueryId = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); + var latestCpu = reader.IsDBNull(1) ? 0.0 : Convert.ToDouble(reader.GetValue(1)); + var latestDur = reader.IsDBNull(2) ? 0.0 : Convert.ToDouble(reader.GetValue(2)); + worstLatestForced = (!reader.IsDBNull(3) && Convert.ToInt32(reader.GetValue(3)) > 0) ? 1 : 0; + worstForceFailures = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)); + var bestCpu = reader.IsDBNull(5) ? 0.0 : Convert.ToDouble(reader.GetValue(5)); + var bestDur = reader.IsDBNull(6) ? 0.0 : Convert.ToDouble(reader.GetValue(6)); + worstFactor = reader.IsDBNull(7) ? 0.0 : Convert.ToDouble(reader.GetValue(7)); + + worstLatestCpu = latestCpu; + worstBestCpu = bestCpu; + var cpuRatio = bestCpu > 0 ? latestCpu / bestCpu : 0.0; + var durRatio = bestDur > 0 ? latestDur / bestDur : 0.0; + worstDimension = cpuRatio >= durRatio ? 1 : 2; // 1 = cpu, 2 = duration + } + offenderCount++; + } + + if (offenderCount == 0) return; + + facts.Add(new Fact + { + Source = "queries", + Key = "PLAN_REGRESSION", + Value = worstFactor, + ServerId = context.ServerId, + Metadata = new Dictionary + { + ["offender_count"] = offenderCount, + ["worst_regression_factor"] = worstFactor, + ["worst_query_id"] = worstQueryId, + ["latest_cpu_per_exec_us"] = worstLatestCpu, + ["best_cpu_per_exec_us"] = worstBestCpu, + ["regressed_dimension"] = worstDimension, + ["latest_is_forced"] = worstLatestForced, + ["force_failure_count"] = worstForceFailures + } + }); + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectPlanRegressionFactsAsync failed", ex); + } + } + + /// + /// Collects procedure stats: top procedure by delta CPU time in the period. + /// + private async Task CollectProcedureStatsFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT + COUNT(DISTINCT object_name) AS distinct_procs, + SUM(execution_count_delta) AS total_executions, + SUM(total_worker_time_delta) AS total_cpu_time_us, + SUM(total_elapsed_time_delta) AS total_elapsed_time_us, + SUM(total_logical_reads_delta) AS total_logical_reads +FROM collect.procedure_stats +WHERE collection_time >= @startTime +AND collection_time <= @endTime +AND execution_count_delta > 0"; + + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + using var reader = await cmd.ExecuteReaderAsync(); + if (!await reader.ReadAsync()) return; + + var distinctProcs = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); + var totalExecs = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); + var totalCpuUs = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); + var totalElapsedUs = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)); + var totalReads = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)); + + if (totalExecs == 0) return; + + facts.Add(new Fact + { + Source = "queries", + Key = "PROCEDURE_STATS", + Value = totalCpuUs, + ServerId = context.ServerId, + Metadata = new Dictionary + { + ["distinct_procedures"] = distinctProcs, + ["total_executions"] = totalExecs, + ["total_cpu_time_us"] = totalCpuUs, + ["total_elapsed_time_us"] = totalElapsedUs, + ["total_logical_reads"] = totalReads, + ["avg_cpu_per_exec_us"] = totalExecs > 0 ? (double)totalCpuUs / totalExecs : 0 + } + }); + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectProcedureStatsFactsAsync failed", ex); + } + } + + /// + /// WS4: plan-XML advisories. Parses the already-collected query plans of the top queries by + /// cost (no live fetch, no DMV) with the shared ShowPlanParser/PlanAnalyzer and emits two + /// advise-only facts — MISSING_INDEX (Value = distinct suggested indexes) and PLAN_WARNING + /// (Value = actionable warnings). The specific indexes/warnings ride in the finding drill-down + /// (SqlServerDrillDownCollector); Fact.Metadata is numeric only. + /// + private async Task CollectPlanAdvisoryFactsAsync(AnalysisContext context, List facts) + { + try + { + var planXmls = new List(); + + using (var connection = new SqlConnection(_connectionString)) + { + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT TOP (10) + plan_xml = CAST(DECOMPRESS(qs.query_plan_text) AS nvarchar(max)) +FROM collect.query_stats AS qs +WHERE qs.collection_time >= @startTime +AND qs.collection_time <= @endTime +AND qs.query_plan_text IS NOT NULL +ORDER BY + qs.total_worker_time DESC;"; + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + using var reader = await cmd.ExecuteReaderAsync(); + while (await reader.ReadAsync()) + { + if (!reader.IsDBNull(0)) + planXmls.Add(reader.GetString(0)); + } + } + + if (planXmls.Count == 0) + return; + + var summary = PlanAdvisoryAggregator.Summarize(planXmls); + + if (summary.MissingIndexCount > 0) + { + facts.Add(new Fact + { + Source = "queries", + Key = "MISSING_INDEX", + Value = summary.MissingIndexCount, + ServerId = context.ServerId, + Metadata = new Dictionary + { + ["index_count"] = summary.MissingIndexCount, + ["max_impact"] = summary.MaxImpact + } + }); + } + + if (summary.WarningCount > 0) + { + facts.Add(new Fact + { + Source = "queries", + Key = "PLAN_WARNING", + Value = summary.WarningCount, + ServerId = context.ServerId, + Metadata = new Dictionary + { + ["warning_count"] = summary.WarningCount, + ["critical_count"] = summary.CriticalCount + } + }); + } + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectPlanAdvisoryFactsAsync failed", ex); + } + } +} diff --git a/Dashboard/Analysis/SqlServerFactCollector.Resources.cs b/Dashboard/Analysis/SqlServerFactCollector.Resources.cs new file mode 100644 index 000000000..2dedac4eb --- /dev/null +++ b/Dashboard/Analysis/SqlServerFactCollector.Resources.cs @@ -0,0 +1,336 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading.Tasks; +using Microsoft.Data.SqlClient; +using PerformanceMonitor.Analysis; +using PerformanceMonitor.PlanAnalysis; +using PerformanceMonitorDashboard.Helpers; + +namespace PerformanceMonitorDashboard.Analysis; + +public partial class SqlServerFactCollector +{ + /// + /// Collects memory stats: total physical RAM, buffer pool size, target memory. + /// These facts enable edition-aware memory recommendations in the config audit. + /// + private async Task CollectMemoryFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT TOP 1 + total_physical_memory_mb, + buffer_pool_mb, + committed_target_memory_mb +FROM collect.memory_stats +WHERE collection_time <= @endTime +ORDER BY collection_time DESC"; + + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + using var reader = await cmd.ExecuteReaderAsync(); + if (!await reader.ReadAsync()) return; + + var totalPhysical = reader.IsDBNull(0) ? 0.0 : Convert.ToDouble(reader.GetValue(0)); + var bufferPool = reader.IsDBNull(1) ? 0.0 : Convert.ToDouble(reader.GetValue(1)); + var targetMemory = reader.IsDBNull(2) ? 0.0 : Convert.ToDouble(reader.GetValue(2)); + + if (totalPhysical > 0) + facts.Add(new Fact { Source = "memory", Key = "MEMORY_TOTAL_PHYSICAL_MB", Value = totalPhysical, ServerId = context.ServerId }); + if (bufferPool > 0) + facts.Add(new Fact { Source = "memory", Key = "MEMORY_BUFFER_POOL_MB", Value = bufferPool, ServerId = context.ServerId }); + if (targetMemory > 0) + facts.Add(new Fact { Source = "memory", Key = "MEMORY_TARGET_MB", Value = targetMemory, ServerId = context.ServerId }); + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectMemoryFactsAsync failed", ex); + } + } + + /// + /// Collects CPU utilization: average and max SQL Server CPU % over the period. + /// Value is average SQL CPU %. Corroborates SOS_SCHEDULER_YIELD. + /// + private async Task CollectCpuUtilizationFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT + AVG(CAST(sqlserver_cpu_utilization AS FLOAT)) AS avg_sql_cpu, + MAX(sqlserver_cpu_utilization) AS max_sql_cpu, + AVG(CAST(other_process_cpu_utilization AS FLOAT)) AS avg_other_cpu, + MAX(other_process_cpu_utilization) AS max_other_cpu, + COUNT(*) AS sample_count +FROM collect.cpu_utilization_stats +WHERE collection_time >= @startTime +AND collection_time <= @endTime"; + + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + using var reader = await cmd.ExecuteReaderAsync(); + if (!await reader.ReadAsync()) return; + + var avgSqlCpu = reader.IsDBNull(0) ? 0.0 : Convert.ToDouble(reader.GetValue(0)); + var maxSqlCpu = reader.IsDBNull(1) ? 0.0 : Convert.ToDouble(reader.GetValue(1)); + var avgOtherCpu = reader.IsDBNull(2) ? 0.0 : Convert.ToDouble(reader.GetValue(2)); + var maxOtherCpu = reader.IsDBNull(3) ? 0.0 : Convert.ToDouble(reader.GetValue(3)); + var sampleCount = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)); + + if (sampleCount == 0) return; + + var cpuMetadata = new Dictionary + { + ["avg_sql_cpu"] = avgSqlCpu, + ["max_sql_cpu"] = maxSqlCpu, + ["avg_other_cpu"] = avgOtherCpu, + ["max_other_cpu"] = maxOtherCpu, + ["avg_total_cpu"] = avgSqlCpu + avgOtherCpu, + ["sample_count"] = sampleCount + }; + + facts.Add(new Fact + { + Source = "cpu", + Key = "CPU_SQL_PERCENT", + Value = avgSqlCpu, + ServerId = context.ServerId, + Metadata = cpuMetadata + }); + + // Emit a CPU_SPIKE fact when max is high and significantly above average. + // This catches bursty CPU events that average-based scoring misses entirely. + // Requires max >= 80% AND at least 3x the average (or avg < 20% with max >= 80%). + if (maxSqlCpu >= 80 && (avgSqlCpu < 20 || maxSqlCpu / Math.Max(avgSqlCpu, 1) >= 3)) + { + facts.Add(new Fact + { + Source = "cpu", + Key = "CPU_SPIKE", + Value = maxSqlCpu, + ServerId = context.ServerId, + Metadata = cpuMetadata + }); + } + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectCpuUtilizationFactsAsync failed", ex); + } + } + + /// + /// Collects memory grant facts from the memory_grant_stats table. + /// Detects grant waiters (sessions waiting for memory) and grant pressure. + /// + private async Task CollectMemoryGrantFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT + MAX(waiter_count) AS max_waiters, + AVG(CAST(waiter_count AS FLOAT)) AS avg_waiters, + MAX(grantee_count) AS max_grantees, + SUM(timeout_error_count_delta) AS total_timeout_errors, + SUM(forced_grant_count_delta) AS total_forced_grants +FROM collect.memory_grant_stats +WHERE collection_time >= @startTime +AND collection_time <= @endTime"; + + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + using var reader = await cmd.ExecuteReaderAsync(); + if (!await reader.ReadAsync()) return; + + var maxWaiters = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); + var avgWaiters = reader.IsDBNull(1) ? 0.0 : Convert.ToDouble(reader.GetValue(1)); + var maxGrantees = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); + var totalTimeouts = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)); + var totalForcedGrants = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)); + + // Only create a fact if there's evidence of grant pressure + if (maxWaiters <= 0 && totalTimeouts <= 0 && totalForcedGrants <= 0) return; + + facts.Add(new Fact + { + Source = "memory", + Key = "MEMORY_GRANT_PENDING", + Value = maxWaiters, + ServerId = context.ServerId, + Metadata = new Dictionary + { + ["max_waiters"] = maxWaiters, + ["avg_waiters"] = avgWaiters, + ["max_grantees"] = maxGrantees, + ["total_timeout_errors"] = totalTimeouts, + ["total_forced_grants"] = totalForcedGrants + } + }); + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectMemoryGrantFactsAsync failed", ex); + } + } + + /// + /// Collects key perfmon throughput counters: Batch Requests/sec, compilations, recompilations. + /// Unscored context that distinguishes a busy server from a sick one (used by the AI surfaces). + /// + private async Task CollectPerfmonFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +;WITH latest AS ( + SELECT + counter_name, + cntr_value, + cntr_value_delta, + ROW_NUMBER() OVER (PARTITION BY counter_name ORDER BY collection_time DESC) AS rn + FROM collect.perfmon_stats + WHERE collection_time >= @startTime + AND collection_time <= @endTime + AND counter_name IN ('Batch Requests/sec', 'SQL Compilations/sec', 'SQL Re-Compilations/sec') +) +SELECT counter_name, cntr_value, cntr_value_delta +FROM latest WHERE rn = 1"; + + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + using var reader = await cmd.ExecuteReaderAsync(); + while (await reader.ReadAsync()) + { + var counterName = reader.GetString(0); + var cntrValue = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); + var deltaValue = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); + + var (factKey, source) = counterName switch + { + "Batch Requests/sec" => ("PERFMON_BATCH_REQ_SEC", "perfmon"), + "SQL Compilations/sec" => ("PERFMON_COMPILATIONS_SEC", "perfmon"), + "SQL Re-Compilations/sec" => ("PERFMON_RECOMPILATIONS_SEC", "perfmon"), + _ => (null, null) + }; + + if (factKey == null) continue; + + // All remaining counters are per-second rates — use the delta. + var value = (double)deltaValue; + + facts.Add(new Fact + { + Source = source!, + Key = factKey, + Value = value, + ServerId = context.ServerId, + Metadata = new Dictionary + { + ["cntr_value"] = cntrValue, + ["delta_cntr_value"] = deltaValue + } + }); + } + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectPerfmonFactsAsync failed", ex); + } + } + + /// + /// Collects top memory clerks by size. Context for understanding where memory is allocated. + /// Dashboard stores pages_kb — convert to MB for consistency with Lite facts. + /// + private async Task CollectMemoryClerkFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +;WITH latest AS ( + SELECT + clerk_type, + SUM(pages_kb) / 1024.0 AS memory_mb, + ROW_NUMBER() OVER (PARTITION BY clerk_type ORDER BY collection_time DESC) AS rn, + collection_time + FROM collect.memory_clerks_stats + WHERE collection_time <= @endTime + GROUP BY clerk_type, collection_time +) +SELECT TOP 10 clerk_type, memory_mb +FROM latest WHERE rn = 1 AND memory_mb > 0 +ORDER BY memory_mb DESC"; + + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + using var reader = await cmd.ExecuteReaderAsync(); + var metadata = new Dictionary(); + var totalMb = 0.0; + var clerkCount = 0; + + while (await reader.ReadAsync()) + { + var clerkType = reader.GetString(0); + var memoryMb = Convert.ToDouble(reader.GetValue(1)); + metadata[clerkType] = memoryMb; + totalMb += memoryMb; + clerkCount++; + } + + if (clerkCount == 0) return; + + metadata["total_top_clerks_mb"] = totalMb; + metadata["clerk_count"] = clerkCount; + + facts.Add(new Fact + { + Source = "memory", + Key = "MEMORY_CLERKS", + Value = totalMb, + ServerId = context.ServerId, + Metadata = metadata + }); + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectMemoryClerkFactsAsync failed", ex); + } + } +} diff --git a/Dashboard/Analysis/SqlServerFactCollector.Storage.cs b/Dashboard/Analysis/SqlServerFactCollector.Storage.cs new file mode 100644 index 000000000..b8124d8ad --- /dev/null +++ b/Dashboard/Analysis/SqlServerFactCollector.Storage.cs @@ -0,0 +1,338 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading.Tasks; +using Microsoft.Data.SqlClient; +using PerformanceMonitor.Analysis; +using PerformanceMonitor.PlanAnalysis; +using PerformanceMonitorDashboard.Helpers; + +namespace PerformanceMonitorDashboard.Analysis; + +public partial class SqlServerFactCollector +{ + /// + /// Collects total database data size from file_io_stats. + /// Sums the latest size_on_disk_bytes across all database files for the server. + /// + private async Task CollectDatabaseSizeFactAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +;WITH latest AS ( + SELECT + database_name, + file_name, + size_on_disk_bytes, + ROW_NUMBER() OVER (PARTITION BY database_name, file_name ORDER BY collection_time DESC) AS rn + FROM collect.file_io_stats + WHERE collection_time <= @endTime + AND size_on_disk_bytes > 0 +) +SELECT SUM(size_on_disk_bytes / 1048576.0) AS total_size_mb +FROM latest +WHERE rn = 1"; + + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + using var reader = await cmd.ExecuteReaderAsync(); + if (!await reader.ReadAsync()) return; + + var totalSize = reader.IsDBNull(0) ? 0.0 : Convert.ToDouble(reader.GetValue(0)); + if (totalSize > 0) + facts.Add(new Fact { Source = "config", Key = "DATABASE_TOTAL_SIZE_MB", Value = totalSize, ServerId = context.ServerId }); + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectDatabaseSizeFactAsync failed", ex); + } + } + + /// + /// Collects I/O latency from file_io_stats delta columns. + /// Computes average read and write latency across all database files. + /// + private async Task CollectIoLatencyFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT + SUM(io_stall_read_ms_delta) AS total_stall_read_ms, + SUM(num_of_reads_delta) AS total_reads, + SUM(io_stall_write_ms_delta) AS total_stall_write_ms, + SUM(num_of_writes_delta) AS total_writes +FROM collect.file_io_stats +WHERE collection_time >= @startTime +AND collection_time <= @endTime +AND (num_of_reads_delta > 0 OR num_of_writes_delta > 0)"; + + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + using var reader = await cmd.ExecuteReaderAsync(); + if (!await reader.ReadAsync()) return; + + var totalStallReadMs = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); + var totalReads = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); + var totalStallWriteMs = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); + var totalWrites = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)); + + if (totalReads > 0) + { + var avgReadLatency = (double)totalStallReadMs / totalReads; + facts.Add(new Fact + { + Source = "io", + Key = "IO_READ_LATENCY_MS", + Value = avgReadLatency, + ServerId = context.ServerId, + Metadata = new Dictionary + { + ["avg_read_latency_ms"] = avgReadLatency, + ["total_stall_read_ms"] = totalStallReadMs, + ["total_reads"] = totalReads + } + }); + } + + if (totalWrites > 0) + { + var avgWriteLatency = (double)totalStallWriteMs / totalWrites; + facts.Add(new Fact + { + Source = "io", + Key = "IO_WRITE_LATENCY_MS", + Value = avgWriteLatency, + ServerId = context.ServerId, + Metadata = new Dictionary + { + ["avg_write_latency_ms"] = avgWriteLatency, + ["total_stall_write_ms"] = totalStallWriteMs, + ["total_writes"] = totalWrites + } + }); + } + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectIoLatencyFactsAsync failed", ex); + } + } + + /// + /// Collects TempDB usage facts: max usage, version store size, and unallocated space. + /// Value is max total_reserved_mb over the period. + /// Dashboard uses computed columns (total_reserved_mb, etc.) from collect.tempdb_stats. + /// + private async Task CollectTempDbFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT + MAX(total_reserved_mb) AS max_total_reserved_mb, + MAX(user_object_reserved_mb) AS max_user_object_mb, + MAX(internal_object_reserved_mb) AS max_internal_object_mb, + MAX(version_store_reserved_mb) AS max_version_store_mb, + MIN(unallocated_mb) AS min_unallocated_mb, + AVG(CAST(total_reserved_mb AS FLOAT)) AS avg_total_reserved_mb +FROM collect.tempdb_stats +WHERE collection_time >= @startTime +AND collection_time <= @endTime"; + + cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + using var reader = await cmd.ExecuteReaderAsync(); + if (!await reader.ReadAsync()) return; + + var maxReserved = reader.IsDBNull(0) ? 0.0 : Convert.ToDouble(reader.GetValue(0)); + var maxUserObj = reader.IsDBNull(1) ? 0.0 : Convert.ToDouble(reader.GetValue(1)); + var maxInternalObj = reader.IsDBNull(2) ? 0.0 : Convert.ToDouble(reader.GetValue(2)); + var maxVersionStore = reader.IsDBNull(3) ? 0.0 : Convert.ToDouble(reader.GetValue(3)); + var minUnallocated = reader.IsDBNull(4) ? 0.0 : Convert.ToDouble(reader.GetValue(4)); + var avgReserved = reader.IsDBNull(5) ? 0.0 : Convert.ToDouble(reader.GetValue(5)); + + if (maxReserved <= 0) return; + + // TempDB usage as fraction of total space (reserved + unallocated) + var totalSpace = maxReserved + minUnallocated; + var usageFraction = totalSpace > 0 ? maxReserved / totalSpace : 0; + + facts.Add(new Fact + { + Source = "tempdb", + Key = "TEMPDB_USAGE", + Value = usageFraction, + ServerId = context.ServerId, + Metadata = new Dictionary + { + ["max_reserved_mb"] = maxReserved, + ["avg_reserved_mb"] = avgReserved, + ["max_user_object_mb"] = maxUserObj, + ["max_internal_object_mb"] = maxInternalObj, + ["max_version_store_mb"] = maxVersionStore, + ["min_unallocated_mb"] = minUnallocated, + ["usage_fraction"] = usageFraction + } + }); + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectTempDbFactsAsync failed", ex); + } + } + + /// + /// Collects the percent-autogrowth-on-large-files config fact (WS3): data/log files set + /// to grow in PERCENTAGE steps that are also large (>= 10 GB), where a single growth is a + /// huge, stalling allocation. Reads the latest snapshot per file from + /// collect.database_size_stats, excludes system databases, and emits ONE aggregate + /// FILE_AUTOGROWTH_PERCENT fact carrying the offending-file/database counts (the per-file + /// detail + copy-paste fix is attached later by the drill-down collector). + /// + private async Task CollectFileAutogrowthFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +;WITH latest AS ( + SELECT + database_name, + file_id, + total_size_mb, + is_percent_growth, + ROW_NUMBER() OVER (PARTITION BY database_name, file_id ORDER BY collection_time DESC) AS rn + FROM collect.database_size_stats + WHERE database_name NOT IN ('master', 'msdb', 'model', 'tempdb') +) +SELECT + file_count = COUNT(*), + database_count = COUNT(DISTINCT database_name) +FROM latest +WHERE rn = 1 +AND is_percent_growth = 1 +AND total_size_mb >= @minSizeMb;"; + + cmd.Parameters.Add(new SqlParameter("@minSizeMb", 10240.0)); /* 10 GB */ + + using var reader = await cmd.ExecuteReaderAsync(); + if (!await reader.ReadAsync()) return; + + var fileCount = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); + if (fileCount == 0) return; + + var databaseCount = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); + + facts.Add(new Fact + { + Source = "config", + Key = "FILE_AUTOGROWTH_PERCENT", + Value = fileCount, + ServerId = context.ServerId, + Metadata = new Dictionary + { + ["file_count"] = fileCount, + ["database_count"] = databaseCount + } + }); + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectFileAutogrowthFactsAsync failed", ex); + } + } + + /// + /// Collects disk space facts from database_size_stats: volume free space, file sizes. + /// + private async Task CollectDiskSpaceFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + cmd.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +;WITH latest AS ( + SELECT + volume_mount_point, + volume_total_mb, + volume_free_mb, + ROW_NUMBER() OVER (PARTITION BY volume_mount_point ORDER BY collection_time DESC) AS rn + FROM collect.database_size_stats + WHERE collection_time <= @endTime + AND volume_total_mb > 0 +) +SELECT + MIN(volume_free_mb * 1.0 / volume_total_mb) AS min_free_pct, + MIN(volume_free_mb) AS min_free_mb, + COUNT(DISTINCT volume_mount_point) AS volume_count, + SUM(volume_total_mb) AS total_volume_mb, + SUM(volume_free_mb) AS total_free_mb +FROM latest WHERE rn = 1"; + + cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + using var reader = await cmd.ExecuteReaderAsync(); + if (!await reader.ReadAsync()) return; + + var minFreePct = reader.IsDBNull(0) ? 1.0 : Convert.ToDouble(reader.GetValue(0)); + var minFreeMb = reader.IsDBNull(1) ? 0.0 : Convert.ToDouble(reader.GetValue(1)); + var volumeCount = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); + var totalVolumeMb = reader.IsDBNull(3) ? 0.0 : Convert.ToDouble(reader.GetValue(3)); + var totalFreeMb = reader.IsDBNull(4) ? 0.0 : Convert.ToDouble(reader.GetValue(4)); + + if (volumeCount == 0) return; + + facts.Add(new Fact + { + Source = "disk", + Key = "DISK_SPACE", + Value = minFreePct, + ServerId = context.ServerId, + Metadata = new Dictionary + { + ["min_free_pct"] = minFreePct, + ["min_free_mb"] = minFreeMb, + ["volume_count"] = volumeCount, + ["total_volume_mb"] = totalVolumeMb, + ["total_free_mb"] = totalFreeMb + } + }); + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectDiskSpaceFactsAsync failed", ex); + } + } +} diff --git a/Dashboard/Analysis/SqlServerFactCollector.Waits.cs b/Dashboard/Analysis/SqlServerFactCollector.Waits.cs new file mode 100644 index 000000000..bd6305a0b --- /dev/null +++ b/Dashboard/Analysis/SqlServerFactCollector.Waits.cs @@ -0,0 +1,274 @@ +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading.Tasks; +using Microsoft.Data.SqlClient; +using PerformanceMonitor.Analysis; +using PerformanceMonitor.PlanAnalysis; +using PerformanceMonitorDashboard.Helpers; + +namespace PerformanceMonitorDashboard.Analysis; + +public partial class SqlServerFactCollector +{ + /// + /// Collects wait stats facts — one Fact per significant wait type. + /// Value is wait_time_ms / period_duration_ms (fraction of examined period). + /// + private async Task CollectWaitStatsFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var command = connection.CreateCommand(); + command.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT + wait_type, + SUM(waiting_tasks_count_delta) AS total_waiting_tasks, + SUM(wait_time_ms_delta) AS total_wait_time_ms, + SUM(signal_wait_time_ms_delta) AS total_signal_wait_time_ms +FROM collect.wait_stats +WHERE collection_time >= @startTime +AND collection_time <= @endTime +AND wait_time_ms_delta > 0 +GROUP BY wait_type +ORDER BY SUM(wait_time_ms_delta) DESC"; + + command.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + command.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + using var reader = await command.ExecuteReaderAsync(); + while (await reader.ReadAsync()) + { + var waitType = reader.GetString(0); + var waitingTasks = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); + var waitTimeMs = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); + var signalWaitTimeMs = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)); + + if (waitTimeMs <= 0) continue; + + var fractionOfPeriod = waitTimeMs / context.PeriodDurationMs; + var avgMsPerWait = waitingTasks > 0 ? (double)waitTimeMs / waitingTasks : 0; + + facts.Add(new Fact + { + Source = "waits", + Key = waitType, + Value = fractionOfPeriod, + ServerId = context.ServerId, + Metadata = new Dictionary + { + ["wait_time_ms"] = waitTimeMs, + ["waiting_tasks_count"] = waitingTasks, + ["signal_wait_time_ms"] = signalWaitTimeMs, + ["resource_wait_time_ms"] = waitTimeMs - signalWaitTimeMs, + ["avg_ms_per_wait"] = avgMsPerWait, + ["period_duration_ms"] = context.PeriodDurationMs + } + }); + } + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectWaitStatsFactsAsync failed", ex); + } + } + + /// + /// Collects blocking facts from blocking_BlockedProcessReport. + /// Produces a single BLOCKING_EVENTS fact with event count, rate, and details. + /// Value is events per hour for threshold comparison. + /// + private async Task CollectBlockingFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var command = connection.CreateCommand(); + command.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT + COUNT(*) AS event_count, + AVG(CAST(wait_time_ms AS FLOAT)) AS avg_wait_time_ms, + MAX(wait_time_ms) AS max_wait_time_ms, + COUNT(DISTINCT spid) AS distinct_head_blockers, + COUNT(CASE WHEN status = 'sleeping' THEN 1 END) AS sleeping_blocker_count +FROM collect.blocking_BlockedProcessReport +WHERE collection_time >= @startTime +AND collection_time <= @endTime"; + + command.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + command.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + using var reader = await command.ExecuteReaderAsync(); + if (!await reader.ReadAsync()) return; + + var eventCount = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); + if (eventCount <= 0) return; + + var avgWaitTimeMs = reader.IsDBNull(1) ? 0.0 : Convert.ToDouble(reader.GetValue(1)); + var maxWaitTimeMs = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); + var distinctHeadBlockers = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)); + var sleepingBlockerCount = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)); + + var periodHours = context.PeriodDurationMs / 3_600_000.0; + var eventsPerHour = periodHours > 0 ? eventCount / periodHours : 0; + + facts.Add(new Fact + { + Source = "blocking", + Key = "BLOCKING_EVENTS", + Value = eventsPerHour, + ServerId = context.ServerId, + Metadata = new Dictionary + { + ["event_count"] = eventCount, + ["events_per_hour"] = eventsPerHour, + ["avg_wait_time_ms"] = avgWaitTimeMs, + ["max_wait_time_ms"] = maxWaitTimeMs, + ["distinct_head_blockers"] = distinctHeadBlockers, + ["sleeping_blocker_count"] = sleepingBlockerCount, + ["period_hours"] = periodHours + } + }); + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectBlockingFactsAsync failed", ex); + } + } + + /// + /// Collects deadlock facts from the deadlocks table. + /// Produces a single DEADLOCKS fact with count and rate. + /// Value is deadlocks per hour for threshold comparison. + /// + private async Task CollectDeadlockFactsAsync(AnalysisContext context, List facts) + { + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var command = connection.CreateCommand(); + command.CommandText = @" +SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; + +SELECT COUNT(*) AS deadlock_count +FROM collect.deadlocks +WHERE collection_time >= @startTime +AND collection_time <= @endTime"; + + command.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); + command.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); + + using var reader = await command.ExecuteReaderAsync(); + if (!await reader.ReadAsync()) return; + + var deadlockCount = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); + if (deadlockCount <= 0) return; + + var periodHours = context.PeriodDurationMs / 3_600_000.0; + var deadlocksPerHour = periodHours > 0 ? deadlockCount / periodHours : 0; + + facts.Add(new Fact + { + Source = "blocking", + Key = "DEADLOCKS", + Value = deadlocksPerHour, + ServerId = context.ServerId, + Metadata = new Dictionary + { + ["deadlock_count"] = deadlockCount, + ["deadlocks_per_hour"] = deadlocksPerHour, + ["period_hours"] = periodHours + } + }); + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectDeadlockFactsAsync failed", ex); + } + } + + /// + /// Reconstructs blocking chains from collect.blocking_BlockedProcessReport (one row + /// per side of each blocking event, dedup'd to activity = 'blocked') and emits + /// one aggregate BLOCKING_CHAIN fact describing the worst chain — apex head blocker, + /// depth, transitive victim count. Structure the BLOCKING_EVENTS rate is blind to. + /// Reads typed blocker-side columns populated by collect.process_blocked_process_xml, + /// so no XML re-parse on the analysis hot path. + /// + private async Task CollectBlockingChainFactsAsync(AnalysisContext context, List facts) + { + const int maxPairs = 5000; + const int maxDepth = 50; + const int stepBudget = 100_000; + + try + { + using var connection = new SqlConnection(_connectionString); + await connection.OpenAsync(); + + using var cmd = connection.CreateCommand(); + // Shared query/filter — see BlockingPairRowQuery. ORDER BY collection_time DESC is a backward + // CIX scan (sort-free); event_time is a residual predicate. Keeping this in lockstep with the + // drill-down + viewer fetch is the whole point: all three agree on the apex. + cmd.CommandText = BlockingPairRowQuery.Sql; + BlockingPairRowQuery.AddParameters(cmd, context.TimeRangeStart, context.TimeRangeEnd); + + var rows = new List(); + using (var reader = await cmd.ExecuteReaderAsync()) + { + while (await reader.ReadAsync()) + rows.Add(BlockingPairRowQuery.Read(reader)); + } + + // Always-on DMV blocking snapshot fallback (works when the blocked-process-report XE is empty, + // e.g. AWS RDS). Merge BEFORE the empty check so DMV-only blocking still produces facts. + await BlockingPairRowQuery.AppendDmvSnapshotRowsAsync(connection, rows, context.TimeRangeStart, context.TimeRangeEnd); + + if (rows.Count == 0) return; + + // Cumulative (not per-scan): merges an episode's re-fires across the window so the severity fact + // keeps window-level depth/victim counts (per-scan scoping would under-count and under-fire). + var reconstruction = BlockingChainReconstructor.Reconstruct(rows, maxDepth, maxPairs, stepBudget, scopeByMonitorLoop: false); + if (reconstruction.Chains.Count == 0) return; + + var worst = reconstruction.Chains[0]; + + facts.Add(new Fact + { + Source = "blocking", + Key = "BLOCKING_CHAIN", + Value = worst.Depth, + ServerId = context.ServerId, + Metadata = new Dictionary + { + ["worst_chain_depth"] = worst.Depth, + ["worst_chain_victim_count"] = worst.VictimCount, + ["worst_apex_spid"] = worst.ApexSpid, + ["worst_apex_sleeping"] = worst.ApexSleeping ? 1 : 0, + ["worst_chain_max_wait_ms"] = worst.MaxWaitMs, + ["total_reconstructed_chains"] = reconstruction.Chains.Count, + ["deepest_chain_overall"] = reconstruction.Chains.Max(c => c.Depth), + ["max_victim_count_overall"] = reconstruction.Chains.Max(c => c.VictimCount), + ["depth_capped"] = reconstruction.DepthCapped ? 1 : 0, + ["traversal_truncated"] = reconstruction.TraversalTruncated ? 1 : 0, + ["cycle_detected"] = reconstruction.CycleDetected ? 1 : 0 + } + }); + } + catch (Exception ex) + { + Logger.Error("SqlServerFactCollector.CollectBlockingChainFactsAsync failed", ex); + } + } +} diff --git a/Dashboard/Analysis/SqlServerFactCollector.cs b/Dashboard/Analysis/SqlServerFactCollector.cs index 29cd1462d..f9bd9af59 100644 --- a/Dashboard/Analysis/SqlServerFactCollector.cs +++ b/Dashboard/Analysis/SqlServerFactCollector.cs @@ -14,7 +14,7 @@ namespace PerformanceMonitorDashboard.Analysis; /// Each fact category has its own collection method, added incrementally. /// Port of DuckDbFactCollector from Lite — queries collect.* tables instead of DuckDB views. /// -public class SqlServerFactCollector : IFactCollector +public partial class SqlServerFactCollector : IFactCollector { private readonly string _connectionString; @@ -28,8 +28,8 @@ public async Task> CollectFactsAsync(AnalysisContext context) var facts = new List(); await CollectWaitStatsFactsAsync(context, facts); - GroupGeneralLockWaits(facts, context); - GroupParallelismWaits(facts, context); + FactCollectorHelpers.GroupGeneralLockWaits(facts, context); + FactCollectorHelpers.GroupParallelismWaits(facts, context); await CollectBlockingFactsAsync(context, facts); await CollectDeadlockFactsAsync(context, facts); await CollectServerConfigFactsAsync(context, facts); @@ -61,1909 +61,6 @@ public async Task> CollectFactsAsync(AnalysisContext context) return facts; } - /// - /// Collects wait stats facts — one Fact per significant wait type. - /// Value is wait_time_ms / period_duration_ms (fraction of examined period). - /// - private async Task CollectWaitStatsFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var command = connection.CreateCommand(); - command.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT - wait_type, - SUM(waiting_tasks_count_delta) AS total_waiting_tasks, - SUM(wait_time_ms_delta) AS total_wait_time_ms, - SUM(signal_wait_time_ms_delta) AS total_signal_wait_time_ms -FROM collect.wait_stats -WHERE collection_time >= @startTime -AND collection_time <= @endTime -AND wait_time_ms_delta > 0 -GROUP BY wait_type -ORDER BY SUM(wait_time_ms_delta) DESC"; - - command.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - command.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - using var reader = await command.ExecuteReaderAsync(); - while (await reader.ReadAsync()) - { - var waitType = reader.GetString(0); - var waitingTasks = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); - var waitTimeMs = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); - var signalWaitTimeMs = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)); - - if (waitTimeMs <= 0) continue; - - var fractionOfPeriod = waitTimeMs / context.PeriodDurationMs; - var avgMsPerWait = waitingTasks > 0 ? (double)waitTimeMs / waitingTasks : 0; - - facts.Add(new Fact - { - Source = "waits", - Key = waitType, - Value = fractionOfPeriod, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["wait_time_ms"] = waitTimeMs, - ["waiting_tasks_count"] = waitingTasks, - ["signal_wait_time_ms"] = signalWaitTimeMs, - ["resource_wait_time_ms"] = waitTimeMs - signalWaitTimeMs, - ["avg_ms_per_wait"] = avgMsPerWait, - ["period_duration_ms"] = context.PeriodDurationMs - } - }); - } - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectWaitStatsFactsAsync failed", ex); - } - } - - /// - /// Collects blocking facts from blocking_BlockedProcessReport. - /// Produces a single BLOCKING_EVENTS fact with event count, rate, and details. - /// Value is events per hour for threshold comparison. - /// - private async Task CollectBlockingFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var command = connection.CreateCommand(); - command.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT - COUNT(*) AS event_count, - AVG(CAST(wait_time_ms AS FLOAT)) AS avg_wait_time_ms, - MAX(wait_time_ms) AS max_wait_time_ms, - COUNT(DISTINCT spid) AS distinct_head_blockers, - COUNT(CASE WHEN status = 'sleeping' THEN 1 END) AS sleeping_blocker_count -FROM collect.blocking_BlockedProcessReport -WHERE collection_time >= @startTime -AND collection_time <= @endTime"; - - command.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - command.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - using var reader = await command.ExecuteReaderAsync(); - if (!await reader.ReadAsync()) return; - - var eventCount = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); - if (eventCount <= 0) return; - - var avgWaitTimeMs = reader.IsDBNull(1) ? 0.0 : Convert.ToDouble(reader.GetValue(1)); - var maxWaitTimeMs = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); - var distinctHeadBlockers = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)); - var sleepingBlockerCount = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)); - - var periodHours = context.PeriodDurationMs / 3_600_000.0; - var eventsPerHour = periodHours > 0 ? eventCount / periodHours : 0; - - facts.Add(new Fact - { - Source = "blocking", - Key = "BLOCKING_EVENTS", - Value = eventsPerHour, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["event_count"] = eventCount, - ["events_per_hour"] = eventsPerHour, - ["avg_wait_time_ms"] = avgWaitTimeMs, - ["max_wait_time_ms"] = maxWaitTimeMs, - ["distinct_head_blockers"] = distinctHeadBlockers, - ["sleeping_blocker_count"] = sleepingBlockerCount, - ["period_hours"] = periodHours - } - }); - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectBlockingFactsAsync failed", ex); - } - } - - /// - /// Collects deadlock facts from the deadlocks table. - /// Produces a single DEADLOCKS fact with count and rate. - /// Value is deadlocks per hour for threshold comparison. - /// - private async Task CollectDeadlockFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var command = connection.CreateCommand(); - command.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT COUNT(*) AS deadlock_count -FROM collect.deadlocks -WHERE collection_time >= @startTime -AND collection_time <= @endTime"; - - command.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - command.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - using var reader = await command.ExecuteReaderAsync(); - if (!await reader.ReadAsync()) return; - - var deadlockCount = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); - if (deadlockCount <= 0) return; - - var periodHours = context.PeriodDurationMs / 3_600_000.0; - var deadlocksPerHour = periodHours > 0 ? deadlockCount / periodHours : 0; - - facts.Add(new Fact - { - Source = "blocking", - Key = "DEADLOCKS", - Value = deadlocksPerHour, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["deadlock_count"] = deadlockCount, - ["deadlocks_per_hour"] = deadlocksPerHour, - ["period_hours"] = periodHours - } - }); - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectDeadlockFactsAsync failed", ex); - } - } - - /// - /// Collects server configuration settings relevant to analysis. - /// These become facts that amplifiers and the config audit tool can reference - /// to make recommendations specific (e.g., "your CTFP is 50" vs "check CTFP"). - /// - private async Task CollectServerConfigFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - // Latest value PER configuration_name (ROW_NUMBER, not TOP N): server_configuration_history - // accumulates a row per collection, so a naive TOP-N-ORDER-BY-time returns the newest N - // ROWS — which collapses to one config when collections are frequent, silently dropping - // settings. Partition by name and take rn = 1 so each requested setting is its latest value. - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -;WITH latest AS ( - SELECT - configuration_name, - CAST(value_in_use AS BIGINT) AS value_in_use, - ROW_NUMBER() OVER (PARTITION BY configuration_name ORDER BY collection_time DESC) AS rn - FROM config.server_configuration_history - WHERE configuration_name IN ( - 'cost threshold for parallelism', - 'max degree of parallelism', - 'max server memory (MB)', - 'min server memory (MB)', - 'max worker threads' - ) -) -SELECT - configuration_name, - value_in_use -FROM latest -WHERE rn = 1"; - - // max/min server memory are read alongside the rooted CONFIG_* facts so the - // narrow-memory derivation below can compare them without a second query. - double? maxMemoryMb = null; - double? minMemoryMb = null; - - using (var reader = await cmd.ExecuteReaderAsync()) - { - while (await reader.ReadAsync()) - { - var configName = reader.GetString(0); - var value = Convert.ToDouble(reader.GetValue(1)); - - if (configName == "max server memory (MB)") maxMemoryMb = value; - if (configName == "min server memory (MB)") minMemoryMb = value; - - var factKey = configName switch - { - "cost threshold for parallelism" => "CONFIG_CTFP", - "max degree of parallelism" => "CONFIG_MAXDOP", - "max server memory (MB)" => "CONFIG_MAX_MEMORY_MB", - "min server memory (MB)" => "CONFIG_MIN_MEMORY_MB", - "max worker threads" => "CONFIG_MAX_WORKER_THREADS", - _ => null - }; - - if (factKey == null) continue; - - facts.Add(new Fact - { - Source = "config", - Key = factKey, - Value = value, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["value_in_use"] = value - } - }); - } - } - - // CONFIG_MIN_MAX_MEMORY_NARROW: emitted only when max is configured AND min is pinned - // near it (shared rule so Dashboard/Lite agree). - var narrow = FactRemediation.BuildNarrowMemoryFact(context.ServerId, maxMemoryMb, minMemoryMb); - if (narrow is not null) - facts.Add(narrow); - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectServerConfigFactsAsync failed", ex); - } - } - - /// - /// Collects memory stats: total physical RAM, buffer pool size, target memory. - /// These facts enable edition-aware memory recommendations in the config audit. - /// - private async Task CollectMemoryFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT TOP 1 - total_physical_memory_mb, - buffer_pool_mb, - committed_target_memory_mb -FROM collect.memory_stats -WHERE collection_time <= @endTime -ORDER BY collection_time DESC"; - - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - using var reader = await cmd.ExecuteReaderAsync(); - if (!await reader.ReadAsync()) return; - - var totalPhysical = reader.IsDBNull(0) ? 0.0 : Convert.ToDouble(reader.GetValue(0)); - var bufferPool = reader.IsDBNull(1) ? 0.0 : Convert.ToDouble(reader.GetValue(1)); - var targetMemory = reader.IsDBNull(2) ? 0.0 : Convert.ToDouble(reader.GetValue(2)); - - if (totalPhysical > 0) - facts.Add(new Fact { Source = "memory", Key = "MEMORY_TOTAL_PHYSICAL_MB", Value = totalPhysical, ServerId = context.ServerId }); - if (bufferPool > 0) - facts.Add(new Fact { Source = "memory", Key = "MEMORY_BUFFER_POOL_MB", Value = bufferPool, ServerId = context.ServerId }); - if (targetMemory > 0) - facts.Add(new Fact { Source = "memory", Key = "MEMORY_TARGET_MB", Value = targetMemory, ServerId = context.ServerId }); - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectMemoryFactsAsync failed", ex); - } - } - - /// - /// Collects total database data size from file_io_stats. - /// Sums the latest size_on_disk_bytes across all database files for the server. - /// - private async Task CollectDatabaseSizeFactAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -;WITH latest AS ( - SELECT - database_name, - file_name, - size_on_disk_bytes, - ROW_NUMBER() OVER (PARTITION BY database_name, file_name ORDER BY collection_time DESC) AS rn - FROM collect.file_io_stats - WHERE collection_time <= @endTime - AND size_on_disk_bytes > 0 -) -SELECT SUM(size_on_disk_bytes / 1048576.0) AS total_size_mb -FROM latest -WHERE rn = 1"; - - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - using var reader = await cmd.ExecuteReaderAsync(); - if (!await reader.ReadAsync()) return; - - var totalSize = reader.IsDBNull(0) ? 0.0 : Convert.ToDouble(reader.GetValue(0)); - if (totalSize > 0) - facts.Add(new Fact { Source = "config", Key = "DATABASE_TOTAL_SIZE_MB", Value = totalSize, ServerId = context.ServerId }); - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectDatabaseSizeFactAsync failed", ex); - } - } - - /// - /// Collects SQL Server edition and major version from the server_properties table. - /// - private async Task CollectServerMetadataFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT TOP 1 - engine_edition, - CAST(LEFT(product_version, CHARINDEX('.', product_version) - 1) AS INT) AS major_version -FROM collect.server_properties -ORDER BY collection_time DESC"; - - using var reader = await cmd.ExecuteReaderAsync(); - if (!await reader.ReadAsync()) return; - - var edition = reader.IsDBNull(0) ? 0 : Convert.ToInt32(reader.GetValue(0)); - var majorVersion = reader.IsDBNull(1) ? 0 : Convert.ToInt32(reader.GetValue(1)); - - if (edition > 0) - facts.Add(new Fact { Source = "config", Key = "SERVER_EDITION", Value = edition, ServerId = context.ServerId }); - if (majorVersion > 0) - facts.Add(new Fact { Source = "config", Key = "SERVER_MAJOR_VERSION", Value = majorVersion, ServerId = context.ServerId }); - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectServerMetadataFactsAsync failed", ex); - } - } - - /// - /// Collects CPU utilization: average and max SQL Server CPU % over the period. - /// Value is average SQL CPU %. Corroborates SOS_SCHEDULER_YIELD. - /// - private async Task CollectCpuUtilizationFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT - AVG(CAST(sqlserver_cpu_utilization AS FLOAT)) AS avg_sql_cpu, - MAX(sqlserver_cpu_utilization) AS max_sql_cpu, - AVG(CAST(other_process_cpu_utilization AS FLOAT)) AS avg_other_cpu, - MAX(other_process_cpu_utilization) AS max_other_cpu, - COUNT(*) AS sample_count -FROM collect.cpu_utilization_stats -WHERE collection_time >= @startTime -AND collection_time <= @endTime"; - - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - using var reader = await cmd.ExecuteReaderAsync(); - if (!await reader.ReadAsync()) return; - - var avgSqlCpu = reader.IsDBNull(0) ? 0.0 : Convert.ToDouble(reader.GetValue(0)); - var maxSqlCpu = reader.IsDBNull(1) ? 0.0 : Convert.ToDouble(reader.GetValue(1)); - var avgOtherCpu = reader.IsDBNull(2) ? 0.0 : Convert.ToDouble(reader.GetValue(2)); - var maxOtherCpu = reader.IsDBNull(3) ? 0.0 : Convert.ToDouble(reader.GetValue(3)); - var sampleCount = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)); - - if (sampleCount == 0) return; - - var cpuMetadata = new Dictionary - { - ["avg_sql_cpu"] = avgSqlCpu, - ["max_sql_cpu"] = maxSqlCpu, - ["avg_other_cpu"] = avgOtherCpu, - ["max_other_cpu"] = maxOtherCpu, - ["avg_total_cpu"] = avgSqlCpu + avgOtherCpu, - ["sample_count"] = sampleCount - }; - - facts.Add(new Fact - { - Source = "cpu", - Key = "CPU_SQL_PERCENT", - Value = avgSqlCpu, - ServerId = context.ServerId, - Metadata = cpuMetadata - }); - - // Emit a CPU_SPIKE fact when max is high and significantly above average. - // This catches bursty CPU events that average-based scoring misses entirely. - // Requires max >= 80% AND at least 3x the average (or avg < 20% with max >= 80%). - if (maxSqlCpu >= 80 && (avgSqlCpu < 20 || maxSqlCpu / Math.Max(avgSqlCpu, 1) >= 3)) - { - facts.Add(new Fact - { - Source = "cpu", - Key = "CPU_SPIKE", - Value = maxSqlCpu, - ServerId = context.ServerId, - Metadata = cpuMetadata - }); - } - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectCpuUtilizationFactsAsync failed", ex); - } - } - - /// - /// Collects I/O latency from file_io_stats delta columns. - /// Computes average read and write latency across all database files. - /// - private async Task CollectIoLatencyFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT - SUM(io_stall_read_ms_delta) AS total_stall_read_ms, - SUM(num_of_reads_delta) AS total_reads, - SUM(io_stall_write_ms_delta) AS total_stall_write_ms, - SUM(num_of_writes_delta) AS total_writes -FROM collect.file_io_stats -WHERE collection_time >= @startTime -AND collection_time <= @endTime -AND (num_of_reads_delta > 0 OR num_of_writes_delta > 0)"; - - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - using var reader = await cmd.ExecuteReaderAsync(); - if (!await reader.ReadAsync()) return; - - var totalStallReadMs = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); - var totalReads = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); - var totalStallWriteMs = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); - var totalWrites = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)); - - if (totalReads > 0) - { - var avgReadLatency = (double)totalStallReadMs / totalReads; - facts.Add(new Fact - { - Source = "io", - Key = "IO_READ_LATENCY_MS", - Value = avgReadLatency, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["avg_read_latency_ms"] = avgReadLatency, - ["total_stall_read_ms"] = totalStallReadMs, - ["total_reads"] = totalReads - } - }); - } - - if (totalWrites > 0) - { - var avgWriteLatency = (double)totalStallWriteMs / totalWrites; - facts.Add(new Fact - { - Source = "io", - Key = "IO_WRITE_LATENCY_MS", - Value = avgWriteLatency, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["avg_write_latency_ms"] = avgWriteLatency, - ["total_stall_write_ms"] = totalStallWriteMs, - ["total_writes"] = totalWrites - } - }); - } - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectIoLatencyFactsAsync failed", ex); - } - } - - /// - /// Collects TempDB usage facts: max usage, version store size, and unallocated space. - /// Value is max total_reserved_mb over the period. - /// Dashboard uses computed columns (total_reserved_mb, etc.) from collect.tempdb_stats. - /// - private async Task CollectTempDbFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT - MAX(total_reserved_mb) AS max_total_reserved_mb, - MAX(user_object_reserved_mb) AS max_user_object_mb, - MAX(internal_object_reserved_mb) AS max_internal_object_mb, - MAX(version_store_reserved_mb) AS max_version_store_mb, - MIN(unallocated_mb) AS min_unallocated_mb, - AVG(CAST(total_reserved_mb AS FLOAT)) AS avg_total_reserved_mb -FROM collect.tempdb_stats -WHERE collection_time >= @startTime -AND collection_time <= @endTime"; - - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - using var reader = await cmd.ExecuteReaderAsync(); - if (!await reader.ReadAsync()) return; - - var maxReserved = reader.IsDBNull(0) ? 0.0 : Convert.ToDouble(reader.GetValue(0)); - var maxUserObj = reader.IsDBNull(1) ? 0.0 : Convert.ToDouble(reader.GetValue(1)); - var maxInternalObj = reader.IsDBNull(2) ? 0.0 : Convert.ToDouble(reader.GetValue(2)); - var maxVersionStore = reader.IsDBNull(3) ? 0.0 : Convert.ToDouble(reader.GetValue(3)); - var minUnallocated = reader.IsDBNull(4) ? 0.0 : Convert.ToDouble(reader.GetValue(4)); - var avgReserved = reader.IsDBNull(5) ? 0.0 : Convert.ToDouble(reader.GetValue(5)); - - if (maxReserved <= 0) return; - - // TempDB usage as fraction of total space (reserved + unallocated) - var totalSpace = maxReserved + minUnallocated; - var usageFraction = totalSpace > 0 ? maxReserved / totalSpace : 0; - - facts.Add(new Fact - { - Source = "tempdb", - Key = "TEMPDB_USAGE", - Value = usageFraction, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["max_reserved_mb"] = maxReserved, - ["avg_reserved_mb"] = avgReserved, - ["max_user_object_mb"] = maxUserObj, - ["max_internal_object_mb"] = maxInternalObj, - ["max_version_store_mb"] = maxVersionStore, - ["min_unallocated_mb"] = minUnallocated, - ["usage_fraction"] = usageFraction - } - }); - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectTempDbFactsAsync failed", ex); - } - } - - /// - /// Collects memory grant facts from the memory_grant_stats table. - /// Detects grant waiters (sessions waiting for memory) and grant pressure. - /// - private async Task CollectMemoryGrantFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT - MAX(waiter_count) AS max_waiters, - AVG(CAST(waiter_count AS FLOAT)) AS avg_waiters, - MAX(grantee_count) AS max_grantees, - SUM(timeout_error_count_delta) AS total_timeout_errors, - SUM(forced_grant_count_delta) AS total_forced_grants -FROM collect.memory_grant_stats -WHERE collection_time >= @startTime -AND collection_time <= @endTime"; - - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - using var reader = await cmd.ExecuteReaderAsync(); - if (!await reader.ReadAsync()) return; - - var maxWaiters = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); - var avgWaiters = reader.IsDBNull(1) ? 0.0 : Convert.ToDouble(reader.GetValue(1)); - var maxGrantees = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); - var totalTimeouts = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)); - var totalForcedGrants = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)); - - // Only create a fact if there's evidence of grant pressure - if (maxWaiters <= 0 && totalTimeouts <= 0 && totalForcedGrants <= 0) return; - - facts.Add(new Fact - { - Source = "memory", - Key = "MEMORY_GRANT_PENDING", - Value = maxWaiters, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["max_waiters"] = maxWaiters, - ["avg_waiters"] = avgWaiters, - ["max_grantees"] = maxGrantees, - ["total_timeout_errors"] = totalTimeouts, - ["total_forced_grants"] = totalForcedGrants - } - }); - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectMemoryGrantFactsAsync failed", ex); - } - } - - /// - /// Collects query-level aggregate facts from query_stats. - /// Focuses on spills (memory grant misestimates) and high-parallelism queries. - /// - private async Task CollectQueryStatsFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT - SUM(total_spills) AS total_spills, - COUNT(CASE WHEN max_dop > 8 THEN 1 END) AS high_dop_queries, - COUNT(CASE WHEN total_spills > 0 THEN 1 END) AS spilling_queries, - SUM(execution_count_delta) AS total_executions, - SUM(total_worker_time_delta) AS total_cpu_time_us -FROM collect.query_stats -WHERE collection_time >= @startTime -AND collection_time <= @endTime -AND execution_count_delta > 0"; - - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - using var reader = await cmd.ExecuteReaderAsync(); - if (!await reader.ReadAsync()) return; - - var totalSpills = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); - var highDopQueries = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); - var spillingQueries = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); - var totalExecutions = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)); - var totalCpuTimeUs = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)); - - if (totalSpills > 0) - { - facts.Add(new Fact - { - Source = "queries", - Key = "QUERY_SPILLS", - Value = totalSpills, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["total_spills"] = totalSpills, - ["spilling_query_count"] = spillingQueries, - ["total_executions"] = totalExecutions - } - }); - } - - if (highDopQueries > 0) - { - facts.Add(new Fact - { - Source = "queries", - Key = "QUERY_HIGH_DOP", - Value = highDopQueries, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["high_dop_query_count"] = highDopQueries, - ["total_cpu_time_us"] = totalCpuTimeUs, - ["total_executions"] = totalExecutions - } - }); - } - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectQueryStatsFactsAsync failed", ex); - } - } - - /// - /// Detects parameter-sensitive cached plans: a single query_plan_hash whose - /// per-execution worker time varies wildly — one plan serving very different - /// parameter values. Emits one aggregate PARAMETER_SENSITIVITY fact. - /// Note min_*/max_* are cumulative over the plan's cached lifetime, so the - /// finding means "this plan, active now, has a history of widely varying cost". - /// - private async Task CollectParameterSensitivityFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -WITH latest AS -( - SELECT - query_hash, - query_plan_hash, - database_name, - execution_count, - creation_time, - min_worker_time, - max_worker_time, - min_grant_kb, - max_grant_kb, - min_spills, - max_spills, - ROW_NUMBER() OVER - ( - PARTITION BY database_name, query_hash, query_plan_hash - ORDER BY collection_time DESC - ) AS rn - FROM collect.query_stats - WHERE collection_time >= @startTime - AND collection_time <= @endTime - AND execution_count_delta > 0 -) -SELECT - min_worker_time, - max_worker_time, - CAST(max_worker_time AS float) / NULLIF(min_worker_time, 0) AS worker_ratio, - CAST(max_grant_kb AS float) / NULLIF(min_grant_kb, 0) AS grant_ratio, - CASE WHEN max_spills > 0 AND min_spills = 0 THEN 1 ELSE 0 END AS spill_divergence -FROM latest -WHERE rn = 1 -AND min_worker_time >= 10000 -AND max_worker_time >= 250000 -AND execution_count >= 20 -AND creation_time <= @startTime -AND CAST(max_worker_time AS float) / NULLIF(min_worker_time, 0) >= 10 -ORDER BY worker_ratio DESC -OFFSET 0 ROWS FETCH NEXT 20 ROWS ONLY"; - - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - var offenderCount = 0; - var worstRatio = 0.0; - var worstMinWorker = 0L; - var worstMaxWorker = 0L; - var worstGrantRatio = 0.0; - var worstSpillDivergence = 0; - - using var reader = await cmd.ExecuteReaderAsync(); - while (await reader.ReadAsync()) - { - // Rows arrive ordered by worker_ratio DESC — the first row is the worst offender. - if (offenderCount == 0) - { - worstMinWorker = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); - worstMaxWorker = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); - worstRatio = reader.IsDBNull(2) ? 0.0 : Convert.ToDouble(reader.GetValue(2)); - worstGrantRatio = reader.IsDBNull(3) ? 0.0 : Convert.ToDouble(reader.GetValue(3)); - worstSpillDivergence = reader.IsDBNull(4) ? 0 : Convert.ToInt32(reader.GetValue(4)); - } - offenderCount++; - } - - if (offenderCount == 0) return; - - facts.Add(new Fact - { - Source = "queries", - Key = "PARAMETER_SENSITIVITY", - Value = worstRatio, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["offender_count"] = offenderCount, - ["worst_ratio"] = worstRatio, - ["worst_min_worker_us"] = worstMinWorker, - ["worst_max_worker_us"] = worstMaxWorker, - ["worst_grant_ratio"] = worstGrantRatio, - ["grant_divergence"] = worstGrantRatio >= 5 ? 1 : 0, - ["spill_divergence"] = worstSpillDivergence - } - }); - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectParameterSensitivityFactsAsync failed", ex); - } - } - - /// - /// Detects plan regressions: a query whose currently-active plan has per-execution - /// cost >= 2x the best plan that query is known to perform well with. Emits one - /// aggregate PLAN_REGRESSION fact. Sourced from Query Store (collect.query_store_data); - /// no fact when Query Store is not enabled on the monitored databases. - /// Unlike other collectors this windows on server_last_execution_time (14-day - /// comparison window), NOT collection_time. - /// - private async Task CollectPlanRegressionFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -WITH deduped AS -( - -- Collapse incremental re-collections of the same open runtime-stats interval: - -- keep only the latest collection_time row per logical interval. - SELECT - database_name, - query_id, - plan_id, - query_plan_hash, - count_executions, - avg_cpu_time, - avg_duration, - server_last_execution_time, - is_forced_plan, - force_failure_count, - ROW_NUMBER() OVER - ( - PARTITION BY database_name, query_id, plan_id, server_first_execution_time - ORDER BY collection_time DESC - ) AS rn - FROM collect.query_store_data - WHERE execution_type_desc = N'Regular' - AND server_last_execution_time >= @windowStart -), -plan_agg AS -( - -- Execution-weighted per-exec cost per plan_id. query_plan_hash is invariant - -- within a plan_id, so include it in the GROUP BY rather than aggregating it - -- (MS Learn's MAX page does not list binary/varbinary in the accepted types). - SELECT - database_name, - query_id, - plan_id, - query_plan_hash, - SUM(count_executions) AS execs, - CASE WHEN SUM(count_executions) > 0 - THEN SUM(avg_cpu_time * count_executions) / NULLIF(SUM(count_executions), 0) - ELSE 0 END AS cpu_per_exec, - CASE WHEN SUM(count_executions) > 0 - THEN SUM(avg_duration * count_executions) / NULLIF(SUM(count_executions), 0) - ELSE 0 END AS dur_per_exec, - MAX(server_last_execution_time) AS last_exec, - MAX(CAST(is_forced_plan AS tinyint)) AS is_forced_plan, - MAX(force_failure_count) AS force_failure_count - FROM deduped - WHERE rn = 1 - GROUP BY database_name, query_id, plan_id, query_plan_hash -), -plan_dedup AS -( - -- Collapse plan_ids that share a query_plan_hash (a recompile can produce an - -- identical plan under a new plan_id); keep only plans with enough executions. - SELECT - database_name, - query_id, - query_plan_hash, - SUM(execs) AS execs, - CASE WHEN SUM(execs) > 0 - THEN SUM(cpu_per_exec * execs) / NULLIF(SUM(execs), 0) - ELSE 0 END AS cpu_per_exec, - CASE WHEN SUM(execs) > 0 - THEN SUM(dur_per_exec * execs) / NULLIF(SUM(execs), 0) - ELSE 0 END AS dur_per_exec, - MAX(last_exec) AS last_exec, - MAX(is_forced_plan) AS is_forced_plan, - MAX(force_failure_count) AS force_failure_count - FROM plan_agg - GROUP BY database_name, query_id, query_plan_hash - HAVING SUM(execs) >= 25 -), -ranked AS -( - SELECT - *, - ROW_NUMBER() OVER (PARTITION BY database_name, query_id ORDER BY last_exec DESC) AS recency, - ROW_NUMBER() OVER (PARTITION BY database_name, query_id ORDER BY cpu_per_exec ASC) AS cheapness - FROM plan_dedup -), -compared AS -( - -- Latest active plan vs the best-performing plan for the same query. - SELECT - l.query_id, - l.cpu_per_exec AS latest_cpu, - l.dur_per_exec AS latest_dur, - l.is_forced_plan AS latest_is_forced, - l.force_failure_count AS force_failure_count, - b.cpu_per_exec AS best_cpu, - b.dur_per_exec AS best_dur, - (SELECT MAX(v) - FROM (VALUES - (CAST(l.cpu_per_exec AS float) / NULLIF(b.cpu_per_exec, 0)), - (CAST(l.dur_per_exec AS float) / NULLIF(b.dur_per_exec, 0)) - ) AS x(v)) AS regression_factor - FROM ranked AS l - JOIN ranked AS b - ON b.database_name = l.database_name - AND b.query_id = l.query_id - AND b.cheapness = 1 - WHERE l.recency = 1 - AND l.query_plan_hash <> b.query_plan_hash -) -SELECT - query_id, - latest_cpu, - latest_dur, - latest_is_forced, - force_failure_count, - best_cpu, - best_dur, - regression_factor -FROM compared -WHERE regression_factor >= 2 -ORDER BY regression_factor DESC -OFFSET 0 ROWS FETCH NEXT 20 ROWS ONLY"; - - cmd.Parameters.Add(new SqlParameter("@windowStart", context.TimeRangeStart.AddDays(-14))); - - var offenderCount = 0; - var worstFactor = 0.0; - var worstQueryId = 0L; - var worstLatestCpu = 0.0; - var worstBestCpu = 0.0; - var worstDimension = 1; - var worstLatestForced = 0; - var worstForceFailures = 0L; - - using var reader = await cmd.ExecuteReaderAsync(); - while (await reader.ReadAsync()) - { - // Rows arrive ordered by regression_factor DESC — the first row is the worst offender. - if (offenderCount == 0) - { - worstQueryId = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); - var latestCpu = reader.IsDBNull(1) ? 0.0 : Convert.ToDouble(reader.GetValue(1)); - var latestDur = reader.IsDBNull(2) ? 0.0 : Convert.ToDouble(reader.GetValue(2)); - worstLatestForced = (!reader.IsDBNull(3) && Convert.ToInt32(reader.GetValue(3)) > 0) ? 1 : 0; - worstForceFailures = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)); - var bestCpu = reader.IsDBNull(5) ? 0.0 : Convert.ToDouble(reader.GetValue(5)); - var bestDur = reader.IsDBNull(6) ? 0.0 : Convert.ToDouble(reader.GetValue(6)); - worstFactor = reader.IsDBNull(7) ? 0.0 : Convert.ToDouble(reader.GetValue(7)); - - worstLatestCpu = latestCpu; - worstBestCpu = bestCpu; - var cpuRatio = bestCpu > 0 ? latestCpu / bestCpu : 0.0; - var durRatio = bestDur > 0 ? latestDur / bestDur : 0.0; - worstDimension = cpuRatio >= durRatio ? 1 : 2; // 1 = cpu, 2 = duration - } - offenderCount++; - } - - if (offenderCount == 0) return; - - facts.Add(new Fact - { - Source = "queries", - Key = "PLAN_REGRESSION", - Value = worstFactor, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["offender_count"] = offenderCount, - ["worst_regression_factor"] = worstFactor, - ["worst_query_id"] = worstQueryId, - ["latest_cpu_per_exec_us"] = worstLatestCpu, - ["best_cpu_per_exec_us"] = worstBestCpu, - ["regressed_dimension"] = worstDimension, - ["latest_is_forced"] = worstLatestForced, - ["force_failure_count"] = worstForceFailures - } - }); - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectPlanRegressionFactsAsync failed", ex); - } - } - - /// - /// Reconstructs blocking chains from collect.blocking_BlockedProcessReport (one row - /// per side of each blocking event, dedup'd to activity = 'blocked') and emits - /// one aggregate BLOCKING_CHAIN fact describing the worst chain — apex head blocker, - /// depth, transitive victim count. Structure the BLOCKING_EVENTS rate is blind to. - /// Reads typed blocker-side columns populated by collect.process_blocked_process_xml, - /// so no XML re-parse on the analysis hot path. - /// - private async Task CollectBlockingChainFactsAsync(AnalysisContext context, List facts) - { - const int maxPairs = 5000; - const int maxDepth = 50; - const int stepBudget = 100_000; - - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - // ORDER BY collection_time DESC is a backward CIX scan (sort-free); event_time - // is a residual predicate. activity='blocked' picks the canonical per-event side. - // blocking_spid IS NOT NULL filters out rows whose source XML had an empty - // (system task / torn-down session) — - // those can't contribute to a reconstructed chain. - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT TOP (5000) - event_time, - database_name, - spid, - last_transaction_started, - blocking_spid, - blocking_last_tran_started, - wait_time_ms, - lock_mode, - blocking_status, - blocked_sql_text, - blocking_sql_text -FROM collect.blocking_BlockedProcessReport -WHERE collection_time >= @collectionWindow -AND event_time >= @startTime -AND event_time <= @endTime -AND activity = 'blocked' -AND blocking_spid IS NOT NULL -ORDER BY collection_time DESC"; - - // Generous bound — analysis window plus an hour — to catch rows whose - // event_time is inside the window but whose collection_time may lag slightly. - cmd.Parameters.Add(new SqlParameter("@collectionWindow", context.TimeRangeStart.AddHours(-1))); - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - var rows = new List(); - using (var reader = await cmd.ExecuteReaderAsync()) - { - while (await reader.ReadAsync()) - { - rows.Add(new BlockingPairRow - { - EventTime = reader.IsDBNull(0) ? default : reader.GetDateTime(0), - DatabaseName = reader.IsDBNull(1) ? string.Empty : reader.GetString(1), - BlockedSpid = reader.IsDBNull(2) ? 0 : Convert.ToInt32(reader.GetValue(2)), - BlockedTranStarted = reader.IsDBNull(3) ? (DateTime?)null : reader.GetDateTime(3), - BlockingSpid = reader.IsDBNull(4) ? 0 : Convert.ToInt32(reader.GetValue(4)), - BlockingTranStarted = reader.IsDBNull(5) ? (DateTime?)null : reader.GetDateTime(5), - WaitTimeMs = reader.IsDBNull(6) ? 0L : Convert.ToInt64(reader.GetValue(6)), - LockMode = reader.IsDBNull(7) ? string.Empty : reader.GetString(7), - BlockingStatus = reader.IsDBNull(8) ? string.Empty : reader.GetString(8), - BlockedSqlText = reader.IsDBNull(9) ? string.Empty : reader.GetString(9), - BlockingSqlText = reader.IsDBNull(10) ? string.Empty : reader.GetString(10) - }); - } - } - - if (rows.Count == 0) return; - - var reconstruction = BlockingChainReconstructor.Reconstruct(rows, maxDepth, maxPairs, stepBudget); - if (reconstruction.Chains.Count == 0) return; - - var worst = reconstruction.Chains[0]; - - facts.Add(new Fact - { - Source = "blocking", - Key = "BLOCKING_CHAIN", - Value = worst.Depth, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["worst_chain_depth"] = worst.Depth, - ["worst_chain_victim_count"] = worst.VictimCount, - ["worst_apex_spid"] = worst.ApexSpid, - ["worst_apex_sleeping"] = worst.ApexSleeping ? 1 : 0, - ["worst_chain_max_wait_ms"] = worst.MaxWaitMs, - ["total_reconstructed_chains"] = reconstruction.Chains.Count, - ["deepest_chain_overall"] = reconstruction.Chains.Max(c => c.Depth), - ["max_victim_count_overall"] = reconstruction.Chains.Max(c => c.VictimCount), - ["depth_capped"] = reconstruction.DepthCapped ? 1 : 0, - ["traversal_truncated"] = reconstruction.TraversalTruncated ? 1 : 0, - ["cycle_detected"] = reconstruction.CycleDetected ? 1 : 0 - } - }); - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectBlockingChainFactsAsync failed", ex); - } - } - - /// - /// Identifies individual queries that are consistently terrible ("bad actors"). - /// These queries don't necessarily cause server-level symptoms but waste resources - /// on every execution. Detection uses execution count tiers x per-execution impact. - /// Top 5 worst offenders become individual BAD_ACTOR facts. - /// Dashboard query_hash is binary(8) — convert to hex string for fact key. - /// - private async Task CollectBadActorFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT TOP 5 - database_name, - CONVERT(VARCHAR(18), query_hash, 1) AS query_hash, - CAST(SUM(execution_count_delta) AS BIGINT) AS exec_count, - CASE WHEN SUM(execution_count_delta) > 0 - THEN CAST(SUM(total_worker_time_delta) AS FLOAT) / SUM(execution_count_delta) / 1000.0 - ELSE 0 END AS avg_cpu_ms, - CASE WHEN SUM(execution_count_delta) > 0 - THEN CAST(SUM(total_elapsed_time_delta) AS FLOAT) / SUM(execution_count_delta) / 1000.0 - ELSE 0 END AS avg_elapsed_ms, - CASE WHEN SUM(execution_count_delta) > 0 - THEN CAST(SUM(total_logical_reads_delta) AS FLOAT) / SUM(execution_count_delta) - ELSE 0 END AS avg_reads, - CAST(SUM(total_worker_time_delta) AS BIGINT) AS total_cpu_us, - CAST(SUM(total_logical_reads_delta) AS BIGINT) AS total_reads, - CAST(SUM(total_spills) AS BIGINT) AS total_spills, - MAX(max_dop) AS max_dop, - LEFT(CAST(DECOMPRESS(MAX(query_text)) AS NVARCHAR(MAX)), 200) AS query_text -FROM collect.query_stats -WHERE collection_time >= @startTime -AND collection_time <= @endTime -AND execution_count_delta > 0 -GROUP BY database_name, query_hash -HAVING SUM(execution_count_delta) >= 100 -ORDER BY CAST(SUM(total_worker_time_delta) AS FLOAT) / NULLIF(SUM(execution_count_delta), 0) * - LOG(NULLIF(SUM(execution_count_delta), 0)) DESC"; - - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - using var reader = await cmd.ExecuteReaderAsync(); - while (await reader.ReadAsync()) - { - var dbName = reader.IsDBNull(0) ? "" : reader.GetString(0); - var queryHash = reader.IsDBNull(1) ? "" : reader.GetString(1); - var execCount = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); - var avgCpuMs = reader.IsDBNull(3) ? 0.0 : Convert.ToDouble(reader.GetValue(3)); - var avgElapsedMs = reader.IsDBNull(4) ? 0.0 : Convert.ToDouble(reader.GetValue(4)); - var avgReads = reader.IsDBNull(5) ? 0.0 : Convert.ToDouble(reader.GetValue(5)); - var totalCpuUs = reader.IsDBNull(6) ? 0L : Convert.ToInt64(reader.GetValue(6)); - var totalReads = reader.IsDBNull(7) ? 0L : Convert.ToInt64(reader.GetValue(7)); - var totalSpills = reader.IsDBNull(8) ? 0L : Convert.ToInt64(reader.GetValue(8)); - var maxDop = reader.IsDBNull(9) ? 0 : Convert.ToInt32(reader.GetValue(9)); - var queryText = reader.IsDBNull(10) ? "" : reader.GetString(10); - - // Skip low-impact queries — need meaningful per-execution cost - if (avgCpuMs < 10 && avgReads < 1000) continue; - - facts.Add(new Fact - { - Source = "bad_actor", - Key = $"BAD_ACTOR_{queryHash}", - Value = avgCpuMs, // Primary scoring dimension - ServerId = context.ServerId, - DatabaseName = dbName, - Metadata = new Dictionary - { - ["execution_count"] = execCount, - ["avg_cpu_ms"] = avgCpuMs, - ["avg_elapsed_ms"] = avgElapsedMs, - ["avg_reads"] = avgReads, - ["total_cpu_us"] = totalCpuUs, - ["total_reads"] = totalReads, - ["total_spills"] = totalSpills, - ["max_dop"] = maxDop - } - }); - } - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectBadActorFactsAsync failed", ex); - } - } - - /// - /// Collects key perfmon counters: Page Life Expectancy, Batch Requests/sec, compilations. - /// PLE is scored; others are throughput context for the AI. - /// - private async Task CollectPerfmonFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -;WITH latest AS ( - SELECT - counter_name, - cntr_value, - cntr_value_delta, - ROW_NUMBER() OVER (PARTITION BY counter_name ORDER BY collection_time DESC) AS rn - FROM collect.perfmon_stats - WHERE collection_time >= @startTime - AND collection_time <= @endTime - AND counter_name IN ('Page life expectancy', 'Batch Requests/sec', 'SQL Compilations/sec', 'SQL Re-Compilations/sec') -) -SELECT counter_name, cntr_value, cntr_value_delta -FROM latest WHERE rn = 1"; - - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - using var reader = await cmd.ExecuteReaderAsync(); - while (await reader.ReadAsync()) - { - var counterName = reader.GetString(0); - var cntrValue = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); - var deltaValue = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); - - var (factKey, source) = counterName switch - { - "Page life expectancy" => ("PERFMON_PLE", "perfmon"), - "Batch Requests/sec" => ("PERFMON_BATCH_REQ_SEC", "perfmon"), - "SQL Compilations/sec" => ("PERFMON_COMPILATIONS_SEC", "perfmon"), - "SQL Re-Compilations/sec" => ("PERFMON_RECOMPILATIONS_SEC", "perfmon"), - _ => (null, null) - }; - - if (factKey == null) continue; - - // For PLE, use the absolute value. For rate counters, use delta. - var value = counterName == "Page life expectancy" ? (double)cntrValue : (double)deltaValue; - - facts.Add(new Fact - { - Source = source!, - Key = factKey, - Value = value, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["cntr_value"] = cntrValue, - ["delta_cntr_value"] = deltaValue - } - }); - } - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectPerfmonFactsAsync failed", ex); - } - } - - /// - /// Collects top memory clerks by size. Context for understanding where memory is allocated. - /// Dashboard stores pages_kb — convert to MB for consistency with Lite facts. - /// - private async Task CollectMemoryClerkFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -;WITH latest AS ( - SELECT - clerk_type, - SUM(pages_kb) / 1024.0 AS memory_mb, - ROW_NUMBER() OVER (PARTITION BY clerk_type ORDER BY collection_time DESC) AS rn, - collection_time - FROM collect.memory_clerks_stats - WHERE collection_time <= @endTime - GROUP BY clerk_type, collection_time -) -SELECT TOP 10 clerk_type, memory_mb -FROM latest WHERE rn = 1 AND memory_mb > 0 -ORDER BY memory_mb DESC"; - - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - using var reader = await cmd.ExecuteReaderAsync(); - var metadata = new Dictionary(); - var totalMb = 0.0; - var clerkCount = 0; - - while (await reader.ReadAsync()) - { - var clerkType = reader.GetString(0); - var memoryMb = Convert.ToDouble(reader.GetValue(1)); - metadata[clerkType] = memoryMb; - totalMb += memoryMb; - clerkCount++; - } - - if (clerkCount == 0) return; - - metadata["total_top_clerks_mb"] = totalMb; - metadata["clerk_count"] = clerkCount; - - facts.Add(new Fact - { - Source = "memory", - Key = "MEMORY_CLERKS", - Value = totalMb, - ServerId = context.ServerId, - Metadata = metadata - }); - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectMemoryClerkFactsAsync failed", ex); - } - } - - /// - /// Collects database configuration facts: RCSI status, auto_shrink, auto_close, - /// recovery model. Aggregates counts across databases. - /// Dashboard stores config as individual setting rows in config.database_configuration_history. - /// We pivot from the per-setting rows into aggregated counts. - /// - private async Task CollectDatabaseConfigFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -;WITH latest AS ( - SELECT - database_name, - setting_name, - setting_value, - ROW_NUMBER() OVER (PARTITION BY database_name, setting_name ORDER BY collection_time DESC) AS rn - FROM config.database_configuration_history - WHERE setting_type = 'DATABASE_PROPERTY' /* collector writes 'DATABASE_PROPERTY' (install/39); 'database_option' matched 0 rows → DB_CONFIG fact never fired */ - AND database_name NOT IN ('master', 'msdb', 'model', 'tempdb') -), -pivoted AS ( - SELECT - database_name, - MAX(CASE WHEN setting_name = 'recovery_model_desc' THEN CAST(setting_value AS NVARCHAR(128)) END) AS recovery_model, - MAX(CASE WHEN setting_name = 'is_auto_shrink_on' THEN CAST(setting_value AS NVARCHAR(10)) END) AS is_auto_shrink_on, - MAX(CASE WHEN setting_name = 'is_auto_close_on' THEN CAST(setting_value AS NVARCHAR(10)) END) AS is_auto_close_on, - MAX(CASE WHEN setting_name = 'is_read_committed_snapshot_on' THEN CAST(setting_value AS NVARCHAR(10)) END) AS is_read_committed_snapshot_on, - MAX(CASE WHEN setting_name = 'is_auto_create_stats_on' THEN CAST(setting_value AS NVARCHAR(10)) END) AS is_auto_create_stats_on, - MAX(CASE WHEN setting_name = 'is_auto_update_stats_on' THEN CAST(setting_value AS NVARCHAR(10)) END) AS is_auto_update_stats_on, - MAX(CASE WHEN setting_name = 'page_verify_option_desc' THEN CAST(setting_value AS NVARCHAR(128)) END) AS page_verify_option, - MAX(CASE WHEN setting_name = 'is_query_store_on' THEN CAST(setting_value AS NVARCHAR(10)) END) AS is_query_store_on - FROM latest - WHERE rn = 1 - GROUP BY database_name -) -SELECT - COUNT(*) AS database_count, - COUNT(CASE WHEN is_auto_shrink_on = '1' OR is_auto_shrink_on = 'True' THEN 1 END) AS auto_shrink_count, - COUNT(CASE WHEN is_auto_close_on = '1' OR is_auto_close_on = 'True' THEN 1 END) AS auto_close_count, - COUNT(CASE WHEN is_read_committed_snapshot_on = '0' OR is_read_committed_snapshot_on = 'False' THEN 1 END) AS rcsi_off_count, - COUNT(CASE WHEN is_auto_create_stats_on = '0' OR is_auto_create_stats_on = 'False' THEN 1 END) AS auto_create_stats_off_count, - COUNT(CASE WHEN is_auto_update_stats_on = '0' OR is_auto_update_stats_on = 'False' THEN 1 END) AS auto_update_stats_off_count, - COUNT(CASE WHEN page_verify_option IS NOT NULL AND page_verify_option != 'CHECKSUM' THEN 1 END) AS page_verify_not_checksum_count, - COUNT(CASE WHEN recovery_model = 'FULL' THEN 1 END) AS full_recovery_count, - COUNT(CASE WHEN recovery_model = 'SIMPLE' THEN 1 END) AS simple_recovery_count, - COUNT(CASE WHEN is_query_store_on = '1' OR is_query_store_on = 'True' THEN 1 END) AS query_store_on_count -FROM pivoted"; - - using var reader = await cmd.ExecuteReaderAsync(); - if (!await reader.ReadAsync()) return; - - var dbCount = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); - if (dbCount == 0) return; - - var autoShrink = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); - var autoClose = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); - var rcsiOff = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)); - var autoCreateOff = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)); - var autoUpdateOff = reader.IsDBNull(5) ? 0L : Convert.ToInt64(reader.GetValue(5)); - var pageVerifyBad = reader.IsDBNull(6) ? 0L : Convert.ToInt64(reader.GetValue(6)); - var fullRecovery = reader.IsDBNull(7) ? 0L : Convert.ToInt64(reader.GetValue(7)); - var simpleRecovery = reader.IsDBNull(8) ? 0L : Convert.ToInt64(reader.GetValue(8)); - var queryStoreOn = reader.IsDBNull(9) ? 0L : Convert.ToInt64(reader.GetValue(9)); - - facts.Add(new Fact - { - Source = "database_config", - Key = "DB_CONFIG", - Value = dbCount, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["database_count"] = dbCount, - ["auto_shrink_on_count"] = autoShrink, - ["auto_close_on_count"] = autoClose, - ["rcsi_off_count"] = rcsiOff, - ["auto_create_stats_off_count"] = autoCreateOff, - ["auto_update_stats_off_count"] = autoUpdateOff, - ["page_verify_not_checksum_count"] = pageVerifyBad, - ["full_recovery_count"] = fullRecovery, - ["simple_recovery_count"] = simpleRecovery, - ["query_store_on_count"] = queryStoreOn - } - }); - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectDatabaseConfigFactsAsync failed", ex); - } - } - - /// - /// Collects the percent-autogrowth-on-large-files config fact (WS3): data/log files set - /// to grow in PERCENTAGE steps that are also large (>= 10 GB), where a single growth is a - /// huge, stalling allocation. Reads the latest snapshot per file from - /// collect.database_size_stats, excludes system databases, and emits ONE aggregate - /// FILE_AUTOGROWTH_PERCENT fact carrying the offending-file/database counts (the per-file - /// detail + copy-paste fix is attached later by the drill-down collector). - /// - private async Task CollectFileAutogrowthFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -;WITH latest AS ( - SELECT - database_name, - file_id, - total_size_mb, - is_percent_growth, - ROW_NUMBER() OVER (PARTITION BY database_name, file_id ORDER BY collection_time DESC) AS rn - FROM collect.database_size_stats - WHERE database_name NOT IN ('master', 'msdb', 'model', 'tempdb') -) -SELECT - file_count = COUNT(*), - database_count = COUNT(DISTINCT database_name) -FROM latest -WHERE rn = 1 -AND is_percent_growth = 1 -AND total_size_mb >= @minSizeMb;"; - - cmd.Parameters.Add(new SqlParameter("@minSizeMb", 10240.0)); /* 10 GB */ - - using var reader = await cmd.ExecuteReaderAsync(); - if (!await reader.ReadAsync()) return; - - var fileCount = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); - if (fileCount == 0) return; - - var databaseCount = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); - - facts.Add(new Fact - { - Source = "config", - Key = "FILE_AUTOGROWTH_PERCENT", - Value = fileCount, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["file_count"] = fileCount, - ["database_count"] = databaseCount - } - }); - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectFileAutogrowthFactsAsync failed", ex); - } - } - - /// - /// Collects procedure stats: top procedure by delta CPU time in the period. - /// - private async Task CollectProcedureStatsFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT - COUNT(DISTINCT object_name) AS distinct_procs, - SUM(execution_count_delta) AS total_executions, - SUM(total_worker_time_delta) AS total_cpu_time_us, - SUM(total_elapsed_time_delta) AS total_elapsed_time_us, - SUM(total_logical_reads_delta) AS total_logical_reads -FROM collect.procedure_stats -WHERE collection_time >= @startTime -AND collection_time <= @endTime -AND execution_count_delta > 0"; - - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - using var reader = await cmd.ExecuteReaderAsync(); - if (!await reader.ReadAsync()) return; - - var distinctProcs = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); - var totalExecs = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); - var totalCpuUs = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); - var totalElapsedUs = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)); - var totalReads = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)); - - if (totalExecs == 0) return; - - facts.Add(new Fact - { - Source = "queries", - Key = "PROCEDURE_STATS", - Value = totalCpuUs, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["distinct_procedures"] = distinctProcs, - ["total_executions"] = totalExecs, - ["total_cpu_time_us"] = totalCpuUs, - ["total_elapsed_time_us"] = totalElapsedUs, - ["total_logical_reads"] = totalReads, - ["avg_cpu_per_exec_us"] = totalExecs > 0 ? (double)totalCpuUs / totalExecs : 0 - } - }); - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectProcedureStatsFactsAsync failed", ex); - } - } - - /// - /// Collects active query snapshot facts: long-running queries, blocked sessions, high DOP. - /// Dashboard query_snapshots table is created by sp_WhoIsActive dynamically. - /// We query it if it exists. - /// - private async Task CollectActiveQueryFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - // Check if the table exists first (created dynamically by sp_WhoIsActive) - using var checkCmd = connection.CreateCommand(); - checkCmd.CommandText = "SELECT OBJECT_ID(N'collect.query_snapshots', N'U')"; - var tableExists = await checkCmd.ExecuteScalarAsync(); - if (tableExists == null || tableExists == DBNull.Value) return; - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT - COUNT(*) AS total_snapshots, - COUNT(CASE WHEN DATEDIFF(MILLISECOND, 0, [elapsed_time]) > 30000 THEN 1 END) AS long_running_count, - COUNT(CASE WHEN [blocking_session_id] IS NOT NULL AND [blocking_session_id] != '' THEN 1 END) AS blocked_count, - MAX(DATEDIFF(MILLISECOND, 0, [elapsed_time])) AS max_elapsed_ms, - COUNT(DISTINCT [session_id]) AS distinct_sessions -FROM collect.query_snapshots -WHERE collection_time >= @startTime -AND collection_time <= @endTime"; - - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - using var reader = await cmd.ExecuteReaderAsync(); - if (!await reader.ReadAsync()) return; - - var totalSnapshots = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); - if (totalSnapshots == 0) return; - - var longRunning = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); - var blocked = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); - var maxElapsed = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)); - var distinctSessions = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)); - - facts.Add(new Fact - { - Source = "queries", - Key = "ACTIVE_QUERIES", - Value = longRunning, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["total_snapshots"] = totalSnapshots, - ["long_running_count"] = longRunning, - ["blocked_count"] = blocked, - ["max_elapsed_ms"] = maxElapsed, - ["distinct_sessions"] = distinctSessions - } - }); - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectActiveQueryFactsAsync failed", ex); - } - } - - /// - /// Collects running job facts: jobs currently running long vs historical averages. - /// - private async Task CollectRunningJobFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT - COUNT(*) AS running_count, - COUNT(CASE WHEN is_running_long = 1 THEN 1 END) AS running_long_count, - MAX(percent_of_average) AS max_percent_of_avg, - MAX(current_duration_seconds) AS max_duration_seconds -FROM collect.running_jobs -WHERE collection_time >= @startTime -AND collection_time <= @endTime"; - - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - using var reader = await cmd.ExecuteReaderAsync(); - if (!await reader.ReadAsync()) return; - - var runningCount = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); - if (runningCount == 0) return; - - var runningLong = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); - var maxPctAvg = reader.IsDBNull(2) ? 0.0 : Convert.ToDouble(reader.GetValue(2)); - var maxDuration = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)); - - facts.Add(new Fact - { - Source = "jobs", - Key = "RUNNING_JOBS", - Value = runningLong, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["running_count"] = runningCount, - ["running_long_count"] = runningLong, - ["max_percent_of_average"] = maxPctAvg, - ["max_duration_seconds"] = maxDuration - } - }); - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectRunningJobFactsAsync failed", ex); - } - } - - /// - /// Collects session stats: connection counts, total connections. - /// Dashboard session_stats is a flat table (not per-program_name), so we adapt. - /// - private async Task CollectSessionFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -;WITH latest AS ( - SELECT - total_sessions, - running_sessions, - sleeping_sessions, - dormant_sessions, - databases_with_connections, - top_application_connections, - ROW_NUMBER() OVER (ORDER BY collection_time DESC) AS rn - FROM collect.session_stats - WHERE collection_time >= @startTime - AND collection_time <= @endTime -) -SELECT - total_sessions AS total_connections, - running_sessions AS total_running, - sleeping_sessions AS total_sleeping, - dormant_sessions AS total_dormant, - databases_with_connections AS distinct_apps, - top_application_connections AS max_app_connections -FROM latest WHERE rn = 1"; - - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - using var reader = await cmd.ExecuteReaderAsync(); - if (!await reader.ReadAsync()) return; - - var totalConns = reader.IsDBNull(0) ? 0L : Convert.ToInt64(reader.GetValue(0)); - if (totalConns == 0) return; - - var totalRunning = reader.IsDBNull(1) ? 0L : Convert.ToInt64(reader.GetValue(1)); - var totalSleeping = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); - var totalDormant = reader.IsDBNull(3) ? 0L : Convert.ToInt64(reader.GetValue(3)); - var distinctApps = reader.IsDBNull(4) ? 0L : Convert.ToInt64(reader.GetValue(4)); - var maxAppConns = reader.IsDBNull(5) ? 0L : Convert.ToInt64(reader.GetValue(5)); - - facts.Add(new Fact - { - Source = "sessions", - Key = "SESSION_STATS", - Value = totalConns, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["total_connections"] = totalConns, - ["total_running"] = totalRunning, - ["total_sleeping"] = totalSleeping, - ["total_dormant"] = totalDormant, - ["distinct_applications"] = distinctApps, - ["max_app_connections"] = maxAppConns - } - }); - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectSessionFactsAsync failed", ex); - } - } - - /// - /// Collects active global trace flags. Context for the AI to factor into recommendations. - /// - private async Task CollectTraceFlagFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -;WITH latest AS ( - SELECT - trace_flag, - status, - ROW_NUMBER() OVER (PARTITION BY trace_flag ORDER BY collection_time DESC) AS rn - FROM config.trace_flags_history - WHERE is_global = 1 -) -SELECT trace_flag -FROM latest WHERE rn = 1 AND status = 1 -ORDER BY trace_flag"; - - using var reader = await cmd.ExecuteReaderAsync(); - var metadata = new Dictionary(); - var flagCount = 0; - - while (await reader.ReadAsync()) - { - var flag = Convert.ToInt32(reader.GetValue(0)); - metadata[$"TF_{flag}"] = 1; - flagCount++; - } - - if (flagCount == 0) return; - - metadata["flag_count"] = flagCount; - - facts.Add(new Fact - { - Source = "config", - Key = "TRACE_FLAGS", - Value = flagCount, - ServerId = context.ServerId, - Metadata = metadata - }); - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectTraceFlagFactsAsync failed", ex); - } - } - /// /// Collects server hardware properties: CPU count, cores, sockets, memory. /// Critical context for MAXDOP and memory recommendations. @@ -2009,422 +106,4 @@ FROM collect.server_properties ORDER BY collection_time DESC"; } - private async Task CollectServerPropertiesFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - // Version-skew resilience: a server whose PerformanceMonitor DB has not yet had the WS5 - // upgrade lacks some or all of the three server-health columns. Probe EACH independently - // (COL_LENGTH returns NULL for an absent column) and reference only the present ones, so - // the core SERVER_HARDWARE read never fails — and keeps flowing — regardless of which - // columns a partially-upgraded or out-of-order schema happens to have. - bool hasLpim, hasIfi, hasDumps; - using (var probe = connection.CreateCommand()) - { - probe.CommandText = $@" -SELECT - COL_LENGTH('collect.server_properties', '{LpimColumn}'), - COL_LENGTH('collect.server_properties', '{IfiColumn}'), - COL_LENGTH('collect.server_properties', '{DumpsColumn}');"; - using var probeReader = await probe.ExecuteReaderAsync(); - await probeReader.ReadAsync(); - hasLpim = !probeReader.IsDBNull(0); - hasIfi = !probeReader.IsDBNull(1); - hasDumps = !probeReader.IsDBNull(2); - } - - using var cmd = connection.CreateCommand(); - cmd.CommandText = BuildServerPropertiesQuery(hasLpim, hasIfi, hasDumps); - - using var reader = await cmd.ExecuteReaderAsync(); - if (!await reader.ReadAsync()) return; - - var cpuCount = reader.IsDBNull(0) ? 0 : Convert.ToInt32(reader.GetValue(0)); - var htRatio = reader.IsDBNull(1) ? 0 : Convert.ToInt32(reader.GetValue(1)); - var physicalMemMb = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); - var socketCount = reader.IsDBNull(3) ? 0 : Convert.ToInt32(reader.GetValue(3)); - var coresPerSocket = reader.IsDBNull(4) ? 0 : Convert.ToInt32(reader.GetValue(4)); - var hadrEnabled = !reader.IsDBNull(5) && Convert.ToBoolean(reader.GetValue(5)); - var edition = reader.IsDBNull(6) ? string.Empty : reader.GetString(6); - - // Read each PRESENT health column by name — the SELECT includes only the columns that - // exist, so their ordinals shift with the subset; GetOrdinal resolves each regardless. - // An absent column stays null, so EmitServerHealthFacts emits nothing for it. - bool? lpim = null; - bool? ifi = null; - int? dumpCount = null; - if (hasLpim) - { - var ord = reader.GetOrdinal(LpimColumn); - lpim = reader.IsDBNull(ord) ? (bool?)null : Convert.ToBoolean(reader.GetValue(ord)); - } - if (hasIfi) - { - var ord = reader.GetOrdinal(IfiColumn); - ifi = reader.IsDBNull(ord) ? (bool?)null : Convert.ToBoolean(reader.GetValue(ord)); - } - if (hasDumps) - { - var ord = reader.GetOrdinal(DumpsColumn); - dumpCount = reader.IsDBNull(ord) ? (int?)null : Convert.ToInt32(reader.GetValue(ord)); - } - - if (cpuCount == 0) return; - - facts.Add(new Fact - { - Source = "config", - Key = "SERVER_HARDWARE", - Value = cpuCount, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["cpu_count"] = cpuCount, - ["hyperthread_ratio"] = htRatio, - ["physical_memory_mb"] = physicalMemMb, - ["socket_count"] = socketCount, - ["cores_per_socket"] = coresPerSocket, - ["hadr_enabled"] = hadrEnabled ? 1 : 0 - } - }); - - // WS5 server-health advisories (advise-only). Gating lives here so a fact that would - // score 0 is simply never emitted (noise control); the scorer then scores the emitted - // fact's Value. Shared with Lite — keep the rules identical (see DuckDbFactCollector). - EmitServerHealthFacts(context, facts, edition, physicalMemMb, lpim, ifi, dumpCount); - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectServerPropertiesFactsAsync failed", ex); - } - } - - // RAM floor below which LPIM-off is not worth flagging — on a small buffer pool the OS paging - // SQL out is not the practical risk it is on a large dedicated host. Shared rule with Lite. - private const long LpimAdvisoryMinPhysicalMemoryMb = 32 * 1024; - - /// - /// Emits the WS5 advise-only server-health facts (IFI off / LPIM off / memory dumps) from the - /// latest server_properties values, applying the noise-control gating both apps share: - /// • IFI: emit whenever the value is known (Value = enabled bit) — universally good advice. - /// • LPIM: emit only on non-Express editions with meaningful RAM (Value = enabled bit) — so a - /// tiny instance never flags. When LPIM is ON the emitted Value scores 0 (harmless). - /// • Dumps: emit whenever the count is known (Value = count) — the scorer flags count > 0. - /// - private static void EmitServerHealthFacts( - AnalysisContext context, List facts, string edition, long physicalMemMb, - bool? lockPagesInMemory, bool? instantFileInit, int? memoryDumpCount) - { - var isExpress = edition.Contains("Express", StringComparison.OrdinalIgnoreCase); - - if (instantFileInit.HasValue) - { - facts.Add(new Fact - { - Source = "config", - Key = "CONFIG_IFI_DISABLED", - Value = instantFileInit.Value ? 1 : 0, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["instant_file_initialization_enabled"] = instantFileInit.Value ? 1 : 0 - } - }); - } - - if (lockPagesInMemory.HasValue && !isExpress && physicalMemMb >= LpimAdvisoryMinPhysicalMemoryMb) - { - facts.Add(new Fact - { - Source = "config", - Key = "CONFIG_LPIM_DISABLED", - Value = lockPagesInMemory.Value ? 1 : 0, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["lock_pages_in_memory"] = lockPagesInMemory.Value ? 1 : 0, - ["physical_memory_mb"] = physicalMemMb - } - }); - } - - if (memoryDumpCount.HasValue) - { - facts.Add(new Fact - { - Source = "config", - Key = "SERVER_MEMORY_DUMPS", - Value = memoryDumpCount.Value, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["memory_dump_count"] = memoryDumpCount.Value - } - }); - } - } - - /// - /// WS4: plan-XML advisories. Parses the already-collected query plans of the top queries by - /// cost (no live fetch, no DMV) with the shared ShowPlanParser/PlanAnalyzer and emits two - /// advise-only facts — MISSING_INDEX (Value = distinct suggested indexes) and PLAN_WARNING - /// (Value = actionable warnings). The specific indexes/warnings ride in the finding drill-down - /// (SqlServerDrillDownCollector); Fact.Metadata is numeric only. - /// - private async Task CollectPlanAdvisoryFactsAsync(AnalysisContext context, List facts) - { - try - { - var planXmls = new List(); - - using (var connection = new SqlConnection(_connectionString)) - { - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -SELECT TOP (10) - plan_xml = CAST(DECOMPRESS(qs.query_plan_text) AS nvarchar(max)) -FROM collect.query_stats AS qs -WHERE qs.collection_time >= @startTime -AND qs.collection_time <= @endTime -AND qs.query_plan_text IS NOT NULL -ORDER BY - qs.total_worker_time DESC;"; - cmd.Parameters.Add(new SqlParameter("@startTime", context.TimeRangeStart)); - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - using var reader = await cmd.ExecuteReaderAsync(); - while (await reader.ReadAsync()) - { - if (!reader.IsDBNull(0)) - planXmls.Add(reader.GetString(0)); - } - } - - if (planXmls.Count == 0) - return; - - var summary = PlanAdvisoryAggregator.Summarize(planXmls); - - if (summary.MissingIndexCount > 0) - { - facts.Add(new Fact - { - Source = "queries", - Key = "MISSING_INDEX", - Value = summary.MissingIndexCount, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["index_count"] = summary.MissingIndexCount, - ["max_impact"] = summary.MaxImpact - } - }); - } - - if (summary.WarningCount > 0) - { - facts.Add(new Fact - { - Source = "queries", - Key = "PLAN_WARNING", - Value = summary.WarningCount, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["warning_count"] = summary.WarningCount, - ["critical_count"] = summary.CriticalCount - } - }); - } - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectPlanAdvisoryFactsAsync failed", ex); - } - } - - /// - /// Collects disk space facts from database_size_stats: volume free space, file sizes. - /// - private async Task CollectDiskSpaceFactsAsync(AnalysisContext context, List facts) - { - try - { - using var connection = new SqlConnection(_connectionString); - await connection.OpenAsync(); - - using var cmd = connection.CreateCommand(); - cmd.CommandText = @" -SET TRANSACTION ISOLATION LEVEL READ UNCOMMITTED; - -;WITH latest AS ( - SELECT - volume_mount_point, - volume_total_mb, - volume_free_mb, - ROW_NUMBER() OVER (PARTITION BY volume_mount_point ORDER BY collection_time DESC) AS rn - FROM collect.database_size_stats - WHERE collection_time <= @endTime - AND volume_total_mb > 0 -) -SELECT - MIN(volume_free_mb * 1.0 / volume_total_mb) AS min_free_pct, - MIN(volume_free_mb) AS min_free_mb, - COUNT(DISTINCT volume_mount_point) AS volume_count, - SUM(volume_total_mb) AS total_volume_mb, - SUM(volume_free_mb) AS total_free_mb -FROM latest WHERE rn = 1"; - - cmd.Parameters.Add(new SqlParameter("@endTime", context.TimeRangeEnd)); - - using var reader = await cmd.ExecuteReaderAsync(); - if (!await reader.ReadAsync()) return; - - var minFreePct = reader.IsDBNull(0) ? 1.0 : Convert.ToDouble(reader.GetValue(0)); - var minFreeMb = reader.IsDBNull(1) ? 0.0 : Convert.ToDouble(reader.GetValue(1)); - var volumeCount = reader.IsDBNull(2) ? 0L : Convert.ToInt64(reader.GetValue(2)); - var totalVolumeMb = reader.IsDBNull(3) ? 0.0 : Convert.ToDouble(reader.GetValue(3)); - var totalFreeMb = reader.IsDBNull(4) ? 0.0 : Convert.ToDouble(reader.GetValue(4)); - - if (volumeCount == 0) return; - - facts.Add(new Fact - { - Source = "disk", - Key = "DISK_SPACE", - Value = minFreePct, - ServerId = context.ServerId, - Metadata = new Dictionary - { - ["min_free_pct"] = minFreePct, - ["min_free_mb"] = minFreeMb, - ["volume_count"] = volumeCount, - ["total_volume_mb"] = totalVolumeMb, - ["total_free_mb"] = totalFreeMb - } - }); - } - catch (Exception ex) - { - Logger.Error("SqlServerFactCollector.CollectDiskSpaceFactsAsync failed", ex); - } - } - - /// - /// Groups general lock waits (X, U, IX, SIX, BU, IU, UIX, etc.) into a single "LCK" fact. - /// Keeps individual facts for: - /// - LCK_M_S, LCK_M_IS (reader/writer blocking -- RCSI signal) - /// - LCK_M_RS_*, LCK_M_RIn_*, LCK_M_RX_* (serializable/repeatable read signal) - /// - SCH_M, SCH_S (schema locks -- DDL/index operations) - /// Individual constituent wait times are preserved in metadata as "{type}_ms" keys. - /// - private static void GroupGeneralLockWaits(List facts, AnalysisContext context) - { - var generalLocks = facts.Where(f => f.Source == "waits" && IsGeneralLockWait(f.Key)).ToList(); - if (generalLocks.Count == 0) return; - - var totalWaitTimeMs = generalLocks.Sum(f => f.Metadata.GetValueOrDefault("wait_time_ms")); - var totalWaitingTasks = generalLocks.Sum(f => f.Metadata.GetValueOrDefault("waiting_tasks_count")); - var totalSignalMs = generalLocks.Sum(f => f.Metadata.GetValueOrDefault("signal_wait_time_ms")); - var avgMsPerWait = totalWaitingTasks > 0 ? totalWaitTimeMs / totalWaitingTasks : 0; - var fractionOfPeriod = totalWaitTimeMs / context.PeriodDurationMs; - - var metadata = new Dictionary - { - ["wait_time_ms"] = totalWaitTimeMs, - ["waiting_tasks_count"] = totalWaitingTasks, - ["signal_wait_time_ms"] = totalSignalMs, - ["resource_wait_time_ms"] = totalWaitTimeMs - totalSignalMs, - ["avg_ms_per_wait"] = avgMsPerWait, - ["period_duration_ms"] = context.PeriodDurationMs, - ["lock_type_count"] = generalLocks.Count - }; - - // Preserve individual constituent wait times for detailed analysis - foreach (var lck in generalLocks) - metadata[$"{lck.Key}_ms"] = lck.Metadata.GetValueOrDefault("wait_time_ms"); - - // Remove individual facts, add grouped fact - foreach (var lck in generalLocks) - facts.Remove(lck); - - facts.Add(new Fact - { - Source = "waits", - Key = "LCK", - Value = fractionOfPeriod, - ServerId = context.ServerId, - Metadata = metadata - }); - } - - /// - /// Groups all CX* parallelism waits (CXPACKET, CXCONSUMER, CXSYNC_PORT, CXSYNC_CONSUMER, etc.) - /// into a single "CXPACKET" fact. They all indicate the same thing: parallel queries are running. - /// Individual wait times are preserved in metadata for detailed analysis. - /// - private static void GroupParallelismWaits(List facts, AnalysisContext context) - { - var cxWaits = facts.Where(f => f.Source == "waits" && f.Key.StartsWith("CX", StringComparison.Ordinal)).ToList(); - if (cxWaits.Count <= 1) return; - - var totalWaitTimeMs = cxWaits.Sum(f => f.Metadata.GetValueOrDefault("wait_time_ms")); - var totalWaitingTasks = cxWaits.Sum(f => f.Metadata.GetValueOrDefault("waiting_tasks_count")); - var totalSignalMs = cxWaits.Sum(f => f.Metadata.GetValueOrDefault("signal_wait_time_ms")); - var avgMsPerWait = totalWaitingTasks > 0 ? totalWaitTimeMs / totalWaitingTasks : 0; - var fractionOfPeriod = totalWaitTimeMs / context.PeriodDurationMs; - - var metadata = new Dictionary - { - ["wait_time_ms"] = totalWaitTimeMs, - ["waiting_tasks_count"] = totalWaitingTasks, - ["signal_wait_time_ms"] = totalSignalMs, - ["resource_wait_time_ms"] = totalWaitTimeMs - totalSignalMs, - ["avg_ms_per_wait"] = avgMsPerWait, - ["period_duration_ms"] = context.PeriodDurationMs - }; - - // Preserve individual constituent wait times for detailed analysis - foreach (var cx in cxWaits) - metadata[$"{cx.Key}_ms"] = cx.Metadata.GetValueOrDefault("wait_time_ms"); - - foreach (var cx in cxWaits) - facts.Remove(cx); - - facts.Add(new Fact - { - Source = "waits", - Key = "CXPACKET", - Value = fractionOfPeriod, - ServerId = cxWaits[0].ServerId, - Metadata = metadata - }); - } - - /// - /// Returns true for general lock waits that should be grouped into "LCK". - /// Excludes reader locks (S, IS), range locks (RS_*, RIn_*, RX_*), and schema locks. - /// - private static bool IsGeneralLockWait(string waitType) - { - if (!waitType.StartsWith("LCK_M_", StringComparison.OrdinalIgnoreCase)) return false; - - // Keep individual: reader/writer locks - if (waitType is "LCK_M_S" or "LCK_M_IS") return false; - - // Keep individual: range locks (serializable/repeatable read) - if (waitType.StartsWith("LCK_M_RS_", StringComparison.OrdinalIgnoreCase) || - waitType.StartsWith("LCK_M_RIn_", StringComparison.OrdinalIgnoreCase) || - waitType.StartsWith("LCK_M_RX_", StringComparison.OrdinalIgnoreCase)) return false; - - // Everything else (X, U, IX, SIX, BU, IU, UIX, etc.) -> group - return true; - } } diff --git a/Dashboard/Analysis/SqlServerFindingStore.cs b/Dashboard/Analysis/SqlServerFindingStore.cs index 0fa8a0e2e..360c11f73 100644 --- a/Dashboard/Analysis/SqlServerFindingStore.cs +++ b/Dashboard/Analysis/SqlServerFindingStore.cs @@ -50,12 +50,13 @@ time_range_end datetime2(7) NULL, category nvarchar(256) NOT NULL, story_path nvarchar(2000) NOT NULL, story_path_hash nvarchar(256) NOT NULL, - story_text nvarchar(4000) NOT NULL, + story_text nvarchar(max) NOT NULL, root_fact_key nvarchar(256) NOT NULL, root_fact_value float NULL, leaf_fact_key nvarchar(256) NULL, leaf_fact_value float NULL, fact_count integer NOT NULL, + incident_id nvarchar(64) NULL, remediation_action_json nvarchar(max) NULL, CONSTRAINT PK_analysis_findings PRIMARY KEY CLUSTERED (finding_id) WITH (DATA_COMPRESSION = PAGE) @@ -89,7 +90,18 @@ ON config.analysis_muted (server_id, story_path_hash) idempotently. The Recommendations surface reads it back to drive Apply + the two-sided consent gate (the built RemediationAction, not raw drill-down). */ IF COL_LENGTH(N'config.analysis_findings', N'remediation_action_json') IS NULL - ALTER TABLE config.analysis_findings ADD remediation_action_json nvarchar(max) NULL;"; + ALTER TABLE config.analysis_findings ADD remediation_action_json nvarchar(max) NULL; + +/* Compose-from-facts: story_text now carries the serialized value-stated advice for EVERY finding + (it was previously written empty), so widen it from nvarchar(4000) to nvarchar(max) on existing + DBs — removes the truncation cliff and matches Lite's unbounded story_text. COL_LENGTH returns + -1 for nvarchar(max); any other value means the column still needs widening. NOT NULL is kept. */ +IF COL_LENGTH(N'config.analysis_findings', N'story_text') <> -1 + ALTER TABLE config.analysis_findings ALTER COLUMN story_text nvarchar(max) NOT NULL; + +/* Correlate-and-focus slice 2: the incident grouping id, added idempotently on existing DBs. */ +IF COL_LENGTH(N'config.analysis_findings', N'incident_id') IS NULL + ALTER TABLE config.analysis_findings ADD incident_id nvarchar(64) NULL;"; await cmd.ExecuteNonQueryAsync(); } @@ -146,6 +158,7 @@ public async Task> FilterMutedFindingsAsync( Category = story.Category, StoryPath = story.StoryPath, StoryPathHash = story.StoryPathHash, + IncidentId = story.IncidentId, StoryText = story.StoryText, RootFactKey = story.RootFactKey, RootFactValue = story.RootFactValue, @@ -219,7 +232,7 @@ SELECT TOP (@limit) time_range_start, time_range_end, severity, confidence, category, story_path, story_path_hash, story_text, root_fact_key, root_fact_value, leaf_fact_key, leaf_fact_value, fact_count, - remediation_action_json + incident_id, remediation_action_json FROM config.analysis_findings WHERE server_id = @serverId AND analysis_time >= @cutoff @@ -264,7 +277,8 @@ public async Task> GetLatestFindingsAsync(int serverId) finding_id, analysis_time, server_id, server_name, database_name, time_range_start, time_range_end, severity, confidence, category, story_path, story_path_hash, story_text, - root_fact_key, root_fact_value, leaf_fact_key, leaf_fact_value, fact_count + root_fact_key, root_fact_value, leaf_fact_key, leaf_fact_value, fact_count, + incident_id FROM config.analysis_findings WHERE server_id = @serverId AND analysis_time = ( @@ -411,13 +425,13 @@ INSERT INTO config.analysis_findings time_range_start, time_range_end, severity, confidence, category, story_path, story_path_hash, story_text, root_fact_key, root_fact_value, leaf_fact_key, leaf_fact_value, fact_count, - remediation_action_json) + incident_id, remediation_action_json) VALUES (@findingId, @analysisTime, @serverId, @serverName, @databaseName, @timeRangeStart, @timeRangeEnd, @severity, @confidence, @category, @storyPath, @storyPathHash, @storyText, @rootFactKey, @rootFactValue, @leafFactKey, @leafFactValue, @factCount, - @remediationActionJson);"; + @incidentId, @remediationActionJson);"; cmd.Parameters.Add(new SqlParameter("@findingId", finding.FindingId)); cmd.Parameters.Add(new SqlParameter("@analysisTime", finding.AnalysisTime)); @@ -437,6 +451,8 @@ INSERT INTO config.analysis_findings cmd.Parameters.Add(new SqlParameter("@leafFactKey", (object?)finding.LeafFactKey ?? DBNull.Value)); cmd.Parameters.Add(new SqlParameter("@leafFactValue", (object?)finding.LeafFactValue ?? DBNull.Value)); cmd.Parameters.Add(new SqlParameter("@factCount", finding.FactCount)); + cmd.Parameters.Add(new SqlParameter("@incidentId", + (object?)(string.IsNullOrEmpty(finding.IncidentId) ? null : finding.IncidentId) ?? DBNull.Value)); // D2: persist the BUILT action (mirrors the alert path's ContextJson) so the // Recommendations reader can drive Apply + consent from a stored finding. cmd.Parameters.Add(new SqlParameter("@remediationActionJson", @@ -474,15 +490,17 @@ private static AnalysisFinding ReadFinding(SqlDataReader reader) RootFactValue = reader.IsDBNull(14) ? null : reader.GetDouble(14), LeafFactKey = reader.IsDBNull(15) ? null : reader.GetString(15), LeafFactValue = reader.IsDBNull(16) ? null : reader.GetDouble(16), - FactCount = reader.GetInt32(17) + FactCount = reader.GetInt32(17), + // incident_id is ordinal 18 in BOTH SELECTs (correlate-and-focus slice 2). + IncidentId = reader.FieldCount > 18 && !reader.IsDBNull(18) ? reader.GetString(18) : string.Empty }; - // D2: only GetRecentFindingsAsync selects remediation_action_json (ordinal 18); - // GetLatestFindingsAsync omits it, so guard by field count before reading. The + // D2: GetRecentFindingsAsync selects remediation_action_json (now ordinal 19, after + // incident_id); GetLatestFindingsAsync omits it, so guard by field count before reading. The // BUILT action is deserialized via the SAME serializer the alert path uses, so the // Recommendations surface can drive Apply + the two-sided consent gate from storage. - if (reader.FieldCount > 18 && !reader.IsDBNull(18)) - finding.Remediation = AlertContextSerializer.DeserializeAction(reader.GetString(18)); + if (reader.FieldCount > 19 && !reader.IsDBNull(19)) + finding.Remediation = AlertContextSerializer.DeserializeAction(reader.GetString(19)); return finding; } diff --git a/Dashboard/App.xaml.cs b/Dashboard/App.xaml.cs index 2f3e066e2..319b4560b 100644 --- a/Dashboard/App.xaml.cs +++ b/Dashboard/App.xaml.cs @@ -22,26 +22,49 @@ namespace PerformanceMonitorDashboard public partial class App : Application { private const string MutexName = "PerformanceMonitorDashboard_SingleInstance"; - private Mutex? _singleInstanceMutex; - private bool _ownsMutex; + /* Version-aware single-instance + upgrade handoff (plans/single-instance-upgrade-handoff.md): + a newer build launched over an older tray-resident one closes it and takes over instead of + being handed back the stale in-memory version. The coordinator owns the mutex + the + exit-for-upgrade listener for the life of the owning process. */ + private const string ExitForUpgradeEventName = "PerformanceMonitorDashboard_ExitForUpgrade"; + private SingleInstanceCoordinator? _instanceCoordinator; protected override void OnStartup(StartupEventArgs e) { NativeMethods.SetAppUserModelId("DarlingData.PerformanceMonitor.Dashboard"); - // Check for existing instance - _singleInstanceMutex = new Mutex(true, MutexName, out _ownsMutex); - - if (!_ownsMutex) + /* Single-instance with upgrade handoff. Runs synchronously at the top of OnStartup before + base.OnStartup and any window/MCP init, so a stale older build is closed before we bind + the MCP port / touch shared config. A same/newer instance just surfaces (today's behavior + via WM_SHOWMONITOR); an older-but-elevated one raises an actionable error. */ + _instanceCoordinator = new SingleInstanceCoordinator(new SingleInstanceOptions + { + MutexName = MutexName, + ProcessName = "PerformanceMonitorDashboard", + ExitEventName = ExitForUpgradeEventName, + SurfaceRunningInstance = NativeMethods.BroadcastShowMessage, + GracefulSelfExit = () => Dispatcher.BeginInvoke(new Action(() => + { + if (MainWindow is MainWindow mw) mw.ExitApplication(); + else Shutdown(); + })), + Prompts = new MessageBoxHandoffPrompts("Performance Monitor Dashboard"), + AutoConfirm = Array.Exists(e.Args, a => string.Equals(a, HandoffArgs.AutoConfirm, StringComparison.OrdinalIgnoreCase)), + Log = msg => { try { Logger.Info($"[SingleInstance] {msg}"); } catch { /* logger not yet initialized */ } }, + }); + + if (!_instanceCoordinator.TryBecomeOwner()) { - // Another instance is already running - activate it and exit - NativeMethods.BroadcastShowMessage(); Shutdown(); return; } base.OnStartup(e); + // Right-click selects the DataGrid row under the cursor app-wide, so context-menu actions + // (e.g. View Plan) act on the clicked row even after an auto-refresh cleared the selection. + PerformanceMonitor.Ui.DataGridRowSelectionBehavior.Enable(); + // #1050: WPF's GPU render thread can zombie its surface across sleep/wake or RDP, leaving a // live-but-blank window. Software rendering removes the GPU dependency entirely. Charts are // unaffected — ScottPlot renders via SkiaSharp (CPU) into a bitmap, not WPF's GPU path. @@ -73,6 +96,13 @@ protected override void OnStartup(StartupEventArgs e) mainWindow.Show(); } + /// + /// Opens the upgrade-handoff "exit" channel once startup is past its risky init. Called by + /// after initialization so a newer build won't signal/kill us mid-init + /// (#single-instance-upgrade-handoff). Safe to call more than once. + /// + public void EnableUpgradeHandoff() => _instanceCoordinator?.EnableUpgradeHandoff(); + protected override void OnExit(ExitEventArgs e) { Logger.Info($"=== Application Exiting (Exit Code: {e.ApplicationExitCode}) ==="); @@ -83,11 +113,8 @@ protected override void OnExit(ExitEventArgs e) mainWin.ExitApplication(); } - if (_ownsMutex) - { - _singleInstanceMutex?.ReleaseMutex(); - } - _singleInstanceMutex?.Dispose(); + /* Releases the mutex + disposes the exit-for-upgrade listener. */ + _instanceCoordinator?.Dispose(); base.OnExit(e); } diff --git a/Dashboard/Controls/AlertsHistoryContent.xaml.cs b/Dashboard/Controls/AlertsHistoryContent.xaml.cs index 5026a6c24..aec46c8d0 100644 --- a/Dashboard/Controls/AlertsHistoryContent.xaml.cs +++ b/Dashboard/Controls/AlertsHistoryContent.xaml.cs @@ -99,10 +99,9 @@ private void LoadAlerts() ThresholdValue = e.ThresholdValue, NotificationType = e.NotificationType, StatusDisplay = GetStatusDisplay(e), - IsResolved = e.MetricName.Contains("Cleared") || e.MetricName.Contains("Resolved"), - IsCritical = e.MetricName.Contains("Deadlock") || e.MetricName.Contains("Poison"), - IsWarning = !e.MetricName.Contains("Cleared") && !e.MetricName.Contains("Resolved") - && !e.MetricName.Contains("Deadlock") && !e.MetricName.Contains("Poison"), + IsResolved = AlertMetricClassifier.IsResolution(e.MetricName), + IsCritical = AlertMetricClassifier.IsCritical(e.MetricName), + IsWarning = AlertMetricClassifier.IsWarning(e.MetricName), Muted = e.Muted, DetailText = e.DetailText, ContextJson = e.ContextJson diff --git a/Dashboard/Controls/CorrelatedTimelineLanesControl.xaml.cs b/Dashboard/Controls/CorrelatedTimelineLanesControl.xaml.cs index a19c736d5..da6a3b7ba 100644 --- a/Dashboard/Controls/CorrelatedTimelineLanesControl.xaml.cs +++ b/Dashboard/Controls/CorrelatedTimelineLanesControl.xaml.cs @@ -20,6 +20,7 @@ using PerformanceMonitorDashboard.Analysis; using PerformanceMonitorDashboard.Helpers; using PerformanceMonitorDashboard.Services; +using PerformanceMonitor.Common; using PerformanceMonitor.Ui; namespace PerformanceMonitorDashboard.Controls; @@ -56,6 +57,7 @@ public void Initialize(DatabaseService dataService, SqlServerBaselineProvider? b TabHelpers.ApplyThemeToChart(chart); // Disable zoom/pan/drag but keep mouse events for crosshair chart.UserInputProcessor.UserActionResponses.Clear(); + SetupLaneDrillDown(chart); } _crosshairManager = new CorrelatedCrosshairManager(); @@ -66,6 +68,58 @@ public void Initialize(DatabaseService dataService, SqlServerBaselineProvider? b _crosshairManager.AddLane(FileIoChart, "I/O Latency", "ms"); } + /// + /// Raised when the user picks "Show Active Queries at This Time" on a lane. The argument is the + /// clicked time in the lanes' (server-local) X-axis space; the host navigates to Active Queries. + /// + public event Action? ShowActiveQueriesRequested; + + /// + /// Adds a minimal right-click menu (just the Active Queries drill-down) to a lane. The lanes are a + /// stripped-chrome view with pan/zoom disabled, so the clicked time is read straight from the X axis. + /// + private void SetupLaneDrillDown(ScottPlot.WPF.WpfPlot chart) + { + var menu = new ContextMenu(); + var item = new MenuItem { Header = "Show Active Queries at This Time" }; + menu.Items.Add(item); + + menu.Opened += (s, _) => + { + try + { + var pos = System.Windows.Input.Mouse.GetPosition(chart); + var dpi = System.Windows.Media.VisualTreeHelper.GetDpi(chart); + var pixel = new ScottPlot.Pixel((float)(pos.X * dpi.DpiScaleX), (float)(pos.Y * dpi.DpiScaleY)); + var t = DateTime.FromOADate(chart.Plot.GetCoordinates(pixel).X); + // Empty-state lanes set the X axis to [-1, 1] (~year 1899); only offer the drill-down + // when the click resolves to a real timestamp. + bool valid = t.Year >= 2000; + item.Tag = valid ? t : (DateTime?)null; + item.IsEnabled = valid; + } + catch + { + item.Tag = null; + item.IsEnabled = false; + } + }; + + item.Click += (s, _) => + { + if (item.Tag is DateTime t) + ShowActiveQueriesRequested?.Invoke(t); + }; + + chart.PreviewMouseRightButtonDown += (s, e) => + { + e.Handled = true; + menu.PlacementTarget = chart; + menu.Placement = System.Windows.Controls.Primitives.PlacementMode.MousePoint; + menu.IsOpen = true; + }; + } + /// /// Refreshes all lane data for the given time range. /// @@ -296,7 +350,7 @@ private void UpdateBlockingLane(List<(double Time, double Value)> blockingData, Position = d.Time, Value = d.Value, Size = barWidth, - FillColor = ScottPlot.Color.FromHex("#E57373"), + FillColor = ScottPlot.Color.FromHex(ChartPalette.SeriesColor("Blocking")), LineWidth = 0 }).ToArray(); BlockingChart.Plot.Add.Bars(bars); @@ -310,7 +364,7 @@ private void UpdateBlockingLane(List<(double Time, double Value)> blockingData, Position = d.Time, Value = d.Value, Size = barWidth * 0.6, - FillColor = ScottPlot.Color.FromHex("#FFD54F"), + FillColor = ScottPlot.Color.FromHex(ChartPalette.SeriesColor("Deadlocks")), LineWidth = 0 }).ToArray(); BlockingChart.Plot.Add.Bars(bars); @@ -332,11 +386,11 @@ private void UpdateBlockingLane(List<(double Time, double Value)> blockingData, if (baseline.EffectiveStdDev > 0) { var band = BlockingChart.Plot.Add.HorizontalSpan(lower, upper); - band.FillStyle.Color = ScottPlot.Color.FromHex("#E57373").WithAlpha(25); + band.FillStyle.Color = ScottPlot.Color.FromHex(ChartPalette.AccentColor("BaselineBlocking")).WithAlpha(25); band.LineStyle.Width = 0; var meanLine = BlockingChart.Plot.Add.HorizontalLine(baseline.Mean); - meanLine.Color = ScottPlot.Color.FromHex("#E57373").WithAlpha(60); + meanLine.Color = ScottPlot.Color.FromHex(ChartPalette.AccentColor("BaselineBlocking")).WithAlpha(60); meanLine.LinePattern = ScottPlot.LinePattern.Dashed; meanLine.LineWidth = 1; } @@ -389,11 +443,11 @@ private void UpdateCpuLane( _crosshairManager?.SetLaneBaseline(CpuChart, lower, upper, 10); var band = CpuChart.Plot.Add.HorizontalSpan(lower, upper); - band.FillStyle.Color = ScottPlot.Color.FromHex("#4FC3F7").WithAlpha(25); + band.FillStyle.Color = ScottPlot.Color.FromHex(ChartPalette.AccentColor("BaselineCpu")).WithAlpha(25); band.LineStyle.Width = 0; var meanLine = CpuChart.Plot.Add.HorizontalLine(baseline.Mean); - meanLine.Color = ScottPlot.Color.FromHex("#4FC3F7").WithAlpha(60); + meanLine.Color = ScottPlot.Color.FromHex(ChartPalette.AccentColor("BaselineCpu")).WithAlpha(60); meanLine.LinePattern = ScottPlot.LinePattern.Dashed; meanLine.LineWidth = 1; @@ -409,7 +463,7 @@ private void UpdateCpuLane( var anomalyTimes = anomalyIndices.Select(i => sqlTimes[i]).ToArray(); var anomalyVals = anomalyIndices.Select(i => sqlValues[i]).ToArray(); var anomalyScatter = CpuChart.Plot.Add.Scatter(anomalyTimes, anomalyVals); - anomalyScatter.Color = ScottPlot.Color.FromHex("#FF5252"); + anomalyScatter.Color = ScottPlot.Color.FromHex(ChartPalette.AccentColor("Anomaly")); anomalyScatter.MarkerSize = 6; anomalyScatter.MarkerShape = ScottPlot.MarkerShape.FilledCircle; anomalyScatter.LineWidth = 0; @@ -419,7 +473,7 @@ private void UpdateCpuLane( if (totalValues.Length > 0) { var totalScatter = CpuChart.Plot.Add.Scatter(totalTimes, totalValues); - totalScatter.Color = ScottPlot.Color.FromHex("#FF7043"); + totalScatter.Color = ScottPlot.Color.FromHex(ChartPalette.SeriesColor("TotalCpu")); totalScatter.MarkerSize = 0; totalScatter.LineWidth = 1.5f; totalScatter.LegendText = "Total"; @@ -429,7 +483,7 @@ private void UpdateCpuLane( if (sqlValues.Length > 0) { var sqlScatter = CpuChart.Plot.Add.Scatter(sqlTimes, sqlValues); - sqlScatter.Color = ScottPlot.Color.FromHex("#4FC3F7"); + sqlScatter.Color = ScottPlot.Color.FromHex(ChartPalette.SeriesColor("SqlCpu")); sqlScatter.MarkerSize = 0; sqlScatter.LineWidth = 1.5f; sqlScatter.LegendText = "SQL"; @@ -496,7 +550,7 @@ private void UpdateLane(ScottPlot.WPF.WpfPlot chart, string title, var anomalyTimes = anomalyIndices.Select(i => times[i]).ToArray(); var anomalyValues = anomalyIndices.Select(i => values[i]).ToArray(); var anomalyScatter = chart.Plot.Add.Scatter(anomalyTimes, anomalyValues); - anomalyScatter.Color = ScottPlot.Color.FromHex("#FF5252"); + anomalyScatter.Color = ScottPlot.Color.FromHex(ChartPalette.AccentColor("Anomaly")); anomalyScatter.MarkerSize = 6; anomalyScatter.MarkerShape = ScottPlot.MarkerShape.FilledCircle; anomalyScatter.LineWidth = 0; @@ -568,7 +622,7 @@ private static void AddGhostLine(ScottPlot.WPF.WpfPlot chart, var values = data.Select(d => d.Value).ToArray(); var scatter = chart.Plot.Add.Scatter(times, values); - scatter.Color = ScottPlot.Colors.White.WithAlpha(140); + scatter.Color = ScottPlot.Color.FromHex(ChartPalette.AccentColor("GhostLine")).WithAlpha(140); scatter.MarkerSize = 0; scatter.LineWidth = 1.5f; scatter.LinePattern = ScottPlot.LinePattern.Dashed; @@ -590,19 +644,27 @@ private static void ClearChart(ScottPlot.WPF.WpfPlot chart) chart.Plot.Clear(); } - private static void ShowEmpty(ScottPlot.WPF.WpfPlot chart, string title) + /* Render an empty lane as a live, gridded chart instead of a dead black box. This is most often + the Blocking/Deadlocking lane on a healthy server (no events = good news), so make it match the + populated lanes: keep the grid and show a 0-1 Y axis. We no longer blank the axes or add the old + "No Data" text (which SyncXAxes pushed off-screen anyway). X limits and the vertical gridlines are + set afterward by SyncXAxes so the time axis aligns with the other lanes; the left axis keeps its + default numeric ticks, so 0/1 labels and horizontal gridlines render. SyncXAxes also issues the + Refresh. The title arg is retained for call-site readability. */ + private void ShowEmpty(ScottPlot.WPF.WpfPlot chart, string title) { + chart.Plot.Axes.DateTimeTicksBottomDateChange(); + // Only the bottom (File I/O) lane shows time labels; the upper lanes hide them (matches UpdateLane). + if (chart != FileIoChart) + chart.Plot.Axes.Bottom.TickLabelStyle.IsVisible = false; + TabHelpers.ReapplyAxisColors(chart); - var text = chart.Plot.Add.Text($"{title}\nNo Data", 0, 0); - text.LabelFontColor = ScottPlot.Color.FromHex("#888888"); - text.LabelFontSize = 12; - text.LabelAlignment = ScottPlot.Alignment.MiddleCenter; - chart.Plot.HideGrid(); - chart.Plot.Axes.SetLimitsX(-1, 1); - chart.Plot.Axes.SetLimitsY(-1, 1); - chart.Plot.Axes.Bottom.TickGenerator = new ScottPlot.TickGenerators.EmptyTickGenerator(); - chart.Plot.Axes.Left.TickGenerator = new ScottPlot.TickGenerators.EmptyTickGenerator(); + + chart.Plot.Title(""); + chart.Plot.YLabel(""); chart.Plot.Legend.IsVisible = false; + chart.Plot.Axes.Margins(bottom: 0); + chart.Plot.Axes.SetLimitsY(0, 1); } /// diff --git a/Dashboard/Controls/FinOpsContent.CopyExport.cs b/Dashboard/Controls/FinOpsContent.CopyExport.cs new file mode 100644 index 000000000..14b05a871 --- /dev/null +++ b/Dashboard/Controls/FinOpsContent.CopyExport.cs @@ -0,0 +1,173 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.IO; +using System.Linq; +using System.Text; +using System.Threading.Tasks; +using System.Windows; +using System.Windows.Controls; +using System.Windows.Controls.Primitives; +using System.Windows.Data; +using System.Windows.Media; +using Microsoft.Win32; +using PerformanceMonitorDashboard.Helpers; +using PerformanceMonitorDashboard.Models; +using PerformanceMonitorDashboard.Services; +using PerformanceMonitor.Ui; +using PerformanceMonitor.Common; + +namespace PerformanceMonitorDashboard.Controls +{ + public partial class FinOpsContent : UserControl + { + + // ============================================ + // Copy / Export Context Menu Handlers + // ============================================ + + private static DataGrid? FindParentDataGrid(DependencyObject? element) + { + while (element != null) + { + if (element is DataGrid dg) return dg; + element = VisualTreeHelper.GetParent(element); + } + return null; + } + + private void CopyCell_Click(object sender, RoutedEventArgs e) + { + if (sender is MenuItem menuItem && menuItem.Parent is ContextMenu contextMenu) + { + var grid = FindParentDataGrid(contextMenu.PlacementTarget); + if (grid != null && grid.CurrentCell.Column != null) + { + var cellContent = TabHelpers.GetCellContent(grid, grid.CurrentCell); + if (!string.IsNullOrEmpty(cellContent)) + { + /* Use SetDataObject with copy=false to avoid WPF's problematic Clipboard.Flush() */ + Clipboard.SetDataObject(cellContent, false); + } + } + } + } + + private void CopyRow_Click(object sender, RoutedEventArgs e) + { + if (sender is MenuItem menuItem && menuItem.Parent is ContextMenu contextMenu) + { + var grid = FindParentDataGrid(contextMenu.PlacementTarget); + if (grid != null && grid.SelectedItem != null) + { + var rowText = TabHelpers.GetRowAsText(grid, grid.SelectedItem); + if (!string.IsNullOrEmpty(rowText)) + { + /* Use SetDataObject with copy=false to avoid WPF's problematic Clipboard.Flush() */ + Clipboard.SetDataObject(rowText, false); + } + } + } + } + + private void CopyAllRows_Click(object sender, RoutedEventArgs e) + { + if (sender is MenuItem menuItem && menuItem.Parent is ContextMenu contextMenu) + { + var grid = FindParentDataGrid(contextMenu.PlacementTarget); + if (grid != null) + { + var sb = new StringBuilder(); + + // Header row + var headers = grid.Columns.Select(c => DataGridClipboardBehavior.GetHeaderText(c)); + sb.AppendLine(string.Join("\t", headers)); + + // Data rows + foreach (var item in grid.Items) + { + var values = new List(); + foreach (var column in grid.Columns) + { + var binding = (column as DataGridBoundColumn)?.Binding as Binding; + if (binding != null) + { + var prop = item.GetType().GetProperty(binding.Path.Path); + var value = prop?.GetValue(item)?.ToString() ?? string.Empty; + values.Add(value); + } + } + sb.AppendLine(string.Join("\t", values)); + } + + if (sb.Length > 0) + { + /* Use SetDataObject with copy=false to avoid WPF's problematic Clipboard.Flush() */ + Clipboard.SetDataObject(sb.ToString(), false); + } + } + } + } + + private void ExportToCsv_Click(object sender, RoutedEventArgs e) + { + if (sender is MenuItem menuItem && menuItem.Parent is ContextMenu contextMenu) + { + var grid = FindParentDataGrid(contextMenu.PlacementTarget); + if (grid != null) + { + var dialog = new SaveFileDialog + { + Filter = "CSV files (*.csv)|*.csv|All files (*.*)|*.*", + DefaultExt = ".csv", + FileName = $"FinOps_Export_{DateTime.Now:yyyyMMdd_HHmmss}.csv" + }; + + if (dialog.ShowDialog() == true) + { + try + { + var sb = new StringBuilder(); + + // Header row + var sep = TabHelpers.CsvSeparator; + var headers = grid.Columns.Select(c => TabHelpers.EscapeCsvField(DataGridClipboardBehavior.GetHeaderText(c), sep)); + sb.AppendLine(string.Join(sep, headers)); + + // Data rows + foreach (var item in grid.Items) + { + var values = new List(); + foreach (var column in grid.Columns) + { + var binding = (column as DataGridBoundColumn)?.Binding as Binding; + if (binding != null) + { + var prop = item.GetType().GetProperty(binding.Path.Path); + values.Add(TabHelpers.EscapeCsvField(TabHelpers.FormatForExport(prop?.GetValue(item)), sep)); + } + } + sb.AppendLine(string.Join(sep, values)); + } + + File.WriteAllText(dialog.FileName, sb.ToString()); + } + catch (Exception ex) + { + Logger.Error($"Error exporting to CSV: {ex.Message}", ex); + MessageBox.Show($"Error exporting to CSV: {ex.Message}", "Export Error", MessageBoxButton.OK, MessageBoxImage.Error); + } + } + } + } + } + + } +} diff --git a/Dashboard/Controls/FinOpsContent.Filters.cs b/Dashboard/Controls/FinOpsContent.Filters.cs new file mode 100644 index 000000000..5d6cf4c3c --- /dev/null +++ b/Dashboard/Controls/FinOpsContent.Filters.cs @@ -0,0 +1,228 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.IO; +using System.Linq; +using System.Text; +using System.Threading.Tasks; +using System.Windows; +using System.Windows.Controls; +using System.Windows.Controls.Primitives; +using System.Windows.Data; +using System.Windows.Media; +using Microsoft.Win32; +using PerformanceMonitorDashboard.Helpers; +using PerformanceMonitorDashboard.Models; +using PerformanceMonitorDashboard.Services; +using PerformanceMonitor.Ui; +using PerformanceMonitor.Common; + +namespace PerformanceMonitorDashboard.Controls +{ + public partial class FinOpsContent : UserControl + { + + private void DatabaseSizesFilter_Click(object sender, RoutedEventArgs e) + { + if (sender is not Button button || button.Tag is not string columnName) return; + + if (_dbSizeFilterPopup == null) + { + _dbSizeFilterPopupContent = new ColumnFilterPopup(); + _dbSizeFilterPopupContent.FilterApplied += FilterPopup_DbSizeFilterApplied; + _dbSizeFilterPopupContent.FilterCleared += FilterPopup_DbSizeFilterCleared; + _dbSizeFilterPopup = new Popup + { + Child = _dbSizeFilterPopupContent, + StaysOpen = false, + Placement = PlacementMode.Bottom, + AllowsTransparency = true + }; + } + + _dbSizesFilterMgr!.Filters.TryGetValue(columnName, out var existingFilter); + _dbSizeFilterPopupContent!.Initialize(columnName, existingFilter); + _dbSizeFilterPopup.PlacementTarget = button; + _dbSizeFilterPopup.IsOpen = true; + } + + private void FilterPopup_DbSizeFilterApplied(object? sender, FilterAppliedEventArgs e) + { + if (_dbSizeFilterPopup != null) + _dbSizeFilterPopup.IsOpen = false; + + _dbSizesFilterMgr!.SetFilter(e.FilterState); + UpdateDbSizeCountUI(); + } + + private void FilterPopup_DbSizeFilterCleared(object? sender, EventArgs e) + { + if (_dbSizeFilterPopup != null) + _dbSizeFilterPopup.IsOpen = false; + } + + // ============================================ + // Column Filtering + // ============================================ + + #region Column Filtering + + private Popup? _filterPopup; + private ColumnFilterPopup? _filterPopupContent; + private DataGrid? _currentFilterGrid; + private readonly Dictionary> _gridFilters = new(); + private readonly Dictionary _gridUnfilteredData = new(); + + private void EnsureFinOpsFilterPopup() + { + if (_filterPopup == null) + { + _filterPopupContent = new ColumnFilterPopup(); + + _filterPopup = new Popup + { + Child = _filterPopupContent, + StaysOpen = false, + Placement = PlacementMode.Bottom, + AllowsTransparency = true + }; + } + } + + private void FinOpsFilter_Click(object sender, RoutedEventArgs e) + { + if (sender is not Button button || button.Tag is not string columnName) return; + + var dataGrid = TabHelpers.FindParent(button); + if (dataGrid == null) return; + + EnsureFinOpsFilterPopup(); + + // Rewire events — remove then add to avoid double-firing + _filterPopupContent!.FilterApplied -= FinOpsFilterPopup_Applied; + _filterPopupContent.FilterCleared -= FinOpsFilterPopup_Cleared; + _filterPopupContent.FilterApplied += FinOpsFilterPopup_Applied; + _filterPopupContent.FilterCleared += FinOpsFilterPopup_Cleared; + + _currentFilterGrid = dataGrid; + + if (!_gridFilters.ContainsKey(dataGrid)) + _gridFilters[dataGrid] = new Dictionary(); + + _gridFilters[dataGrid].TryGetValue(columnName, out var existing); + _filterPopupContent.Initialize(columnName, existing); + + _filterPopup!.PlacementTarget = button; + _filterPopup.IsOpen = true; + } + + private void FinOpsFilterPopup_Applied(object? sender, FilterAppliedEventArgs e) + { + if (_filterPopup != null) + _filterPopup.IsOpen = false; + + if (_currentFilterGrid == null) return; + + if (!_gridFilters.ContainsKey(_currentFilterGrid)) + _gridFilters[_currentFilterGrid] = new Dictionary(); + + if (e.FilterState.IsActive) + { + _gridFilters[_currentFilterGrid][e.FilterState.ColumnName] = e.FilterState; + } + else + { + _gridFilters[_currentFilterGrid].Remove(e.FilterState.ColumnName); + } + + ApplyFinOpsFilters(_currentFilterGrid); + UpdateFinOpsFilterButtonStyles(_currentFilterGrid); + } + + private void FinOpsFilterPopup_Cleared(object? sender, EventArgs e) + { + if (_filterPopup != null) + _filterPopup.IsOpen = false; + } + + private void ApplyFinOpsFilters(DataGrid dataGrid) + { + // Capture unfiltered data on first filter application + if (!_gridUnfilteredData.TryGetValue(dataGrid, out var cached) || cached == null) + { + cached = dataGrid.ItemsSource; + _gridUnfilteredData[dataGrid] = cached; + } + + var unfilteredData = cached; + if (unfilteredData == null) return; + + if (!_gridFilters.TryGetValue(dataGrid, out var filters) || filters.Count == 0) + { + dataGrid.ItemsSource = unfilteredData; + return; + } + + // Generic filtering: cast to IEnumerable, filter each item using reflection-based MatchesFilter + var sourceList = unfilteredData.Cast().ToList(); + var filteredData = sourceList.Where(item => + { + foreach (var filter in filters.Values) + { + if (filter.IsActive && !DataGridFilterService.MatchesFilter(item, filter)) + { + return false; + } + } + return true; + }).ToList(); + + dataGrid.ItemsSource = filteredData; + } + + /// + /// Updates filter button styles (gold when active, white when inactive) for any FinOps DataGrid. + /// + private void UpdateFinOpsFilterButtonStyles(DataGrid dataGrid) + { + if (!_gridFilters.TryGetValue(dataGrid, out var filters)) + filters = new Dictionary(); + + foreach (var column in dataGrid.Columns) + { + if (column.Header is StackPanel headerPanel) + { + var filterButton = headerPanel.Children.OfType