diff --git a/.claude-plugin/plugin.json b/.claude-plugin/plugin.json index 5b5dbfe31..f6cde1b78 100644 --- a/.claude-plugin/plugin.json +++ b/.claude-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "agentic-qe", - "version": "3.14.2", + "version": "3.14.3", "description": "Agentic Quality Engineering — AI-powered QE platform with 60 specialized agents, 75+ skills, sublinear coverage analysis, ReasoningBank pattern learning, and deep MCP integration for Claude Code and 11 coding agent platforms", "author": { "name": "Agentic QE Team", diff --git a/.claude/skills/skills-manifest.json b/.claude/skills/skills-manifest.json index 79efdadeb..c7575cf9d 100644 --- a/.claude/skills/skills-manifest.json +++ b/.claude/skills/skills-manifest.json @@ -940,7 +940,7 @@ }, "metadata": { "generatedBy": "Agentic QE Fleet", - "fleetVersion": "3.14.2", + "fleetVersion": "3.14.3", "manifestVersion": "1.4.0", "lastUpdated": "2026-04-13T00:00:00.000Z", "contributors": [ diff --git a/CHANGELOG.md b/CHANGELOG.md index 0e75ebcd8..836194f87 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -5,6 +5,78 @@ All notable changes to the Agentic QE project will be documented in this file. The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/), and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html). +## [3.14.3] - 2026-09-22 + +This patch makes test, security, coherence, and quality-gate results honest +about failure: broken test runs, incomplete security scans, and unmeasured +quality gates no longer report as passing. It adds Vitest 5 support to every +test-runner integration and stops the CI pipeline itself from masking failures. + +### Added + +- Supported Vitest 5 in every test-runner integration (`aqe test execute`, the + `test_execute_parallel` MCP tool, retry, flaky detection, and scheduled + runs). Vitest 5 writes its JSON report to a file instead of stdout; runners + now request an explicit `--outputFile` and read it back, which behaves + identically on Vitest 4 ([#700]). +- Attached per-file and per-engine execution receipts to security scans. An + unreadable file, an unsupported language, or a missing or failed Semgrep is + now reported as `partial` or `unavailable` in the CLI, the + `security_scan_comprehensive` MCP tool, and saved JSON, SARIF, and Markdown + reports, instead of being mistaken for a clean result ([#705], refs [#694]). + +### Fixed + +- Reported broken test runs as execution errors rather than passes. A suite + that fails to import, a crashing teardown hook, an unhandled rejection, or an + empty suite now yields CLI exit 1 and an MCP `success: false`. Ordinary + assertion failures, skips, and todos keep their existing behavior, and glob + test paths are rejected explicitly instead of counted as a zero-test success + ([#699]). +- Made the registered `quality_assess({ runGate: true })` MCP tool run the same + fail-closed, timestamped evidence evaluator as `aqe quality --gate`, so a + static score can no longer approve a build with failing tests or coverage + and the session cache cannot replay a stale approval ([#702]). +- Made the coherence adapter actually reach the native sheaf engine for + connected graphs by serializing the restriction maps and dimensions it + requires, reading obstruction records by their real field names, and + rebuilding position-based edge indices after node removal; graphs beyond the + native cell budget fall back explicitly ([#703]). +- Stopped counting a guidance lookup as a successful experience reuse. + Application counts and token-savings statistics are recorded only when a + caller reports an outcome, so learning confidence is no longer inflated by + evidence that never happened ([#698]). +- Kept a judge response whose "unmet" list references no recognized checklist + item inconclusive instead of treating it as full coverage ([#696]), and made + the frontier-judge adapter use the requested model or return inconclusive + rather than silently routing to a cheaper one ([#697]). +- Fixed CI so the coherence job really executes its tests (an invalid Vitest + flag hidden by `continue-on-error` had prevented every run), runner exit + codes and timeouts propagate instead of being masked as success, and the + coverage threshold step reads a summary that actually exists ([#704]). Fork + pull requests no longer fail on the report-comment step ([#701]). + +### Changed + +- Upgraded the development toolchain to Vitest 5.0.1 with a matching + `@vitest/coverage-v8`, raised the `adm-zip` floor to 0.6.1, updated + `js-yaml` to 4.3.2, and refreshed `@vitest/mocker` in the embedded Ruflo + trees ([#687], [#700], [#708]). + +[#687]: https://github.com/proffesor-for-testing/agentic-qe/pull/687 +[#694]: https://github.com/proffesor-for-testing/agentic-qe/issues/694 +[#696]: https://github.com/proffesor-for-testing/agentic-qe/pull/696 +[#697]: https://github.com/proffesor-for-testing/agentic-qe/pull/697 +[#698]: https://github.com/proffesor-for-testing/agentic-qe/pull/698 +[#699]: https://github.com/proffesor-for-testing/agentic-qe/pull/699 +[#700]: https://github.com/proffesor-for-testing/agentic-qe/pull/700 +[#701]: https://github.com/proffesor-for-testing/agentic-qe/pull/701 +[#702]: https://github.com/proffesor-for-testing/agentic-qe/pull/702 +[#703]: https://github.com/proffesor-for-testing/agentic-qe/pull/703 +[#704]: https://github.com/proffesor-for-testing/agentic-qe/pull/704 +[#705]: https://github.com/proffesor-for-testing/agentic-qe/pull/705 +[#708]: https://github.com/proffesor-for-testing/agentic-qe/pull/708 + ## [3.14.2] - 2026-09-13 This patch keeps the quality daemon alive in foreground mode, publishes the diff --git a/assets/skills/skills-manifest.json b/assets/skills/skills-manifest.json index 79efdadeb..c7575cf9d 100644 --- a/assets/skills/skills-manifest.json +++ b/assets/skills/skills-manifest.json @@ -940,7 +940,7 @@ }, "metadata": { "generatedBy": "Agentic QE Fleet", - "fleetVersion": "3.14.2", + "fleetVersion": "3.14.3", "manifestVersion": "1.4.0", "lastUpdated": "2026-04-13T00:00:00.000Z", "contributors": [ diff --git a/docs/releases/README.md b/docs/releases/README.md index 95eeeb06a..982684e30 100644 --- a/docs/releases/README.md +++ b/docs/releases/README.md @@ -4,6 +4,7 @@ All Agentic QE release notes organized by version. | Version | Date | Highlights | |---------|------|------------| +| [v3.14.3](v3.14.3.md) | 2026-09-22 | Honest test, security, coherence, and quality-gate failures; Vitest 5 support. | | [v3.14.2](v3.14.2.md) | 2026-09-13 | Reliable foreground daemon lifetime, conservative MCP safety metadata, and patched Hono/Sharp dependencies. | | [v3.14.1](v3.14.1.md) | 2026-09-07 | Node 22+ support and stronger verification, recovery, and learning evidence. | | [v3.14.0](v3.14.0.md) | 2026-08-30 | Qualified learning evidence, embedding provenance, fail-closed RVF recovery, and packaged QE Court. | diff --git a/docs/releases/v3.14.3.md b/docs/releases/v3.14.3.md new file mode 100644 index 000000000..4f231ccf6 --- /dev/null +++ b/docs/releases/v3.14.3.md @@ -0,0 +1,84 @@ +# v3.14.3 Release Notes + +**Release Date:** 2026-09-22 + +## Highlights + +Agentic QE 3.14.3 makes test, security, coherence, and quality-gate results +honest about failure. Broken test runs, incomplete security scans, and +unmeasured quality gates no longer report as passing; every test-runner +integration supports Vitest 5; and the CI pipeline stops masking its own +failures. Most of this release was contributed by [@rudycelekli](https://github.com/rudycelekli). + +## Test execution + +- A broken test run is reported as an execution error, not a pass. A suite that + fails to import, a crashing `afterAll`, an unhandled rejection, or an empty + suite now gives `aqe test execute` exit code 1 and the `test_execute_parallel` + MCP tool `success: false`. Ordinary assertion failures, skips, and todos keep + their existing behavior. +- Glob patterns in `testFiles` are rejected with a clear message. They were + already silently rejected before; they just looked like a zero-test success. +- Vitest 5 is supported everywhere Agentic QE spawns a runner. Vitest 5 writes + its JSON report to a file instead of stdout, so the executor, MCP task + handler, retry handler, flaky detector, and scheduled runs now request an + explicit `--outputFile` and read it back. Vitest 4 behaves identically under + the same flag. + +## Security scanning + +- Every scan carries per-file and per-engine execution receipts. An unreadable + file, an unsupported language, or a missing or failed Semgrep engine is + reported as `partial` or `unavailable` in the CLI, the + `security_scan_comprehensive` MCP tool, and the saved JSON, SARIF, and + Markdown reports. +- `rulesApplied` now counts the unique patterns actually evaluated, `--dast` + without a target reports `not-run` with a non-zero exit, and SARIF marks + legacy receipt-less results as not executed successfully. See + `docs/security/scan-execution-evidence.md` for the schema. + +## Quality gates and coherence + +- The registered `quality_assess({ runGate: true })` MCP tool runs the same + fail-closed, timestamped evidence evaluator as `aqe quality --gate`. Missing, + malformed, or stale evidence is rejected, all seven checks must pass, and the + session cache never replays a gate result. +- The coherence adapter now reaches the native sheaf engine for connected + graphs. Edges carry the restriction maps and dimensions the engine requires, + obstruction records are read by their real field names, and edge indices are + rebuilt after node removal. Graphs beyond the native cell budget fall back + explicitly. + +## Learning and verification + +- Looking up past experience guidance no longer counts as a successful reuse. + Application counts and token-savings statistics are recorded only when a + caller reports an outcome. +- A judge response whose "unmet" list references no recognized checklist item + stays inconclusive, and the frontier-judge adapter uses the requested model or + returns inconclusive instead of silently routing to a cheaper one. + +## CI pipeline + +- The coherence job really executes its tests. An invalid Vitest flag hidden by + `continue-on-error` had prevented every run since the Vitest 4 upgrade. +- Runner exit codes and timeouts propagate instead of being masked as success, + and the coverage threshold step reads a summary file that actually exists. +- Fork pull requests no longer fail on the report-comment step; reports and + gates still run for contributors. + +## Dependencies + +- Development toolchain moves to Vitest 5.0.1 with a matching + `@vitest/coverage-v8`. `adm-zip` floor raised to 0.6.1, `js-yaml` updated to + 4.3.2, and `@vitest/mocker` refreshed in the embedded Ruflo trees. + +## Upgrade notes + +No database migration is required. Scripts that relied on `aqe security --dast` +exiting 0 without a target, or on `rulesApplied` equalling the sum of declared +rule counts, should be updated. If you run `aqe test execute` against a project +on Vitest 5, this is the first release that reads its results correctly. + +Production packages are published only from merged `main` through the GitHub +release workflow and its pre-publish corpus gate. diff --git a/package-lock.json b/package-lock.json index b145f9956..6ec7886ad 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "agentic-qe", - "version": "3.14.2", + "version": "3.14.3", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "agentic-qe", - "version": "3.14.2", + "version": "3.14.3", "hasInstallScript": true, "license": "MIT", "dependencies": { diff --git a/package.json b/package.json index a9e9d59af..fb8639f68 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "agentic-qe", - "version": "3.14.2", + "version": "3.14.3", "description": "Agentic Quality Engineering V3 - Domain-Driven Design Architecture with 13 Bounded Contexts, O(log n) coverage analysis, ReasoningBank learning, 60 specialized QE agents, mathematical Coherence verification, deep Claude Flow integration", "type": "module", "main": "./dist/index.js",