diff --git a/.github/scripts/run-native-fullsend-evals.sh b/.github/scripts/run-native-fullsend-evals.sh new file mode 100644 index 000000000..95efce2c0 --- /dev/null +++ b/.github/scripts/run-native-fullsend-evals.sh @@ -0,0 +1,131 @@ +#!/usr/bin/env bash +# Trusted CI setup/run wrapper; PR plugin files are only sandbox test subjects. +set -euo pipefail + +cache="${RUNNER_TEMP:?}/tc6726-deps" +upstream="${GITHUB_WORKSPACE:?}/upstream-fullsend" +test "$(git -C "$upstream" rev-parse HEAD)" = d5f36921ac754705619f38c637ef692873809fbc + +reviewed="${GITHUB_WORKSPACE}/native-eval-source" +: "${NATIVE_EVAL_SOURCE_SHA:?Reviewed eval source pin is required}" +if [ "$(git -C "$reviewed" rev-parse HEAD)" != "$NATIVE_EVAL_SOURCE_SHA" ]; then + echo '::error::Reviewed native eval source changed' + exit 1 +fi +runner="$reviewed/evals/fullsend/run.py" + +case "${1:?setup or run}" in + setup) + python3.12 "$runner" setup --cache "$cache" + source "$upstream/.github/scripts/openshell-version.sh" + mkdir -p "$HOME/.config/openshell" + echo 'OPENSHELL_BIND_ADDRESS=0.0.0.0' > "$HOME/.config/openshell/gateway.env" + cat > "$HOME/.config/openshell/gateway.toml" < "$RUNNER_TEMP/tc6726-podman.log" 2>&1 & + for _i in $(seq 1 30); do + if [ -S "$socket_path" ] && podman --url "unix://${socket_path}" info >/dev/null 2>&1; then + break + fi + sleep 1 + done + test -S "$socket_path" + bash "$upstream/.github/scripts/install-openshell.sh" + export PATH="$cache/venv/bin:$PATH" + python3.12 "$runner" preflight --cache "$cache" + ;; + run) + : "${GOOGLE_APPLICATION_CREDENTIALS:?WIF host ADC is required}" + : "${ANTHROPIC_VERTEX_PROJECT_ID:?Vertex project is required}" + : "${CLOUD_ML_REGION:?Vertex region is required}" + : "${TC6726_HEAD_SHA:?Exact source is required}" + HOST_GOOGLE_APPLICATION_CREDENTIALS="$GOOGLE_APPLICATION_CREDENTIALS" + # The native synthetic job requires external-account WIF; no key fallback. + test "$(jq -r '.type' "$HOST_GOOGLE_APPLICATION_CREDENTIALS")" = external_account + for credential_var in GOOGLE_APPLICATION_CREDENTIALS GOOGLE_GHA_CREDS_PATH CLOUDSDK_AUTH_CREDENTIAL_FILE_OVERRIDE; do + credential_value="${!credential_var:-}" + if [ -n "$credential_value" ]; then echo "::add-mask::$credential_value"; fi + done + # Reuse upstream conversion without replacing the scoring process's ADC. + # Parse only known outputs as data; never source an environment file. + prepared_env="$RUNNER_TEMP/tc6726-sandbox.env" + GITHUB_ENV="$prepared_env" bash "$upstream/internal/scaffold/fullsend-repo/scripts/prepare-sandbox-credentials.sh" + while IFS= read -r credential_line || [ -n "$credential_line" ]; do + if [ -z "$credential_line" ]; then continue; fi + if [[ "$credential_line" == *'<<'* && "${credential_line%%<<*}" != *'='* ]]; then + credential_name="${credential_line%%<<*}" + credential_delimiter="${credential_line#*<<}" + credential_value='' + credential_separator='' + credential_closed=false + while IFS= read -r credential_line || [ -n "$credential_line" ]; do + if [ "$credential_line" = "$credential_delimiter" ]; then + credential_closed=true + break + fi + credential_value+="${credential_separator}${credential_line}" + credential_separator=$'\n' + done + if [ "$credential_closed" != true ]; then + echo '::error::Unterminated upstream credential value' + exit 1 + fi + elif [[ "$credential_line" == *'='* ]]; then + credential_name="${credential_line%%=*}" + credential_value="${credential_line#*=}" + else + continue + fi + case "$credential_name" in + GOOGLE_APPLICATION_CREDENTIALS|GCP_OIDC_TOKEN_FILE|FULLSEND_GCP_OIDC_URL|FULLSEND_GCP_OIDC_AUTH_FILE) ;; + *) echo '::error::Unexpected upstream credential output'; exit 1 ;; + esac + # Escape multiline mask data so its lines cannot become workflow commands. + credential_mask="${credential_value//%/%25}" + credential_mask="${credential_mask//$'\r'/%0D}" + credential_mask="${credential_mask//$'\n'/%0A}" + printf '::add-mask::%s\n' "$credential_mask" + if [[ "$credential_value" == *$'\n'* ]]; then + while IFS= read -r credential_mask_line || [ -n "$credential_mask_line" ]; do + if [ -z "$credential_mask_line" ]; then continue; fi + credential_mask_line="${credential_mask_line//%/%25}" + credential_mask_line="${credential_mask_line//$'\r'/%0D}" + printf '::add-mask::%s\n' "$credential_mask_line" + done <<< "$credential_value" + fi + if [ "$credential_name" = GOOGLE_APPLICATION_CREDENTIALS ]; then + export TC6726_SANDBOX_CREDENTIALS="$credential_value" + else + export "$credential_name=$credential_value" + fi + done < "$prepared_env" + : "${TC6726_SANDBOX_CREDENTIALS:?Prepared sandbox ADC is required}" + : "${GCP_OIDC_TOKEN_FILE:?Native OIDC mount is required}" + : "${FULLSEND_GCP_OIDC_URL:?Native OIDC refresh is required}" + : "${FULLSEND_GCP_OIDC_AUTH_FILE:?Native OIDC refresh authentication is required}" + export GOOGLE_APPLICATION_CREDENTIALS="$HOST_GOOGLE_APPLICATION_CREDENTIALS" + export PATH="$cache/venv/bin:$PATH" + # Native stdout/stderr/transcripts can contain credential paths or arbitrary + # PR-generated bytes. Keep raw logs private; only export allowlisted results. + status=0 + python3.12 "$runner" run --cache "$cache" \ + --plugin-root "$GITHUB_WORKSPACE/pr-head/plugins/sdlc-workflow" \ + --output "$RUNNER_TEMP/tc6726-private" --report-dir "$RUNNER_TEMP/tc6726-safe" \ + > "$RUNNER_TEMP/tc6726-private-run.log" 2>&1 || status=$? + echo "Native Fullsend execution/scoring finished (exit $status); safe source-bound result only." + exit "$status" + ;; + *) echo '::error::Expected setup or run'; exit 1 ;; +esac diff --git a/.github/workflows/eval-pr-run.yml b/.github/workflows/eval-pr-run.yml index 068e6b143..7b04b70a7 100644 --- a/.github/workflows/eval-pr-run.yml +++ b/.github/workflows/eval-pr-run.yml @@ -18,18 +18,28 @@ # See docs/specs/2026-05-07-eval-pr-fork-dispatch-design.md for full design. name: Eval PR Run +run-name: Eval PR Run ${{ github.event.workflow_run.head_sha }} on: workflow_run: workflows: ["Eval PR"] types: [completed] +# Serialize the complete ordinary/native publication sequence for one PR head. +concurrency: + group: eval-pr-run-${{ github.event.workflow_run.head_sha }} + cancel-in-progress: false + permissions: contents: read pull-requests: write actions: read statuses: write +env: + # Reviewed PR299 suite; normal activation switches this to trusted github.sha. + NATIVE_EVAL_SOURCE_SHA: b46b2bf647be8451b6aedda58dc8e51448f0eab6 + jobs: discover: name: Discover PR Context @@ -40,8 +50,49 @@ jobs: pr_number: ${{ steps.pr.outputs.pr_number }} author: ${{ steps.pr.outputs.author }} trusted: ${{ steps.pr.outputs.trusted }} + native: ${{ steps.discover.outputs.native }} + head_sha: ${{ steps.pr.outputs.head_sha }} + merge_sha: ${{ steps.pr.outputs.merge_sha }} + base_sha: ${{ steps.pr.outputs.base_sha }} + source_repo: ${{ steps.pr.outputs.source_repo }} + source_branch: ${{ steps.pr.outputs.source_branch }} steps: + - &publication-guard + name: Check latest run before publishing + id: publication + if: always() + env: + HEAD_SHA: ${{ github.event.workflow_run.head_sha }} + uses: actions/github-script@v9 + with: + script: &latest-run-check | + core.setOutput('latest', 'false'); + core.setOutput('error', 'false'); + try { + const {data: current} = await github.rest.actions.getWorkflowRun({ + ...context.repo, run_id: context.runId + }); + // workflow_run's own head_sha is main, not the tested PR head. + // The API run title binds this consumer to the triggering PR head. + const runs = await github.paginate(github.rest.actions.listWorkflowRuns, { + ...context.repo, workflow_id: current.workflow_id, event: 'workflow_run', + created: `>=${current.created_at}`, per_page: 100 + }); + const matching = runs.filter(r => r.display_title === `Eval PR Run ${process.env.HEAD_SHA}`); + const latest = matching.sort((a,b) => b.run_number - a.run_number)[0]; + // The list can lag behind getWorkflowRun for a newly started run. + if ((latest && latest.run_number > current.run_number) || current.run_attempt !== context.runAttempt) { + core.info('Superseded run or attempt; skipping publication'); + return; + } + core.setOutput('latest', 'true'); + } catch (error) { + core.setOutput('error', 'true'); + core.setFailed('Cannot determine latest eval run; refusing publication'); + } + - name: Set pending commit status + if: steps.publication.outputs.latest == 'true' uses: actions/github-script@v9 with: script: | @@ -77,6 +128,39 @@ jobs: core.setFailed(`No open PR targeting main found for commit ${headSha}`); return; } + let { data: current } = await github.rest.pulls.get({ + ...context.repo, pull_number: pr.number + }); + // GitHub can return null while computing the merge after a push. + for (let attempt = 1; current.merge_commit_sha === null && attempt < 3; attempt++) { + await new Promise(resolve => setTimeout(resolve, 2000)); + ({ data: current } = await github.rest.pulls.get({ + ...context.repo, pull_number: pr.number + })); + } + const shaPattern = /^[0-9a-f]{40}$/; + if (current.state !== 'open' || current.base?.ref !== 'main' || + current.head?.sha !== headSha || + current.head?.repo?.full_name !== context.payload.workflow_run.head_repository?.full_name || + !shaPattern.test(current.base?.sha || '') || !shaPattern.test(current.merge_commit_sha || '')) { + core.setFailed('PR identity/revision changed or merge source unavailable'); + return; + } + // Verify the exact merge commit is constructed from this API-associated + // head and base. No floating pull ref reaches a credentialed job. + const { data: merge } = await github.rest.git.getCommit({ + ...context.repo, commit_sha: current.merge_commit_sha + }); + if (merge.parents?.length !== 2 || merge.parents[0]?.sha !== current.base.sha || + merge.parents[1]?.sha !== headSha) { + core.setFailed('Merge source is not the event-associated head/base'); + return; + } + core.setOutput('head_sha', headSha); + core.setOutput('merge_sha', current.merge_commit_sha); + core.setOutput('base_sha', current.base.sha); + core.setOutput('source_repo', current.head.repo.full_name); + core.setOutput('source_branch', current.head.ref); core.setOutput('pr_number', pr.number.toString()); core.setOutput('author', pr.user.login); @@ -97,10 +181,14 @@ jobs: - name: Discover changed skills id: discover + env: + PR_NUMBER: ${{ steps.pr.outputs.pr_number }} + SOURCE_BRANCH: ${{ steps.pr.outputs.source_branch }} + MERGE_SHA: ${{ steps.pr.outputs.merge_sha }} uses: actions/github-script@v9 with: script: | - const prNumber = parseInt('${{ steps.pr.outputs.pr_number }}'); + const prNumber = Number(process.env.PR_NUMBER); const files = await github.paginate(github.rest.pulls.listFiles, { owner: context.repo.owner, repo: context.repo.repo, @@ -118,7 +206,7 @@ jobs: for (const f of files) { const match = f.filename.match(skillPattern) || f.filename.match(evalPattern); - if (match) candidates.add(match[1]); + if (match && match[1] !== 'fullsend') candidates.add(match[1]); } const confirmed = []; @@ -132,7 +220,7 @@ jobs: owner: context.repo.owner, repo: context.repo.repo, path: `evals/${skill}/evals.json`, - ref: `refs/pull/${prNumber}/merge` + ref: process.env.MERGE_SHA }); confirmed.push(skill); } catch { @@ -141,11 +229,31 @@ jobs: } const skills = confirmed.join(','); + // TC-6726 bootstrap restriction: remove only this identity condition + // during the separately reviewed PR299 activation delivery. + const bootstrap = prNumber === 299 && process.env.SOURCE_BRANCH === 'verify-pr-fullsend'; + const nativeRelevant = files.some(f => + /^evals\/fullsend\//.test(f.filename) || + /^evals\/triage-security\//.test(f.filename) || + /^plugins\/sdlc-workflow\/skills\/triage-security\//.test(f.filename) || + /^plugins\/sdlc-workflow\/(policies\/triage-security\.yaml|providers\/vertex-ai\.yaml|profiles\/fullsend-vertex-ai\.yaml|env\/gcp-vertex\.env|schemas\/triage-security-(input|result)\.schema\.json|scripts\/(validate-output-schema\.sh|strip_extra_properties\.py|test_(fullsend_gate_eval|native_fullsend_eval_ci)\.py))$/.test(f.filename) || + /^\.github\/(workflows\/eval-pr(-run)?\.yml|scripts\/run-native-fullsend-evals\.sh)$/.test(f.filename) + ); + core.setOutput('native', String(bootstrap && nativeRelevant)); core.setOutput('skills', skills); console.log(`Discovered changed skills with evals: ${skills || 'none'}`); + - name: Recheck latest run before approval status + id: gate-publication + if: always() + env: + HEAD_SHA: ${{ github.event.workflow_run.head_sha }} + uses: actions/github-script@v9 + with: + script: *latest-run-check + - name: Update status for approval gate - if: steps.pr.outputs.trusted != 'true' && steps.pr.outputs.pr_number != '' + if: steps.gate-publication.outputs.latest == 'true' && steps.pr.outputs.trusted != 'true' && steps.pr.outputs.pr_number != '' uses: actions/github-script@v9 with: script: | @@ -160,16 +268,19 @@ jobs: }); gate: - name: Approval Gate + name: Approval Gate for ${{ needs.discover.outputs.head_sha }} needs: discover if: >- needs.discover.result == 'success' && needs.discover.outputs.trusted != 'true' && - needs.discover.outputs.skills != '' + (needs.discover.outputs.skills != '' || needs.discover.outputs.native == 'true') runs-on: ubuntu-latest environment: eval-protected steps: - - run: echo "Approved by reviewer" + - env: + APPROVED_HEAD: ${{ needs.discover.outputs.head_sha }} + APPROVED_MERGE: ${{ needs.discover.outputs.merge_sha }} + run: echo "Approved revision $APPROVED_HEAD / merge $APPROVED_MERGE" run-evals: name: Run PR Evals @@ -183,15 +294,20 @@ jobs: permissions: contents: read pull-requests: write + actions: read id-token: write steps: - name: Checkout base branch uses: actions/checkout@v7 + with: + ref: ${{ github.sha }} + persist-credentials: false - name: Checkout PR merge commit into subdirectory uses: actions/checkout@v7 with: - ref: refs/pull/${{ needs.discover.outputs.pr_number }}/merge + ref: ${{ needs.discover.outputs.merge_sha }} + persist-credentials: false path: pr-head fetch-depth: 0 allow-unsafe-pr-checkout: true @@ -279,10 +395,15 @@ jobs: exit 1 fi + - *publication-guard + - name: Post eval results review + if: steps.publication.outputs.latest == 'true' env: SKILLS_CSV: ${{ needs.discover.outputs.skills }} PR_NUMBER: ${{ needs.discover.outputs.pr_number }} + HEAD_SHA: ${{ needs.discover.outputs.head_sha }} + MERGE_SHA: ${{ needs.discover.outputs.merge_sha }} uses: actions/github-script@v9 with: script: | @@ -290,7 +411,7 @@ jobs: const skills = process.env.SKILLS_CSV.split(',').filter(Boolean); const prNumber = parseInt(process.env.PR_NUMBER); - let body = '## Eval Results\n\n'; + let body = `## Eval Results\n\nSource head: ${process.env.HEAD_SHA}\nMerge: ${process.env.MERGE_SHA}\n\n`; for (const skill of skills) { const summaryPath = `/tmp/${skill}-eval-pr/summary.md`; if (fs.existsSync(summaryPath)) { @@ -300,14 +421,14 @@ jobs: } } - const { data: reviews } = await github.rest.pulls.listReviews({ + const reviews = await github.paginate(github.rest.pulls.listReviews, { owner: context.repo.owner, repo: context.repo.repo, pull_number: prNumber }); const marker = '## Eval Results'; const existing = reviews.find(r => - r.user?.login === 'github-actions[bot]' && r.body?.startsWith(marker) + r.user?.login === 'github-actions[bot]' && r.commit_id === process.env.HEAD_SHA && r.body?.startsWith(marker) ); if (existing) { @@ -324,21 +445,199 @@ jobs: repo: context.repo.repo, pull_number: prNumber, event: 'COMMENT', + commit_id: process.env.HEAD_SHA, body }); } + run-native-evals: + name: Run Native Fullsend Evals + needs: [discover, gate] + if: >- + !cancelled() && + needs.discover.result == 'success' && + needs.discover.outputs.native == 'true' && + (needs.discover.outputs.trusted == 'true' || needs.gate.result == 'success') + runs-on: ubuntu-24.04 + timeout-minutes: 90 + permissions: + contents: read + pull-requests: read + id-token: write + steps: + - name: Checkout trusted base + uses: actions/checkout@v7 + with: + ref: ${{ github.sha }} + persist-credentials: false + + - name: Checkout exact tested merge + uses: actions/checkout@v7 + with: + ref: ${{ needs.discover.outputs.merge_sha }} + path: pr-head + persist-credentials: false + allow-unsafe-pr-checkout: true + + - name: Checkout reviewed native eval source + uses: actions/checkout@v7 + with: + repository: ${{ github.repository }} + ref: ${{ env.NATIVE_EVAL_SOURCE_SHA }} + path: native-eval-source + persist-credentials: false + + - name: Checkout pinned Fullsend infrastructure + uses: actions/checkout@v7 + with: + repository: fullsend-ai/fullsend + ref: d5f36921ac754705619f38c637ef692873809fbc + path: upstream-fullsend + persist-credentials: false + + - name: Install Python3.12 + uses: actions/setup-python@v6 + with: + python-version: '3.12' + + - name: Install trusted native dependencies + run: bash .github/scripts/run-native-fullsend-evals.sh setup + + - name: Recheck approved revision before WIF + id: revision + env: + PR_NUMBER: ${{ needs.discover.outputs.pr_number }} + HEAD_SHA: ${{ needs.discover.outputs.head_sha }} + MERGE_SHA: ${{ needs.discover.outputs.merge_sha }} + BASE_SHA: ${{ needs.discover.outputs.base_sha }} + SOURCE_REPO: ${{ needs.discover.outputs.source_repo }} + SOURCE_BRANCH: ${{ needs.discover.outputs.source_branch }} + uses: actions/github-script@v9 + with: + script: | + const { data: pr } = await github.rest.pulls.get({ + ...context.repo, pull_number: Number(process.env.PR_NUMBER) + }); + if (pr.state !== 'open' || pr.number !== 299 || pr.base?.ref !== 'main' || + pr.head?.ref !== 'verify-pr-fullsend' || + pr.head?.sha !== process.env.HEAD_SHA || pr.base?.sha !== process.env.BASE_SHA || + pr.merge_commit_sha !== process.env.MERGE_SHA || + pr.head?.repo?.full_name !== process.env.SOURCE_REPO || + pr.head?.ref !== process.env.SOURCE_BRANCH) { + core.setFailed('Approved PR revision changed; require a new source-bound run'); + } + + - name: Pre-mask host credential path + run: echo "::add-mask::${GITHUB_WORKSPACE}/gha-creds-" + + - name: Authenticate native inference with existing Fullsend WIF + uses: google-github-actions/auth@7c6bc770dae815cd3e89ee6cdf493a5fab2cc093 + with: + workload_identity_provider: ${{ secrets.FULLSEND_GCP_WIF_PROVIDER }} + project_id: ${{ secrets.FULLSEND_GCP_PROJECT_ID }} + + - name: Run trusted native suite against sandbox plugin + env: + ANTHROPIC_VERTEX_PROJECT_ID: ${{ secrets.FULLSEND_GCP_PROJECT_ID }} + GOOGLE_CLOUD_PROJECT: ${{ secrets.FULLSEND_GCP_PROJECT_ID }} + CLOUD_ML_REGION: ${{ vars.FULLSEND_GCP_REGION }} + CLAUDE_CODE_USE_VERTEX: '1' + TC6726_PR_NUMBER: ${{ needs.discover.outputs.pr_number }} + TC6726_HEAD_SHA: ${{ needs.discover.outputs.head_sha }} + TC6726_MERGE_SHA: ${{ needs.discover.outputs.merge_sha }} + TC6726_BASE_SHA: ${{ needs.discover.outputs.base_sha }} + TC6726_TRUSTED_SHA: ${{ github.sha }} + TC6726_EVAL_SOURCE_SHA: ${{ env.NATIVE_EVAL_SOURCE_SHA }} + run: bash .github/scripts/run-native-fullsend-evals.sh run + + - name: Upload only allowlisted source-bound native result + if: always() && steps.revision.outcome == 'success' + uses: actions/upload-artifact@v7 + with: + name: native-fullsend-result + path: ${{ runner.temp }}/tc6726-safe/native-result.json + if-no-files-found: error + retention-days: 14 + report-status: name: Report Status - needs: [discover, gate, run-evals] + needs: [discover, gate, run-evals, run-native-evals] if: always() runs-on: ubuntu-latest steps: + - *publication-guard + + - name: Download safe native result + continue-on-error: true + if: needs.discover.outputs.native == 'true' && needs.run-native-evals.result != 'skipped' + uses: actions/download-artifact@v8 + with: + name: native-fullsend-result + path: native-report + + - name: Publish native result alongside ordinary review + id: native-report + if: always() && steps.publication.outputs.latest == 'true' && needs.discover.outputs.native == 'true' + env: + PR_NUMBER: ${{ needs.discover.outputs.pr_number }} + HEAD_SHA: ${{ needs.discover.outputs.head_sha }} + MERGE_SHA: ${{ needs.discover.outputs.merge_sha }} + BASE_SHA: ${{ needs.discover.outputs.base_sha }} + TRUSTED_SHA: ${{ github.sha }} + EVAL_SOURCE_SHA: ${{ env.NATIVE_EVAL_SOURCE_SHA }} + NATIVE_RESULT: ${{ needs.run-native-evals.result }} + uses: actions/github-script@v9 + with: + script: | + const fs = require('fs'); + const expected = {pr_number: Number(process.env.PR_NUMBER), head_sha: process.env.HEAD_SHA, + merge_sha: process.env.MERGE_SHA, base_sha: process.env.BASE_SHA, trusted_sha: process.env.TRUSTED_SHA, eval_source_sha: process.env.EVAL_SOURCE_SHA}; + let body = `## Native Fullsend Eval Results\n\nHead: ${expected.head_sha}\nMerge: ${expected.merge_sha}\nTrusted workflow: ${expected.trusted_sha}\nReviewed eval source: ${expected.eval_source_sha}\n\n`; + let valid = false; + if (fs.existsSync('native-report/native-result.json')) { + const report = JSON.parse(fs.readFileSync('native-report/native-result.json', 'utf8')); + const counts = {'033-absent': 4, '034-empty': 5, '035-malformed': 5, '036-valid': 7}; + const bound = Object.entries(expected).every(([k,v]) => report.source?.[k] === v); + const complete = report.complete === true && report.total === 21 && + Object.keys(report.outcomes || {}).length === 4 && Object.entries(counts).every(([k,n]) => + Object.keys(report.outcomes?.[k] || {}).length === n && Array.from({length:n}, (_,i) => + report.outcomes?.[k]?.[`assertion_${i+1}`]).every(v => typeof v === 'boolean')); + if (!bound) { core.setFailed('Native report source mismatch'); return; } + const passed = Object.values(report.outcomes || {}).flatMap(v => Object.values(v)).filter(v => v === true).length; + valid = complete && passed === 21 && report.exit_code === 0; + body += `Native job: ${process.env.NATIVE_RESULT}; complete: ${complete}; ${passed}/21 passed.\n`; + } else { + body += 'No safe native result was produced; native execution/approval failed.\n'; + } + // Only constructed scalar/count data enters the review. No arbitrary + // rationale, transcript, credential path or PR-controlled Markdown. + const reviews = await github.paginate(github.rest.pulls.listReviews, { + ...context.repo, pull_number: expected.pr_number + }); + const marker = '## Native Fullsend Eval Results'; + const existing = reviews.find(r => + r.user?.login === 'github-actions[bot]' && r.commit_id === expected.head_sha && r.body?.startsWith(marker) + ); + if (existing) { + await github.rest.pulls.updateReview({...context.repo, pull_number: expected.pr_number, + review_id: existing.id, body}); + } else { + await github.rest.pulls.createReview({...context.repo, pull_number: expected.pr_number, + commit_id: expected.head_sha, event: 'COMMENT', body}); + } + if (!valid) core.setFailed('Native evidence incomplete, failed, or missing'); + - name: Set final commit status + if: always() && (steps.publication.outputs.latest == 'true' || steps.publication.outputs.error == 'true') env: + PUBLICATION_GUARD_ERROR: ${{ steps.publication.outputs.error }} DISCOVER_RESULT: ${{ needs.discover.result }} EVALS_RESULT: ${{ needs.run-evals.result }} GATE_RESULT: ${{ needs.gate.result }} + NATIVE_RESULT: ${{ needs.run-native-evals.result }} + NATIVE_REQUESTED: ${{ needs.discover.outputs.native }} + NATIVE_REPORT_RESULT: ${{ steps.native-report.outcome }} + SKILLS_CSV: ${{ needs.discover.outputs.skills }} uses: actions/github-script@v9 with: script: | @@ -346,23 +645,17 @@ jobs: const evalsResult = process.env.EVALS_RESULT; const gateResult = process.env.GATE_RESULT; - let state, description; - if (evalsResult === 'success') { - state = 'success'; - description = 'Eval run completed — results posted as PR review'; - } else if (discoverResult !== 'success') { - state = 'failure'; - description = 'Eval run failed — could not resolve PR context'; - } else if (gateResult === 'failure') { - state = 'failure'; - description = 'Approval was rejected in eval-protected environment'; - } else if (evalsResult === 'skipped') { - state = 'success'; - description = 'No evals to run for changed files'; - } else { - state = 'failure'; - description = 'Eval run failed — check workflow logs'; - } + const nativeRequested = process.env.NATIVE_REQUESTED === 'true'; + const ordinaryRequested = Boolean(process.env.SKILLS_CSV); + const nativeOk = !nativeRequested || (process.env.NATIVE_RESULT === 'success' && + process.env.NATIVE_REPORT_RESULT === 'success'); + const ordinaryOk = !ordinaryRequested || evalsResult === 'success'; + const publicationError = process.env.PUBLICATION_GUARD_ERROR === 'true'; + const state = !publicationError && discoverResult === 'success' && gateResult !== 'failure' && gateResult !== 'cancelled' && + nativeOk && ordinaryOk ? 'success' : 'failure'; + const description = publicationError ? 'Could not verify latest eval run; rerun required' : + state === 'success' ? 'Requested eval suites completed successfully' : + 'Eval execution, approval or native evidence failed'; await github.rest.repos.createCommitStatus({ owner: context.repo.owner, diff --git a/.github/workflows/eval-pr.yml b/.github/workflows/eval-pr.yml index cc676c384..cb0e5780c 100644 --- a/.github/workflows/eval-pr.yml +++ b/.github/workflows/eval-pr.yml @@ -12,6 +12,19 @@ on: paths: - 'plugins/sdlc-workflow/skills/**/*.md' - 'evals/**/evals.json' + - 'evals/fullsend/**' + - 'evals/triage-security/**' + - 'plugins/sdlc-workflow/policies/triage-security.yaml' + - 'plugins/sdlc-workflow/providers/vertex-ai.yaml' + - 'plugins/sdlc-workflow/profiles/fullsend-vertex-ai.yaml' + - 'plugins/sdlc-workflow/env/gcp-vertex.env' + - 'plugins/sdlc-workflow/schemas/triage-security-*.schema.json' + - 'plugins/sdlc-workflow/scripts/validate-output-schema.sh' + - 'plugins/sdlc-workflow/scripts/strip_extra_properties.py' + - 'plugins/sdlc-workflow/scripts/test_fullsend_gate_eval.py' + - 'plugins/sdlc-workflow/scripts/test_native_fullsend_eval_ci.py' + - '.github/scripts/run-native-fullsend-evals.sh' + - '.github/workflows/eval-pr-run.yml' - '.github/workflows/eval-pr.yml' jobs: diff --git a/plugins/sdlc-workflow/scripts/test_native_fullsend_eval_ci.py b/plugins/sdlc-workflow/scripts/test_native_fullsend_eval_ci.py new file mode 100644 index 000000000..cfe0db9f6 --- /dev/null +++ b/plugins/sdlc-workflow/scripts/test_native_fullsend_eval_ci.py @@ -0,0 +1,489 @@ +"""Trusted workflow contracts; no native suite or model execution lives here.""" + +import json +import os +from pathlib import Path +import subprocess + +import pytest +import yaml + +ROOT = Path(__file__).resolve().parents[3] + +def workflow(): + """Read the actual trusted workflow rather than a duplicate implementation.""" + return yaml.safe_load((ROOT / ".github/workflows/eval-pr-run.yml").read_text()) + + +def script_step(job, name): + """Select a workflow script by its human-readable step name.""" + return next(step for step in workflow()["jobs"][job]["steps"] if step["name"] == name) + + +def run_js(script, data, env=None): + """SYNTHETIC TEST DATA — double only GitHub API responses, execute real JS.""" + code = r''' +const data = JSON.parse(process.argv[1]); +const outputs = {}, statuses = [], errors = [], reviews = [], updates = []; +const storedReviews = data.existingReviews || []; +let prReads = 0; +const delays = []; +const setTimeout = (fn,ms)=>{delays.push(ms);fn();}; +const require = name => {if (name !== 'fs') throw Error('unexpected module'); + return {existsSync:()=>Boolean(data.report),readFileSync:()=>JSON.stringify(data.report)};}; +const core = {setOutput: (k,v) => outputs[k]=v, setFailed: x => errors.push(x), info:()=>{}}; +const context = {repo:{owner:'RHEcosystemAppEng',repo:'sdlc-plugins'}, + payload:{workflow_run:{head_sha:data.eventHead || 'a'.repeat(40), + head_repository:{full_name:'mrizzi/sdlc-plugins'}}}, serverUrl:'https://github.com',runId:1,runAttempt:1}; +const pr = {number:data.number || 299,state:'open',user:{login:'synthetic'}, + head:{sha:data.head || 'a'.repeat(40),ref:data.branch || 'verify-pr-fullsend',repo:{full_name:'mrizzi/sdlc-plugins'}}, + base:{sha:'b'.repeat(40),ref:data.base || 'main'},merge_commit_sha:'c'.repeat(40)}; +const github = {paginate:async (fn,args)=>{const response=await fn(args);return response.data || response;},rest:{ + actions:{getWorkflowRun:async()=>{ + if(data.apiError && data.apiError !== 'list') throw Error('API unavailable'); + return {data:{id:1,run_number:1,workflow_id:10,run_attempt:data.attempt || 1,created_at:'2026-10-06T07:00:00Z'}};}, + listWorkflowRuns:async()=> {if(data.apiError === 'list') throw Error('API unavailable'); + return (data.runs || []).map(r=>({...r, + display_title:`Eval PR Run ${r.other_head?'b'.repeat(40):process.env.HEAD_SHA}`}));}}, + pulls:{list:async()=>[pr],get:async()=>{ + const revision=(data.revisions || [])[Math.min(prReads,(data.revisions || []).length-1)] || {}; + prReads++; + return {data:{...pr,...revision}};}, + listReviews:async()=>({data:storedReviews}), + updateReview:async r=>{updates.push(r);storedReviews.find(s=>s.id===r.review_id).body=r.body;}, + createReview:async r=>{reviews.push(r);storedReviews.push({...r,id:storedReviews.length+1,user:{login:'github-actions[bot]'}});}, + listFiles:async()=> (data.paths||[]).map(filename=>({filename}))}, + git:{getCommit:async()=>({data:{parents:(data.parents||['b'.repeat(40),'a'.repeat(40)]).map(sha=>({sha}))}})}, + repos:{getCollaboratorPermissionLevel:async()=>({data:{permission:data.permission||'read'}}), + getContent:async()=>({data:{}}),createCommitStatus:async s=>statuses.push(s)}}}; +(async()=>{for(let i=0;i<(data.repeat || 1);i++){SCRIPT +}})().then(()=>process.stdout.write(JSON.stringify({outputs,statuses,errors,reviews,updates,prReads,delays}))) +.catch(e=>{process.stdout.write(JSON.stringify({outputs,statuses,errors:[...errors,e.message],reviews,updates}));}); +'''.replace("SCRIPT", script) + result = subprocess.run(["node", "-e", code, json.dumps(data)], + env=dict(os.environ, **(env or {})), capture_output=True, text=True, check=True) + return json.loads(result.stdout.splitlines()[-1]) + + +@pytest.mark.parametrize("permission,trusted", [("admin", "true"), ("write", "true"), ("read", "false")]) +def test_source_resolution_pins_api_associated_revision(permission, trusted): + """The triggering head, merge parents and base revision form one immutable source.""" + result = run_js(script_step("discover", "Resolve PR identity and check trust")["with"]["script"], + {"permission": permission}) + assert result["errors"] == [] + assert {key: result["outputs"].get(key) for key in ["head_sha", "merge_sha", "base_sha", "trusted"]} == { + "head_sha": "a" * 40, "merge_sha": "c" * 40, "base_sha": "b" * 40, "trusted": trusted} + + +@pytest.mark.parametrize("defect", [{"head": "d" * 40}, {"parents": ["b" * 40, "d" * 40]}, {"base": "other"}]) +def test_source_resolution_refuses_stale_or_unrelated_revisions(defect): + """A new head or unrelated merge cannot inherit the event's authorization.""" + result = run_js(script_step("discover", "Resolve PR identity and check trust")["with"]["script"], defect) + assert result["errors"] + assert not result["outputs"].get("merge_sha") + + +@pytest.mark.parametrize("number,branch,path,expected", [ + (299, "verify-pr-fullsend", "evals/fullsend/triage-security/cases/033-absent/input.yaml", "true"), + (299, "verify-pr-fullsend", "plugins/sdlc-workflow/skills/triage-security/SKILL.md", "true"), + (299, "verify-pr-fullsend", "plugins/sdlc-workflow/schemas/triage-security-input.schema.json", "true"), + (299, "verify-pr-fullsend", "plugins/sdlc-workflow/providers/vertex-ai.yaml", "true"), + (299, "verify-pr-fullsend", ".github/scripts/run-native-fullsend-evals.sh", "true"), + (300, "verify-pr-fullsend", "evals/fullsend/run.py", "false"), + (299, "wrong", "evals/fullsend/run.py", "false"), + (299, "verify-pr-fullsend", "README.md", "false"), +]) +def test_native_discovery_is_relevant_and_bootstrap_only(number, branch, path, expected): + """Only relevant changes on exact PR299/source branch activate bootstrap native CI.""" + result = run_js(script_step("discover", "Discover changed skills")["with"]["script"], {"paths": [path]}, + {"PR_NUMBER": str(number), "SOURCE_BRANCH": branch, "MERGE_SHA": "c" * 40}) + assert result["errors"] == [] + assert result["outputs"].get("native") == expected + + +def test_native_execution_uses_trusted_setup_and_readonly_github_permissions(): + """Credentialed host executes base scripts, and the tested revision is immutable data.""" + jobs = workflow()["jobs"] + assert "run-native-evals" in jobs, "Native CI is not implemented" + native = jobs["run-native-evals"] + assert native["permissions"] == {"contents": "read", "pull-requests": "read", "id-token": "write"} + assert native["needs"] == ["discover", "gate"] + assert "needs.gate.result == 'success'" in native["if"] + assert "needs.discover.outputs.native == 'true'" in native["if"] + assert "needs.discover.outputs.native == 'true'" in jobs["gate"]["if"] + assert jobs["gate"]["environment"] == "eval-protected" + checkouts = [s for s in native["steps"] if s.get("uses", "").startswith("actions/checkout@")] + assert [s["with"]["ref"] for s in checkouts] == ["${{ github.sha }}", "${{ needs.discover.outputs.merge_sha }}", + "${{ env.NATIVE_EVAL_SOURCE_SHA }}", "d5f36921ac754705619f38c637ef692873809fbc"] + assert all(s["with"]["persist-credentials"] is False for s in checkouts) + assert next(s for s in native["steps"] if s.get("id") == "revision")["name"] == "Recheck approved revision before WIF" + auth = next(s for s in native["steps"] if s.get("uses", "").startswith("google-github-actions/auth@")) + assert auth["with"] == {"workload_identity_provider": "${{ secrets.FULLSEND_GCP_WIF_PROVIDER }}", + "project_id": "${{ secrets.FULLSEND_GCP_PROJECT_ID }}"} + wrapper = (ROOT / ".github/scripts/run-native-fullsend-evals.sh").read_text() + assert "prepare-sandbox-credentials.sh" in wrapper + assert "HOST_GOOGLE_APPLICATION_CREDENTIALS" in wrapper + assert "TC6726_SANDBOX_CREDENTIALS" in wrapper + assert "--plugin-root" in wrapper and "pr-head/plugins/sdlc-workflow" in wrapper + assert "pr-head/" not in wrapper.replace("pr-head/plugins/sdlc-workflow", "") + + +@pytest.mark.parametrize("head", ["a" * 40, "d" * 40]) +def test_revision_rechecked_after_approval_before_wif(head): + """An approval waiting on an older revision must fail before authentication.""" + result = run_js(script_step("run-native-evals", "Recheck approved revision before WIF")["with"]["script"], + {"head": head}, {"PR_NUMBER": "299", "HEAD_SHA": "a" * 40, "MERGE_SHA": "c" * 40, + "BASE_SHA": "b" * 40, "SOURCE_REPO": "mrizzi/sdlc-plugins", + "SOURCE_BRANCH": "verify-pr-fullsend"}) + assert bool(result["errors"]) == (head != "a" * 40) + + +@pytest.mark.parametrize("ordinary,native,requested,expected", [ + ("success", "success", "true", "success"), ("success", "failure", "true", "failure"), + ("success", "skipped", "true", "failure"), ("failure", "success", "true", "failure"), + ("skipped", "success", "true", "success"), ("success", "skipped", "false", "success"), +]) +def test_combined_status_fails_on_requested_native_failure(ordinary, native, requested, expected): + """Neither a native failure nor a requested skip is masked by ordinary success.""" + result = run_js(script_step("report-status", "Set final commit status")["with"]["script"], {}, { + "DISCOVER_RESULT": "success", "EVALS_RESULT": ordinary, "GATE_RESULT": "skipped", + "NATIVE_RESULT": native, "NATIVE_REQUESTED": requested, + "NATIVE_REPORT_RESULT": "success", + "SKILLS_CSV": "triage-security" if ordinary != "skipped" else ""}) + assert result["statuses"][0]["state"] == expected + + +@pytest.mark.parametrize("report_result", ["failure", "skipped", "cancelled"]) +def test_combined_status_requires_successful_native_reporting(report_result): + """Source-bound publication is mandatory even when the execution job succeeded.""" + result = run_js(script_step("report-status", "Set final commit status")["with"]["script"], {}, { + "DISCOVER_RESULT": "success", "EVALS_RESULT": "success", "GATE_RESULT": "skipped", + "NATIVE_RESULT": "success", "NATIVE_REQUESTED": "true", "NATIVE_REPORT_RESULT": report_result, + "SKILLS_CSV": "triage-security"}) + assert result["statuses"][0]["state"] == "failure" + + +@pytest.mark.parametrize("trusted,gate,expected", [("true", "skipped", True), ("false", "success", True), + ("false", "failure", False), ("false", "skipped", False)]) +def test_native_job_requires_collaborator_or_current_run_approval(trusted, gate, expected): + """Evaluate the real job condition for approved/unapproved author combinations.""" + condition = workflow()["jobs"]["run-native-evals"]["if"] + values = {"needs.discover.result": "success", "needs.discover.outputs.native": "true", + "needs.discover.outputs.trusted": trusted, "needs.gate.result": gate} + for key, value in values.items(): + condition = condition.replace(key, json.dumps(value)) + condition = condition.replace("!cancelled()", "true") + result = subprocess.run(["node", "-e", f"process.stdout.write(JSON.stringify(Boolean({condition})));"], + capture_output=True, text=True, check=True) + assert json.loads(result.stdout) is expected + + +@pytest.mark.parametrize("defect", [None, "wrong-source", "wrong-eval-source", "missing-outcome", "null", "false", "scorer-failed", "missing-report"]) +def test_reporting_verifies_source_and_boolean_outcomes(defect): + """Missing/incomplete/scorer failure cannot be published as successful native evidence.""" + source = {"pr_number": 299, "head_sha": "a" * 40, "merge_sha": "c" * 40, + "base_sha": "b" * 40, "trusted_sha": "e" * 40, "eval_source_sha": "f" * 40} + outcomes = {case: {f"assertion_{i}": True for i in range(1, n+1)} + for case,n in {"033-absent": 4, "034-empty": 5, "035-malformed": 5, "036-valid": 7}.items()} + report = {"source": source, "outcomes": outcomes, "complete": True, "total": 21, "exit_code": 0, + "rationale": "SECRET /tmp/gha-creds-evil"} + if defect == "wrong-source": source["head_sha"] = "d" * 40 + elif defect == "wrong-eval-source": source["eval_source_sha"] = "d" * 40 + elif defect == "missing-outcome": del outcomes["033-absent"]["assertion_1"] + elif defect in {"null", "false"}: outcomes["033-absent"]["assertion_1"] = None if defect == "null" else False + elif defect == "scorer-failed": report["exit_code"] = 7 + elif defect == "missing-report": report = None + env = {"PR_NUMBER": "299", "HEAD_SHA": "a" * 40, "MERGE_SHA": "c" * 40, + "BASE_SHA": "b" * 40, "TRUSTED_SHA": "e" * 40, "EVAL_SOURCE_SHA": "f" * 40, "NATIVE_RESULT": "success"} + result = run_js(script_step("report-status", "Publish native result alongside ordinary review")["with"]["script"], + {"report": report}, env) + assert bool(result["errors"]) == (defect is not None) + if result["reviews"]: + assert result["reviews"][0]["commit_id"] == "a" * 40 + assert "SECRET" not in result["reviews"][0]["body"] + + + +def test_wrapper_rejects_different_reviewed_suite_before_credentials(tmp_path): + """A mismatched suite checkout must stop before setup or credential access.""" + tools = tmp_path / "tools" + tools.mkdir() + git = tools / "git" + git.write_text("#!/bin/sh\ncase \"$2\" in *upstream-fullsend) echo d5f36921ac754705619f38c637ef692873809fbc;; *) echo aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa;; esac\n") + git.chmod(0o755) + environment = dict(os.environ, GITHUB_WORKSPACE=str(tmp_path), RUNNER_TEMP=str(tmp_path), + NATIVE_EVAL_SOURCE_SHA="b" * 40, PATH=str(tools) + os.pathsep + os.environ["PATH"]) + environment.pop("GOOGLE_APPLICATION_CREDENTIALS", None) + result = subprocess.run(["bash", str(ROOT / ".github/scripts/run-native-fullsend-evals.sh"), "run"], + env=environment, capture_output=True, text=True) + assert result.returncode != 0 + assert "Reviewed native eval source changed" in result.stdout + assert "WIF host ADC" not in result.stderr + + +@pytest.mark.parametrize("job", ["discover", "run-evals", "report-status"]) +def test_all_publication_jobs_check_latest_run(job): + """Every status/review write requires a successful latest-run check.""" + job_data = workflow()["jobs"][job] + guard = next(s for s in job_data["steps"] if s.get("id") == "publication") + assert guard["env"]["HEAD_SHA"] == "${{ github.event.workflow_run.head_sha }}" + assert workflow()["run-name"] == "Eval PR Run ${{ github.event.workflow_run.head_sha }}" + if "permissions" in job_data: + assert job_data["permissions"]["actions"] == "read" + for step in job_data["steps"]: + script = step.get("with", {}).get("script", "") + if any(api in script for api in ["createCommitStatus(", "createReview(", "updateReview("]): + assert any(f"steps.{name}.outputs.latest == 'true'" in step["if"] + for name in ["publication", "gate-publication"]) + + +@pytest.mark.parametrize("runs,attempt,api_error,expected", [ + ([{"id": 1, "run_number": 1}], 1, False, "true"), + ([{"id": 1, "run_number": 1}, {"id": 2, "run_number": 2}], 1, False, "false"), + ([{"id": 1, "run_number": 1}, {"id": 2, "run_number": 2, "other_head": True}], 1, False, "true"), + ([], 1, False, "true"), + ([{"id": 1, "run_number": 1}], 2, False, "false"), + ([{"id": 1, "run_number": 1}], 1, True, "false"), +]) +def test_latest_run_guard_refuses_superseded_or_unidentifiable_runs(runs, attempt, api_error, expected): + """Execute the real guard for newer runs, other heads, reruns and API failures.""" + script = script_step("discover", "Check latest run before publishing")["with"]["script"] + result = run_js(script, {"runs": runs, "attempt": attempt, "apiError": api_error}, {"HEAD_SHA": "a" * 40}) + assert result["outputs"].get("latest") == expected + assert bool(result["errors"]) == api_error + + +def test_same_pr_head_cannot_publish_concurrently(): + """The workflow lock covers both ordinary and native publication sequences.""" + concurrency = workflow().get("concurrency", {}) + assert concurrency.get("group") == "eval-pr-run-${{ github.event.workflow_run.head_sha }}" + assert concurrency["cancel-in-progress"] is False + + +@pytest.mark.parametrize("native", [True, False]) +@pytest.mark.parametrize("existing_kind", ["none", "matching", "wrong-head", "human", "other-suite", "later-page"]) +def test_review_reruns_reuse_only_matching_bot_head_review(native, existing_kind): + """Both publishers create once, update reruns and leave unrelated reviews alone.""" + job, name = ("report-status", "Publish native result alongside ordinary review") if native else ( + "run-evals", "Post eval results review") + script = script_step(job, name)["with"]["script"] + marker = "## Native Fullsend Eval Results" if native else "## Eval Results" + existing = {"id": 888, "user": {"login": "github-actions[bot]"}, "commit_id": "a" * 40, "body": marker} + if existing_kind == "wrong-head": existing["commit_id"] = "b" * 40 + elif existing_kind == "human": existing["user"]["login"] = "human" + elif existing_kind == "other-suite": existing["body"] = "## Eval Results" if native else "## Native Fullsend Eval Results" + stored = [] if existing_kind == "none" else [existing] + if existing_kind == "later-page": + stored = [{"id": i, "user": {"login": "human"}} for i in range(100)] + stored + assert "github.paginate(github.rest.pulls.listReviews" in script + source = {"pr_number": 299, "head_sha": "a" * 40, "merge_sha": "c" * 40, + "base_sha": "b" * 40, "trusted_sha": "e" * 40, "eval_source_sha": "f" * 40} + report = {"source": source, "complete": True, "total": 21, "exit_code": 0, + "outcomes": {case: {f"assertion_{i}": True for i in range(1, n+1)} + for case,n in {"033-absent": 4, "034-empty": 5, "035-malformed": 5, "036-valid": 7}.items()}} + result = run_js(script, {"report": report, "existingReviews": stored, "repeat": 2}, { + "PR_NUMBER": "299", "HEAD_SHA": "a" * 40, "MERGE_SHA": "c" * 40, + "BASE_SHA": "b" * 40, "TRUSTED_SHA": "e" * 40, "EVAL_SOURCE_SHA": "f" * 40, + "NATIVE_RESULT": "success", "SKILLS_CSV": "triage-security"}) + reuse = existing_kind in {"matching", "later-page"} + assert result["errors"] == [] + assert len(result["reviews"]) == (0 if reuse else 1) + assert len(result["updates"]) == (2 if reuse else 1) + assert all(r["body"].startswith(marker) for r in result["reviews"] + result["updates"]) + if reuse: + assert all(r["review_id"] == 888 for r in result["updates"]) + else: + assert all(r["review_id"] != 888 for r in result["updates"]) + + +@pytest.mark.parametrize("revisions,parents,expected_sha,reads", [ + ([{"merge_commit_sha": None}, {"merge_commit_sha": "c" * 40}], None, "c" * 40, 2), + ([{"merge_commit_sha": None}], None, None, 3), + ([{"merge_commit_sha": None}, {"merge_commit_sha": "c" * 40, "head": {"sha": "d" * 40}}], None, None, 2), + ([{"merge_commit_sha": None}, {"merge_commit_sha": "c" * 40}], ["b" * 40, "d" * 40], None, 2), + ([{"merge_commit_sha": "invalid"}], None, None, 1), +]) +def test_merge_source_poll_is_bounded_and_preserves_revision_checks(revisions, parents, expected_sha, reads): + """Transient nulls recover; persistent nulls and changed revisions fail closed.""" + # Given API mergeability responses and exact merge-parent evidence + data = {"revisions": revisions} + if parents is not None: + data["parents"] = parents + # When resolving the event-associated PR in the real workflow script + result = run_js(script_step("discover", "Resolve PR identity and check trust")["with"]["script"], data) + # Then retries are bounded and only the verified merge reaches downstream jobs + assert result["outputs"].get("merge_sha") == expected_sha + assert bool(result["errors"]) == (expected_sha is None) + assert result["prReads"] == reads + assert result["delays"] == [2000] * (reads - 1) + if revisions == [{"merge_commit_sha": None}]: + assert result["errors"] == ["PR identity/revision changed or merge source unavailable"] + + +def test_native_artifact_download_failure_keeps_controlled_reporting(): + """Missing native artifacts are nonfatal downloads while evidence still fails closed.""" + # Given the reporting job's native artifact download + step = script_step("report-status", "Download safe native result") + # Then its existing guard is preserved and download errors can reach the reporter + assert step.get("continue-on-error") is True + assert step["if"] == "needs.discover.outputs.native == 'true' && needs.run-native-evals.result != 'skipped'" + assert step["with"] == {"name": "native-fullsend-result", "path": "native-report"} + # When no artifact is available, the real publisher emits controlled failure evidence + result = run_js(script_step("report-status", "Publish native result alongside ordinary review")["with"]["script"], {}, { + "PR_NUMBER": "299", "HEAD_SHA": "a" * 40, "MERGE_SHA": "c" * 40, + "BASE_SHA": "b" * 40, "TRUSTED_SHA": "e" * 40, "EVAL_SOURCE_SHA": "f" * 40, "NATIVE_RESULT": "failure"}) + assert result["errors"] == ["Native evidence incomplete, failed, or missing"] + assert "No safe native result was produced; native execution/approval failed." in result["reviews"][0]["body"] + + +def run_credential_wrapper(tmp_path, output): + """SYNTHETIC TEST DATA — run the real wrapper with local credential/inference doubles.""" + tools = tmp_path / "tools" + tools.mkdir() + scripts = tmp_path / "upstream-fullsend/internal/scaffold/fullsend-repo/scripts" + scripts.mkdir(parents=True) + fixture = tmp_path / "prepared-output.txt" + fixture.write_text(output) + (scripts / "prepare-sandbox-credentials.sh").write_text( + '#!/bin/sh\n# SYNTHETIC TEST DATA — emit the test environment file\ncat "$TC6742_FIXTURE" >> "$GITHUB_ENV"\n') + doubles = { + "git": '#!/bin/sh\n# SYNTHETIC TEST DATA — immutable checkout identities\ncase "$2" in *upstream-fullsend) echo d5f36921ac754705619f38c637ef692873809fbc;; *) echo bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb;; esac\n', + "jq": '#!/bin/sh\n# SYNTHETIC TEST DATA — host ADC type\necho external_account\n', + "python3.12": '#!/usr/bin/env python3\n# SYNTHETIC TEST DATA — capture parser output without inference\nimport json, os\nfrom pathlib import Path\nPath(os.environ["TC6742_CAPTURE"]).write_text(json.dumps({k: os.environ.get(k) for k in ["GOOGLE_APPLICATION_CREDENTIALS", "TC6726_SANDBOX_CREDENTIALS", "GCP_OIDC_TOKEN_FILE", "FULLSEND_GCP_OIDC_URL", "FULLSEND_GCP_OIDC_AUTH_FILE", "TC6742_UNEXPECTED"]}))\n', + } + for name, content in doubles.items(): + path = tools / name + path.write_text(content) + path.chmod(0o755) + capture = tmp_path / "captured.json" + environment = dict(os.environ, GITHUB_WORKSPACE=str(tmp_path), RUNNER_TEMP=str(tmp_path), + GOOGLE_APPLICATION_CREDENTIALS="synthetic-host-adc", ANTHROPIC_VERTEX_PROJECT_ID="synthetic", + CLOUD_ML_REGION="global", TC6726_HEAD_SHA="a" * 40, NATIVE_EVAL_SOURCE_SHA="b" * 40, + TC6742_FIXTURE=str(fixture), TC6742_CAPTURE=str(capture), + PATH=str(tools) + os.pathsep + os.environ["PATH"]) + for name in ["TC6726_SANDBOX_CREDENTIALS", "GCP_OIDC_TOKEN_FILE", "FULLSEND_GCP_OIDC_URL", "FULLSEND_GCP_OIDC_AUTH_FILE", "TC6742_UNEXPECTED"]: + environment.pop(name, None) + result = subprocess.run(["bash", str(ROOT / ".github/scripts/run-native-fullsend-evals.sh"), "run"], + env=environment, capture_output=True, text=True) + return result, json.loads(capture.read_text()) if capture.exists() else None + + +@pytest.mark.parametrize("heredoc", [False, True]) +def test_credential_parser_accepts_blank_lines_and_heredoc_values(tmp_path, heredoc): + """Valid environment syntax captures/masks required values and preserves host ADC.""" + # Given synthetic credentials, with optional multiline output + expected = {"TC6726_SANDBOX_CREDENTIALS": "synthetic-sandbox-adc", "GCP_OIDC_TOKEN_FILE": "synthetic-token-file", + "FULLSEND_GCP_OIDC_URL": "https://synthetic.invalid/?audience=eval", "FULLSEND_GCP_OIDC_AUTH_FILE": "synthetic-auth-file"} + names = {"GOOGLE_APPLICATION_CREDENTIALS": "TC6726_SANDBOX_CREDENTIALS", **{k: k for k in expected if k != "TC6726_SANDBOX_CREDENTIALS"}} + if heredoc: + expected["FULLSEND_GCP_OIDC_AUTH_FILE"] = "synthetic-auth%file\nsynthetic-second-line" + output = "\n".join(f"{name}<