diff --git a/evals/triage-security/evals.json b/evals/triage-security/evals.json index 1e495495c..46572474f 100644 --- a/evals/triage-security/evals.json +++ b/evals/triage-security/evals.json @@ -21,8 +21,7 @@ "Remediation tasks include Repository, Target Branch, acceptance criteria, and dependency linkage between upstream and downstream tasks (Important Rule 9)", "For Cargo ecosystem, remediation creates two tasks: upstream backport task and downstream propagation subtask with Blocks dependency (Important Rule 8)", "The analysis presents triage recommendations as proposed actions — framing Affects Versions changes, label additions, and status transitions as proposals rather than executed mutations (Guardrails)", - "The data-extraction output describes Step 0.8 actions before the Step 1 data table: assigning the CVE issue to the current user and transitioning it to Assigned status (Step 0.8)", - "With a schema-valid Fullsend input bundle and authorization.mutation_authorized true, the standard triage serializes its assignment, Affects Versions correction, labels, transitions, comments, two remediation tasks, and Blocks/Depend links as ordered schema actions with stable triage-security markers rather than performing the Jira writes" + "The data-extraction output describes Step 0.8 actions before the Step 1 data table: assigning the CVE issue to the current user and transitioning it to Assigned status (Step 0.8)" ] }, { @@ -39,8 +38,7 @@ "Triage outcome recommends closing as Not a Bug with resolution, not creating remediation tasks (Step 8 Case C)", "The close recommendation includes a comment documenting the version impact evidence — listing each version and its dependency version as proof of non-impact", "VEX Justification is set to 'Component not Present' or appropriate value when the VEX field is configured (Step 8 VEX Justification)", - "The analysis does NOT create any remediation tasks — no upstream backport, no downstream propagation (Step 8 Case C closes the issue)", - "In a valid Fullsend bundle with authorization.mutation_authorized false, the already-fixed closure, VEX update, labels, transition, and evidence comment are named as withheld in exactly one evidence-backed report-only action and no mutation action is serialized" + "The analysis does NOT create any remediation tasks — no upstream backport, no downstream propagation (Step 8 Case C closes the issue)" ] }, { @@ -57,8 +55,7 @@ "Duplicate check correctly classifies TC-7999 as a same-stream sibling (both have stream suffix [rhtpa-2.2]) and identifies it as a duplicate (Step 4.1)", "Triage outcome recommends closing TC-8003 as Duplicate with resolution 'Duplicate', referencing TC-7999 as the original (Step 4.1)", "The close recommendation includes a comment documenting why it is a duplicate — same CVE tracked for the same stream (Step 4.1)", - "The analysis does NOT proceed to remediation task creation — duplicate detection short-circuits the triage flow", - "With Fullsend authorization, the duplicate resolution and its evidence comment are serialized as status-transition and comment actions using trusted sibling evidence; no Jira search, Jira write, or remediation-task action is performed in the sandbox" + "The analysis does NOT proceed to remediation task creation — duplicate detection short-circuits the triage flow" ] }, { @@ -75,8 +72,7 @@ "Affects Versions correction includes only the actually affected versions (2.1.x versions), not the unaffected 2.2.x versions (Step 3.2)", "Remediation tasks are created only for the affected stream (2.1.x), not for 2.2.x which already ships the patched version (Step 8 Case A scoping)", "The analysis treats the issue as unscoped (no stream suffix) and checks ALL versions across all streams (Step 1 Stream scope resolution, unscoped path)", - "Cross-stream impact notice is NOT generated because the issue is unscoped — it covers all streams by definition (Step 8 Case B applies only to scoped issues)", - "In Fullsend mode the mixed-stream analysis consumes only validated matrix and source-evidence records, does not call git or Jira or write a matrix, and serializes the 2.1.x correction and remediation actions only when authorization permits them" + "Cross-stream impact notice is NOT generated because the issue is unscoped — it covers all streams by definition (Step 8 Case B applies only to scoped issues)" ] }, { @@ -94,8 +90,7 @@ "Dependency chain context identifies openssl-libs as present in rpms.lock.yaml and uses this to determine the package origin — the lock file presence is the primary classification signal (Step 2.3.5)", "Remediation creates a single task for the Konflux release repo — not the two-task upstream backport + downstream propagation flow used for source dependency ecosystems (Step 8 Remediation Task Creation — system package path)", "Remediation task description follows task-description-template.md format with Repository, Target Branch, Description, Implementation Notes, and Acceptance Criteria sections — Files to Modify is intentionally omitted per remediation-templates.md (Step 8 Remediation Task Creation)", - "The version impact output addresses SBOM verification in the dependency chain context — either noting that verification was skipped because cosign is not available or because external tools are prohibited, or presenting the SBOM classification result (Step 2.3.5 Optional SBOM verification)", - "For Fullsend RPM triage, lock and any supplied SBOM evidence are parsed only from the valid trusted bundle: the sandbox does not invoke cosign, git, Jira, or a network lookup and does not write local evidence or security-matrix.md" + "The version impact output addresses SBOM verification in the dependency chain context — either noting that verification was skipped because cosign is not available or because external tools are prohibited, or presenting the SBOM classification result (Step 2.3.5 Optional SBOM verification)" ] }, { @@ -149,8 +144,7 @@ "Triage outcome recommends closing the issue because existing remediation task TC-8009 already bumps axios past the fix threshold — no new remediation task is created", "Step 4.3 creates a Related link between the current CVE (TC-8010) and the related CVE (TC-8008) with an idempotency check on existing issuelinks before creating", "Step 4.3 creates a Depend link from the covering remediation task (TC-8009) to the current CVE (TC-8010) with an idempotency check on existing issuelinks before creating", - "A comment is posted on the current CVE documenting the cross-CVE overlap finding — including the related CVE key (TC-8008), covering task key (TC-8009), library (axios), bump version (1.9.0), and fix threshold (1.8.2)", - "With authorized Fullsend input, the covered-overlap outcome serializes idempotent Related and Depend link actions, the overlap comment action, and the close transition from trusted evidence; it does not call Jira and omits actions whose markers or trusted existing links already exist" + "A comment is posted on the current CVE documenting the cross-CVE overlap finding — including the related CVE key (TC-8008), covering task key (TC-8009), library (axios), bump version (1.9.0), and fix threshold (1.8.2)" ] }, { @@ -167,8 +161,7 @@ "Step 4.3 filters search results by matching PS Component (customfield_10669 = pscomponent:org/rhtpa-ui) and Stream (customfield_10832 = rhtpa-2.2)", "Step 4.3 traverses issuelinks on TC-8012 to find remediation task TC-8013 via the 'Depend' link type", "Step 4.3 compares the remediation task's bump version (5.96.1) against the current CVE's fix threshold (5.98.0) and determines the fix does NOT cover this CVE", - "The overlap table is presented to the engineer showing TC-8012 and TC-8013 with 'Covers This CVE? = No' — the triage proceeds to create new remediation tasks rather than recommending closure", - "With authorized Fullsend input, the insufficient-overlap path serializes each new remediation-task with a stable marker and reference before its dependent links and later comments, without external calls or invented Jira keys" + "The overlap table is presented to the engineer showing TC-8012 and TC-8013 with 'Covers This CVE? = No' — the triage proceeds to create new remediation tasks rather than recommending closure" ] }, { @@ -202,8 +195,7 @@ "Step 4.4 filters results to match the current stream (rhtpa-2.1) by checking the task summary for the stream name", "Step 4.4 links the new CVE Jira TC-8021 to the preemptive task TC-8022 with 'Depend' link type (standard remediation linkage)", "Step 4.4 removes the 'security-preemptive' label from TC-8022 — the task is now a standard remediation task linked to a proper CVE Jira", - "Step 8 skips remediation task creation for this stream because Step 4.4 reconciliation already linked an existing task", - "In authorized Fullsend reconciliation, the trusted preemptive task produces a Depend link action and security-preemptive label-removal field-edit action with stable markers; it never mutates Jira directly and omits marker-recorded ordinary actions on retry" + "Step 8 skips remediation task creation for this stream because Step 4.4 reconciliation already linked an existing task" ] }, { @@ -220,8 +212,7 @@ "Step 1.5 queries MITRE CVE API and extracts lessThan 0.4.8 as the fix threshold from the affected[].versions[] field (§1.62)", "Step 1.5 queries OSV.dev API and extracts fixed version 0.4.8 from the affected[].ranges[].events field (§1.62)", "Step 1.5 cross-validation shows agreement between MITRE and OSV.dev — the enriched fix threshold (0.4.8) is used as the authoritative value for Step 2.3", - "Step 2.3 version impact comparison uses the enriched fix threshold (0.4.8) from Step 1.5, not the imprecise Jira description data", - "In Fullsend mode enrichment reads MITRE and OSV responses only from valid external_evidence in the trusted input bundle; it makes no API or network request and reports a missing required enrichment record as an evidence limitation" + "Step 2.3 version impact comparison uses the enriched fix threshold (0.4.8) from Step 1.5, not the imprecise Jira description data" ] }, { @@ -324,8 +315,7 @@ "The ai-cve-triaged label is already present on the issue — the triage does NOT attempt to add it again (label idempotency)", "The issue is already In Progress — the triage does NOT attempt to transition the status again or warns that the issue is already in a post-triage state (status idempotency)", "The existing description digest comment is detected — no duplicate digest comment is posted (comment idempotency)", - "The existing post-triage summary comment is detected — no duplicate summary comment is posted (comment idempotency)", - "On an authorized Fullsend rerun, existing idempotency.action_markers suppress duplicate ordinary field, transition, comment, and link actions, while an existing remediation task with a missing description digest is still emitted with the same remediation-task marker and ref so the executor reuses it, repairs the digest once, and resolves later placeholders without duplicate creation" + "The existing post-triage summary comment is detected — no duplicate summary comment is posted (comment idempotency)" ] }, { @@ -615,6 +605,136 @@ "The report uses the supplied SBOM evidence and explicitly does not invoke cosign, git, Jira, a network lookup, local evidence, or security-matrix.md", "The result is report-only with exactly one report-only action because trusted authorization is false" ] + }, + { + "id": 37, + "prompt": "Run triage-security for TC-8001 in an isolated Fullsend fixture sandbox using fullsend-eval-1-trusted-input.json. Before invoking the skill, actually set FULLSEND_OUTPUT_DIR to an isolated sandbox output directory outside the assigned grader outputs directory in your shell/Python execution environment (not just in prose). Copy this JSON file byte-for-byte to an isolated eval workspace .pre-script/triage-security-input.json; this is the local fixture mount mapping for /sandbox/workspace/.pre-script/triage-security-input.json. The eval runner must record invocation.json outside both the sandbox and grader outputs, including UTC clock, actual gate presence/value, mounted input path, the complete mounted scenario JSON and authorization. Set INVOCATION_PATH to this runner-owned audit file and GRADER_OUTPUTS_DIR to the assigned outputs directory; these are runner handoff variables, not skill configuration. Validate the mounted document using pre_triage_security.validate_bundle before analysis and record its actual outcome in invocation.json. Invoke the triage-security skill with this mapping and use only the validated mounted bundle for scenario evidence. Do not read local CLAUDE.md, a local matrix, or unrelated fixtures, call Jira/GitHub/WebFetch/git/cosign/network tools, execute a Jira action plan, or write security-matrix.md. Analyze the standard quinn-proto scenario and serialize assignment, Affects Versions correction, labels, transitions, evidence comments, two upstream/downstream remediation tasks and their Blocks/Depend links in skill step order. Use the trusted current_user account ID; keep new task keys symbolic. Write only the final accumulated agent-result.json to $FULLSEND_OUTPUT_DIR and validate it against the unchanged triage-security-result.schema.json before finishing. Authorized mode serializes proposed actions only; it does not execute them. After the skill finishes, the eval runner must record the actual result-schema validation outcome and sandbox directory listing in invocation.json, then execute this handoff with the same actual FULLSEND_OUTPUT_DIR gate. Copy the two artifacts byte-for-byte into the grader's assigned outputs directory; do not place runner evidence in the sandbox or modify the skill result. The grader receives outputs/invocation.json and outputs/agent-result.json directly, without needing files outside outputs/. This handoff is runner work after skill execution, not a sandbox-side skill write.\n```python\nimport os\nimport shutil\nfrom pathlib import Path\n\ngrader_outputs = Path(GRADER_OUTPUTS_DIR).resolve()\nsandbox_outputs = Path(os.environ[\"FULLSEND_OUTPUT_DIR\"]).resolve()\ninvocation = Path(INVOCATION_PATH).resolve()\nassert not grader_outputs.is_relative_to(sandbox_outputs)\nassert not invocation.is_relative_to(sandbox_outputs)\nfor source in [invocation, sandbox_outputs / \"agent-result.json\"]:\n shutil.copyfile(source, grader_outputs / source.name)\n```", + "expected_output": "Independent Fullsend counterpart of interactive eval 1: Analyze the standard quinn-proto scenario and serialize assignment, Affects Versions correction, labels, transitions, evidence comments, two upstream/downstream remediation tasks and their Blocks/Depend links in skill step order. Use the trusted current_user account ID; keep new task keys symbolic.", + "files": [ + "files/fullsend-eval-1-trusted-input.json" + ], + "assertions": [ + "With a schema-valid Fullsend input bundle and authorization.mutation_authorized true, the standard triage serializes its assignment, Affects Versions correction, labels, transitions, comments, two remediation tasks, and Blocks/Depend links as ordered schema actions with stable triage-security markers rather than performing the Jira writes", + "The runner records actual non-empty FULLSEND_OUTPUT_DIR gate state, clock and mounted scenario JSON in invocation.json; the input passes pre_triage_security.validate_bundle and the observable agent-result.json passes triage-security-result.schema.json with its report issue and mode bound to the trusted subject and authorization.", + "The sandbox uses only the mounted scenario evidence, performs no external operation or Jira mutation, and writes only agent-result.json; every serialized action has a unique stable triage-security marker." + ] + }, + { + "id": 38, + "prompt": "Run triage-security for TC-8002 in an isolated Fullsend fixture sandbox using fullsend-eval-2-trusted-input.json. Before invoking the skill, actually set FULLSEND_OUTPUT_DIR to an isolated sandbox output directory outside the assigned grader outputs directory in your shell/Python execution environment (not just in prose). Copy this JSON file byte-for-byte to an isolated eval workspace .pre-script/triage-security-input.json; this is the local fixture mount mapping for /sandbox/workspace/.pre-script/triage-security-input.json. The eval runner must record invocation.json outside both the sandbox and grader outputs, including UTC clock, actual gate presence/value, mounted input path, the complete mounted scenario JSON and authorization. Set INVOCATION_PATH to this runner-owned audit file and GRADER_OUTPUTS_DIR to the assigned outputs directory; these are runner handoff variables, not skill configuration. Validate the mounted document using pre_triage_security.validate_bundle before analysis and record its actual outcome in invocation.json. Invoke the triage-security skill with this mapping and use only the validated mounted bundle for scenario evidence. Do not read local CLAUDE.md, a local matrix, or unrelated fixtures, call Jira/GitHub/WebFetch/git/cosign/network tools, execute a Jira action plan, or write security-matrix.md. All supplied serde_json versions are already patched. Report the Not a Bug closure, VEX update, labels, transition and evidence comment as withheld because authorization is false. Produce exactly one report-only action; create no remediation. Write only the final accumulated agent-result.json to $FULLSEND_OUTPUT_DIR and validate it against the unchanged triage-security-result.schema.json before finishing. Authorized mode serializes proposed actions only; it does not execute them. After the skill finishes, the eval runner must record the actual result-schema validation outcome and sandbox directory listing in invocation.json, then execute this handoff with the same actual FULLSEND_OUTPUT_DIR gate. Copy the two artifacts byte-for-byte into the grader's assigned outputs directory; do not place runner evidence in the sandbox or modify the skill result. The grader receives outputs/invocation.json and outputs/agent-result.json directly, without needing files outside outputs/. This handoff is runner work after skill execution, not a sandbox-side skill write.\n```python\nimport os\nimport shutil\nfrom pathlib import Path\n\ngrader_outputs = Path(GRADER_OUTPUTS_DIR).resolve()\nsandbox_outputs = Path(os.environ[\"FULLSEND_OUTPUT_DIR\"]).resolve()\ninvocation = Path(INVOCATION_PATH).resolve()\nassert not grader_outputs.is_relative_to(sandbox_outputs)\nassert not invocation.is_relative_to(sandbox_outputs)\nfor source in [invocation, sandbox_outputs / \"agent-result.json\"]:\n shutil.copyfile(source, grader_outputs / source.name)\n```", + "expected_output": "Independent Fullsend counterpart of interactive eval 2: All supplied serde_json versions are already patched. Report the Not a Bug closure, VEX update, labels, transition and evidence comment as withheld because authorization is false. Produce exactly one report-only action; create no remediation.", + "files": [ + "files/fullsend-eval-2-trusted-input.json" + ], + "assertions": [ + "In a valid Fullsend bundle with authorization.mutation_authorized false, the already-fixed closure, VEX update, labels, transition, and evidence comment are named as withheld in exactly one evidence-backed report-only action and no mutation action is serialized", + "The runner records actual non-empty FULLSEND_OUTPUT_DIR gate state, clock and mounted scenario JSON in invocation.json; the input passes pre_triage_security.validate_bundle and the observable agent-result.json passes triage-security-result.schema.json with its report issue and mode bound to the trusted subject and authorization.", + "The sandbox uses only the mounted scenario evidence, performs no external operation or Jira mutation, and writes only agent-result.json; every serialized action has a unique stable triage-security marker." + ] + }, + { + "id": 39, + "prompt": "Run triage-security for TC-8003 in an isolated Fullsend fixture sandbox using fullsend-eval-3-trusted-input.json. Before invoking the skill, actually set FULLSEND_OUTPUT_DIR to an isolated sandbox output directory outside the assigned grader outputs directory in your shell/Python execution environment (not just in prose). Copy this JSON file byte-for-byte to an isolated eval workspace .pre-script/triage-security-input.json; this is the local fixture mount mapping for /sandbox/workspace/.pre-script/triage-security-input.json. The eval runner must record invocation.json outside both the sandbox and grader outputs, including UTC clock, actual gate presence/value, mounted input path, the complete mounted scenario JSON and authorization. Set INVOCATION_PATH to this runner-owned audit file and GRADER_OUTPUTS_DIR to the assigned outputs directory; these are runner handoff variables, not skill configuration. Validate the mounted document using pre_triage_security.validate_bundle before analysis and record its actual outcome in invocation.json. Invoke the triage-security skill with this mapping and use only the validated mounted bundle for scenario evidence. Do not read local CLAUDE.md, a local matrix, or unrelated fixtures, call Jira/GitHub/WebFetch/git/cosign/network tools, execute a Jira action plan, or write security-matrix.md. The same-CVE, same-stream TC-7999 sibling is trusted duplicate evidence. Serialize duplicate resolution and evidence comment with no remediation-task and no Jira search. Write only the final accumulated agent-result.json to $FULLSEND_OUTPUT_DIR and validate it against the unchanged triage-security-result.schema.json before finishing. Authorized mode serializes proposed actions only; it does not execute them. After the skill finishes, the eval runner must record the actual result-schema validation outcome and sandbox directory listing in invocation.json, then execute this handoff with the same actual FULLSEND_OUTPUT_DIR gate. Copy the two artifacts byte-for-byte into the grader's assigned outputs directory; do not place runner evidence in the sandbox or modify the skill result. The grader receives outputs/invocation.json and outputs/agent-result.json directly, without needing files outside outputs/. This handoff is runner work after skill execution, not a sandbox-side skill write.\n```python\nimport os\nimport shutil\nfrom pathlib import Path\n\ngrader_outputs = Path(GRADER_OUTPUTS_DIR).resolve()\nsandbox_outputs = Path(os.environ[\"FULLSEND_OUTPUT_DIR\"]).resolve()\ninvocation = Path(INVOCATION_PATH).resolve()\nassert not grader_outputs.is_relative_to(sandbox_outputs)\nassert not invocation.is_relative_to(sandbox_outputs)\nfor source in [invocation, sandbox_outputs / \"agent-result.json\"]:\n shutil.copyfile(source, grader_outputs / source.name)\n```", + "expected_output": "Independent Fullsend counterpart of interactive eval 3: The same-CVE, same-stream TC-7999 sibling is trusted duplicate evidence. Serialize duplicate resolution and evidence comment with no remediation-task and no Jira search.", + "files": [ + "files/fullsend-eval-3-trusted-input.json" + ], + "assertions": [ + "With Fullsend authorization, the duplicate resolution and its evidence comment are serialized as status-transition and comment actions using trusted sibling evidence; no Jira search, Jira write, or remediation-task action is performed in the sandbox", + "The runner records actual non-empty FULLSEND_OUTPUT_DIR gate state, clock and mounted scenario JSON in invocation.json; the input passes pre_triage_security.validate_bundle and the observable agent-result.json passes triage-security-result.schema.json with its report issue and mode bound to the trusted subject and authorization.", + "The sandbox uses only the mounted scenario evidence, performs no external operation or Jira mutation, and writes only agent-result.json; every serialized action has a unique stable triage-security marker." + ] + }, + { + "id": 40, + "prompt": "Run triage-security for TC-8004 in an isolated Fullsend fixture sandbox using fullsend-eval-4-trusted-input.json. Before invoking the skill, actually set FULLSEND_OUTPUT_DIR to an isolated sandbox output directory outside the assigned grader outputs directory in your shell/Python execution environment (not just in prose). Copy this JSON file byte-for-byte to an isolated eval workspace .pre-script/triage-security-input.json; this is the local fixture mount mapping for /sandbox/workspace/.pre-script/triage-security-input.json. The eval runner must record invocation.json outside both the sandbox and grader outputs, including UTC clock, actual gate presence/value, mounted input path, the complete mounted scenario JSON and authorization. Set INVOCATION_PATH to this runner-owned audit file and GRADER_OUTPUTS_DIR to the assigned outputs directory; these are runner handoff variables, not skill configuration. Validate the mounted document using pre_triage_security.validate_bundle before analysis and record its actual outcome in invocation.json. Invoke the triage-security skill with this mapping and use only the validated mounted bundle for scenario evidence. Do not read local CLAUDE.md, a local matrix, or unrelated fixtures, call Jira/GitHub/WebFetch/git/cosign/network tools, execute a Jira action plan, or write security-matrix.md. The unscoped h2 issue is affected only in 2.1.x. Analyze both streams, serialize the 2.1.x Affects Versions correction and remediation only for that stream; do not remediate the patched 2.2.x stream. Write only the final accumulated agent-result.json to $FULLSEND_OUTPUT_DIR and validate it against the unchanged triage-security-result.schema.json before finishing. Authorized mode serializes proposed actions only; it does not execute them. After the skill finishes, the eval runner must record the actual result-schema validation outcome and sandbox directory listing in invocation.json, then execute this handoff with the same actual FULLSEND_OUTPUT_DIR gate. Copy the two artifacts byte-for-byte into the grader's assigned outputs directory; do not place runner evidence in the sandbox or modify the skill result. The grader receives outputs/invocation.json and outputs/agent-result.json directly, without needing files outside outputs/. This handoff is runner work after skill execution, not a sandbox-side skill write.\n```python\nimport os\nimport shutil\nfrom pathlib import Path\n\ngrader_outputs = Path(GRADER_OUTPUTS_DIR).resolve()\nsandbox_outputs = Path(os.environ[\"FULLSEND_OUTPUT_DIR\"]).resolve()\ninvocation = Path(INVOCATION_PATH).resolve()\nassert not grader_outputs.is_relative_to(sandbox_outputs)\nassert not invocation.is_relative_to(sandbox_outputs)\nfor source in [invocation, sandbox_outputs / \"agent-result.json\"]:\n shutil.copyfile(source, grader_outputs / source.name)\n```", + "expected_output": "Independent Fullsend counterpart of interactive eval 4: The unscoped h2 issue is affected only in 2.1.x. Analyze both streams, serialize the 2.1.x Affects Versions correction and remediation only for that stream; do not remediate the patched 2.2.x stream.", + "files": [ + "files/fullsend-eval-4-trusted-input.json" + ], + "assertions": [ + "In Fullsend mode the mixed-stream analysis consumes only validated matrix and source-evidence records, does not call git or Jira or write a matrix, and serializes the 2.1.x correction and remediation actions only when authorization permits them", + "The runner records actual non-empty FULLSEND_OUTPUT_DIR gate state, clock and mounted scenario JSON in invocation.json; the input passes pre_triage_security.validate_bundle and the observable agent-result.json passes triage-security-result.schema.json with its report issue and mode bound to the trusted subject and authorization.", + "The sandbox uses only the mounted scenario evidence, performs no external operation or Jira mutation, and writes only agent-result.json; every serialized action has a unique stable triage-security marker." + ] + }, + { + "id": 41, + "prompt": "Run triage-security for TC-8005 in an isolated Fullsend fixture sandbox using fullsend-eval-5-trusted-input.json. Before invoking the skill, actually set FULLSEND_OUTPUT_DIR to an isolated sandbox output directory outside the assigned grader outputs directory in your shell/Python execution environment (not just in prose). Copy this JSON file byte-for-byte to an isolated eval workspace .pre-script/triage-security-input.json; this is the local fixture mount mapping for /sandbox/workspace/.pre-script/triage-security-input.json. The eval runner must record invocation.json outside both the sandbox and grader outputs, including UTC clock, actual gate presence/value, mounted input path, the complete mounted scenario JSON and authorization. Set INVOCATION_PATH to this runner-owned audit file and GRADER_OUTPUTS_DIR to the assigned outputs directory; these are runner handoff variables, not skill configuration. Validate the mounted document using pre_triage_security.validate_bundle before analysis and record its actual outcome in invocation.json. Invoke the triage-security skill with this mapping and use only the validated mounted bundle for scenario evidence. Do not read local CLAUDE.md, a local matrix, or unrelated fixtures, call Jira/GitHub/WebFetch/git/cosign/network tools, execute a Jira action plan, or write security-matrix.md. Parse openssl-libs RPM lock records and the supplied SBOM comparison. Authorization is false: report the affected 2.2.0/2.2.1/2.2.2 and patched 2.2.3/2.2.4 releases with exactly one report-only action. Do not run cosign or write local evidence. Write only the final accumulated agent-result.json to $FULLSEND_OUTPUT_DIR and validate it against the unchanged triage-security-result.schema.json before finishing. Authorized mode serializes proposed actions only; it does not execute them. After the skill finishes, the eval runner must record the actual result-schema validation outcome and sandbox directory listing in invocation.json, then execute this handoff with the same actual FULLSEND_OUTPUT_DIR gate. Copy the two artifacts byte-for-byte into the grader's assigned outputs directory; do not place runner evidence in the sandbox or modify the skill result. The grader receives outputs/invocation.json and outputs/agent-result.json directly, without needing files outside outputs/. This handoff is runner work after skill execution, not a sandbox-side skill write.\n```python\nimport os\nimport shutil\nfrom pathlib import Path\n\ngrader_outputs = Path(GRADER_OUTPUTS_DIR).resolve()\nsandbox_outputs = Path(os.environ[\"FULLSEND_OUTPUT_DIR\"]).resolve()\ninvocation = Path(INVOCATION_PATH).resolve()\nassert not grader_outputs.is_relative_to(sandbox_outputs)\nassert not invocation.is_relative_to(sandbox_outputs)\nfor source in [invocation, sandbox_outputs / \"agent-result.json\"]:\n shutil.copyfile(source, grader_outputs / source.name)\n```", + "expected_output": "Independent Fullsend counterpart of interactive eval 5: Parse openssl-libs RPM lock records and the supplied SBOM comparison. Authorization is false: report the affected 2.2.0/2.2.1/2.2.2 and patched 2.2.3/2.2.4 releases with exactly one report-only action. Do not run cosign or write local evidence.", + "files": [ + "files/fullsend-eval-5-trusted-input.json" + ], + "assertions": [ + "For Fullsend RPM triage, lock and any supplied SBOM evidence are parsed only from the valid trusted bundle: the sandbox does not invoke cosign, git, Jira, or a network lookup and does not write local evidence or security-matrix.md", + "The runner records actual non-empty FULLSEND_OUTPUT_DIR gate state, clock and mounted scenario JSON in invocation.json; the input passes pre_triage_security.validate_bundle and the observable agent-result.json passes triage-security-result.schema.json with its report issue and mode bound to the trusted subject and authorization.", + "The sandbox uses only the mounted scenario evidence, performs no external operation or Jira mutation, and writes only agent-result.json; every serialized action has a unique stable triage-security marker." + ] + }, + { + "id": 42, + "prompt": "Run triage-security for TC-8010 in an isolated Fullsend fixture sandbox using fullsend-eval-8-trusted-input.json. Before invoking the skill, actually set FULLSEND_OUTPUT_DIR to an isolated sandbox output directory outside the assigned grader outputs directory in your shell/Python execution environment (not just in prose). Copy this JSON file byte-for-byte to an isolated eval workspace .pre-script/triage-security-input.json; this is the local fixture mount mapping for /sandbox/workspace/.pre-script/triage-security-input.json. The eval runner must record invocation.json outside both the sandbox and grader outputs, including UTC clock, actual gate presence/value, mounted input path, the complete mounted scenario JSON and authorization. Set INVOCATION_PATH to this runner-owned audit file and GRADER_OUTPUTS_DIR to the assigned outputs directory; these are runner handoff variables, not skill configuration. Validate the mounted document using pre_triage_security.validate_bundle before analysis and record its actual outcome in invocation.json. Invoke the triage-security skill with this mapping and use only the validated mounted bundle for scenario evidence. Do not read local CLAUDE.md, a local matrix, or unrelated fixtures, call Jira/GitHub/WebFetch/git/cosign/network tools, execute a Jira action plan, or write security-matrix.md. TC-8009 bumps axios to 1.9.0, covering the current 1.8.2 fix threshold. Serialize new Related/Depend links, the overlap evidence comment and close transition, suppressing the pre-recorded Related link marker. No new remediation is needed. Write only the final accumulated agent-result.json to $FULLSEND_OUTPUT_DIR and validate it against the unchanged triage-security-result.schema.json before finishing. Authorized mode serializes proposed actions only; it does not execute them. After the skill finishes, the eval runner must record the actual result-schema validation outcome and sandbox directory listing in invocation.json, then execute this handoff with the same actual FULLSEND_OUTPUT_DIR gate. Copy the two artifacts byte-for-byte into the grader's assigned outputs directory; do not place runner evidence in the sandbox or modify the skill result. The grader receives outputs/invocation.json and outputs/agent-result.json directly, without needing files outside outputs/. This handoff is runner work after skill execution, not a sandbox-side skill write.\n```python\nimport os\nimport shutil\nfrom pathlib import Path\n\ngrader_outputs = Path(GRADER_OUTPUTS_DIR).resolve()\nsandbox_outputs = Path(os.environ[\"FULLSEND_OUTPUT_DIR\"]).resolve()\ninvocation = Path(INVOCATION_PATH).resolve()\nassert not grader_outputs.is_relative_to(sandbox_outputs)\nassert not invocation.is_relative_to(sandbox_outputs)\nfor source in [invocation, sandbox_outputs / \"agent-result.json\"]:\n shutil.copyfile(source, grader_outputs / source.name)\n```", + "expected_output": "Independent Fullsend counterpart of interactive eval 8: TC-8009 bumps axios to 1.9.0, covering the current 1.8.2 fix threshold. Serialize new Related/Depend links, the overlap evidence comment and close transition, suppressing the pre-recorded Related link marker. No new remediation is needed.", + "files": [ + "files/fullsend-eval-8-trusted-input.json" + ], + "assertions": [ + "With authorized Fullsend input, the covered-overlap outcome serializes idempotent Related and Depend link actions, the overlap comment action, and the close transition from trusted evidence; it does not call Jira and omits actions whose markers or trusted existing links already exist", + "The runner records actual non-empty FULLSEND_OUTPUT_DIR gate state, clock and mounted scenario JSON in invocation.json; the input passes pre_triage_security.validate_bundle and the observable agent-result.json passes triage-security-result.schema.json with its report issue and mode bound to the trusted subject and authorization.", + "The sandbox uses only the mounted scenario evidence, performs no external operation or Jira mutation, and writes only agent-result.json; every serialized action has a unique stable triage-security marker." + ] + }, + { + "id": 43, + "prompt": "Run triage-security for TC-8011 in an isolated Fullsend fixture sandbox using fullsend-eval-9-trusted-input.json. Before invoking the skill, actually set FULLSEND_OUTPUT_DIR to an isolated sandbox output directory outside the assigned grader outputs directory in your shell/Python execution environment (not just in prose). Copy this JSON file byte-for-byte to an isolated eval workspace .pre-script/triage-security-input.json; this is the local fixture mount mapping for /sandbox/workspace/.pre-script/triage-security-input.json. The eval runner must record invocation.json outside both the sandbox and grader outputs, including UTC clock, actual gate presence/value, mounted input path, the complete mounted scenario JSON and authorization. Set INVOCATION_PATH to this runner-owned audit file and GRADER_OUTPUTS_DIR to the assigned outputs directory; these are runner handoff variables, not skill configuration. Validate the mounted document using pre_triage_security.validate_bundle before analysis and record its actual outcome in invocation.json. Invoke the triage-security skill with this mapping and use only the validated mounted bundle for scenario evidence. Do not read local CLAUDE.md, a local matrix, or unrelated fixtures, call Jira/GitHub/WebFetch/git/cosign/network tools, execute a Jira action plan, or write security-matrix.md. TC-8013 bumps webpack only to 5.96.1, below the current 5.98.0 threshold. Serialize new remediation tasks using stable refs before their dependent links and later comments. Keep task identities symbolic; do not reuse insufficient remediation. Write only the final accumulated agent-result.json to $FULLSEND_OUTPUT_DIR and validate it against the unchanged triage-security-result.schema.json before finishing. Authorized mode serializes proposed actions only; it does not execute them. After the skill finishes, the eval runner must record the actual result-schema validation outcome and sandbox directory listing in invocation.json, then execute this handoff with the same actual FULLSEND_OUTPUT_DIR gate. Copy the two artifacts byte-for-byte into the grader's assigned outputs directory; do not place runner evidence in the sandbox or modify the skill result. The grader receives outputs/invocation.json and outputs/agent-result.json directly, without needing files outside outputs/. This handoff is runner work after skill execution, not a sandbox-side skill write.\n```python\nimport os\nimport shutil\nfrom pathlib import Path\n\ngrader_outputs = Path(GRADER_OUTPUTS_DIR).resolve()\nsandbox_outputs = Path(os.environ[\"FULLSEND_OUTPUT_DIR\"]).resolve()\ninvocation = Path(INVOCATION_PATH).resolve()\nassert not grader_outputs.is_relative_to(sandbox_outputs)\nassert not invocation.is_relative_to(sandbox_outputs)\nfor source in [invocation, sandbox_outputs / \"agent-result.json\"]:\n shutil.copyfile(source, grader_outputs / source.name)\n```", + "expected_output": "Independent Fullsend counterpart of interactive eval 9: TC-8013 bumps webpack only to 5.96.1, below the current 5.98.0 threshold. Serialize new remediation tasks using stable refs before their dependent links and later comments. Keep task identities symbolic; do not reuse insufficient remediation.", + "files": [ + "files/fullsend-eval-9-trusted-input.json" + ], + "assertions": [ + "With authorized Fullsend input, the insufficient-overlap path serializes each new remediation-task with a stable marker and reference before its dependent links and later comments, without external calls or invented Jira keys", + "The runner records actual non-empty FULLSEND_OUTPUT_DIR gate state, clock and mounted scenario JSON in invocation.json; the input passes pre_triage_security.validate_bundle and the observable agent-result.json passes triage-security-result.schema.json with its report issue and mode bound to the trusted subject and authorization.", + "The sandbox uses only the mounted scenario evidence, performs no external operation or Jira mutation, and writes only agent-result.json; every serialized action has a unique stable triage-security marker." + ] + }, + { + "id": 44, + "prompt": "Run triage-security for TC-8021 in an isolated Fullsend fixture sandbox using fullsend-eval-11-trusted-input.json. Before invoking the skill, actually set FULLSEND_OUTPUT_DIR to an isolated sandbox output directory outside the assigned grader outputs directory in your shell/Python execution environment (not just in prose). Copy this JSON file byte-for-byte to an isolated eval workspace .pre-script/triage-security-input.json; this is the local fixture mount mapping for /sandbox/workspace/.pre-script/triage-security-input.json. The eval runner must record invocation.json outside both the sandbox and grader outputs, including UTC clock, actual gate presence/value, mounted input path, the complete mounted scenario JSON and authorization. Set INVOCATION_PATH to this runner-owned audit file and GRADER_OUTPUTS_DIR to the assigned outputs directory; these are runner handoff variables, not skill configuration. Validate the mounted document using pre_triage_security.validate_bundle before analysis and record its actual outcome in invocation.json. Invoke the triage-security skill with this mapping and use only the validated mounted bundle for scenario evidence. Do not read local CLAUDE.md, a local matrix, or unrelated fixtures, call Jira/GitHub/WebFetch/git/cosign/network tools, execute a Jira action plan, or write security-matrix.md. Reconcile trusted preemptive task TC-8022 with this 2.1.x CVE: serialize the Depend link and security-preemptive label removal on TC-8022. Suppress marker-recorded assignment and transition; create no duplicate remediation. Write only the final accumulated agent-result.json to $FULLSEND_OUTPUT_DIR and validate it against the unchanged triage-security-result.schema.json before finishing. Authorized mode serializes proposed actions only; it does not execute them. After the skill finishes, the eval runner must record the actual result-schema validation outcome and sandbox directory listing in invocation.json, then execute this handoff with the same actual FULLSEND_OUTPUT_DIR gate. Copy the two artifacts byte-for-byte into the grader's assigned outputs directory; do not place runner evidence in the sandbox or modify the skill result. The grader receives outputs/invocation.json and outputs/agent-result.json directly, without needing files outside outputs/. This handoff is runner work after skill execution, not a sandbox-side skill write.\n```python\nimport os\nimport shutil\nfrom pathlib import Path\n\ngrader_outputs = Path(GRADER_OUTPUTS_DIR).resolve()\nsandbox_outputs = Path(os.environ[\"FULLSEND_OUTPUT_DIR\"]).resolve()\ninvocation = Path(INVOCATION_PATH).resolve()\nassert not grader_outputs.is_relative_to(sandbox_outputs)\nassert not invocation.is_relative_to(sandbox_outputs)\nfor source in [invocation, sandbox_outputs / \"agent-result.json\"]:\n shutil.copyfile(source, grader_outputs / source.name)\n```", + "expected_output": "Independent Fullsend counterpart of interactive eval 11: Reconcile trusted preemptive task TC-8022 with this 2.1.x CVE: serialize the Depend link and security-preemptive label removal on TC-8022. Suppress marker-recorded assignment and transition; create no duplicate remediation.", + "files": [ + "files/fullsend-eval-11-trusted-input.json" + ], + "assertions": [ + "In authorized Fullsend reconciliation, the trusted preemptive task produces a Depend link action and security-preemptive label-removal field-edit action with stable markers; it never mutates Jira directly and omits marker-recorded ordinary actions on retry", + "The runner records actual non-empty FULLSEND_OUTPUT_DIR gate state, clock and mounted scenario JSON in invocation.json; the input passes pre_triage_security.validate_bundle and the observable agent-result.json passes triage-security-result.schema.json with its report issue and mode bound to the trusted subject and authorization.", + "The sandbox uses only the mounted scenario evidence, performs no external operation or Jira mutation, and writes only agent-result.json; every serialized action has a unique stable triage-security marker." + ] + }, + { + "id": 45, + "prompt": "Run triage-security for TC-8030 in an isolated Fullsend fixture sandbox using fullsend-eval-12-trusted-input.json. Before invoking the skill, actually set FULLSEND_OUTPUT_DIR to an isolated sandbox output directory outside the assigned grader outputs directory in your shell/Python execution environment (not just in prose). Copy this JSON file byte-for-byte to an isolated eval workspace .pre-script/triage-security-input.json; this is the local fixture mount mapping for /sandbox/workspace/.pre-script/triage-security-input.json. The eval runner must record invocation.json outside both the sandbox and grader outputs, including UTC clock, actual gate presence/value, mounted input path, the complete mounted scenario JSON and authorization. Set INVOCATION_PATH to this runner-owned audit file and GRADER_OUTPUTS_DIR to the assigned outputs directory; these are runner handoff variables, not skill configuration. Validate the mounted document using pre_triage_security.validate_bundle before analysis and record its actual outcome in invocation.json. Invoke the triage-security skill with this mapping and use only the validated mounted bundle for scenario evidence. Do not read local CLAUDE.md, a local matrix, or unrelated fixtures, call Jira/GitHub/WebFetch/git/cosign/network tools, execute a Jira action plan, or write security-matrix.md. Derive h2 fix threshold 0.4.8 from external_evidence.mitre and .osv only and cross-validate them. Describe empty or unavailable enrichment records as evidence limitations rather than fetching them. Authorization is false: report the assessment and withheld mutations in one report-only action. Write only the final accumulated agent-result.json to $FULLSEND_OUTPUT_DIR and validate it against the unchanged triage-security-result.schema.json before finishing. Authorized mode serializes proposed actions only; it does not execute them. After the skill finishes, the eval runner must record the actual result-schema validation outcome and sandbox directory listing in invocation.json, then execute this handoff with the same actual FULLSEND_OUTPUT_DIR gate. Copy the two artifacts byte-for-byte into the grader's assigned outputs directory; do not place runner evidence in the sandbox or modify the skill result. The grader receives outputs/invocation.json and outputs/agent-result.json directly, without needing files outside outputs/. This handoff is runner work after skill execution, not a sandbox-side skill write.\n```python\nimport os\nimport shutil\nfrom pathlib import Path\n\ngrader_outputs = Path(GRADER_OUTPUTS_DIR).resolve()\nsandbox_outputs = Path(os.environ[\"FULLSEND_OUTPUT_DIR\"]).resolve()\ninvocation = Path(INVOCATION_PATH).resolve()\nassert not grader_outputs.is_relative_to(sandbox_outputs)\nassert not invocation.is_relative_to(sandbox_outputs)\nfor source in [invocation, sandbox_outputs / \"agent-result.json\"]:\n shutil.copyfile(source, grader_outputs / source.name)\n```", + "expected_output": "Independent Fullsend counterpart of interactive eval 12: Derive h2 fix threshold 0.4.8 from external_evidence.mitre and .osv only and cross-validate them. Describe empty or unavailable enrichment records as evidence limitations rather than fetching them. Authorization is false: report the assessment and withheld mutations in one report-only action.", + "files": [ + "files/fullsend-eval-12-trusted-input.json" + ], + "assertions": [ + "In Fullsend mode enrichment reads MITRE and OSV responses only from valid external_evidence in the trusted input bundle; it makes no API or network request and reports a missing required enrichment record as an evidence limitation", + "The runner records actual non-empty FULLSEND_OUTPUT_DIR gate state, clock and mounted scenario JSON in invocation.json; the input passes pre_triage_security.validate_bundle and the observable agent-result.json passes triage-security-result.schema.json with its report issue and mode bound to the trusted subject and authorization.", + "The sandbox uses only the mounted scenario evidence, performs no external operation or Jira mutation, and writes only agent-result.json; every serialized action has a unique stable triage-security marker." + ] + }, + { + "id": 46, + "prompt": "Run triage-security for TC-8001 in an isolated Fullsend fixture sandbox using fullsend-eval-18-trusted-input.json. Before invoking the skill, actually set FULLSEND_OUTPUT_DIR to an isolated sandbox output directory outside the assigned grader outputs directory in your shell/Python execution environment (not just in prose). Copy this JSON file byte-for-byte to an isolated eval workspace .pre-script/triage-security-input.json; this is the local fixture mount mapping for /sandbox/workspace/.pre-script/triage-security-input.json. The eval runner must record invocation.json outside both the sandbox and grader outputs, including UTC clock, actual gate presence/value, mounted input path, the complete mounted scenario JSON and authorization. Set INVOCATION_PATH to this runner-owned audit file and GRADER_OUTPUTS_DIR to the assigned outputs directory; these are runner handoff variables, not skill configuration. Validate the mounted document using pre_triage_security.validate_bundle before analysis and record its actual outcome in invocation.json. Invoke the triage-security skill with this mapping and use only the validated mounted bundle for scenario evidence. Do not read local CLAUDE.md, a local matrix, or unrelated fixtures, call Jira/GitHub/WebFetch/git/cosign/network tools, execute a Jira action plan, or write security-matrix.md. This partial retry has two existing remediation tasks. TC-8100 has NO description digest; TC-8101 has one. Emit the planned remediation-task actions using the exact existing summaries and stable refs upstream/downstream even though their markers exist. The trusted executor must reuse them, repair TC-8100 digest once and resolve later placeholders. Omit all marker-recorded ordinary field, status, comment and existing Depend/Blocks link actions. The downstream Depend link was not yet applied: emit it after the remediation tasks using {{downstream.key}} so reference resolution is observable. Do not serialize a standalone digest comment or recreate tasks. Write only the final accumulated agent-result.json to $FULLSEND_OUTPUT_DIR and validate it against the unchanged triage-security-result.schema.json before finishing. Authorized mode serializes proposed actions only; it does not execute them. After the skill finishes, the eval runner must record the actual result-schema validation outcome and sandbox directory listing in invocation.json, then execute this handoff with the same actual FULLSEND_OUTPUT_DIR gate. Copy the two artifacts byte-for-byte into the grader's assigned outputs directory; do not place runner evidence in the sandbox or modify the skill result. The grader receives outputs/invocation.json and outputs/agent-result.json directly, without needing files outside outputs/. This handoff is runner work after skill execution, not a sandbox-side skill write.\n```python\nimport os\nimport shutil\nfrom pathlib import Path\n\ngrader_outputs = Path(GRADER_OUTPUTS_DIR).resolve()\nsandbox_outputs = Path(os.environ[\"FULLSEND_OUTPUT_DIR\"]).resolve()\ninvocation = Path(INVOCATION_PATH).resolve()\nassert not grader_outputs.is_relative_to(sandbox_outputs)\nassert not invocation.is_relative_to(sandbox_outputs)\nfor source in [invocation, sandbox_outputs / \"agent-result.json\"]:\n shutil.copyfile(source, grader_outputs / source.name)\n```", + "expected_output": "Independent Fullsend counterpart of interactive eval 18: This partial retry has two existing remediation tasks. TC-8100 has NO description digest; TC-8101 has one. Emit the planned remediation-task actions using the exact existing summaries and stable refs upstream/downstream even though their markers exist. The trusted executor must reuse them, repair TC-8100 digest once and resolve later placeholders. Omit all marker-recorded ordinary field, status, comment and existing Depend/Blocks link actions. The downstream Depend link was not yet applied: emit it after the remediation tasks using {{downstream.key}} so reference resolution is observable. Do not serialize a standalone digest comment or recreate tasks.", + "files": [ + "files/fullsend-eval-18-trusted-input.json" + ], + "assertions": [ + "On an authorized Fullsend rerun, existing idempotency.action_markers suppress duplicate ordinary field, transition, comment, and link actions, while an existing remediation task with a missing description digest is still emitted with the same remediation-task marker and ref so the executor reuses it, repairs the digest once, and resolves later placeholders without duplicate creation", + "The runner records actual non-empty FULLSEND_OUTPUT_DIR gate state, clock and mounted scenario JSON in invocation.json; the input passes pre_triage_security.validate_bundle and the observable agent-result.json passes triage-security-result.schema.json with its report issue and mode bound to the trusted subject and authorization.", + "The sandbox uses only the mounted scenario evidence, performs no external operation or Jira mutation, and writes only agent-result.json; every serialized action has a unique stable triage-security marker." + ] } ] } diff --git a/evals/triage-security/files/fullsend-eval-1-trusted-input.json b/evals/triage-security/files/fullsend-eval-1-trusted-input.json new file mode 100644 index 000000000..bbdd40824 --- /dev/null +++ b/evals/triage-security/files/fullsend-eval-1-trusted-input.json @@ -0,0 +1,336 @@ +{ + "schema_version": "1", + "issue": { + "key": "TC-8001", + "summary": "CVE-2026-31812 quinn-proto - Panic on large stream counts [rhtpa-2.2]", + "description": { + "type": "doc", + "version": 1, + "content": [ + { + "type": "paragraph", + "content": [ + { + "type": "text", + "text": "A vulnerability was found in quinn-proto. The quinn-proto crate before version 0.11.14 allows a remote attacker to cause a panic by sending a QUIC transport frame that creates an excessive number of streams. This vulnerability is classified as a denial of service (DoS).\n\n**Affected package**: quinn-proto\n**Affected versions**: versions before 0.11.14\n**Fixed version**: 0.11.14\n**CVSS**: 7.5 (High)\n\nThe vulnerability exists because quinn-proto does not properly validate the number of streams requested in a STREAMS frame. An attacker can send a specially crafted frame that causes the server to allocate an unbounded number of stream state objects, leading to a panic when the allocation exceeds internal limits.\n\n### References\n\n- https://github.com/advisories/GHSA-2026-qp73-x4mq\n- https://rustsec.org/advisories/RUSTSEC-2026-0042.html" + } + ] + } + ] + }, + "status": "New", + "labels": [ + "CVE-2026-31812", + "pscomponent:org/rhtpa-server" + ], + "versions": [], + "reporter": { + "account_id": "synthetic-reporter", + "display_name": "Synthetic Reporter" + }, + "comments": [], + "fields": { + "fixture_purpose": "SYNTHETIC TEST DATA — independent Fullsend counterpart of interactive eval 1; all identities, URLs, CVEs, commit pins and evidence are deliberate public test material, not live advisory evidence.", + "current_user": { + "accountId": "synthetic-engineer", + "displayName": "Synthetic Engineer" + }, + "ecosystem": "Cargo", + "affected_package": "quinn-proto", + "fixed_version": "0.11.14", + "affected_range": "< 0.11.14", + "affectsVersions": [ + { + "id": "version-stale", + "name": "RHTPA 2.0.0" + } + ], + "assignee": null, + "issuelinks": [], + "customfield_10632": "quinn-proto" + } + }, + "remote_links": [ + { + "url": "https://example.com/advisories/CVE-2026-31812", + "title": "Synthetic CVE-2026-31812 advisory" + } + ], + "configuration": { + "project_key": "TC", + "jira_version_prefix": "RHTPA", + "vulnerability_issue_type_id": "10016", + "component_label_pattern": "pscomponent:", + "version_streams": [ + { + "name": "2.1.x", + "matrix_path": "2.1.x/security-matrix.md", + "release_repository": "release-repo" + }, + { + "name": "2.2.x", + "matrix_path": "2.2.x/security-matrix.md", + "release_repository": "release-repo" + } + ], + "source_repositories": [ + { + "name": "backend", + "url": "https://github.com/example/backend", + "deployment_context": "customer-shipped" + } + ], + "vex_justification_field": "customfield_10665", + "upstream_affected_component_field": "customfield_10632", + "stream_field": "customfield_10832" + }, + "external_evidence": { + "mitre": { + "source_url": "https://example.com/mitre/CVE-2026-31812", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "cveMetadata": { + "cveId": "CVE-2026-31812" + }, + "containers": { + "cna": { + "affected": [ + { + "product": "quinn-proto", + "versions": [ + { + "status": "affected", + "lessThan": "0.11.14", + "versionType": "semver" + } + ] + } + ] + } + } + } + }, + "osv": { + "source_url": "https://example.com/osv/CVE-2026-31812", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "id": "CVE-2026-31812", + "affected": [ + { + "package": { + "ecosystem": "Cargo", + "name": "quinn-proto" + }, + "ranges": [ + { + "type": "SEMVER", + "events": [ + { + "introduced": "0" + }, + { + "fixed": "0.11.14" + } + ] + } + ] + } + ] + } + }, + "lifecycle": { + "source_url": "https://example.com/lifecycle", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "supported_streams": [ + "2.1.x", + "2.2.x" + ], + "eol_streams": [] + } + } + }, + "matrix": { + "streams": [ + { + "name": "2.1.x", + "matrix_source": "SYNTHETIC TEST DATA — trusted 2.1.x matrix snapshot", + "rows": [ + { + "version": "2.1.0", + "source_commits": { + "backend": "a003008" + }, + "retag_of": null + }, + { + "version": "2.1.1", + "source_commits": { + "backend": "a003012" + }, + "retag_of": null + } + ] + }, + { + "name": "2.2.x", + "matrix_source": "SYNTHETIC TEST DATA — trusted 2.2.x matrix snapshot", + "rows": [ + { + "version": "2.2.0", + "source_commits": { + "backend": "a004005" + }, + "retag_of": null + }, + { + "version": "2.2.1", + "source_commits": { + "backend": "a004008" + }, + "retag_of": null + }, + { + "version": "2.2.2", + "source_commits": { + "backend": "a004008" + }, + "retag_of": "2.2.1" + }, + { + "version": "2.2.3", + "source_commits": { + "backend": "a004011" + }, + "retag_of": null + }, + { + "version": "2.2.4", + "source_commits": { + "backend": "a004012" + }, + "retag_of": null + } + ] + } + ] + }, + "source_evidence": { + "lock_files": [ + { + "repository": "backend", + "ref": "a003008", + "path": "Cargo.lock", + "command": "git show a003008:Cargo.lock", + "content": "[[package]]\nname = \"quinn-proto\"\nversion = \"0.11.14\"\n" + }, + { + "repository": "backend", + "ref": "a003012", + "path": "Cargo.lock", + "command": "git show a003012:Cargo.lock", + "content": "[[package]]\nname = \"quinn-proto\"\nversion = \"0.11.14\"\n" + }, + { + "repository": "backend", + "ref": "a004005", + "path": "Cargo.lock", + "command": "git show a004005:Cargo.lock", + "content": "[[package]]\nname = \"quinn-proto\"\nversion = \"0.11.9\"\n" + }, + { + "repository": "backend", + "ref": "a004008", + "path": "Cargo.lock", + "command": "git show a004008:Cargo.lock", + "content": "[[package]]\nname = \"quinn-proto\"\nversion = \"0.11.12\"\n" + }, + { + "repository": "backend", + "ref": "a004011", + "path": "Cargo.lock", + "command": "git show a004011:Cargo.lock", + "content": "[[package]]\nname = \"quinn-proto\"\nversion = \"0.11.14\"\n" + }, + { + "repository": "backend", + "ref": "a004012", + "path": "Cargo.lock", + "command": "git show a004012:Cargo.lock", + "content": "[[package]]\nname = \"quinn-proto\"\nversion = \"0.11.14\"\n" + } + ], + "development_streams": [ + { + "repository": "backend", + "ref": "release/0.3.z", + "path": "Cargo.lock", + "command": "git show release/0.3.z:Cargo.lock", + "content": "[[package]]\nname = \"quinn-proto\"\nversion = \"0.11.14\"\n" + }, + { + "repository": "backend", + "ref": "release/0.4.z", + "path": "Cargo.lock", + "command": "git show release/0.4.z:Cargo.lock", + "content": "[[package]]\nname = \"quinn-proto\"\nversion = \"0.11.12\"\n" + } + ] + }, + "jira_metadata": { + "versions": [ + { + "id": "version-2-1-0", + "name": "RHTPA 2.1.0", + "released": true + }, + { + "id": "version-2-1-1", + "name": "RHTPA 2.1.1", + "released": true + }, + { + "id": "version-2-2-0", + "name": "RHTPA 2.2.0", + "released": true + }, + { + "id": "version-2-2-1", + "name": "RHTPA 2.2.1", + "released": true + }, + { + "id": "version-2-2-2", + "name": "RHTPA 2.2.2", + "released": true + }, + { + "id": "version-2-2-3", + "name": "RHTPA 2.2.3", + "released": true + }, + { + "id": "version-2-2-4", + "name": "RHTPA 2.2.4", + "released": true + }, + { + "id": "version-dev", + "name": "RHTPA 2.2.5", + "released": false + } + ], + "sibling_searches": [], + "related_issues": [] + }, + "idempotency": { + "action_markers": [], + "existing_remediation": [] + }, + "authorization": { + "mutation_authorized": true + } +} diff --git a/evals/triage-security/files/fullsend-eval-11-trusted-input.json b/evals/triage-security/files/fullsend-eval-11-trusted-input.json new file mode 100644 index 000000000..45842db53 --- /dev/null +++ b/evals/triage-security/files/fullsend-eval-11-trusted-input.json @@ -0,0 +1,379 @@ +{ + "schema_version": "1", + "issue": { + "key": "TC-8021", + "summary": "CVE-2026-55123 tokio - Use-after-free in task abort [rhtpa-2.1]", + "description": { + "type": "doc", + "version": 1, + "content": [ + { + "type": "paragraph", + "content": [ + { + "type": "text", + "text": "A vulnerability was found in the tokio crate. Versions of tokio before 1.42.0 are vulnerable to a use-after-free when a spawned task is aborted while holding a borrowed reference. This can lead to memory corruption and potential code execution.\n\n**Affected package**: tokio\n**Affected versions**: versions before 1.42.0\n**Fixed version**: 1.42.0\n**CVSS**: 8.1 (High)\n\nThis issue is scoped to stream [rhtpa-2.1]. A proactive remediation task (TC-8022) already exists for this stream, created by a prior cross-stream triage of TC-8020 (stream [rhtpa-2.2]).\n\n### Existing preemptive task (provided by Step 4.4 JQL search)\n\nA JQL search for `labels = 'security-preemptive' AND labels = 'CVE-2026-55123'` returns:\n\n- **TC-8022** — Remediate CVE-2026-55123: bump tokio to 1.42.0 (rhtpa-2.1)\n - **Status**: Open\n - **Labels**: ai-generated-jira, Security, CVE-2026-55123, security-preemptive\n - **Issue Links**:\n - **Related**: TC-8020 (originating CVE Jira, stream [rhtpa-2.2])\n\n### References\n\n- https://github.com/advisories/GHSA-2026-tk91-v5pp\n- https://rustsec.org/advisories/RUSTSEC-2026-0088.html" + } + ] + } + ] + }, + "status": "New", + "labels": [ + "CVE-2026-55123", + "pscomponent:org/rhtpa-server" + ], + "versions": [], + "reporter": { + "account_id": "synthetic-reporter", + "display_name": "Synthetic Reporter" + }, + "comments": [], + "fields": { + "fixture_purpose": "SYNTHETIC TEST DATA — independent Fullsend counterpart of interactive eval 11; all identities, URLs, CVEs, commit pins and evidence are deliberate public test material, not live advisory evidence.", + "current_user": { + "accountId": "synthetic-engineer", + "displayName": "Synthetic Engineer" + }, + "ecosystem": "Cargo", + "affected_package": "tokio", + "fixed_version": "1.42.0", + "affected_range": "< 1.42.0", + "affectsVersions": [ + { + "id": "version-stale", + "name": "RHTPA 2.0.0" + } + ], + "assignee": null, + "issuelinks": [], + "customfield_10632": "tokio" + } + }, + "remote_links": [ + { + "url": "https://example.com/advisories/CVE-2026-55123", + "title": "Synthetic CVE-2026-55123 advisory" + } + ], + "configuration": { + "project_key": "TC", + "jira_version_prefix": "RHTPA", + "vulnerability_issue_type_id": "10016", + "component_label_pattern": "pscomponent:", + "version_streams": [ + { + "name": "2.1.x", + "matrix_path": "2.1.x/security-matrix.md", + "release_repository": "release-repo" + }, + { + "name": "2.2.x", + "matrix_path": "2.2.x/security-matrix.md", + "release_repository": "release-repo" + } + ], + "source_repositories": [ + { + "name": "backend", + "url": "https://github.com/example/backend", + "deployment_context": "customer-shipped" + } + ], + "vex_justification_field": "customfield_10665", + "upstream_affected_component_field": "customfield_10632", + "stream_field": "customfield_10832" + }, + "external_evidence": { + "mitre": { + "source_url": "https://example.com/mitre/CVE-2026-55123", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "cveMetadata": { + "cveId": "CVE-2026-55123" + }, + "containers": { + "cna": { + "affected": [ + { + "product": "tokio", + "versions": [ + { + "status": "affected", + "lessThan": "1.42.0", + "versionType": "semver" + } + ] + } + ] + } + } + } + }, + "osv": { + "source_url": "https://example.com/osv/CVE-2026-55123", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "id": "CVE-2026-55123", + "affected": [ + { + "package": { + "ecosystem": "Cargo", + "name": "tokio" + }, + "ranges": [ + { + "type": "SEMVER", + "events": [ + { + "introduced": "0" + }, + { + "fixed": "1.42.0" + } + ] + } + ] + } + ] + } + }, + "lifecycle": { + "source_url": "https://example.com/lifecycle", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "supported_streams": [ + "2.1.x", + "2.2.x" + ], + "eol_streams": [] + } + } + }, + "matrix": { + "streams": [ + { + "name": "2.1.x", + "matrix_source": "SYNTHETIC TEST DATA — trusted 2.1.x matrix snapshot", + "rows": [ + { + "version": "2.1.0", + "source_commits": { + "backend": "a003008" + }, + "retag_of": null + }, + { + "version": "2.1.1", + "source_commits": { + "backend": "a003012" + }, + "retag_of": null + } + ] + }, + { + "name": "2.2.x", + "matrix_source": "SYNTHETIC TEST DATA — trusted 2.2.x matrix snapshot", + "rows": [ + { + "version": "2.2.0", + "source_commits": { + "backend": "a004005" + }, + "retag_of": null + }, + { + "version": "2.2.1", + "source_commits": { + "backend": "a004008" + }, + "retag_of": null + }, + { + "version": "2.2.2", + "source_commits": { + "backend": "a004008" + }, + "retag_of": "2.2.1" + }, + { + "version": "2.2.3", + "source_commits": { + "backend": "a004011" + }, + "retag_of": null + }, + { + "version": "2.2.4", + "source_commits": { + "backend": "a004012" + }, + "retag_of": null + } + ] + } + ] + }, + "source_evidence": { + "lock_files": [ + { + "repository": "backend", + "ref": "a003008", + "path": "Cargo.lock", + "command": "git show a003008:Cargo.lock", + "content": "[[package]]\nname = \"tokio\"\nversion = \"1.40.0\"\n" + }, + { + "repository": "backend", + "ref": "a003012", + "path": "Cargo.lock", + "command": "git show a003012:Cargo.lock", + "content": "[[package]]\nname = \"tokio\"\nversion = \"1.40.0\"\n" + }, + { + "repository": "backend", + "ref": "a004005", + "path": "Cargo.lock", + "command": "git show a004005:Cargo.lock", + "content": "[[package]]\nname = \"tokio\"\nversion = \"1.42.0\"\n" + }, + { + "repository": "backend", + "ref": "a004008", + "path": "Cargo.lock", + "command": "git show a004008:Cargo.lock", + "content": "[[package]]\nname = \"tokio\"\nversion = \"1.42.0\"\n" + }, + { + "repository": "backend", + "ref": "a004011", + "path": "Cargo.lock", + "command": "git show a004011:Cargo.lock", + "content": "[[package]]\nname = \"tokio\"\nversion = \"1.42.0\"\n" + }, + { + "repository": "backend", + "ref": "a004012", + "path": "Cargo.lock", + "command": "git show a004012:Cargo.lock", + "content": "[[package]]\nname = \"tokio\"\nversion = \"1.42.0\"\n" + } + ], + "development_streams": [ + { + "repository": "backend", + "ref": "release/0.3.z", + "path": "Cargo.lock", + "command": "git show release/0.3.z:Cargo.lock", + "content": "[[package]]\nname = \"tokio\"\nversion = \"1.42.0\"\n" + }, + { + "repository": "backend", + "ref": "release/0.4.z", + "path": "Cargo.lock", + "command": "git show release/0.4.z:Cargo.lock", + "content": "[[package]]\nname = \"tokio\"\nversion = \"1.42.0\"\n" + } + ] + }, + "jira_metadata": { + "versions": [ + { + "id": "version-2-1-0", + "name": "RHTPA 2.1.0", + "released": true + }, + { + "id": "version-2-1-1", + "name": "RHTPA 2.1.1", + "released": true + }, + { + "id": "version-2-2-0", + "name": "RHTPA 2.2.0", + "released": true + }, + { + "id": "version-2-2-1", + "name": "RHTPA 2.2.1", + "released": true + }, + { + "id": "version-2-2-2", + "name": "RHTPA 2.2.2", + "released": true + }, + { + "id": "version-2-2-3", + "name": "RHTPA 2.2.3", + "released": true + }, + { + "id": "version-2-2-4", + "name": "RHTPA 2.2.4", + "released": true + }, + { + "id": "version-dev", + "name": "RHTPA 2.2.5", + "released": false + } + ], + "sibling_searches": [ + { + "purpose": "preemptive-remediation", + "jql": "project = TC AND labels = \"security-preemptive\" AND labels = \"CVE-2026-55123\"", + "issues": [ + { + "key": "TC-8022", + "summary": "Remediate CVE-2026-55123: bump tokio to 1.42.0 (rhtpa-2.1)", + "status": "Open", + "labels": [ + "ai-generated-jira", + "Security", + "CVE-2026-55123", + "security-preemptive" + ], + "description": { + "type": "doc", + "version": 1, + "content": [ + { + "type": "paragraph", + "content": [ + { + "type": "text", + "text": "Prior cross-stream remediation from TC-8020." + } + ] + } + ] + }, + "comments": [], + "links": [ + { + "type": "Related", + "outward": "TC-8020" + } + ] + } + ] + } + ], + "related_issues": [] + }, + "idempotency": { + "action_markers": [ + "triage-security:tc-8021:assignment", + "triage-security:tc-8021:status:assigned" + ], + "existing_remediation": [] + }, + "authorization": { + "mutation_authorized": true + } +} diff --git a/evals/triage-security/files/fullsend-eval-12-trusted-input.json b/evals/triage-security/files/fullsend-eval-12-trusted-input.json new file mode 100644 index 000000000..7a74072dd --- /dev/null +++ b/evals/triage-security/files/fullsend-eval-12-trusted-input.json @@ -0,0 +1,313 @@ +{ + "schema_version": "1", + "issue": { + "key": "TC-8030", + "summary": "CVE-2026-48901 h2 - HTTP/2 CONTINUATION flood [rhtpa-2.2]", + "description": { + "type": "doc", + "version": 1, + "content": [ + { + "type": "paragraph", + "content": [ + { + "type": "text", + "text": "A vulnerability was found in h2. The h2 crate is affected by an HTTP/2 CONTINUATION flood vulnerability. An attacker can send a large number of CONTINUATION frames that causes excessive memory allocation and CPU usage.\n\n**Affected package**: h2\n**Affected versions**: versions prior to the fix\n**Fixed version**: see advisory\n**CVSS**: 7.5 (High)\n\nThe vulnerability exists because h2 does not limit the number of CONTINUATION frames that can follow a HEADERS frame. An attacker can exploit this to cause a denial of service.\n\n### References\n\n- https://github.com/advisories/GHSA-2026-r7f2-kk9p\n- https://rustsec.org/advisories/RUSTSEC-2026-0089.html" + } + ] + } + ] + }, + "status": "New", + "labels": [ + "CVE-2026-48901", + "pscomponent:org/rhtpa-server" + ], + "versions": [], + "reporter": { + "account_id": "synthetic-reporter", + "display_name": "Synthetic Reporter" + }, + "comments": [], + "fields": { + "fixture_purpose": "SYNTHETIC TEST DATA — independent Fullsend counterpart of interactive eval 12; all identities, URLs, CVEs, commit pins and evidence are deliberate public test material, not live advisory evidence.", + "current_user": { + "accountId": "synthetic-engineer", + "displayName": "Synthetic Engineer" + }, + "ecosystem": "Cargo", + "affected_package": "h2", + "fixed_version": "0.4.8", + "affected_range": "< 0.4.8", + "affectsVersions": [ + { + "id": "version-stale", + "name": "RHTPA 2.0.0" + } + ], + "assignee": null, + "issuelinks": [], + "customfield_10632": "h2" + } + }, + "remote_links": [ + { + "url": "https://example.com/advisories/CVE-2026-48901", + "title": "Synthetic CVE-2026-48901 advisory" + } + ], + "configuration": { + "project_key": "TC", + "jira_version_prefix": "RHTPA", + "vulnerability_issue_type_id": "10016", + "component_label_pattern": "pscomponent:", + "version_streams": [ + { + "name": "2.1.x", + "matrix_path": "2.1.x/security-matrix.md", + "release_repository": "release-repo" + }, + { + "name": "2.2.x", + "matrix_path": "2.2.x/security-matrix.md", + "release_repository": "release-repo" + } + ], + "source_repositories": [ + { + "name": "backend", + "url": "https://github.com/example/backend", + "deployment_context": "customer-shipped" + } + ], + "vex_justification_field": "customfield_10665", + "upstream_affected_component_field": "customfield_10632", + "stream_field": "customfield_10832" + }, + "external_evidence": { + "mitre": { + "source_url": "https://example.com/mitre/CVE-2026-48901", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "cveMetadata": { + "cveId": "CVE-2026-48901" + }, + "containers": { + "cna": { + "affected": [ + { + "product": "h2", + "versions": [ + { + "status": "affected", + "lessThan": "0.4.8", + "versionType": "semver" + } + ] + } + ] + } + } + } + }, + "osv": { + "source_url": "https://example.com/osv/CVE-2026-48901", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 503, + "body": {} + }, + "lifecycle": { + "source_url": "https://example.com/lifecycle", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "supported_streams": [ + "2.1.x", + "2.2.x" + ], + "eol_streams": [] + } + } + }, + "matrix": { + "streams": [ + { + "name": "2.1.x", + "matrix_source": "SYNTHETIC TEST DATA — trusted 2.1.x matrix snapshot", + "rows": [ + { + "version": "2.1.0", + "source_commits": { + "backend": "a003008" + }, + "retag_of": null + }, + { + "version": "2.1.1", + "source_commits": { + "backend": "a003012" + }, + "retag_of": null + } + ] + }, + { + "name": "2.2.x", + "matrix_source": "SYNTHETIC TEST DATA — trusted 2.2.x matrix snapshot", + "rows": [ + { + "version": "2.2.0", + "source_commits": { + "backend": "a004005" + }, + "retag_of": null + }, + { + "version": "2.2.1", + "source_commits": { + "backend": "a004008" + }, + "retag_of": null + }, + { + "version": "2.2.2", + "source_commits": { + "backend": "a004008" + }, + "retag_of": "2.2.1" + }, + { + "version": "2.2.3", + "source_commits": { + "backend": "a004011" + }, + "retag_of": null + }, + { + "version": "2.2.4", + "source_commits": { + "backend": "a004012" + }, + "retag_of": null + } + ] + } + ] + }, + "source_evidence": { + "lock_files": [ + { + "repository": "backend", + "ref": "a003008", + "path": "Cargo.lock", + "command": "git show a003008:Cargo.lock", + "content": "[[package]]\nname = \"h2\"\nversion = \"0.4.5\"\n" + }, + { + "repository": "backend", + "ref": "a003012", + "path": "Cargo.lock", + "command": "git show a003012:Cargo.lock", + "content": "[[package]]\nname = \"h2\"\nversion = \"0.4.5\"\n" + }, + { + "repository": "backend", + "ref": "a004005", + "path": "Cargo.lock", + "command": "git show a004005:Cargo.lock", + "content": "[[package]]\nname = \"h2\"\nversion = \"0.4.8\"\n" + }, + { + "repository": "backend", + "ref": "a004008", + "path": "Cargo.lock", + "command": "git show a004008:Cargo.lock", + "content": "[[package]]\nname = \"h2\"\nversion = \"0.4.8\"\n" + }, + { + "repository": "backend", + "ref": "a004011", + "path": "Cargo.lock", + "command": "git show a004011:Cargo.lock", + "content": "[[package]]\nname = \"h2\"\nversion = \"0.4.9\"\n" + }, + { + "repository": "backend", + "ref": "a004012", + "path": "Cargo.lock", + "command": "git show a004012:Cargo.lock", + "content": "[[package]]\nname = \"h2\"\nversion = \"0.4.9\"\n" + } + ], + "development_streams": [ + { + "repository": "backend", + "ref": "release/0.3.z", + "path": "Cargo.lock", + "command": "git show release/0.3.z:Cargo.lock", + "content": "[[package]]\nname = \"h2\"\nversion = \"0.4.8\"\n" + }, + { + "repository": "backend", + "ref": "release/0.4.z", + "path": "Cargo.lock", + "command": "git show release/0.4.z:Cargo.lock", + "content": "[[package]]\nname = \"h2\"\nversion = \"0.4.8\"\n" + } + ] + }, + "jira_metadata": { + "versions": [ + { + "id": "version-2-1-0", + "name": "RHTPA 2.1.0", + "released": true + }, + { + "id": "version-2-1-1", + "name": "RHTPA 2.1.1", + "released": true + }, + { + "id": "version-2-2-0", + "name": "RHTPA 2.2.0", + "released": true + }, + { + "id": "version-2-2-1", + "name": "RHTPA 2.2.1", + "released": true + }, + { + "id": "version-2-2-2", + "name": "RHTPA 2.2.2", + "released": true + }, + { + "id": "version-2-2-3", + "name": "RHTPA 2.2.3", + "released": true + }, + { + "id": "version-2-2-4", + "name": "RHTPA 2.2.4", + "released": true + }, + { + "id": "version-dev", + "name": "RHTPA 2.2.5", + "released": false + } + ], + "sibling_searches": [], + "related_issues": [] + }, + "idempotency": { + "action_markers": [], + "existing_remediation": [] + }, + "authorization": { + "mutation_authorized": false + } +} diff --git a/evals/triage-security/files/fullsend-eval-18-trusted-input.json b/evals/triage-security/files/fullsend-eval-18-trusted-input.json new file mode 100644 index 000000000..d033b9fef --- /dev/null +++ b/evals/triage-security/files/fullsend-eval-18-trusted-input.json @@ -0,0 +1,461 @@ +{ + "schema_version": "1", + "issue": { + "key": "TC-8001", + "summary": "CVE-2026-31812 quinn-proto - Panic on large stream counts [rhtpa-2.2]", + "description": { + "type": "doc", + "version": 1, + "content": [ + { + "type": "paragraph", + "content": [ + { + "type": "text", + "text": "A vulnerability was found in quinn-proto. The quinn-proto crate before version 0.11.14 allows a remote attacker to cause a panic by sending a QUIC transport frame that creates an excessive number of streams. This vulnerability is classified as a denial of service (DoS).\n\n**Affected package**: quinn-proto\n**Affected versions**: versions before 0.11.14\n**Fixed version**: 0.11.14\n**CVSS**: 7.5 (High)\n\nThe vulnerability exists because quinn-proto does not properly validate the number of streams requested in a STREAMS frame. An attacker can send a specially crafted frame that causes the server to allocate an unbounded number of stream state objects, leading to a panic when the allocation exceeds internal limits.\n\n### References\n\n- https://github.com/advisories/GHSA-2026-qp73-x4mq\n- https://rustsec.org/advisories/RUSTSEC-2026-0042.html" + } + ] + } + ] + }, + "status": "In Progress", + "labels": [ + "CVE-2026-31812", + "pscomponent:org/rhtpa-server", + "ai-cve-triaged" + ], + "versions": [], + "reporter": { + "account_id": "synthetic-reporter", + "display_name": "Synthetic Reporter" + }, + "comments": [ + { + "body": { + "type": "doc", + "version": 1, + "content": [ + { + "type": "paragraph", + "content": [ + { + "type": "text", + "text": "Triage summary already posted." + } + ] + } + ] + } + } + ], + "fields": { + "fixture_purpose": "SYNTHETIC TEST DATA — independent Fullsend counterpart of interactive eval 18; all identities, URLs, CVEs, commit pins and evidence are deliberate public test material, not live advisory evidence.", + "current_user": { + "accountId": "synthetic-engineer", + "displayName": "Synthetic Engineer" + }, + "ecosystem": "Cargo", + "affected_package": "quinn-proto", + "fixed_version": "0.11.14", + "affected_range": "< 0.11.14", + "affectsVersions": [ + { + "id": "version-2-2-0", + "name": "RHTPA 2.2.0", + "released": true + }, + { + "id": "version-2-2-1", + "name": "RHTPA 2.2.1", + "released": true + }, + { + "id": "version-2-2-2", + "name": "RHTPA 2.2.2", + "released": true + } + ], + "assignee": { + "accountId": "synthetic-engineer" + }, + "issuelinks": [ + { + "type": { + "name": "Depend" + }, + "outwardIssue": { + "key": "TC-8100" + } + } + ], + "customfield_10632": "quinn-proto" + } + }, + "remote_links": [ + { + "url": "https://example.com/advisories/CVE-2026-31812", + "title": "Synthetic CVE-2026-31812 advisory" + } + ], + "configuration": { + "project_key": "TC", + "jira_version_prefix": "RHTPA", + "vulnerability_issue_type_id": "10016", + "component_label_pattern": "pscomponent:", + "version_streams": [ + { + "name": "2.1.x", + "matrix_path": "2.1.x/security-matrix.md", + "release_repository": "release-repo" + }, + { + "name": "2.2.x", + "matrix_path": "2.2.x/security-matrix.md", + "release_repository": "release-repo" + } + ], + "source_repositories": [ + { + "name": "backend", + "url": "https://github.com/example/backend", + "deployment_context": "customer-shipped" + } + ], + "vex_justification_field": "customfield_10665", + "upstream_affected_component_field": "customfield_10632", + "stream_field": "customfield_10832" + }, + "external_evidence": { + "mitre": { + "source_url": "https://example.com/mitre/CVE-2026-31812", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "cveMetadata": { + "cveId": "CVE-2026-31812" + }, + "containers": { + "cna": { + "affected": [ + { + "product": "quinn-proto", + "versions": [ + { + "status": "affected", + "lessThan": "0.11.14", + "versionType": "semver" + } + ] + } + ] + } + } + } + }, + "osv": { + "source_url": "https://example.com/osv/CVE-2026-31812", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "id": "CVE-2026-31812", + "affected": [ + { + "package": { + "ecosystem": "Cargo", + "name": "quinn-proto" + }, + "ranges": [ + { + "type": "SEMVER", + "events": [ + { + "introduced": "0" + }, + { + "fixed": "0.11.14" + } + ] + } + ] + } + ] + } + }, + "lifecycle": { + "source_url": "https://example.com/lifecycle", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "supported_streams": [ + "2.1.x", + "2.2.x" + ], + "eol_streams": [] + } + } + }, + "matrix": { + "streams": [ + { + "name": "2.1.x", + "matrix_source": "SYNTHETIC TEST DATA — trusted 2.1.x matrix snapshot", + "rows": [ + { + "version": "2.1.0", + "source_commits": { + "backend": "a003008" + }, + "retag_of": null + }, + { + "version": "2.1.1", + "source_commits": { + "backend": "a003012" + }, + "retag_of": null + } + ] + }, + { + "name": "2.2.x", + "matrix_source": "SYNTHETIC TEST DATA — trusted 2.2.x matrix snapshot", + "rows": [ + { + "version": "2.2.0", + "source_commits": { + "backend": "a004005" + }, + "retag_of": null + }, + { + "version": "2.2.1", + "source_commits": { + "backend": "a004008" + }, + "retag_of": null + }, + { + "version": "2.2.2", + "source_commits": { + "backend": "a004008" + }, + "retag_of": "2.2.1" + }, + { + "version": "2.2.3", + "source_commits": { + "backend": "a004011" + }, + "retag_of": null + }, + { + "version": "2.2.4", + "source_commits": { + "backend": "a004012" + }, + "retag_of": null + } + ] + } + ] + }, + "source_evidence": { + "lock_files": [ + { + "repository": "backend", + "ref": "a003008", + "path": "Cargo.lock", + "command": "git show a003008:Cargo.lock", + "content": "[[package]]\nname = \"quinn-proto\"\nversion = \"0.11.14\"\n" + }, + { + "repository": "backend", + "ref": "a003012", + "path": "Cargo.lock", + "command": "git show a003012:Cargo.lock", + "content": "[[package]]\nname = \"quinn-proto\"\nversion = \"0.11.14\"\n" + }, + { + "repository": "backend", + "ref": "a004005", + "path": "Cargo.lock", + "command": "git show a004005:Cargo.lock", + "content": "[[package]]\nname = \"quinn-proto\"\nversion = \"0.11.9\"\n" + }, + { + "repository": "backend", + "ref": "a004008", + "path": "Cargo.lock", + "command": "git show a004008:Cargo.lock", + "content": "[[package]]\nname = \"quinn-proto\"\nversion = \"0.11.12\"\n" + }, + { + "repository": "backend", + "ref": "a004011", + "path": "Cargo.lock", + "command": "git show a004011:Cargo.lock", + "content": "[[package]]\nname = \"quinn-proto\"\nversion = \"0.11.14\"\n" + }, + { + "repository": "backend", + "ref": "a004012", + "path": "Cargo.lock", + "command": "git show a004012:Cargo.lock", + "content": "[[package]]\nname = \"quinn-proto\"\nversion = \"0.11.14\"\n" + } + ], + "development_streams": [ + { + "repository": "backend", + "ref": "release/0.3.z", + "path": "Cargo.lock", + "command": "git show release/0.3.z:Cargo.lock", + "content": "[[package]]\nname = \"quinn-proto\"\nversion = \"0.11.14\"\n" + }, + { + "repository": "backend", + "ref": "release/0.4.z", + "path": "Cargo.lock", + "command": "git show release/0.4.z:Cargo.lock", + "content": "[[package]]\nname = \"quinn-proto\"\nversion = \"0.11.14\"\n" + } + ] + }, + "jira_metadata": { + "versions": [ + { + "id": "version-2-1-0", + "name": "RHTPA 2.1.0", + "released": true + }, + { + "id": "version-2-1-1", + "name": "RHTPA 2.1.1", + "released": true + }, + { + "id": "version-2-2-0", + "name": "RHTPA 2.2.0", + "released": true + }, + { + "id": "version-2-2-1", + "name": "RHTPA 2.2.1", + "released": true + }, + { + "id": "version-2-2-2", + "name": "RHTPA 2.2.2", + "released": true + }, + { + "id": "version-2-2-3", + "name": "RHTPA 2.2.3", + "released": true + }, + { + "id": "version-2-2-4", + "name": "RHTPA 2.2.4", + "released": true + }, + { + "id": "version-dev", + "name": "RHTPA 2.2.5", + "released": false + } + ], + "sibling_searches": [], + "related_issues": [] + }, + "idempotency": { + "action_markers": [ + "triage-security:tc-8001:assignment", + "triage-security:tc-8001:affects-versions", + "triage-security:tc-8001:labels", + "triage-security:tc-8001:status:assigned", + "triage-security:tc-8001:status:in-progress", + "triage-security:tc-8001:comment:summary", + "triage-security:tc-8001:remediation:upstream", + "triage-security:tc-8001:remediation:downstream", + "triage-security:tc-8001:link:depend:upstream", + "triage-security:tc-8001:link:blocks:upstream-downstream" + ], + "existing_remediation": [ + { + "key": "TC-8100", + "summary": "Remediate CVE-2026-31812: upstream quinn-proto >= 0.11.14 [rhtpa-2.2]", + "status": "In Progress", + "labels": [ + "ai-generated-jira", + "Security", + "CVE-2026-31812" + ], + "description": { + "type": "doc", + "version": 1, + "content": [ + { + "type": "paragraph", + "content": [ + { + "type": "text", + "text": "Existing upstream remediation task." + } + ] + } + ] + }, + "comments": [], + "links": [] + }, + { + "key": "TC-8101", + "summary": "Remediate CVE-2026-31812: downstream quinn-proto >= 0.11.14 [rhtpa-2.2]", + "status": "In Progress", + "labels": [ + "ai-generated-jira", + "Security", + "CVE-2026-31812" + ], + "description": { + "type": "doc", + "version": 1, + "content": [ + { + "type": "paragraph", + "content": [ + { + "type": "text", + "text": "Existing downstream remediation task." + } + ] + } + ] + }, + "comments": [ + { + "body": { + "type": "doc", + "version": 1, + "content": [ + { + "type": "paragraph", + "content": [ + { + "type": "text", + "text": "[sdlc-workflow] Description digest: sha256-adf:aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" + } + ] + } + ] + } + } + ], + "links": [] + } + ] + }, + "authorization": { + "mutation_authorized": true + } +} diff --git a/evals/triage-security/files/fullsend-eval-2-trusted-input.json b/evals/triage-security/files/fullsend-eval-2-trusted-input.json new file mode 100644 index 000000000..0d6ca7c96 --- /dev/null +++ b/evals/triage-security/files/fullsend-eval-2-trusted-input.json @@ -0,0 +1,336 @@ +{ + "schema_version": "1", + "issue": { + "key": "TC-8002", + "summary": "CVE-2026-28940 serde_json - Stack overflow on deeply nested input [rhtpa-2.2]", + "description": { + "type": "doc", + "version": 1, + "content": [ + { + "type": "paragraph", + "content": [ + { + "type": "text", + "text": "A vulnerability was found in serde_json. Versions of serde_json before 1.0.135 are vulnerable to a stack overflow when deserializing deeply nested JSON input. An attacker can craft a JSON payload with thousands of nested arrays or objects that causes unbounded recursion during deserialization, leading to a stack overflow and process crash.\n\n**Affected package**: serde_json\n**Affected versions**: versions before 1.0.135\n**Fixed version**: 1.0.135\n**CVSS**: 5.3 (Medium)\n\nThe fix introduces a configurable recursion limit that defaults to 128 levels of nesting.\n\n### References\n\n- https://github.com/advisories/GHSA-2026-j9r2-m5vk\n- https://rustsec.org/advisories/RUSTSEC-2026-0019.html" + } + ] + } + ] + }, + "status": "New", + "labels": [ + "CVE-2026-28940", + "pscomponent:org/rhtpa-server" + ], + "versions": [], + "reporter": { + "account_id": "synthetic-reporter", + "display_name": "Synthetic Reporter" + }, + "comments": [], + "fields": { + "fixture_purpose": "SYNTHETIC TEST DATA — independent Fullsend counterpart of interactive eval 2; all identities, URLs, CVEs, commit pins and evidence are deliberate public test material, not live advisory evidence.", + "current_user": { + "accountId": "synthetic-engineer", + "displayName": "Synthetic Engineer" + }, + "ecosystem": "Cargo", + "affected_package": "serde_json", + "fixed_version": "1.0.135", + "affected_range": "< 1.0.135", + "affectsVersions": [ + { + "id": "version-stale", + "name": "RHTPA 2.0.0" + } + ], + "assignee": null, + "issuelinks": [], + "customfield_10632": "serde_json" + } + }, + "remote_links": [ + { + "url": "https://example.com/advisories/CVE-2026-28940", + "title": "Synthetic CVE-2026-28940 advisory" + } + ], + "configuration": { + "project_key": "TC", + "jira_version_prefix": "RHTPA", + "vulnerability_issue_type_id": "10016", + "component_label_pattern": "pscomponent:", + "version_streams": [ + { + "name": "2.1.x", + "matrix_path": "2.1.x/security-matrix.md", + "release_repository": "release-repo" + }, + { + "name": "2.2.x", + "matrix_path": "2.2.x/security-matrix.md", + "release_repository": "release-repo" + } + ], + "source_repositories": [ + { + "name": "backend", + "url": "https://github.com/example/backend", + "deployment_context": "customer-shipped" + } + ], + "vex_justification_field": "customfield_10665", + "upstream_affected_component_field": "customfield_10632", + "stream_field": "customfield_10832" + }, + "external_evidence": { + "mitre": { + "source_url": "https://example.com/mitre/CVE-2026-28940", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "cveMetadata": { + "cveId": "CVE-2026-28940" + }, + "containers": { + "cna": { + "affected": [ + { + "product": "serde_json", + "versions": [ + { + "status": "affected", + "lessThan": "1.0.135", + "versionType": "semver" + } + ] + } + ] + } + } + } + }, + "osv": { + "source_url": "https://example.com/osv/CVE-2026-28940", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "id": "CVE-2026-28940", + "affected": [ + { + "package": { + "ecosystem": "Cargo", + "name": "serde_json" + }, + "ranges": [ + { + "type": "SEMVER", + "events": [ + { + "introduced": "0" + }, + { + "fixed": "1.0.135" + } + ] + } + ] + } + ] + } + }, + "lifecycle": { + "source_url": "https://example.com/lifecycle", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "supported_streams": [ + "2.1.x", + "2.2.x" + ], + "eol_streams": [] + } + } + }, + "matrix": { + "streams": [ + { + "name": "2.1.x", + "matrix_source": "SYNTHETIC TEST DATA — trusted 2.1.x matrix snapshot", + "rows": [ + { + "version": "2.1.0", + "source_commits": { + "backend": "a003008" + }, + "retag_of": null + }, + { + "version": "2.1.1", + "source_commits": { + "backend": "a003012" + }, + "retag_of": null + } + ] + }, + { + "name": "2.2.x", + "matrix_source": "SYNTHETIC TEST DATA — trusted 2.2.x matrix snapshot", + "rows": [ + { + "version": "2.2.0", + "source_commits": { + "backend": "a004005" + }, + "retag_of": null + }, + { + "version": "2.2.1", + "source_commits": { + "backend": "a004008" + }, + "retag_of": null + }, + { + "version": "2.2.2", + "source_commits": { + "backend": "a004008" + }, + "retag_of": "2.2.1" + }, + { + "version": "2.2.3", + "source_commits": { + "backend": "a004011" + }, + "retag_of": null + }, + { + "version": "2.2.4", + "source_commits": { + "backend": "a004012" + }, + "retag_of": null + } + ] + } + ] + }, + "source_evidence": { + "lock_files": [ + { + "repository": "backend", + "ref": "a003008", + "path": "Cargo.lock", + "command": "git show a003008:Cargo.lock", + "content": "[[package]]\nname = \"serde_json\"\nversion = \"1.0.137\"\n" + }, + { + "repository": "backend", + "ref": "a003012", + "path": "Cargo.lock", + "command": "git show a003012:Cargo.lock", + "content": "[[package]]\nname = \"serde_json\"\nversion = \"1.0.137\"\n" + }, + { + "repository": "backend", + "ref": "a004005", + "path": "Cargo.lock", + "command": "git show a004005:Cargo.lock", + "content": "[[package]]\nname = \"serde_json\"\nversion = \"1.0.138\"\n" + }, + { + "repository": "backend", + "ref": "a004008", + "path": "Cargo.lock", + "command": "git show a004008:Cargo.lock", + "content": "[[package]]\nname = \"serde_json\"\nversion = \"1.0.138\"\n" + }, + { + "repository": "backend", + "ref": "a004011", + "path": "Cargo.lock", + "command": "git show a004011:Cargo.lock", + "content": "[[package]]\nname = \"serde_json\"\nversion = \"1.0.139\"\n" + }, + { + "repository": "backend", + "ref": "a004012", + "path": "Cargo.lock", + "command": "git show a004012:Cargo.lock", + "content": "[[package]]\nname = \"serde_json\"\nversion = \"1.0.139\"\n" + } + ], + "development_streams": [ + { + "repository": "backend", + "ref": "release/0.3.z", + "path": "Cargo.lock", + "command": "git show release/0.3.z:Cargo.lock", + "content": "[[package]]\nname = \"serde_json\"\nversion = \"1.0.135\"\n" + }, + { + "repository": "backend", + "ref": "release/0.4.z", + "path": "Cargo.lock", + "command": "git show release/0.4.z:Cargo.lock", + "content": "[[package]]\nname = \"serde_json\"\nversion = \"1.0.135\"\n" + } + ] + }, + "jira_metadata": { + "versions": [ + { + "id": "version-2-1-0", + "name": "RHTPA 2.1.0", + "released": true + }, + { + "id": "version-2-1-1", + "name": "RHTPA 2.1.1", + "released": true + }, + { + "id": "version-2-2-0", + "name": "RHTPA 2.2.0", + "released": true + }, + { + "id": "version-2-2-1", + "name": "RHTPA 2.2.1", + "released": true + }, + { + "id": "version-2-2-2", + "name": "RHTPA 2.2.2", + "released": true + }, + { + "id": "version-2-2-3", + "name": "RHTPA 2.2.3", + "released": true + }, + { + "id": "version-2-2-4", + "name": "RHTPA 2.2.4", + "released": true + }, + { + "id": "version-dev", + "name": "RHTPA 2.2.5", + "released": false + } + ], + "sibling_searches": [], + "related_issues": [] + }, + "idempotency": { + "action_markers": [], + "existing_remediation": [] + }, + "authorization": { + "mutation_authorized": false + } +} diff --git a/evals/triage-security/files/fullsend-eval-3-trusted-input.json b/evals/triage-security/files/fullsend-eval-3-trusted-input.json new file mode 100644 index 000000000..d6009f170 --- /dev/null +++ b/evals/triage-security/files/fullsend-eval-3-trusted-input.json @@ -0,0 +1,368 @@ +{ + "schema_version": "1", + "issue": { + "key": "TC-8003", + "summary": "CVE-2026-31812 quinn-proto - Panic on large stream counts [rhtpa-2.2]", + "description": { + "type": "doc", + "version": 1, + "content": [ + { + "type": "paragraph", + "content": [ + { + "type": "text", + "text": "A vulnerability was found in quinn-proto. The quinn-proto crate before version 0.11.14 allows a remote attacker to cause a panic by sending a QUIC transport frame that creates an excessive number of streams. This vulnerability is classified as a denial of service (DoS).\n\n**Affected package**: quinn-proto\n**Affected versions**: versions before 0.11.14\n**Fixed version**: 0.11.14\n**CVSS**: 7.5 (High)\n\nThe vulnerability exists because quinn-proto does not properly validate the number of streams requested in a STREAMS frame. An attacker can send a specially crafted frame that causes the server to allocate an unbounded number of stream state objects, leading to a panic when the allocation exceeds internal limits.\n\n### References\n\n- https://github.com/advisories/GHSA-2026-qp73-x4mq\n- https://rustsec.org/advisories/RUSTSEC-2026-0042.html\n\n### Sibling Issues (provided by JQL search)\n\nThe following sibling issue exists for the same CVE:\n\n- **TC-7999** — CVE-2026-31812 quinn-proto - Panic on large stream counts [rhtpa-2.2]\n - **Status**: In Progress\n - **Labels**: CVE-2026-31812, pscomponent:org/rhtpa-server\n - **Affects Versions**: RHTPA 2.2.0, RHTPA 2.2.1\n - **Stream suffix**: [rhtpa-2.2] (same stream as TC-8003)" + } + ] + } + ] + }, + "status": "New", + "labels": [ + "CVE-2026-31812", + "pscomponent:org/rhtpa-server" + ], + "versions": [], + "reporter": { + "account_id": "synthetic-reporter", + "display_name": "Synthetic Reporter" + }, + "comments": [], + "fields": { + "fixture_purpose": "SYNTHETIC TEST DATA — independent Fullsend counterpart of interactive eval 3; all identities, URLs, CVEs, commit pins and evidence are deliberate public test material, not live advisory evidence.", + "current_user": { + "accountId": "synthetic-engineer", + "displayName": "Synthetic Engineer" + }, + "ecosystem": "Cargo", + "affected_package": "quinn-proto", + "fixed_version": "0.11.14", + "affected_range": "< 0.11.14", + "affectsVersions": [ + { + "id": "version-stale", + "name": "RHTPA 2.0.0" + } + ], + "assignee": null, + "issuelinks": [], + "customfield_10632": "quinn-proto" + } + }, + "remote_links": [ + { + "url": "https://example.com/advisories/CVE-2026-31812", + "title": "Synthetic CVE-2026-31812 advisory" + } + ], + "configuration": { + "project_key": "TC", + "jira_version_prefix": "RHTPA", + "vulnerability_issue_type_id": "10016", + "component_label_pattern": "pscomponent:", + "version_streams": [ + { + "name": "2.1.x", + "matrix_path": "2.1.x/security-matrix.md", + "release_repository": "release-repo" + }, + { + "name": "2.2.x", + "matrix_path": "2.2.x/security-matrix.md", + "release_repository": "release-repo" + } + ], + "source_repositories": [ + { + "name": "backend", + "url": "https://github.com/example/backend", + "deployment_context": "customer-shipped" + } + ], + "vex_justification_field": "customfield_10665", + "upstream_affected_component_field": "customfield_10632", + "stream_field": "customfield_10832" + }, + "external_evidence": { + "mitre": { + "source_url": "https://example.com/mitre/CVE-2026-31812", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "cveMetadata": { + "cveId": "CVE-2026-31812" + }, + "containers": { + "cna": { + "affected": [ + { + "product": "quinn-proto", + "versions": [ + { + "status": "affected", + "lessThan": "0.11.14", + "versionType": "semver" + } + ] + } + ] + } + } + } + }, + "osv": { + "source_url": "https://example.com/osv/CVE-2026-31812", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "id": "CVE-2026-31812", + "affected": [ + { + "package": { + "ecosystem": "Cargo", + "name": "quinn-proto" + }, + "ranges": [ + { + "type": "SEMVER", + "events": [ + { + "introduced": "0" + }, + { + "fixed": "0.11.14" + } + ] + } + ] + } + ] + } + }, + "lifecycle": { + "source_url": "https://example.com/lifecycle", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "supported_streams": [ + "2.1.x", + "2.2.x" + ], + "eol_streams": [] + } + } + }, + "matrix": { + "streams": [ + { + "name": "2.1.x", + "matrix_source": "SYNTHETIC TEST DATA — trusted 2.1.x matrix snapshot", + "rows": [ + { + "version": "2.1.0", + "source_commits": { + "backend": "a003008" + }, + "retag_of": null + }, + { + "version": "2.1.1", + "source_commits": { + "backend": "a003012" + }, + "retag_of": null + } + ] + }, + { + "name": "2.2.x", + "matrix_source": "SYNTHETIC TEST DATA — trusted 2.2.x matrix snapshot", + "rows": [ + { + "version": "2.2.0", + "source_commits": { + "backend": "a004005" + }, + "retag_of": null + }, + { + "version": "2.2.1", + "source_commits": { + "backend": "a004008" + }, + "retag_of": null + }, + { + "version": "2.2.2", + "source_commits": { + "backend": "a004008" + }, + "retag_of": "2.2.1" + }, + { + "version": "2.2.3", + "source_commits": { + "backend": "a004011" + }, + "retag_of": null + }, + { + "version": "2.2.4", + "source_commits": { + "backend": "a004012" + }, + "retag_of": null + } + ] + } + ] + }, + "source_evidence": { + "lock_files": [ + { + "repository": "backend", + "ref": "a003008", + "path": "Cargo.lock", + "command": "git show a003008:Cargo.lock", + "content": "[[package]]\nname = \"quinn-proto\"\nversion = \"0.11.9\"\n" + }, + { + "repository": "backend", + "ref": "a003012", + "path": "Cargo.lock", + "command": "git show a003012:Cargo.lock", + "content": "[[package]]\nname = \"quinn-proto\"\nversion = \"0.11.9\"\n" + }, + { + "repository": "backend", + "ref": "a004005", + "path": "Cargo.lock", + "command": "git show a004005:Cargo.lock", + "content": "[[package]]\nname = \"quinn-proto\"\nversion = \"0.11.9\"\n" + }, + { + "repository": "backend", + "ref": "a004008", + "path": "Cargo.lock", + "command": "git show a004008:Cargo.lock", + "content": "[[package]]\nname = \"quinn-proto\"\nversion = \"0.11.12\"\n" + }, + { + "repository": "backend", + "ref": "a004011", + "path": "Cargo.lock", + "command": "git show a004011:Cargo.lock", + "content": "[[package]]\nname = \"quinn-proto\"\nversion = \"0.11.14\"\n" + }, + { + "repository": "backend", + "ref": "a004012", + "path": "Cargo.lock", + "command": "git show a004012:Cargo.lock", + "content": "[[package]]\nname = \"quinn-proto\"\nversion = \"0.11.14\"\n" + } + ], + "development_streams": [ + { + "repository": "backend", + "ref": "release/0.3.z", + "path": "Cargo.lock", + "command": "git show release/0.3.z:Cargo.lock", + "content": "[[package]]\nname = \"quinn-proto\"\nversion = \"0.11.14\"\n" + }, + { + "repository": "backend", + "ref": "release/0.4.z", + "path": "Cargo.lock", + "command": "git show release/0.4.z:Cargo.lock", + "content": "[[package]]\nname = \"quinn-proto\"\nversion = \"0.11.14\"\n" + } + ] + }, + "jira_metadata": { + "versions": [ + { + "id": "version-2-1-0", + "name": "RHTPA 2.1.0", + "released": true + }, + { + "id": "version-2-1-1", + "name": "RHTPA 2.1.1", + "released": true + }, + { + "id": "version-2-2-0", + "name": "RHTPA 2.2.0", + "released": true + }, + { + "id": "version-2-2-1", + "name": "RHTPA 2.2.1", + "released": true + }, + { + "id": "version-2-2-2", + "name": "RHTPA 2.2.2", + "released": true + }, + { + "id": "version-2-2-3", + "name": "RHTPA 2.2.3", + "released": true + }, + { + "id": "version-2-2-4", + "name": "RHTPA 2.2.4", + "released": true + }, + { + "id": "version-dev", + "name": "RHTPA 2.2.5", + "released": false + } + ], + "sibling_searches": [ + { + "purpose": "same-cve-siblings", + "jql": "project = TC AND labels = \"CVE-2026-31812\" AND key != \"TC-8003\"", + "issues": [ + { + "key": "TC-7999", + "summary": "CVE-2026-31812 quinn-proto [rhtpa-2.2]", + "status": "In Progress", + "labels": [ + "CVE-2026-31812" + ], + "description": { + "type": "doc", + "version": 1, + "content": [ + { + "type": "paragraph", + "content": [ + { + "type": "text", + "text": "Same CVE, same 2.2.x stream; Affects Versions RHTPA 2.2.0, RHTPA 2.2.1." + } + ] + } + ] + }, + "comments": [], + "links": [] + } + ] + } + ], + "related_issues": [] + }, + "idempotency": { + "action_markers": [], + "existing_remediation": [] + }, + "authorization": { + "mutation_authorized": true + } +} diff --git a/evals/triage-security/files/fullsend-eval-4-trusted-input.json b/evals/triage-security/files/fullsend-eval-4-trusted-input.json new file mode 100644 index 000000000..b30800f63 --- /dev/null +++ b/evals/triage-security/files/fullsend-eval-4-trusted-input.json @@ -0,0 +1,336 @@ +{ + "schema_version": "1", + "issue": { + "key": "TC-8004", + "summary": "CVE-2026-33501 h2 - Memory exhaustion via CONTINUATION frames", + "description": { + "type": "doc", + "version": 1, + "content": [ + { + "type": "paragraph", + "content": [ + { + "type": "text", + "text": "A vulnerability was found in the h2 crate. Versions of h2 before 0.4.8 are vulnerable to memory exhaustion caused by a peer sending an excessive number of CONTINUATION frames following a HEADERS frame. The h2 library accumulates all CONTINUATION frame data without enforcing a size limit on the accumulated header block, allowing an attacker to consume unbounded memory on the server.\n\n**Affected package**: h2\n**Affected versions**: versions before 0.4.8\n**Fixed version**: 0.4.8\n**CVSS**: 7.5 (High)\n\nThis issue is distinct from CVE-2024-2758 (httpd CONTINUATION flood) — this CVE specifically affects the Rust h2 library's header accumulation logic. The fix adds a configurable maximum header list size that defaults to 16 KiB.\n\nNote: This issue has NO stream suffix in the summary — it is unscoped and covers all streams. The version impact analysis should check all streams and create remediation only for actually affected streams.\n\n### References\n\n- https://github.com/advisories/GHSA-2026-kv8p-r3n7\n- https://rustsec.org/advisories/RUSTSEC-2026-0055.html" + } + ] + } + ] + }, + "status": "New", + "labels": [ + "CVE-2026-33501", + "pscomponent:org/rhtpa-server" + ], + "versions": [], + "reporter": { + "account_id": "synthetic-reporter", + "display_name": "Synthetic Reporter" + }, + "comments": [], + "fields": { + "fixture_purpose": "SYNTHETIC TEST DATA — independent Fullsend counterpart of interactive eval 4; all identities, URLs, CVEs, commit pins and evidence are deliberate public test material, not live advisory evidence.", + "current_user": { + "accountId": "synthetic-engineer", + "displayName": "Synthetic Engineer" + }, + "ecosystem": "Cargo", + "affected_package": "h2", + "fixed_version": "0.4.8", + "affected_range": "< 0.4.8", + "affectsVersions": [ + { + "id": "version-stale", + "name": "RHTPA 2.0.0" + } + ], + "assignee": null, + "issuelinks": [], + "customfield_10632": "h2" + } + }, + "remote_links": [ + { + "url": "https://example.com/advisories/CVE-2026-33501", + "title": "Synthetic CVE-2026-33501 advisory" + } + ], + "configuration": { + "project_key": "TC", + "jira_version_prefix": "RHTPA", + "vulnerability_issue_type_id": "10016", + "component_label_pattern": "pscomponent:", + "version_streams": [ + { + "name": "2.1.x", + "matrix_path": "2.1.x/security-matrix.md", + "release_repository": "release-repo" + }, + { + "name": "2.2.x", + "matrix_path": "2.2.x/security-matrix.md", + "release_repository": "release-repo" + } + ], + "source_repositories": [ + { + "name": "backend", + "url": "https://github.com/example/backend", + "deployment_context": "customer-shipped" + } + ], + "vex_justification_field": "customfield_10665", + "upstream_affected_component_field": "customfield_10632", + "stream_field": "customfield_10832" + }, + "external_evidence": { + "mitre": { + "source_url": "https://example.com/mitre/CVE-2026-33501", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "cveMetadata": { + "cveId": "CVE-2026-33501" + }, + "containers": { + "cna": { + "affected": [ + { + "product": "h2", + "versions": [ + { + "status": "affected", + "lessThan": "0.4.8", + "versionType": "semver" + } + ] + } + ] + } + } + } + }, + "osv": { + "source_url": "https://example.com/osv/CVE-2026-33501", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "id": "CVE-2026-33501", + "affected": [ + { + "package": { + "ecosystem": "Cargo", + "name": "h2" + }, + "ranges": [ + { + "type": "SEMVER", + "events": [ + { + "introduced": "0" + }, + { + "fixed": "0.4.8" + } + ] + } + ] + } + ] + } + }, + "lifecycle": { + "source_url": "https://example.com/lifecycle", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "supported_streams": [ + "2.1.x", + "2.2.x" + ], + "eol_streams": [] + } + } + }, + "matrix": { + "streams": [ + { + "name": "2.1.x", + "matrix_source": "SYNTHETIC TEST DATA — trusted 2.1.x matrix snapshot", + "rows": [ + { + "version": "2.1.0", + "source_commits": { + "backend": "a003008" + }, + "retag_of": null + }, + { + "version": "2.1.1", + "source_commits": { + "backend": "a003012" + }, + "retag_of": null + } + ] + }, + { + "name": "2.2.x", + "matrix_source": "SYNTHETIC TEST DATA — trusted 2.2.x matrix snapshot", + "rows": [ + { + "version": "2.2.0", + "source_commits": { + "backend": "a004005" + }, + "retag_of": null + }, + { + "version": "2.2.1", + "source_commits": { + "backend": "a004008" + }, + "retag_of": null + }, + { + "version": "2.2.2", + "source_commits": { + "backend": "a004008" + }, + "retag_of": "2.2.1" + }, + { + "version": "2.2.3", + "source_commits": { + "backend": "a004011" + }, + "retag_of": null + }, + { + "version": "2.2.4", + "source_commits": { + "backend": "a004012" + }, + "retag_of": null + } + ] + } + ] + }, + "source_evidence": { + "lock_files": [ + { + "repository": "backend", + "ref": "a003008", + "path": "Cargo.lock", + "command": "git show a003008:Cargo.lock", + "content": "[[package]]\nname = \"h2\"\nversion = \"0.4.5\"\n" + }, + { + "repository": "backend", + "ref": "a003012", + "path": "Cargo.lock", + "command": "git show a003012:Cargo.lock", + "content": "[[package]]\nname = \"h2\"\nversion = \"0.4.5\"\n" + }, + { + "repository": "backend", + "ref": "a004005", + "path": "Cargo.lock", + "command": "git show a004005:Cargo.lock", + "content": "[[package]]\nname = \"h2\"\nversion = \"0.4.8\"\n" + }, + { + "repository": "backend", + "ref": "a004008", + "path": "Cargo.lock", + "command": "git show a004008:Cargo.lock", + "content": "[[package]]\nname = \"h2\"\nversion = \"0.4.8\"\n" + }, + { + "repository": "backend", + "ref": "a004011", + "path": "Cargo.lock", + "command": "git show a004011:Cargo.lock", + "content": "[[package]]\nname = \"h2\"\nversion = \"0.4.9\"\n" + }, + { + "repository": "backend", + "ref": "a004012", + "path": "Cargo.lock", + "command": "git show a004012:Cargo.lock", + "content": "[[package]]\nname = \"h2\"\nversion = \"0.4.9\"\n" + } + ], + "development_streams": [ + { + "repository": "backend", + "ref": "release/0.3.z", + "path": "Cargo.lock", + "command": "git show release/0.3.z:Cargo.lock", + "content": "[[package]]\nname = \"h2\"\nversion = \"0.4.5\"\n" + }, + { + "repository": "backend", + "ref": "release/0.4.z", + "path": "Cargo.lock", + "command": "git show release/0.4.z:Cargo.lock", + "content": "[[package]]\nname = \"h2\"\nversion = \"0.4.8\"\n" + } + ] + }, + "jira_metadata": { + "versions": [ + { + "id": "version-2-1-0", + "name": "RHTPA 2.1.0", + "released": true + }, + { + "id": "version-2-1-1", + "name": "RHTPA 2.1.1", + "released": true + }, + { + "id": "version-2-2-0", + "name": "RHTPA 2.2.0", + "released": true + }, + { + "id": "version-2-2-1", + "name": "RHTPA 2.2.1", + "released": true + }, + { + "id": "version-2-2-2", + "name": "RHTPA 2.2.2", + "released": true + }, + { + "id": "version-2-2-3", + "name": "RHTPA 2.2.3", + "released": true + }, + { + "id": "version-2-2-4", + "name": "RHTPA 2.2.4", + "released": true + }, + { + "id": "version-dev", + "name": "RHTPA 2.2.5", + "released": false + } + ], + "sibling_searches": [], + "related_issues": [] + }, + "idempotency": { + "action_markers": [], + "existing_remediation": [] + }, + "authorization": { + "mutation_authorized": true + } +} diff --git a/evals/triage-security/files/fullsend-eval-5-trusted-input.json b/evals/triage-security/files/fullsend-eval-5-trusted-input.json new file mode 100644 index 000000000..b5df4b9d0 --- /dev/null +++ b/evals/triage-security/files/fullsend-eval-5-trusted-input.json @@ -0,0 +1,366 @@ +{ + "schema_version": "1", + "issue": { + "key": "TC-8005", + "summary": "CVE-2026-40215 openssl-libs - Buffer over-read in X.509 certificate verification [rhtpa-2.2]", + "description": { + "type": "doc", + "version": 1, + "content": [ + { + "type": "paragraph", + "content": [ + { + "type": "text", + "text": "A vulnerability was found in openssl-libs. Versions of openssl before 3.0.7-28.el9_4 are vulnerable to a buffer over-read during X.509 certificate chain verification. A remote attacker can craft a certificate with a malformed extension that triggers an out-of-bounds read, potentially leaking sensitive memory contents or causing a crash.\n\n**Affected package**: openssl-libs\n**Affected versions**: versions before 3.0.7-28.el9_4\n**Fixed version**: 3.0.7-28.el9_4\n**CVSS**: 7.1 (High)\n\nThe vulnerability exists in the `X509_verify_cert()` code path where the extension parser does not properly validate the length field of a Subject Alternative Name extension. The fix adds bounds checking before reading extension data.\n\n### References\n\n- https://www.cve.org/CVERecord?id=CVE-2026-40215\n- https://access.redhat.com/errata/RHSA-2026:4021" + } + ] + } + ] + }, + "status": "New", + "labels": [ + "CVE-2026-40215", + "pscomponent:org/rhtpa-server" + ], + "versions": [], + "reporter": { + "account_id": "synthetic-reporter", + "display_name": "Synthetic Reporter" + }, + "comments": [], + "fields": { + "fixture_purpose": "SYNTHETIC TEST DATA — independent Fullsend counterpart of interactive eval 5; all identities, URLs, CVEs, commit pins and evidence are deliberate public test material, not live advisory evidence.", + "current_user": { + "accountId": "synthetic-engineer", + "displayName": "Synthetic Engineer" + }, + "ecosystem": "RPM", + "affected_package": "openssl-libs", + "fixed_version": "3.0.7-28.el9_4", + "affected_range": "< 3.0.7-28.el9_4", + "affectsVersions": [ + { + "id": "version-stale", + "name": "RHTPA 2.0.0" + } + ], + "assignee": null, + "issuelinks": [], + "customfield_10632": "openssl-libs", + "sbom_evidence": { + "source": "trusted pre-retrieved SBOM comparison", + "packages": [ + { + "name": "openssl-libs", + "version": "3.0.7-25.el9_3", + "release": "2.2.0" + }, + { + "name": "openssl-libs", + "version": "3.0.7-27.el9_4", + "release": "2.2.1" + }, + { + "name": "openssl-libs", + "version": "3.0.7-27.el9_4", + "release": "2.2.2" + }, + { + "name": "openssl-libs", + "version": "3.0.7-28.el9_4", + "release": "2.2.3" + }, + { + "name": "openssl-libs", + "version": "3.0.7-28.el9_4", + "release": "2.2.4" + } + ] + } + } + }, + "remote_links": [ + { + "url": "https://example.com/advisories/CVE-2026-40215", + "title": "Synthetic CVE-2026-40215 advisory" + } + ], + "configuration": { + "project_key": "TC", + "jira_version_prefix": "RHTPA", + "vulnerability_issue_type_id": "10016", + "component_label_pattern": "pscomponent:", + "version_streams": [ + { + "name": "2.1.x", + "matrix_path": "2.1.x/security-matrix.md", + "release_repository": "release-repo" + }, + { + "name": "2.2.x", + "matrix_path": "2.2.x/security-matrix.md", + "release_repository": "release-repo" + } + ], + "source_repositories": [ + { + "name": "release-repo", + "url": "https://github.com/example/release-repo", + "deployment_context": "customer-shipped" + } + ], + "vex_justification_field": "customfield_10665", + "upstream_affected_component_field": "customfield_10632", + "stream_field": "customfield_10832" + }, + "external_evidence": { + "mitre": { + "source_url": "https://example.com/mitre/CVE-2026-40215", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "cveMetadata": { + "cveId": "CVE-2026-40215" + }, + "containers": { + "cna": { + "affected": [ + { + "product": "openssl-libs", + "versions": [ + { + "status": "affected", + "lessThan": "3.0.7-28.el9_4", + "versionType": "semver" + } + ] + } + ] + } + } + } + }, + "osv": { + "source_url": "https://example.com/osv/CVE-2026-40215", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "id": "CVE-2026-40215", + "affected": [ + { + "package": { + "ecosystem": "RPM", + "name": "openssl-libs" + }, + "ranges": [ + { + "type": "SEMVER", + "events": [ + { + "introduced": "0" + }, + { + "fixed": "3.0.7-28.el9_4" + } + ] + } + ] + } + ] + } + }, + "lifecycle": { + "source_url": "https://example.com/lifecycle", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "supported_streams": [ + "2.1.x", + "2.2.x" + ], + "eol_streams": [] + } + } + }, + "matrix": { + "streams": [ + { + "name": "2.1.x", + "matrix_source": "SYNTHETIC TEST DATA — trusted 2.1.x matrix snapshot", + "rows": [ + { + "version": "2.1.0", + "source_commits": { + "release-repo": "a003008" + }, + "retag_of": null + }, + { + "version": "2.1.1", + "source_commits": { + "release-repo": "a003012" + }, + "retag_of": null + } + ] + }, + { + "name": "2.2.x", + "matrix_source": "SYNTHETIC TEST DATA — trusted 2.2.x matrix snapshot", + "rows": [ + { + "version": "2.2.0", + "source_commits": { + "release-repo": "a004005" + }, + "retag_of": null + }, + { + "version": "2.2.1", + "source_commits": { + "release-repo": "a004008" + }, + "retag_of": null + }, + { + "version": "2.2.2", + "source_commits": { + "release-repo": "a004008" + }, + "retag_of": "2.2.1" + }, + { + "version": "2.2.3", + "source_commits": { + "release-repo": "a004011" + }, + "retag_of": null + }, + { + "version": "2.2.4", + "source_commits": { + "release-repo": "a004012" + }, + "retag_of": null + } + ] + } + ] + }, + "source_evidence": { + "lock_files": [ + { + "repository": "release-repo", + "ref": "a003008", + "path": "rpms.lock.yaml", + "command": "git show a003008:rpms.lock.yaml", + "content": "packages:\n - name: openssl-libs\n version: 3.0.7-24.el9\n" + }, + { + "repository": "release-repo", + "ref": "a003012", + "path": "rpms.lock.yaml", + "command": "git show a003012:rpms.lock.yaml", + "content": "packages:\n - name: openssl-libs\n version: 3.0.7-24.el9\n" + }, + { + "repository": "release-repo", + "ref": "a004005", + "path": "rpms.lock.yaml", + "command": "git show a004005:rpms.lock.yaml", + "content": "packages:\n - name: openssl-libs\n version: 3.0.7-25.el9_3\n" + }, + { + "repository": "release-repo", + "ref": "a004008", + "path": "rpms.lock.yaml", + "command": "git show a004008:rpms.lock.yaml", + "content": "packages:\n - name: openssl-libs\n version: 3.0.7-27.el9_4\n" + }, + { + "repository": "release-repo", + "ref": "a004011", + "path": "rpms.lock.yaml", + "command": "git show a004011:rpms.lock.yaml", + "content": "packages:\n - name: openssl-libs\n version: 3.0.7-28.el9_4\n" + }, + { + "repository": "release-repo", + "ref": "a004012", + "path": "rpms.lock.yaml", + "command": "git show a004012:rpms.lock.yaml", + "content": "packages:\n - name: openssl-libs\n version: 3.0.7-28.el9_4\n" + } + ], + "development_streams": [ + { + "repository": "release-repo", + "ref": "release/0.3.z", + "path": "rpms.lock.yaml", + "command": "git show release/0.3.z:rpms.lock.yaml", + "content": "packages:\n - name: openssl-libs\n version: 3.0.7-28.el9_4\n" + }, + { + "repository": "release-repo", + "ref": "release/0.4.z", + "path": "rpms.lock.yaml", + "command": "git show release/0.4.z:rpms.lock.yaml", + "content": "packages:\n - name: openssl-libs\n version: 3.0.7-28.el9_4\n" + } + ] + }, + "jira_metadata": { + "versions": [ + { + "id": "version-2-1-0", + "name": "RHTPA 2.1.0", + "released": true + }, + { + "id": "version-2-1-1", + "name": "RHTPA 2.1.1", + "released": true + }, + { + "id": "version-2-2-0", + "name": "RHTPA 2.2.0", + "released": true + }, + { + "id": "version-2-2-1", + "name": "RHTPA 2.2.1", + "released": true + }, + { + "id": "version-2-2-2", + "name": "RHTPA 2.2.2", + "released": true + }, + { + "id": "version-2-2-3", + "name": "RHTPA 2.2.3", + "released": true + }, + { + "id": "version-2-2-4", + "name": "RHTPA 2.2.4", + "released": true + }, + { + "id": "version-dev", + "name": "RHTPA 2.2.5", + "released": false + } + ], + "sibling_searches": [], + "related_issues": [] + }, + "idempotency": { + "action_markers": [], + "existing_remediation": [] + }, + "authorization": { + "mutation_authorized": false + } +} diff --git a/evals/triage-security/files/fullsend-eval-8-trusted-input.json b/evals/triage-security/files/fullsend-eval-8-trusted-input.json new file mode 100644 index 000000000..e8660f6d6 --- /dev/null +++ b/evals/triage-security/files/fullsend-eval-8-trusted-input.json @@ -0,0 +1,434 @@ +{ + "schema_version": "1", + "issue": { + "key": "TC-8010", + "summary": "CVE-2026-44492 axios - Server-Side Request Forgery via crafted URL [rhtpa-2.2]", + "description": { + "type": "doc", + "version": 1, + "content": [ + { + "type": "paragraph", + "content": [ + { + "type": "text", + "text": "A vulnerability was found in axios. The axios package before version 1.8.2 is vulnerable to Server-Side Request Forgery (SSRF) via a crafted URL that bypasses hostname validation. An attacker can exploit this to make requests to internal services.\n\n**Affected package**: axios\n**Affected versions**: versions before 1.8.2\n**Fixed version**: 1.8.2\n**CVSS**: 8.1 (High)\n\nThe vulnerability exists because axios does not properly validate the hostname in URLs when following redirects. An attacker can craft a URL that initially resolves to an external host but redirects to an internal service.\n\n### References\n\n- https://github.com/advisories/GHSA-2026-ax91-r7pp\n\n### Related CVE Jiras (provided by JQL search on customfield_10632 = \"axios\")\n\nThe following related CVE Jira was found affecting the same upstream component:\n\n- **TC-8008** — CVE-2026-42035 axios - Prototype Pollution via header parsing [rhtpa-2.2]\n - **Status**: In Progress\n - **Labels**: CVE-2026-42035, pscomponent:org/rhtpa-ui\n - **customfield_10632**: axios\n - **customfield_10669**: pscomponent:org/rhtpa-ui\n - **customfield_10832**: rhtpa-2.2\n - **Issue Links**:\n - **Depend**: TC-8009 (remediation Task)\n - **Summary**: Bump axios to 1.9.0 in rhtpa-ui [rhtpa-2.2]\n - **Status**: In Progress\n - **Description excerpt**: \"Bump axios from 1.7.4 to 1.9.0 to resolve CVE-2026-42035. The fix requires axios >= 1.8.0.\"" + } + ] + } + ] + }, + "status": "New", + "labels": [ + "CVE-2026-44492", + "pscomponent:org/rhtpa-ui" + ], + "versions": [], + "reporter": { + "account_id": "synthetic-reporter", + "display_name": "Synthetic Reporter" + }, + "comments": [], + "fields": { + "fixture_purpose": "SYNTHETIC TEST DATA — independent Fullsend counterpart of interactive eval 8; all identities, URLs, CVEs, commit pins and evidence are deliberate public test material, not live advisory evidence.", + "current_user": { + "accountId": "synthetic-engineer", + "displayName": "Synthetic Engineer" + }, + "ecosystem": "npm", + "affected_package": "axios", + "fixed_version": "1.8.2", + "affected_range": "< 1.8.2", + "affectsVersions": [ + { + "id": "version-stale", + "name": "RHTPA 2.0.0" + } + ], + "assignee": null, + "issuelinks": [ + { + "type": { + "name": "Related" + }, + "outwardIssue": { + "key": "TC-8008" + } + } + ], + "customfield_10632": "axios" + } + }, + "remote_links": [ + { + "url": "https://example.com/advisories/CVE-2026-44492", + "title": "Synthetic CVE-2026-44492 advisory" + } + ], + "configuration": { + "project_key": "TC", + "jira_version_prefix": "RHTPA", + "vulnerability_issue_type_id": "10016", + "component_label_pattern": "pscomponent:", + "version_streams": [ + { + "name": "2.1.x", + "matrix_path": "2.1.x/security-matrix.md", + "release_repository": "release-repo" + }, + { + "name": "2.2.x", + "matrix_path": "2.2.x/security-matrix.md", + "release_repository": "release-repo" + } + ], + "source_repositories": [ + { + "name": "ui", + "url": "https://github.com/example/ui", + "deployment_context": "customer-shipped" + } + ], + "vex_justification_field": "customfield_10665", + "upstream_affected_component_field": "customfield_10632", + "stream_field": "customfield_10832" + }, + "external_evidence": { + "mitre": { + "source_url": "https://example.com/mitre/CVE-2026-44492", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "cveMetadata": { + "cveId": "CVE-2026-44492" + }, + "containers": { + "cna": { + "affected": [ + { + "product": "axios", + "versions": [ + { + "status": "affected", + "lessThan": "1.8.2", + "versionType": "semver" + } + ] + } + ] + } + } + } + }, + "osv": { + "source_url": "https://example.com/osv/CVE-2026-44492", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "id": "CVE-2026-44492", + "affected": [ + { + "package": { + "ecosystem": "npm", + "name": "axios" + }, + "ranges": [ + { + "type": "SEMVER", + "events": [ + { + "introduced": "0" + }, + { + "fixed": "1.8.2" + } + ] + } + ] + } + ] + } + }, + "lifecycle": { + "source_url": "https://example.com/lifecycle", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "supported_streams": [ + "2.1.x", + "2.2.x" + ], + "eol_streams": [] + } + } + }, + "matrix": { + "streams": [ + { + "name": "2.1.x", + "matrix_source": "SYNTHETIC TEST DATA — trusted 2.1.x matrix snapshot", + "rows": [ + { + "version": "2.1.0", + "source_commits": { + "ui": "a003008" + }, + "retag_of": null + }, + { + "version": "2.1.1", + "source_commits": { + "ui": "a003012" + }, + "retag_of": null + } + ] + }, + { + "name": "2.2.x", + "matrix_source": "SYNTHETIC TEST DATA — trusted 2.2.x matrix snapshot", + "rows": [ + { + "version": "2.2.0", + "source_commits": { + "ui": "a004005" + }, + "retag_of": null + }, + { + "version": "2.2.1", + "source_commits": { + "ui": "a004008" + }, + "retag_of": null + }, + { + "version": "2.2.2", + "source_commits": { + "ui": "a004008" + }, + "retag_of": "2.2.1" + }, + { + "version": "2.2.3", + "source_commits": { + "ui": "a004011" + }, + "retag_of": null + }, + { + "version": "2.2.4", + "source_commits": { + "ui": "a004012" + }, + "retag_of": null + } + ] + } + ] + }, + "source_evidence": { + "lock_files": [ + { + "repository": "ui", + "ref": "a003008", + "path": "package-lock.json", + "command": "git show a003008:package-lock.json", + "content": "{\"lockfileVersion\": 3, \"packages\": {\"node_modules/axios\": {\"version\": \"1.7.4\"}}}" + }, + { + "repository": "ui", + "ref": "a003012", + "path": "package-lock.json", + "command": "git show a003012:package-lock.json", + "content": "{\"lockfileVersion\": 3, \"packages\": {\"node_modules/axios\": {\"version\": \"1.7.4\"}}}" + }, + { + "repository": "ui", + "ref": "a004005", + "path": "package-lock.json", + "command": "git show a004005:package-lock.json", + "content": "{\"lockfileVersion\": 3, \"packages\": {\"node_modules/axios\": {\"version\": \"1.7.4\"}}}" + }, + { + "repository": "ui", + "ref": "a004008", + "path": "package-lock.json", + "command": "git show a004008:package-lock.json", + "content": "{\"lockfileVersion\": 3, \"packages\": {\"node_modules/axios\": {\"version\": \"1.7.4\"}}}" + }, + { + "repository": "ui", + "ref": "a004011", + "path": "package-lock.json", + "command": "git show a004011:package-lock.json", + "content": "{\"lockfileVersion\": 3, \"packages\": {\"node_modules/axios\": {\"version\": \"1.9.0\"}}}" + }, + { + "repository": "ui", + "ref": "a004012", + "path": "package-lock.json", + "command": "git show a004012:package-lock.json", + "content": "{\"lockfileVersion\": 3, \"packages\": {\"node_modules/axios\": {\"version\": \"1.9.0\"}}}" + } + ], + "development_streams": [ + { + "repository": "ui", + "ref": "release/0.3.z", + "path": "package-lock.json", + "command": "git show release/0.3.z:package-lock.json", + "content": "{\"lockfileVersion\": 3, \"packages\": {\"node_modules/axios\": {\"version\": \"1.8.2\"}}}" + }, + { + "repository": "ui", + "ref": "release/0.4.z", + "path": "package-lock.json", + "command": "git show release/0.4.z:package-lock.json", + "content": "{\"lockfileVersion\": 3, \"packages\": {\"node_modules/axios\": {\"version\": \"1.8.2\"}}}" + } + ] + }, + "jira_metadata": { + "versions": [ + { + "id": "version-2-1-0", + "name": "RHTPA 2.1.0", + "released": true + }, + { + "id": "version-2-1-1", + "name": "RHTPA 2.1.1", + "released": true + }, + { + "id": "version-2-2-0", + "name": "RHTPA 2.2.0", + "released": true + }, + { + "id": "version-2-2-1", + "name": "RHTPA 2.2.1", + "released": true + }, + { + "id": "version-2-2-2", + "name": "RHTPA 2.2.2", + "released": true + }, + { + "id": "version-2-2-3", + "name": "RHTPA 2.2.3", + "released": true + }, + { + "id": "version-2-2-4", + "name": "RHTPA 2.2.4", + "released": true + }, + { + "id": "version-dev", + "name": "RHTPA 2.2.5", + "released": false + } + ], + "sibling_searches": [ + { + "purpose": "cross-cve-overlap", + "jql": "project = TC AND cf[10632] ~ \"axios\"", + "issues": [ + { + "key": "TC-8008", + "summary": "Other CVE in axios [rhtpa-2.2]", + "status": "In Progress", + "labels": [], + "description": { + "type": "doc", + "version": 1, + "content": [ + { + "type": "paragraph", + "content": [ + { + "type": "text", + "text": "Same upstream component axios, stream 2.2.x." + } + ] + } + ] + }, + "comments": [], + "links": [ + { + "type": "Depend", + "outward": "TC-8009" + } + ] + } + ] + } + ], + "related_issues": [ + { + "key": "TC-8008", + "summary": "Other CVE in axios [rhtpa-2.2]", + "status": "In Progress", + "labels": [], + "description": { + "type": "doc", + "version": 1, + "content": [ + { + "type": "paragraph", + "content": [ + { + "type": "text", + "text": "Same upstream component axios, stream 2.2.x." + } + ] + } + ] + }, + "comments": [], + "links": [ + { + "type": "Depend", + "outward": "TC-8009" + } + ] + }, + { + "key": "TC-8009", + "summary": "Bump axios to 1.9.0 in ui [rhtpa-2.2]", + "status": "In Progress", + "labels": [], + "description": { + "type": "doc", + "version": 1, + "content": [ + { + "type": "paragraph", + "content": [ + { + "type": "text", + "text": "Bump axios to 1.9.0; same component and 2.2.x stream." + } + ] + } + ] + }, + "comments": [], + "links": [] + } + ] + }, + "idempotency": { + "action_markers": [ + "triage-security:tc-8010:link:related:tc-8008" + ], + "existing_remediation": [] + }, + "authorization": { + "mutation_authorized": true + } +} diff --git a/evals/triage-security/files/fullsend-eval-9-trusted-input.json b/evals/triage-security/files/fullsend-eval-9-trusted-input.json new file mode 100644 index 000000000..c0823bb2d --- /dev/null +++ b/evals/triage-security/files/fullsend-eval-9-trusted-input.json @@ -0,0 +1,423 @@ +{ + "schema_version": "1", + "issue": { + "key": "TC-8011", + "summary": "CVE-2026-45678 webpack - Arbitrary Code Execution via loader chain [rhtpa-2.2]", + "description": { + "type": "doc", + "version": 1, + "content": [ + { + "type": "paragraph", + "content": [ + { + "type": "text", + "text": "A vulnerability was found in webpack. The webpack package before version 5.98.0 allows arbitrary code execution through a specially crafted loader chain configuration. An attacker with control over a project's webpack configuration can execute arbitrary code during the build process.\n\n**Affected package**: webpack\n**Affected versions**: versions before 5.98.0\n**Fixed version**: 5.98.0\n**CVSS**: 7.8 (High)\n\nThe vulnerability exists because webpack does not properly sanitize loader paths when resolving the loader chain, allowing path traversal to execute arbitrary modules.\n\n### References\n\n- https://github.com/advisories/GHSA-2026-wk55-m3rr\n\n### Related CVE Jiras (provided by JQL search on customfield_10632 = \"webpack\")\n\nThe following related CVE Jira was found affecting the same upstream component:\n\n- **TC-8012** — CVE-2026-43210 webpack - ReDoS in chunk name validation [rhtpa-2.2]\n - **Status**: Closed (Done)\n - **Labels**: CVE-2026-43210, pscomponent:org/rhtpa-ui\n - **customfield_10632**: webpack\n - **customfield_10669**: pscomponent:org/rhtpa-ui\n - **customfield_10832**: rhtpa-2.2\n - **Issue Links**:\n - **Depend**: TC-8013 (remediation Task)\n - **Summary**: Bump webpack to 5.96.1 in rhtpa-ui [rhtpa-2.2]\n - **Status**: Closed (Done)\n - **Description excerpt**: \"Bump webpack from 5.95.0 to 5.96.1 to resolve CVE-2026-43210. The fix requires webpack >= 5.96.0.\"" + } + ] + } + ] + }, + "status": "New", + "labels": [ + "CVE-2026-45678", + "pscomponent:org/rhtpa-ui" + ], + "versions": [], + "reporter": { + "account_id": "synthetic-reporter", + "display_name": "Synthetic Reporter" + }, + "comments": [], + "fields": { + "fixture_purpose": "SYNTHETIC TEST DATA — independent Fullsend counterpart of interactive eval 9; all identities, URLs, CVEs, commit pins and evidence are deliberate public test material, not live advisory evidence.", + "current_user": { + "accountId": "synthetic-engineer", + "displayName": "Synthetic Engineer" + }, + "ecosystem": "npm", + "affected_package": "webpack", + "fixed_version": "5.98.0", + "affected_range": "< 5.98.0", + "affectsVersions": [ + { + "id": "version-stale", + "name": "RHTPA 2.0.0" + } + ], + "assignee": null, + "issuelinks": [], + "customfield_10632": "webpack" + } + }, + "remote_links": [ + { + "url": "https://example.com/advisories/CVE-2026-45678", + "title": "Synthetic CVE-2026-45678 advisory" + } + ], + "configuration": { + "project_key": "TC", + "jira_version_prefix": "RHTPA", + "vulnerability_issue_type_id": "10016", + "component_label_pattern": "pscomponent:", + "version_streams": [ + { + "name": "2.1.x", + "matrix_path": "2.1.x/security-matrix.md", + "release_repository": "release-repo" + }, + { + "name": "2.2.x", + "matrix_path": "2.2.x/security-matrix.md", + "release_repository": "release-repo" + } + ], + "source_repositories": [ + { + "name": "ui", + "url": "https://github.com/example/ui", + "deployment_context": "customer-shipped" + } + ], + "vex_justification_field": "customfield_10665", + "upstream_affected_component_field": "customfield_10632", + "stream_field": "customfield_10832" + }, + "external_evidence": { + "mitre": { + "source_url": "https://example.com/mitre/CVE-2026-45678", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "cveMetadata": { + "cveId": "CVE-2026-45678" + }, + "containers": { + "cna": { + "affected": [ + { + "product": "webpack", + "versions": [ + { + "status": "affected", + "lessThan": "5.98.0", + "versionType": "semver" + } + ] + } + ] + } + } + } + }, + "osv": { + "source_url": "https://example.com/osv/CVE-2026-45678", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "id": "CVE-2026-45678", + "affected": [ + { + "package": { + "ecosystem": "npm", + "name": "webpack" + }, + "ranges": [ + { + "type": "SEMVER", + "events": [ + { + "introduced": "0" + }, + { + "fixed": "5.98.0" + } + ] + } + ] + } + ] + } + }, + "lifecycle": { + "source_url": "https://example.com/lifecycle", + "retrieved_at": "2026-09-30T12:00:00Z", + "status": 200, + "body": { + "supported_streams": [ + "2.1.x", + "2.2.x" + ], + "eol_streams": [] + } + } + }, + "matrix": { + "streams": [ + { + "name": "2.1.x", + "matrix_source": "SYNTHETIC TEST DATA — trusted 2.1.x matrix snapshot", + "rows": [ + { + "version": "2.1.0", + "source_commits": { + "ui": "a003008" + }, + "retag_of": null + }, + { + "version": "2.1.1", + "source_commits": { + "ui": "a003012" + }, + "retag_of": null + } + ] + }, + { + "name": "2.2.x", + "matrix_source": "SYNTHETIC TEST DATA — trusted 2.2.x matrix snapshot", + "rows": [ + { + "version": "2.2.0", + "source_commits": { + "ui": "a004005" + }, + "retag_of": null + }, + { + "version": "2.2.1", + "source_commits": { + "ui": "a004008" + }, + "retag_of": null + }, + { + "version": "2.2.2", + "source_commits": { + "ui": "a004008" + }, + "retag_of": "2.2.1" + }, + { + "version": "2.2.3", + "source_commits": { + "ui": "a004011" + }, + "retag_of": null + }, + { + "version": "2.2.4", + "source_commits": { + "ui": "a004012" + }, + "retag_of": null + } + ] + } + ] + }, + "source_evidence": { + "lock_files": [ + { + "repository": "ui", + "ref": "a003008", + "path": "package-lock.json", + "command": "git show a003008:package-lock.json", + "content": "{\"lockfileVersion\": 3, \"packages\": {\"node_modules/webpack\": {\"version\": \"5.95.0\"}}}" + }, + { + "repository": "ui", + "ref": "a003012", + "path": "package-lock.json", + "command": "git show a003012:package-lock.json", + "content": "{\"lockfileVersion\": 3, \"packages\": {\"node_modules/webpack\": {\"version\": \"5.95.0\"}}}" + }, + { + "repository": "ui", + "ref": "a004005", + "path": "package-lock.json", + "command": "git show a004005:package-lock.json", + "content": "{\"lockfileVersion\": 3, \"packages\": {\"node_modules/webpack\": {\"version\": \"5.95.0\"}}}" + }, + { + "repository": "ui", + "ref": "a004008", + "path": "package-lock.json", + "command": "git show a004008:package-lock.json", + "content": "{\"lockfileVersion\": 3, \"packages\": {\"node_modules/webpack\": {\"version\": \"5.95.0\"}}}" + }, + { + "repository": "ui", + "ref": "a004011", + "path": "package-lock.json", + "command": "git show a004011:package-lock.json", + "content": "{\"lockfileVersion\": 3, \"packages\": {\"node_modules/webpack\": {\"version\": \"5.96.1\"}}}" + }, + { + "repository": "ui", + "ref": "a004012", + "path": "package-lock.json", + "command": "git show a004012:package-lock.json", + "content": "{\"lockfileVersion\": 3, \"packages\": {\"node_modules/webpack\": {\"version\": \"5.96.1\"}}}" + } + ], + "development_streams": [ + { + "repository": "ui", + "ref": "release/0.3.z", + "path": "package-lock.json", + "command": "git show release/0.3.z:package-lock.json", + "content": "{\"lockfileVersion\": 3, \"packages\": {\"node_modules/webpack\": {\"version\": \"5.98.0\"}}}" + }, + { + "repository": "ui", + "ref": "release/0.4.z", + "path": "package-lock.json", + "command": "git show release/0.4.z:package-lock.json", + "content": "{\"lockfileVersion\": 3, \"packages\": {\"node_modules/webpack\": {\"version\": \"5.96.1\"}}}" + } + ] + }, + "jira_metadata": { + "versions": [ + { + "id": "version-2-1-0", + "name": "RHTPA 2.1.0", + "released": true + }, + { + "id": "version-2-1-1", + "name": "RHTPA 2.1.1", + "released": true + }, + { + "id": "version-2-2-0", + "name": "RHTPA 2.2.0", + "released": true + }, + { + "id": "version-2-2-1", + "name": "RHTPA 2.2.1", + "released": true + }, + { + "id": "version-2-2-2", + "name": "RHTPA 2.2.2", + "released": true + }, + { + "id": "version-2-2-3", + "name": "RHTPA 2.2.3", + "released": true + }, + { + "id": "version-2-2-4", + "name": "RHTPA 2.2.4", + "released": true + }, + { + "id": "version-dev", + "name": "RHTPA 2.2.5", + "released": false + } + ], + "sibling_searches": [ + { + "purpose": "cross-cve-overlap", + "jql": "project = TC AND cf[10632] ~ \"webpack\"", + "issues": [ + { + "key": "TC-8012", + "summary": "Other CVE in webpack [rhtpa-2.2]", + "status": "In Progress", + "labels": [], + "description": { + "type": "doc", + "version": 1, + "content": [ + { + "type": "paragraph", + "content": [ + { + "type": "text", + "text": "Same upstream component webpack, stream 2.2.x." + } + ] + } + ] + }, + "comments": [], + "links": [ + { + "type": "Depend", + "outward": "TC-8013" + } + ] + } + ] + } + ], + "related_issues": [ + { + "key": "TC-8012", + "summary": "Other CVE in webpack [rhtpa-2.2]", + "status": "In Progress", + "labels": [], + "description": { + "type": "doc", + "version": 1, + "content": [ + { + "type": "paragraph", + "content": [ + { + "type": "text", + "text": "Same upstream component webpack, stream 2.2.x." + } + ] + } + ] + }, + "comments": [], + "links": [ + { + "type": "Depend", + "outward": "TC-8013" + } + ] + }, + { + "key": "TC-8013", + "summary": "Bump webpack to 5.96.1 in ui [rhtpa-2.2]", + "status": "In Progress", + "labels": [], + "description": { + "type": "doc", + "version": 1, + "content": [ + { + "type": "paragraph", + "content": [ + { + "type": "text", + "text": "Bump webpack to 5.96.1; same component and 2.2.x stream." + } + ] + } + ] + }, + "comments": [], + "links": [] + } + ] + }, + "idempotency": { + "action_markers": [], + "existing_remediation": [] + }, + "authorization": { + "mutation_authorized": true + } +} diff --git a/plugins/sdlc-workflow/scripts/test_triage_security_fullsend.py b/plugins/sdlc-workflow/scripts/test_triage_security_fullsend.py index a23fa2b24..8043a0b88 100644 --- a/plugins/sdlc-workflow/scripts/test_triage_security_fullsend.py +++ b/plugins/sdlc-workflow/scripts/test_triage_security_fullsend.py @@ -259,3 +259,158 @@ def test_retry_snapshot_suppresses_existing_field_and_link_actions(): # Then _already_applied independently detects both existing operations assert executor._already_applied(field_action, field_snapshot) assert executor._already_applied(resolved_link, field_snapshot) + + +@pytest.mark.parametrize("source_id", [1, 2, 3, 4, 5, 8, 9, 11, 12, 18]) +@requires_format_extra +def test_conditional_fullsend_evals_have_matching_executable_inputs(source_id): + """Every conditional contract gets an independent gated, subject-bound JSON input.""" + # Given the executable eval cases, rather than grader-only assertions + evals = json.loads((FIXTURE_DIR.parent / "evals.json").read_text())["evals"] + fixture = "files/fullsend-eval-{}-trusted-input.json".format(source_id) + matches = [case for case in evals if fixture in case.get("files", [])] + + # When the executor receives its prompt and mounted input + assert len(matches) == 1, "conditional contract lacks a matching Fullsend invocation" + case = matches[0] + assert case["id"] > 36 + assert "FULLSEND_OUTPUT_DIR" in case["prompt"] + assert ".pre-script/triage-security-input.json" in case["prompt"] + assert "invocation.json" in case["prompt"] + bundle = _trusted_input(Path(fixture).name) + pre_triage.validate_bundle(bundle) + + # Then identity and authorization belong to this scenario, not a generic sample + subjects = {1: "TC-8001", 2: "TC-8002", 3: "TC-8003", 4: "TC-8004", + 5: "TC-8005", 8: "TC-8010", 9: "TC-8011", 11: "TC-8021", + 12: "TC-8030", 18: "TC-8001"} + assert bundle["issue"]["key"] == subjects[source_id] + assert subjects[source_id] in case["prompt"] + assert bundle["authorization"]["mutation_authorized"] is (source_id not in [2, 5, 12]) + assert "SYNTHETIC TEST DATA" in bundle["issue"]["fields"]["fixture_purpose"] + + +@pytest.mark.parametrize("source_id", [1, 2, 3, 4, 5, 8, 9, 11, 12, 18]) +def test_conditional_runner_handoff_exports_evidence_without_sandbox_writes(source_id, tmp_path, monkeypatch): + """Runner handoff exposes exact audit/result bytes to the grader without extra sandbox files.""" + # Given distinct runner, grader and sandbox directories and an actual output gate + evals = json.loads((FIXTURE_DIR.parent / "evals.json").read_text())["evals"] + fixture = "files/fullsend-eval-{}-trusted-input.json".format(source_id) + case = next(case for case in evals if fixture in case.get("files", [])) + sandbox = tmp_path / "sandbox-outputs" + outputs = tmp_path / "outputs" + sandbox.mkdir() + outputs.mkdir() + monkeypatch.setenv("FULLSEND_OUTPUT_DIR", str(sandbox)) + result_bytes = b'{"test_handoff_only": true}\n' + invocation_bytes = (FIXTURE_DIR / Path(fixture).name).read_bytes() + (sandbox / "agent-result.json").write_bytes(result_bytes) + invocation = tmp_path / "invocation.json" + invocation.write_bytes(invocation_bytes) + + # When the runner executes the handoff supplied by the eval prompt (not the skill) + assert "```python\n" in case["prompt"], "missing executable grader evidence handoff" + handoff = case["prompt"].split("```python\n", 1)[1].split("\n```", 1)[0] + exec(handoff, {"GRADER_OUTPUTS_DIR": str(outputs), "INVOCATION_PATH": str(invocation)}) + + # Then the grader receives both artifacts byte-for-byte and the sandbox stays isolated + assert (outputs / "agent-result.json").read_bytes() == result_bytes + assert (outputs / "invocation.json").read_bytes() == invocation_bytes + assert sorted(path.name for path in outputs.iterdir()) == ["agent-result.json", "invocation.json"] + assert [path.name for path in sandbox.iterdir()] == ["agent-result.json"] + assert (sandbox / "agent-result.json").read_bytes() == result_bytes + + +def test_conditional_retry_input_has_an_existing_task_without_a_digest(): + """The new partial retry is distinct from the retained fully triaged interactive case.""" + # Given a trusted snapshot of an interrupted remediation creation + bundle = _trusted_input("fullsend-eval-18-trusted-input.json") + existing = bundle["idempotency"]["existing_remediation"] + + # When existing remediation identity and ordinary markers are inspected + assert [item["key"] for item in existing] == ["TC-8100", "TC-8101"] + assert existing[0]["comments"] == [] + assert executor._has_description_digest(existing[0]) is False + assert executor._has_description_digest(existing[1]) is True + + # Then the digest repair path retains its stable task reference and retry markers + assert "triage-security:tc-8001:remediation:upstream" in bundle["idempotency"]["action_markers"] + assert bundle["issue"]["status"] == "In Progress" + assert "ai-cve-triaged" in bundle["issue"]["labels"] + + +def test_conditional_inputs_supply_each_scenarios_distinct_evidence(): + """Scenario inputs retain actual impact, duplicate, overlap, RPM and enrichment facts.""" + # Given independent JSON inputs rather than a shared generic evidence sample + bundles = {source_id: _trusted_input("fullsend-eval-{}-trusted-input.json".format(source_id)) + for source_id in [1, 2, 3, 4, 5, 8, 9, 11, 12, 18]} + + # When trusted pins are joined with their actual lock evidence + for bundle in bundles.values(): + reads = {(read["repository"], read["ref"]): read + for read in bundle["source_evidence"]["lock_files"]} + for stream in bundle["matrix"]["streams"]: + for row in stream["rows"]: + assert all((repo, ref) in reads for repo, ref in row["source_commits"].items()) + + # Then each scenario's decision is supported by its own supplied facts + assert bundles[1]["issue"]["fields"]["affected_package"] == "quinn-proto" + assert bundles[1]["issue"]["fields"]["current_user"]["accountId"] == "synthetic-engineer" + assert all('version = "1.0.13' in read["content"] + for read in bundles[2]["source_evidence"]["lock_files"]) + assert bundles[3]["jira_metadata"]["sibling_searches"][0]["issues"][0]["key"] == "TC-7999" + split_reads = bundles[4]["source_evidence"]["lock_files"] + assert [read["content"].split('version = "')[1].split('"')[0] for read in split_reads] == [ + "0.4.5", "0.4.5", "0.4.8", "0.4.8", "0.4.9", "0.4.9"] + assert bundles[5]["issue"]["fields"]["sbom_evidence"]["packages"] == [ + {"name": "openssl-libs", "version": version, "release": release} + for release, version in [("2.2.0", "3.0.7-25.el9_3"), ("2.2.1", "3.0.7-27.el9_4"), + ("2.2.2", "3.0.7-27.el9_4"), ("2.2.3", "3.0.7-28.el9_4"), + ("2.2.4", "3.0.7-28.el9_4")]] + assert "1.9.0" in bundles[8]["jira_metadata"]["related_issues"][1]["summary"] + assert bundles[8]["issue"]["fields"]["fixed_version"] == "1.8.2" + assert bundles[8]["idempotency"]["action_markers"] == ["triage-security:tc-8010:link:related:tc-8008"] + assert "5.96.1" in bundles[9]["jira_metadata"]["related_issues"][1]["summary"] + assert bundles[9]["issue"]["fields"]["fixed_version"] == "5.98.0" + preemptive = bundles[11]["jira_metadata"]["sibling_searches"][0]["issues"][0] + assert preemptive["key"] == "TC-8022" + assert "security-preemptive" in preemptive["labels"] + mitre = bundles[12]["external_evidence"]["mitre"]["body"] + assert mitre["containers"]["cna"]["affected"][0]["versions"][0]["lessThan"] == "0.4.8" + assert bundles[12]["external_evidence"]["osv"]["status"] == 503 + assert bundles[12]["external_evidence"]["osv"]["body"] == {} + + +def test_conditional_retry_fixture_repairs_digest_before_resolving_new_link(recorder): + """The real partial-retry input repairs one digest and resolves its unapplied link.""" + # Given the mounted partial-retry scenario and its existing task identities + bundle = _trusted_input("fullsend-eval-18-trusted-input.json") + tasks = bundle["idempotency"]["existing_remediation"] + actions = [ + {"type": "remediation-task", "marker": "triage-security:tc-8001:remediation:" + ref, + "ref": ref, "project": "TC", "summary": task["summary"], + "description_adf": task["description"], "labels": task["labels"]} + for ref, task in zip(["upstream", "downstream"], tasks) + ] + actions.append({"type": "link", "marker": "triage-security:tc-8001:link:depend:downstream", + "link_type": "Depend", "inward": "TC-8001", "outward": "{{downstream.key}}"}) + result = {"schema_version": "1", "mode": "mutation-authorized", + "report": {"issue": "TC-8001", "outcome": "affected", "summary_markdown": "Partial retry.", + "evidence": [{"source": "trusted-input", "detail": "Existing tasks, one missing digest."}]}, + "actions": actions} + + # When the trusted executor processes the repair against a recorded Jira boundary + registry = executor.execute_plan(result, bundle) + + # Then task creation is skipped, one digest precedes the resolved downstream link + assert [call[0] for call in recorder.calls] == ["get-issue", "digest", "link"] + assert recorder.calls[0] == ("get-issue", "TC-8100") + assert recorder.calls[-1] == ("link", "TC-8001", "TC-8101", "Depend") + assert {ref: value["key"] for ref, value in registry.items()} == {"upstream": "TC-8100", "downstream": "TC-8101"} + + # Given a refreshed snapshot after the repair, the next retry performs no writes + tasks[0]["comments"] = [{"body": recorder.calls[1][3]}] + bundle["idempotency"]["action_markers"].append(actions[-1]["marker"]) + recorder.calls.clear() + assert executor.execute_plan(result, bundle) == registry + assert recorder.calls == []