diff --git a/pipeline/runs/openshell-openclaw-pipelinerun.yaml b/pipeline/runs/openshell-openclaw-pipelinerun.yaml index 7fe633a..f46241e 100644 --- a/pipeline/runs/openshell-openclaw-pipelinerun.yaml +++ b/pipeline/runs/openshell-openclaw-pipelinerun.yaml @@ -9,6 +9,9 @@ spec: pipelineRef: name: abevalflow-pipeline-openshell params: + # [Optional] Add a submission repository for this run; this replaces the deployed Pipeline default. + # - name: repo-url + # value: "" - name: submission-dir value: "openclaw-forge" - name: eval-engine @@ -19,6 +22,11 @@ spec: value: "feat/aeh-openshell-openclaw" - name: pipeline-repo-revision value: "feat/aeh-openshell-openclaw" + # [Optional] Add a harness repository or revision for this run; these replace Pipeline defaults. + # - name: agent-eval-harness-repo-url + # value: "" + # - name: agent-eval-harness-repo-revision + # value: "" - name: openshell-gateway-endpoint # Certificate-valid hostname, resolved to this namespace's Service below. value: "https://host.containers.internal:17670" @@ -26,6 +34,7 @@ spec: value: "openshell-gateway-mtls" - name: openshell-ai-gateway-ca-secret value: "forge-agent-upstream-tls" + # To use another sandbox image, replace the value below with . - name: openshell-sandbox-image value: "ghcr.io/rh-forge/openclaw-saw-agent@sha256:bcc55e9b7a36d5f65e8ffc75962496f8b3617762a4cdb37fd1cf54611b72d41a" - name: openshell-provider diff --git a/submissions/openclaw-forge/eval.yaml b/submissions/openclaw-forge/eval.yaml index 565750f..4a5f30d 100644 --- a/submissions/openclaw-forge/eval.yaml +++ b/submissions/openclaw-forge/eval.yaml @@ -21,6 +21,8 @@ runner: # public OpenAI embeddings endpoint with the namespace inference key. # Disable it so the only model request is the intended GLM call. settings: + # Separate in-sandbox provider check; does not change the agent's maxTokens. + llm_preflight_max_tokens: 512 plugins: entries: memory-core: @@ -48,6 +50,13 @@ runner: - id: rits/zai-org/glm-5-3 name: GLM 5.3 api: openai-completions + # When a PipelineRun selects this exact id, keep the intended output + # budget; otherwise AEH adds it with its 8192-token default. + - id: rits/zai-org/GLM-5-3-Flash + name: GLM 5.3 Flash + api: openai-completions + reasoning: true + maxTokens: 128000 execution: mode: case prompt: "{{ input.prompt }}" @@ -76,6 +85,8 @@ outputs: - path: output schema: | response.txt: agent final response (morning briefing or analysis panel) + - path: brief.json + schema: Canonical published morning briefing, when produced judges: # --- Prioritization judges (morning-briefing case) --- @@ -376,11 +387,29 @@ judges: - name: response_received check: | response = outputs.get("output_content", "") or "" - return len(response.strip()) > 0 + return bool(response.strip()) and "The tool run finished, but no final summary was produced" not in response + feedback_type: bool + + - name: published_brief + if: "annotations.get('expected_top_of_mind')" + check: | + import json + files = outputs.get("files") or {} + briefs = [content for path, content in files.items() if path.endswith("brief.json")] + if len(briefs) != 1: + return False + try: + brief = json.loads(briefs[0]) if isinstance(briefs[0], str) else briefs[0] + except (TypeError, ValueError): + return False + return (isinstance(brief, dict) and bool(brief.get("evidenceId")) + and brief.get("scope") == "full") feedback_type: bool -# Thresholds / regression gating disabled for now — scores are still computed -# and shown in the report; the run will not exit 1 on low rubric means. +thresholds: + published_brief: {min_pass_rate: 1.0, max_error_rate: 0.0} + +# Qualitative rubric thresholds remain disabled while we diagnose agent output. # thresholds: # prioritization_recall: {min_mean: 5.0} # prioritization_precision: {min_mean: 5.0}