From 947be686341141f3fb8c42f3d17fe7c4573082dc Mon Sep 17 00:00:00 2001 From: "Sana Fayyazgit config --global user.email sanafayyaz315@gmail.comgit config --global user.name Sana" Date: Tue, 29 Sep 2026 10:43:12 +0100 Subject: [PATCH 1/6] Configure OpenClaw Forge briefing evaluation and final brief check --- .../runs/openshell-openclaw-pipelinerun.yaml | 8 +++++ submissions/openclaw-forge/eval.yaml | 35 +++++++++++++++++-- 2 files changed, 40 insertions(+), 3 deletions(-) diff --git a/pipeline/runs/openshell-openclaw-pipelinerun.yaml b/pipeline/runs/openshell-openclaw-pipelinerun.yaml index 7fe633a..fc922d0 100644 --- a/pipeline/runs/openshell-openclaw-pipelinerun.yaml +++ b/pipeline/runs/openshell-openclaw-pipelinerun.yaml @@ -9,6 +9,8 @@ spec: pipelineRef: name: abevalflow-pipeline-openshell params: + - name: repo-url + value: "https://github.com/RHEcosystemAppEng/agentic_eval_flow.git" - name: submission-dir value: "openclaw-forge" - name: eval-engine @@ -19,6 +21,10 @@ spec: value: "feat/aeh-openshell-openclaw" - name: pipeline-repo-revision value: "feat/aeh-openshell-openclaw" + - name: agent-eval-harness-repo-url + value: "https://github.com/GuyZivRH/agent-eval-harness.git" + - name: agent-eval-harness-repo-revision + value: "feat/aeh-openshell-openclaw" - name: openshell-gateway-endpoint # Certificate-valid hostname, resolved to this namespace's Service below. value: "https://host.containers.internal:17670" @@ -36,6 +42,8 @@ spec: value: "rits/zai-org/glm-5-3" - name: llm-api-base value: "http://litellm.gz-forge-eval.svc.cluster.local:4000" + - name: llm-base-url + value: "http://litellm.gz-forge-eval.svc.cluster.local:4000/v1" - name: llm-api-key value: "mock" - name: aeh-model-override diff --git a/submissions/openclaw-forge/eval.yaml b/submissions/openclaw-forge/eval.yaml index 565750f..4a5f30d 100644 --- a/submissions/openclaw-forge/eval.yaml +++ b/submissions/openclaw-forge/eval.yaml @@ -21,6 +21,8 @@ runner: # public OpenAI embeddings endpoint with the namespace inference key. # Disable it so the only model request is the intended GLM call. settings: + # Separate in-sandbox provider check; does not change the agent's maxTokens. + llm_preflight_max_tokens: 512 plugins: entries: memory-core: @@ -48,6 +50,13 @@ runner: - id: rits/zai-org/glm-5-3 name: GLM 5.3 api: openai-completions + # When a PipelineRun selects this exact id, keep the intended output + # budget; otherwise AEH adds it with its 8192-token default. + - id: rits/zai-org/GLM-5-3-Flash + name: GLM 5.3 Flash + api: openai-completions + reasoning: true + maxTokens: 128000 execution: mode: case prompt: "{{ input.prompt }}" @@ -76,6 +85,8 @@ outputs: - path: output schema: | response.txt: agent final response (morning briefing or analysis panel) + - path: brief.json + schema: Canonical published morning briefing, when produced judges: # --- Prioritization judges (morning-briefing case) --- @@ -376,11 +387,29 @@ judges: - name: response_received check: | response = outputs.get("output_content", "") or "" - return len(response.strip()) > 0 + return bool(response.strip()) and "The tool run finished, but no final summary was produced" not in response + feedback_type: bool + + - name: published_brief + if: "annotations.get('expected_top_of_mind')" + check: | + import json + files = outputs.get("files") or {} + briefs = [content for path, content in files.items() if path.endswith("brief.json")] + if len(briefs) != 1: + return False + try: + brief = json.loads(briefs[0]) if isinstance(briefs[0], str) else briefs[0] + except (TypeError, ValueError): + return False + return (isinstance(brief, dict) and bool(brief.get("evidenceId")) + and brief.get("scope") == "full") feedback_type: bool -# Thresholds / regression gating disabled for now — scores are still computed -# and shown in the report; the run will not exit 1 on low rubric means. +thresholds: + published_brief: {min_pass_rate: 1.0, max_error_rate: 0.0} + +# Qualitative rubric thresholds remain disabled while we diagnose agent output. # thresholds: # prioritization_recall: {min_mean: 5.0} # prioritization_precision: {min_mean: 5.0} From 7f35ae23ad008ea7a0eabe8c79ea237138500fa0 Mon Sep 17 00:00:00 2001 From: "Sana Fayyazgit config --global user.email sanafayyaz315@gmail.comgit config --global user.name Sana" Date: Tue, 29 Sep 2026 11:26:22 +0100 Subject: [PATCH 2/6] Show optional PipelineRun overrides as comments --- .../runs/openshell-openclaw-pipelinerun.yaml | 20 +++++++++++-------- 1 file changed, 12 insertions(+), 8 deletions(-) diff --git a/pipeline/runs/openshell-openclaw-pipelinerun.yaml b/pipeline/runs/openshell-openclaw-pipelinerun.yaml index fc922d0..f87d723 100644 --- a/pipeline/runs/openshell-openclaw-pipelinerun.yaml +++ b/pipeline/runs/openshell-openclaw-pipelinerun.yaml @@ -9,8 +9,9 @@ spec: pipelineRef: name: abevalflow-pipeline-openshell params: - - name: repo-url - value: "https://github.com/RHEcosystemAppEng/agentic_eval_flow.git" + # Optional: override the submission repository for this run. + # - name: repo-url + # value: "https://github.com/RHEcosystemAppEng/agentic_eval_flow.git" - name: submission-dir value: "openclaw-forge" - name: eval-engine @@ -21,10 +22,11 @@ spec: value: "feat/aeh-openshell-openclaw" - name: pipeline-repo-revision value: "feat/aeh-openshell-openclaw" - - name: agent-eval-harness-repo-url - value: "https://github.com/GuyZivRH/agent-eval-harness.git" - - name: agent-eval-harness-repo-revision - value: "feat/aeh-openshell-openclaw" + # Optional: choose a different harness repository or revision for this run. + # - name: agent-eval-harness-repo-url + # value: "https://github.com/GuyZivRH/agent-eval-harness.git" + # - name: agent-eval-harness-repo-revision + # value: "main" # replace with the commit or branch to evaluate - name: openshell-gateway-endpoint # Certificate-valid hostname, resolved to this namespace's Service below. value: "https://host.containers.internal:17670" @@ -32,6 +34,7 @@ spec: value: "openshell-gateway-mtls" - name: openshell-ai-gateway-ca-secret value: "forge-agent-upstream-tls" + # Replace this digest to evaluate a different OpenClaw sandbox image. - name: openshell-sandbox-image value: "ghcr.io/rh-forge/openclaw-saw-agent@sha256:bcc55e9b7a36d5f65e8ffc75962496f8b3617762a4cdb37fd1cf54611b72d41a" - name: openshell-provider @@ -42,8 +45,9 @@ spec: value: "rits/zai-org/glm-5-3" - name: llm-api-base value: "http://litellm.gz-forge-eval.svc.cluster.local:4000" - - name: llm-base-url - value: "http://litellm.gz-forge-eval.svc.cluster.local:4000/v1" + # Optional: override the prepare-stage AI generation endpoint when enabled. + # - name: llm-base-url + # value: "http://litellm.gz-forge-eval.svc.cluster.local:4000/v1" - name: llm-api-key value: "mock" - name: aeh-model-override From 19d62d8cde6c333b5d6118bde66f8f3a951a125e Mon Sep 17 00:00:00 2001 From: "Sana Fayyazgit config --global user.email sanafayyaz315@gmail.comgit config --global user.name Sana" Date: Tue, 29 Sep 2026 11:31:37 +0100 Subject: [PATCH 3/6] Use placeholders in optional PipelineRun override comments --- pipeline/runs/openshell-openclaw-pipelinerun.yaml | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/pipeline/runs/openshell-openclaw-pipelinerun.yaml b/pipeline/runs/openshell-openclaw-pipelinerun.yaml index f87d723..983393c 100644 --- a/pipeline/runs/openshell-openclaw-pipelinerun.yaml +++ b/pipeline/runs/openshell-openclaw-pipelinerun.yaml @@ -11,7 +11,7 @@ spec: params: # Optional: override the submission repository for this run. # - name: repo-url - # value: "https://github.com/RHEcosystemAppEng/agentic_eval_flow.git" + # value: "" - name: submission-dir value: "openclaw-forge" - name: eval-engine @@ -24,9 +24,9 @@ spec: value: "feat/aeh-openshell-openclaw" # Optional: choose a different harness repository or revision for this run. # - name: agent-eval-harness-repo-url - # value: "https://github.com/GuyZivRH/agent-eval-harness.git" + # value: "" # - name: agent-eval-harness-repo-revision - # value: "main" # replace with the commit or branch to evaluate + # value: "" - name: openshell-gateway-endpoint # Certificate-valid hostname, resolved to this namespace's Service below. value: "https://host.containers.internal:17670" @@ -34,7 +34,7 @@ spec: value: "openshell-gateway-mtls" - name: openshell-ai-gateway-ca-secret value: "forge-agent-upstream-tls" - # Replace this digest to evaluate a different OpenClaw sandbox image. + # To use another sandbox image, replace the value below with . - name: openshell-sandbox-image value: "ghcr.io/rh-forge/openclaw-saw-agent@sha256:bcc55e9b7a36d5f65e8ffc75962496f8b3617762a4cdb37fd1cf54611b72d41a" - name: openshell-provider @@ -47,7 +47,7 @@ spec: value: "http://litellm.gz-forge-eval.svc.cluster.local:4000" # Optional: override the prepare-stage AI generation endpoint when enabled. # - name: llm-base-url - # value: "http://litellm.gz-forge-eval.svc.cluster.local:4000/v1" + # value: "" - name: llm-api-key value: "mock" - name: aeh-model-override From 2ed976ac1dff4689884d240950337008f26033f9 Mon Sep 17 00:00:00 2001 From: "Sana Fayyazgit config --global user.email sanafayyaz315@gmail.comgit config --global user.name Sana" Date: Tue, 29 Sep 2026 11:37:07 +0100 Subject: [PATCH 4/6] Clarify optional PipelineRun parameter examples --- pipeline/runs/openshell-openclaw-pipelinerun.yaml | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/pipeline/runs/openshell-openclaw-pipelinerun.yaml b/pipeline/runs/openshell-openclaw-pipelinerun.yaml index 983393c..5163461 100644 --- a/pipeline/runs/openshell-openclaw-pipelinerun.yaml +++ b/pipeline/runs/openshell-openclaw-pipelinerun.yaml @@ -9,7 +9,7 @@ spec: pipelineRef: name: abevalflow-pipeline-openshell params: - # Optional: override the submission repository for this run. + # Add a submission repository for this run; this replaces the deployed Pipeline default. # - name: repo-url # value: "" - name: submission-dir @@ -22,7 +22,7 @@ spec: value: "feat/aeh-openshell-openclaw" - name: pipeline-repo-revision value: "feat/aeh-openshell-openclaw" - # Optional: choose a different harness repository or revision for this run. + # Add a harness repository or revision for this run; these replace Pipeline defaults. # - name: agent-eval-harness-repo-url # value: "" # - name: agent-eval-harness-repo-revision @@ -45,7 +45,7 @@ spec: value: "rits/zai-org/glm-5-3" - name: llm-api-base value: "http://litellm.gz-forge-eval.svc.cluster.local:4000" - # Optional: override the prepare-stage AI generation endpoint when enabled. + # Add a prepare-stage AI endpoint for this run when generation is enabled. # - name: llm-base-url # value: "" - name: llm-api-key From 56517337e79f15a615e01daa5a7eaa32eb672cf5 Mon Sep 17 00:00:00 2001 From: "Sana Fayyazgit config --global user.email sanafayyaz315@gmail.comgit config --global user.name Sana" Date: Tue, 29 Sep 2026 11:40:53 +0100 Subject: [PATCH 5/6] Remove unrelated prepare endpoint example --- pipeline/runs/openshell-openclaw-pipelinerun.yaml | 3 --- 1 file changed, 3 deletions(-) diff --git a/pipeline/runs/openshell-openclaw-pipelinerun.yaml b/pipeline/runs/openshell-openclaw-pipelinerun.yaml index 5163461..1ec4785 100644 --- a/pipeline/runs/openshell-openclaw-pipelinerun.yaml +++ b/pipeline/runs/openshell-openclaw-pipelinerun.yaml @@ -45,9 +45,6 @@ spec: value: "rits/zai-org/glm-5-3" - name: llm-api-base value: "http://litellm.gz-forge-eval.svc.cluster.local:4000" - # Add a prepare-stage AI endpoint for this run when generation is enabled. - # - name: llm-base-url - # value: "" - name: llm-api-key value: "mock" - name: aeh-model-override From 6597931f01b87defff772c190d921dfab123348a Mon Sep 17 00:00:00 2001 From: "Sana Fayyazgit config --global user.email sanafayyaz315@gmail.comgit config --global user.name Sana" Date: Tue, 29 Sep 2026 11:41:31 +0100 Subject: [PATCH 6/6] Mark optional PipelineRun examples --- pipeline/runs/openshell-openclaw-pipelinerun.yaml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/pipeline/runs/openshell-openclaw-pipelinerun.yaml b/pipeline/runs/openshell-openclaw-pipelinerun.yaml index 1ec4785..f46241e 100644 --- a/pipeline/runs/openshell-openclaw-pipelinerun.yaml +++ b/pipeline/runs/openshell-openclaw-pipelinerun.yaml @@ -9,7 +9,7 @@ spec: pipelineRef: name: abevalflow-pipeline-openshell params: - # Add a submission repository for this run; this replaces the deployed Pipeline default. + # [Optional] Add a submission repository for this run; this replaces the deployed Pipeline default. # - name: repo-url # value: "" - name: submission-dir @@ -22,7 +22,7 @@ spec: value: "feat/aeh-openshell-openclaw" - name: pipeline-repo-revision value: "feat/aeh-openshell-openclaw" - # Add a harness repository or revision for this run; these replace Pipeline defaults. + # [Optional] Add a harness repository or revision for this run; these replace Pipeline defaults. # - name: agent-eval-harness-repo-url # value: "" # - name: agent-eval-harness-repo-revision