Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 9 additions & 0 deletions pipeline/runs/openshell-openclaw-pipelinerun.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,9 @@ spec:
pipelineRef:
name: abevalflow-pipeline-openshell
params:
# [Optional] Add a submission repository for this run; this replaces the deployed Pipeline default.
# - name: repo-url
# value: "<submission-repo-url>"
- name: submission-dir
value: "openclaw-forge"
- name: eval-engine
Expand All @@ -19,13 +22,19 @@ spec:
value: "feat/aeh-openshell-openclaw"
- name: pipeline-repo-revision
value: "feat/aeh-openshell-openclaw"
# [Optional] Add a harness repository or revision for this run; these replace Pipeline defaults.
# - name: agent-eval-harness-repo-url
# value: "<harness-repo-url>"
# - name: agent-eval-harness-repo-revision
# value: "<harness-commit-or-branch>"
- name: openshell-gateway-endpoint
# Certificate-valid hostname, resolved to this namespace's Service below.
value: "https://host.containers.internal:17670"
- name: openshell-mtls-secret
value: "openshell-gateway-mtls"
- name: openshell-ai-gateway-ca-secret
value: "forge-agent-upstream-tls"
# To use another sandbox image, replace the value below with <image@sha256:digest>.
- name: openshell-sandbox-image
value: "ghcr.io/rh-forge/openclaw-saw-agent@sha256:bcc55e9b7a36d5f65e8ffc75962496f8b3617762a4cdb37fd1cf54611b72d41a"
- name: openshell-provider
Expand Down
35 changes: 32 additions & 3 deletions submissions/openclaw-forge/eval.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -21,6 +21,8 @@ runner:
# public OpenAI embeddings endpoint with the namespace inference key.
# Disable it so the only model request is the intended GLM call.
settings:
# Separate in-sandbox provider check; does not change the agent's maxTokens.
llm_preflight_max_tokens: 512
plugins:
entries:
memory-core:
Expand Down Expand Up @@ -48,6 +50,13 @@ runner:
- id: rits/zai-org/glm-5-3
name: GLM 5.3
api: openai-completions
# When a PipelineRun selects this exact id, keep the intended output
# budget; otherwise AEH adds it with its 8192-token default.
- id: rits/zai-org/GLM-5-3-Flash
name: GLM 5.3 Flash
api: openai-completions
reasoning: true
maxTokens: 128000
execution:
mode: case
prompt: "{{ input.prompt }}"
Expand Down Expand Up @@ -76,6 +85,8 @@ outputs:
- path: output
schema: |
response.txt: agent final response (morning briefing or analysis panel)
- path: brief.json
schema: Canonical published morning briefing, when produced

judges:
# --- Prioritization judges (morning-briefing case) ---
Expand Down Expand Up @@ -376,11 +387,29 @@ judges:
- name: response_received
check: |
response = outputs.get("output_content", "") or ""
return len(response.strip()) > 0
return bool(response.strip()) and "The tool run finished, but no final summary was produced" not in response
feedback_type: bool

- name: published_brief
if: "annotations.get('expected_top_of_mind')"
check: |
import json
files = outputs.get("files") or {}
briefs = [content for path, content in files.items() if path.endswith("brief.json")]
if len(briefs) != 1:
return False
try:
brief = json.loads(briefs[0]) if isinstance(briefs[0], str) else briefs[0]
except (TypeError, ValueError):
return False
return (isinstance(brief, dict) and bool(brief.get("evidenceId"))
and brief.get("scope") == "full")
feedback_type: bool

# Thresholds / regression gating disabled for now — scores are still computed
# and shown in the report; the run will not exit 1 on low rubric means.
thresholds:
published_brief: {min_pass_rate: 1.0, max_error_rate: 0.0}

# Qualitative rubric thresholds remain disabled while we diagnose agent output.
# thresholds:
# prioritization_recall: {min_mean: 5.0}
# prioritization_precision: {min_mean: 5.0}
Expand Down
Loading