diff --git a/config/forge-saw/README.md b/config/forge-saw/README.md index 6f1ea23..b2b566c 100644 --- a/config/forge-saw/README.md +++ b/config/forge-saw/README.md @@ -148,6 +148,24 @@ oc get secret openshell-credentials -n guy-ziv-evalflow -o json \ `aeh-openshell-eval` TCP-checks `:17670` and optionally runs `openshell sandbox list`. If SAW is down the **run fails**. That step does not install SAW. +## NetworkPolicy (Forge shared namespace) + +When SAW VMs and Tekton share a namespace with default-deny policies, evaluate +pods must be allowed to reach the agent gateway on `:17670`. Use the canonical +Pipeline name `abevalflow-pipeline-openshell` (so `tekton.dev/pipeline` matches) +and label PipelineRuns with `app.kubernetes.io/part-of: abevalflow`. + +Same-namespace template: + +```bash +sed "s/NAMESPACE/${EVAL_NS}/g" config/forge-saw/networkpolicy-ci-openshell.yaml \ + | oc apply -f - +``` + +Do not create ad-hoc copies of the Pipeline under a different name unless those +PipelineRuns also carry the `part-of=abevalflow` label and the NetworkPolicies +above are applied — otherwise gateway preflight times out. + ## Image Stock `agent-eval-harness:v1.0.x` cannot import `agent_eval.openshell`. Point diff --git a/config/forge-saw/networkpolicy-ci-openshell.yaml b/config/forge-saw/networkpolicy-ci-openshell.yaml new file mode 100644 index 0000000..5bd6f18 --- /dev/null +++ b/config/forge-saw/networkpolicy-ci-openshell.yaml @@ -0,0 +1,213 @@ +# NetworkPolicies for OpenShell OpenClaw evals when SAW VMs and the Tekton +# pipeline share one namespace (Forge workspace style). +# +# Select evaluate pods by either: +# - tekton.dev/pipeline=abevalflow-pipeline-openshell (canonical Pipeline name) +# - app.kubernetes.io/part-of=abevalflow (set on PipelineRun labels) +# +# Do NOT rename/copy the Pipeline to an ad-hoc name without also labeling the +# PipelineRun with app.kubernetes.io/part-of=abevalflow — otherwise default-deny +# blocks TCP to the gateway on :17670 and evaluate fails preflight. +# +# Apply after substituting NAMESPACE (and AGENT_VM / INTEG_VM if different): +# sed "s/NAMESPACE/forge-nommen/g" networkpolicy-ci-openshell.yaml | oc apply -f - +--- +apiVersion: networking.k8s.io/v1 +kind: NetworkPolicy +metadata: + name: abevalflow-allow-ci-agent-gateway + namespace: NAMESPACE + labels: + app.kubernetes.io/part-of: abevalflow + app.kubernetes.io/component: forge-saw +spec: + podSelector: + matchLabels: + app.kubernetes.io/name: openshell-saw-agent + vm.kubevirt.io/name: openshell-saw-agent + policyTypes: [Ingress] + ingress: + - from: + - podSelector: + matchLabels: + tekton.dev/pipeline: abevalflow-pipeline-openshell + - podSelector: + matchLabels: + app.kubernetes.io/part-of: abevalflow + ports: + - protocol: TCP + port: 17670 +--- +apiVersion: networking.k8s.io/v1 +kind: NetworkPolicy +metadata: + name: abevalflow-allow-ci-integ-gateway + namespace: NAMESPACE + labels: + app.kubernetes.io/part-of: abevalflow + app.kubernetes.io/component: forge-saw +spec: + podSelector: + matchLabels: + app.kubernetes.io/name: openshell-saw-integ + vm.kubevirt.io/name: openshell-saw-integ + policyTypes: [Ingress] + ingress: + - from: + - podSelector: + matchLabels: + tekton.dev/pipeline: abevalflow-pipeline-openshell + - podSelector: + matchLabels: + app.kubernetes.io/part-of: abevalflow + ports: + - protocol: TCP + port: 17670 + - protocol: TCP + port: 18082 +--- +apiVersion: networking.k8s.io/v1 +kind: NetworkPolicy +metadata: + name: abevalflow-allow-ci-egress-by-pipeline + namespace: NAMESPACE + labels: + app.kubernetes.io/part-of: abevalflow + app.kubernetes.io/component: forge-saw +spec: + podSelector: + matchLabels: + tekton.dev/pipeline: abevalflow-pipeline-openshell + policyTypes: [Egress] + egress: + - to: + - podSelector: + matchLabels: + app.kubernetes.io/name: openshell-saw-agent + vm.kubevirt.io/name: openshell-saw-agent + ports: + - protocol: TCP + port: 17670 + - protocol: TCP + port: 8443 + - to: + - podSelector: + matchLabels: + app.kubernetes.io/name: openshell-saw-integ + vm.kubevirt.io/name: openshell-saw-integ + ports: + - protocol: TCP + port: 17670 + - protocol: TCP + port: 18082 + - protocol: TCP + port: 8443 + - to: + - podSelector: + matchLabels: + app.kubernetes.io/name: litellm + ports: + - protocol: TCP + port: 4000 + - to: + - podSelector: + matchLabels: + app.kubernetes.io/name: mlflow + app.kubernetes.io/part-of: abevalflow + ports: + - protocol: TCP + port: 5000 + - to: + - podSelector: + matchLabels: + app.kubernetes.io/name: minio + app.kubernetes.io/part-of: abevalflow + ports: + - protocol: TCP + port: 9000 + - to: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: openshift-dns + ports: + - protocol: UDP + port: 53 + - protocol: TCP + port: 53 + - protocol: UDP + port: 5353 + - protocol: TCP + port: 5353 +--- +apiVersion: networking.k8s.io/v1 +kind: NetworkPolicy +metadata: + name: abevalflow-allow-ci-egress-by-part-of + namespace: NAMESPACE + labels: + app.kubernetes.io/part-of: abevalflow + app.kubernetes.io/component: forge-saw +spec: + podSelector: + matchLabels: + app.kubernetes.io/part-of: abevalflow + policyTypes: [Egress] + egress: + - to: + - podSelector: + matchLabels: + app.kubernetes.io/name: openshell-saw-agent + vm.kubevirt.io/name: openshell-saw-agent + ports: + - protocol: TCP + port: 17670 + - protocol: TCP + port: 8443 + - to: + - podSelector: + matchLabels: + app.kubernetes.io/name: openshell-saw-integ + vm.kubevirt.io/name: openshell-saw-integ + ports: + - protocol: TCP + port: 17670 + - protocol: TCP + port: 18082 + - protocol: TCP + port: 8443 + - to: + - podSelector: + matchLabels: + app.kubernetes.io/name: litellm + ports: + - protocol: TCP + port: 4000 + - to: + - podSelector: + matchLabels: + app.kubernetes.io/name: mlflow + app.kubernetes.io/part-of: abevalflow + ports: + - protocol: TCP + port: 5000 + - to: + - podSelector: + matchLabels: + app.kubernetes.io/name: minio + app.kubernetes.io/part-of: abevalflow + ports: + - protocol: TCP + port: 9000 + - to: + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: openshift-dns + ports: + - protocol: UDP + port: 53 + - protocol: TCP + port: 53 + - protocol: UDP + port: 5353 + - protocol: TCP + port: 5353 diff --git a/config/forge-saw/networkpolicy-gateway-from-abeval.yaml b/config/forge-saw/networkpolicy-gateway-from-abeval.yaml index af1aaef..9071391 100644 --- a/config/forge-saw/networkpolicy-gateway-from-abeval.yaml +++ b/config/forge-saw/networkpolicy-gateway-from-abeval.yaml @@ -1,6 +1,11 @@ # Allow evaluate Task pods to reach the OpenShell gateway on :17670. # bootstrap.sh rewrites namespace / from-namespace (SAW_NS, EVAL_NS). # Apply in the SAW namespace, not from the evaluate PipelineRun. +# +# Prefer matching Tekton pods by the canonical Pipeline name or the +# app.kubernetes.io/part-of=abevalflow label on PipelineRuns. A bare +# podSelector: {} from the whole EVAL_NS is broader than needed when SAW +# and eval share a Forge workspace namespace — see networkpolicy-ci-openshell.yaml. apiVersion: networking.k8s.io/v1 kind: NetworkPolicy metadata: @@ -20,7 +25,15 @@ spec: - namespaceSelector: matchLabels: kubernetes.io/metadata.name: guy-ziv-evalflow - - podSelector: {} + podSelector: + matchLabels: + tekton.dev/pipeline: abevalflow-pipeline-openshell + - namespaceSelector: + matchLabels: + kubernetes.io/metadata.name: guy-ziv-evalflow + podSelector: + matchLabels: + app.kubernetes.io/part-of: abevalflow ports: - protocol: TCP port: 17670 diff --git a/pipeline/pipelines/ci-pipeline-openshell.yaml b/pipeline/pipelines/ci-pipeline-openshell.yaml index 07a41c5..63032d7 100644 --- a/pipeline/pipelines/ci-pipeline-openshell.yaml +++ b/pipeline/pipelines/ci-pipeline-openshell.yaml @@ -5,6 +5,7 @@ metadata: namespace: ab-eval-flow labels: app.kubernetes.io/name: abevalflow + app.kubernetes.io/part-of: abevalflow app.kubernetes.io/version: "2.0" app.kubernetes.io/profile: aeh-openshell-openclaw spec: @@ -102,8 +103,11 @@ spec: default: "https://github.com/GuyZivRH/agent-eval-harness.git" - name: agent-eval-harness-repo-revision type: string - default: "feat/aeh-openshell-openclaw" - description: Harness revision with agent_eval.openshell (feature-branch pin) + default: "main" + description: >- + Harness revision with agent_eval.openshell. Use a tip that includes the + OpenClaw brief-reader/gateway fixes (GuyZivRH/agent-eval-harness#9) until + that lands on upstream main. - name: openshell-cli-version type: string default: "0.0.116" @@ -112,7 +116,7 @@ spec: default: "https://openshell.openshell.svc.cluster.local:17670" - name: openshell-sandbox-image type: string - default: "ghcr.io/rh-forge/openclaw-saw-agent@sha256:bcc55e9b7a36d5f65e8ffc75962496f8b3617762a4cdb37fd1cf54611b72d41a" + default: "ghcr.io/rh-forge/openclaw-saw-agent@sha256:b47b92a6b3fd03327c1f2093a5c28aba0fdf3cb620e9154335688900191fe2b9" - name: openshell-provider type: string default: "forge-ai-gateway,m365-read-intervm,slack-read-proxy,drafts-service-agent" diff --git a/pipeline/runs/openshell-openclaw-pipelinerun.yaml b/pipeline/runs/openshell-openclaw-pipelinerun.yaml index 7fe633a..9b9b1de 100644 --- a/pipeline/runs/openshell-openclaw-pipelinerun.yaml +++ b/pipeline/runs/openshell-openclaw-pipelinerun.yaml @@ -5,10 +5,18 @@ apiVersion: tekton.dev/v1 kind: PipelineRun metadata: generateName: aeh-openshell-openclaw- + labels: + # Required when NetworkPolicies select on part-of (and recommended always). + # Keep pipelineRef.name=abevalflow-pipeline-openshell so tekton.dev/pipeline + # also matches the canonical CI NetworkPolicies. + app.kubernetes.io/part-of: abevalflow spec: pipelineRef: name: abevalflow-pipeline-openshell params: + # [Optional] Add a submission repository for this run; this replaces the deployed Pipeline default. + # - name: repo-url + # value: "" - name: submission-dir value: "openclaw-forge" - name: eval-engine @@ -19,6 +27,11 @@ spec: value: "feat/aeh-openshell-openclaw" - name: pipeline-repo-revision value: "feat/aeh-openshell-openclaw" + # [Optional] Add a harness repository or revision for this run; these replace Pipeline defaults. + # - name: agent-eval-harness-repo-url + # value: "" + # - name: agent-eval-harness-repo-revision + # value: "" - name: openshell-gateway-endpoint # Certificate-valid hostname, resolved to this namespace's Service below. value: "https://host.containers.internal:17670" @@ -26,8 +39,9 @@ spec: value: "openshell-gateway-mtls" - name: openshell-ai-gateway-ca-secret value: "forge-agent-upstream-tls" + # To use another sandbox image, replace the value below with . - name: openshell-sandbox-image - value: "ghcr.io/rh-forge/openclaw-saw-agent@sha256:bcc55e9b7a36d5f65e8ffc75962496f8b3617762a4cdb37fd1cf54611b72d41a" + value: "ghcr.io/rh-forge/openclaw-saw-agent@sha256:b47b92a6b3fd03327c1f2093a5c28aba0fdf3cb620e9154335688900191fe2b9" - name: openshell-provider value: "forge-ai-gateway,m365-read-intervm,slack-read-proxy,drafts-service-agent" - name: aeh-openshell-image diff --git a/pipeline/tasks/phases/evaluate.yaml b/pipeline/tasks/phases/evaluate.yaml index a50eff9..2e4187d 100644 --- a/pipeline/tasks/phases/evaluate.yaml +++ b/pipeline/tasks/phases/evaluate.yaml @@ -184,7 +184,7 @@ spec: default: "https://github.com/GuyZivRH/agent-eval-harness.git" - name: agent-eval-harness-repo-revision type: string - default: "feat/aeh-openshell-openclaw" + default: "main" description: >- Branch or tag of GuyZivRH/agent-eval-harness to clone. Must include the OpenShell create keep-alive fix (`--detach` + `sleep infinity`). @@ -196,7 +196,7 @@ spec: Override only if using a different gateway (e.g. forge-saw :17670). - name: openshell-sandbox-image type: string - default: "ghcr.io/rh-forge/openclaw-saw-agent@sha256:bcc55e9b7a36d5f65e8ffc75962496f8b3617762a4cdb37fd1cf54611b72d41a" + default: "ghcr.io/rh-forge/openclaw-saw-agent@sha256:b47b92a6b3fd03327c1f2093a5c28aba0fdf3cb620e9154335688900191fe2b9" description: >- Immutable SAW Chief-of-Staff image with the daily-briefing skill and governed M365/Slack tools. @@ -1826,6 +1826,46 @@ spec: PY chmod 600 "$HOME/.config/openshell/gateways/$GW_NAME/oidc_token.json" echo "OpenShell OIDC token cache populated via client credentials" + # A case can run for 15 minutes, while the client-credentials token + # expires after roughly five. Refresh the CLI cache throughout this + # step so later artifact downloads and sandbox creation stay authorized. + OIDC_TOKEN_CACHE="$HOME/.config/openshell/gateways/$GW_NAME/oidc_token.json" + ( + umask 077 + trap 'rm -f "${OIDC_TOKEN_JSON}.refresh"' EXIT + while sleep 120; do + if curl -ksSf --max-time 20 \ + -u "${OPENSHELL_OIDC_CLIENT_ID}:${OPENSHELL_OIDC_CLIENT_SECRET}" \ + -d grant_type=client_credentials \ + "${OPENSHELL_OIDC_ISSUER%/}/protocol/openid-connect/token" \ + >"${OIDC_TOKEN_JSON}.refresh" 2>/dev/null; then + python3 - "${OIDC_TOKEN_JSON}.refresh" "$OIDC_TOKEN_CACHE" \ + "$OPENSHELL_OIDC_ISSUER" "$OPENSHELL_OIDC_CLIENT_ID" <<'REFRESH' || true + import json, os, sys, tempfile, time + src, dst, issuer, client_id = sys.argv[1:] + with open(src, encoding="utf-8") as stream: + payload = json.load(stream) + cache = {"access_token": payload["access_token"], + "expires_at": int(time.time()) + int(payload.get("expires_in", 300)), + "issuer": issuer, "client_id": client_id} + fd, tmp = tempfile.mkstemp(dir=os.path.dirname(dst), prefix=".oidc-token-") + try: + with os.fdopen(fd, "w", encoding="utf-8") as stream: + json.dump(cache, stream) + os.chmod(tmp, 0o600) + os.replace(tmp, dst) + finally: + if os.path.exists(tmp): + os.unlink(tmp) + REFRESH + else + echo "WARN: OpenShell OIDC token refresh failed; retrying" >&2 + fi + done + ) & + OIDC_REFRESH_PID=$! + trap 'kill "$OIDC_REFRESH_PID" 2>/dev/null || true' EXIT + echo "OpenShell OIDC token refresher started" export OPENSHELL_GATEWAY="$GW_NAME" echo "OpenShell OIDC gateway configured: ${GW_NAME}" else @@ -2070,6 +2110,13 @@ spec: echo "Model: $MODEL" export AGENT_EVAL_HARNESS_ROOT="$HARNESS_DIR" + # A published Forge briefing needs installation identity. Keep this + # synthetic fixture scoped to submissions that supply one; an explicit + # harness setting takes precedence for other deployments. + if [ -z "${AGENT_EVAL_FORGE_USER_FILE:-}" ] && [ -f "$SUBMISSION_DIR/fixtures/USER.md" ]; then + export AGENT_EVAL_FORGE_USER_FILE="$SUBMISSION_DIR/fixtures/USER.md" + echo "Using submission installation profile fixture" + fi if [ "$(params.enable-mlflow)" = "true" ] && [ -n "$(params.mlflow-tracking-uri)" ]; then export MLFLOW_TRACKING_URI="$(params.mlflow-tracking-uri)" echo "MLFLOW_TRACKING_URI set for AEH harness + CI logger" diff --git a/pipeline/tasks/phases/prepare.yaml b/pipeline/tasks/phases/prepare.yaml index 33f0f21..9da41f6 100644 --- a/pipeline/tasks/phases/prepare.yaml +++ b/pipeline/tasks/phases/prepare.yaml @@ -122,16 +122,21 @@ spec: echo "=== PREPARE PHASE: Clone Pipeline Repo ===" PIPELINE_DIR="$(workspaces.source.path)/_pipeline" + PIPELINE_REV="$(params.pipeline-repo-revision)" if [ -d "$PIPELINE_DIR/.git" ]; then echo "Pipeline repo already cloned, updating..." cd "$PIPELINE_DIR" - git fetch origin "$(params.pipeline-repo-revision)" - git checkout "$(params.pipeline-repo-revision)" 2>/dev/null || git checkout FETCH_HEAD + git fetch origin "$PIPELINE_REV" --depth 1 2>/dev/null \ + || git fetch origin "$PIPELINE_REV" + git checkout "$PIPELINE_REV" 2>/dev/null || git checkout FETCH_HEAD else - git clone --depth 1 --branch "$(params.pipeline-repo-revision)" \ - "$(params.pipeline-repo-url)" "$PIPELINE_DIR" + if ! git clone --depth 1 --branch "$PIPELINE_REV" \ + "$(params.pipeline-repo-url)" "$PIPELINE_DIR" 2>/dev/null; then + git clone "$(params.pipeline-repo-url)" "$PIPELINE_DIR" + git -C "$PIPELINE_DIR" checkout "$PIPELINE_REV" + fi fi - echo "Pipeline repo ready at $(params.pipeline-repo-revision)" + echo "Pipeline repo ready at $(git -C "$PIPELINE_DIR" rev-parse --short HEAD)" # Step 3: Generate tests (conditional) - name: generate-tests diff --git a/submissions/openclaw-forge/eval.yaml b/submissions/openclaw-forge/eval.yaml index 565750f..4a5f30d 100644 --- a/submissions/openclaw-forge/eval.yaml +++ b/submissions/openclaw-forge/eval.yaml @@ -21,6 +21,8 @@ runner: # public OpenAI embeddings endpoint with the namespace inference key. # Disable it so the only model request is the intended GLM call. settings: + # Separate in-sandbox provider check; does not change the agent's maxTokens. + llm_preflight_max_tokens: 512 plugins: entries: memory-core: @@ -48,6 +50,13 @@ runner: - id: rits/zai-org/glm-5-3 name: GLM 5.3 api: openai-completions + # When a PipelineRun selects this exact id, keep the intended output + # budget; otherwise AEH adds it with its 8192-token default. + - id: rits/zai-org/GLM-5-3-Flash + name: GLM 5.3 Flash + api: openai-completions + reasoning: true + maxTokens: 128000 execution: mode: case prompt: "{{ input.prompt }}" @@ -76,6 +85,8 @@ outputs: - path: output schema: | response.txt: agent final response (morning briefing or analysis panel) + - path: brief.json + schema: Canonical published morning briefing, when produced judges: # --- Prioritization judges (morning-briefing case) --- @@ -376,11 +387,29 @@ judges: - name: response_received check: | response = outputs.get("output_content", "") or "" - return len(response.strip()) > 0 + return bool(response.strip()) and "The tool run finished, but no final summary was produced" not in response + feedback_type: bool + + - name: published_brief + if: "annotations.get('expected_top_of_mind')" + check: | + import json + files = outputs.get("files") or {} + briefs = [content for path, content in files.items() if path.endswith("brief.json")] + if len(briefs) != 1: + return False + try: + brief = json.loads(briefs[0]) if isinstance(briefs[0], str) else briefs[0] + except (TypeError, ValueError): + return False + return (isinstance(brief, dict) and bool(brief.get("evidenceId")) + and brief.get("scope") == "full") feedback_type: bool -# Thresholds / regression gating disabled for now — scores are still computed -# and shown in the report; the run will not exit 1 on low rubric means. +thresholds: + published_brief: {min_pass_rate: 1.0, max_error_rate: 0.0} + +# Qualitative rubric thresholds remain disabled while we diagnose agent output. # thresholds: # prioritization_recall: {min_mean: 5.0} # prioritization_precision: {min_mean: 5.0} diff --git a/submissions/openclaw-forge/fixtures/USER.md b/submissions/openclaw-forge/fixtures/USER.md new file mode 100644 index 0000000..e88fe80 --- /dev/null +++ b/submissions/openclaw-forge/fixtures/USER.md @@ -0,0 +1,7 @@ +# Synthetic evaluation identity + +- Display name: Alex Rivera +- Role: Engineering executive +- Initials: AR +- Primary email: tbx-demo2@dev.mscloud.ibm.com +- Time zone: America/New_York diff --git a/tests/test_openshell_pipeline_profile.py b/tests/test_openshell_pipeline_profile.py index f2a6f9e..79790cc 100644 --- a/tests/test_openshell_pipeline_profile.py +++ b/tests/test_openshell_pipeline_profile.py @@ -42,7 +42,7 @@ def test_openshell_defaults(self): assert defaults["aeh-runner"] == "openshell" assert defaults["aeh-openshell-image"] == "registry.access.redhat.com/ubi9/python-311:9.6" assert defaults["openshell-sandbox-image"] == ( - "ghcr.io/rh-forge/openclaw-saw-agent@sha256:bcc55e9b7a36d5f65e8ffc75962496f8b3617762a4cdb37fd1cf54611b72d41a" + "ghcr.io/rh-forge/openclaw-saw-agent@sha256:b47b92a6b3fd03327c1f2093a5c28aba0fdf3cb620e9154335688900191fe2b9" ) assert defaults["enable-mlflow"] == "true" assert defaults["mlflow-tracking-uri"] == ("http://abevalflow-mlflow.gz-forge-eval.svc.cluster.local:5000") @@ -61,6 +61,24 @@ def test_evaluate_openshell_step_logs_mlflow(self): assert uri_note in openshell assert openshell.find(uri_note) < openshell.find("scripts/run_aeh.py") + def test_forge_briefing_has_installation_user_fixture(self): + fixture = REPO / "submissions" / "openclaw-forge" / "fixtures" / "USER.md" + fields = {} + for line in fixture.read_text().splitlines(): + if line.startswith("- ") and ": " in line: + key, value = line[2:].split(": ", 1) + fields[key.lower()] = value.strip() + assert all(fields.get(key) and not fields[key].startswith("<") + for key in ("display name", "role", "initials")) + scene = _load(REPO / "submissions" / "openclaw-forge" / "scenes" / "monday-acquisition.yaml") + assert fields["primary email"] == scene["m365"]["user"] + + task = _load(REPO / "pipeline" / "tasks" / "phases" / "evaluate.yaml") + step = next(s for s in task["spec"]["steps"] if s["name"] == "aeh-openshell-eval") + script = step["script"] + assert 'AGENT_EVAL_FORGE_USER_FILE="$SUBMISSION_DIR/fixtures/USER.md"' in script + assert script.index('[ -f "$SUBMISSION_DIR/fixtures/USER.md" ]') < script.index("scripts/run_aeh.py") + def test_harbor_profiles_still_include_test(self): for path in (CI, CI_DEV): names = _task_names(_load(path)["spec"])