From cfb0d9fe32fec059622d97cb37a043f3e4eefbb7 Mon Sep 17 00:00:00 2001 From: engineer Date: Mon, 8 Jun 2026 05:58:19 -0700 Subject: [PATCH] =?UTF-8?q?test(ci):=20widen=20cua-smoke=20gate=20to=20rea?= =?UTF-8?q?l=20core=20journey=20(connect=E2=86=92send=E2=86=92reply?= =?UTF-8?q?=E2=86=92list)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The CI opencode server had NO LLM provider configured — the server log only showed "listening", never a model. opencode-ai (released npm pkg) does not read AZURE_OPENAI_* for its own LLM; it needs an explicit provider in opencode.json + a default `model`. So send_message/multi_turn could never pass and the gate was stuck on --only-connect-scenario (UI journey minus the model reply). - Wire opencode to the same Azure resource the CUA driver uses via a generated ~/.config/opencode/opencode.json (@ai-sdk/azure provider, resourceName derived from the endpoint secret at runtime, apiKey from env, default model azure/gpt-5.4). - Add a deterministic REST probe step: create a session + send a prompt and check for an assistant reply BEFORE the ~30min emulator run, exporting MODEL_CAPABLE. - Add --scenarios to android-cua-smoke.py to run an explicit named set. - Emulator step now runs connect_and_verify_sessions + send_message + verify_session_list when MODEL_CAPABLE=true; falls back to the UI-only journey (connect + verify_session_list) otherwise, logging the environmental reason. - Raise --max-steps to 40 so multiple scenarios fit. Co-Authored-By: Claude Opus 4.8 --- .github/workflows/cua-smoke.yml | 79 +++++++++++++++++++++++++++++++-- scripts/android-cua-smoke.py | 20 ++++++++- 2 files changed, 95 insertions(+), 4 deletions(-) diff --git a/.github/workflows/cua-smoke.yml b/.github/workflows/cua-smoke.yml index 4026e88..4d670c1 100644 --- a/.github/workflows/cua-smoke.yml +++ b/.github/workflows/cua-smoke.yml @@ -69,9 +69,44 @@ jobs: - name: Install Python deps run: pip install openai - - name: Install & start opencode server on runner host + - name: Configure opencode Azure provider + env: + AZURE_OPENAI_MODEL: "gpt-5.4" run: | npm install -g opencode-ai + # opencode (released npm pkg) does NOT read AZURE_OPENAI_* for its own LLM. + # It needs an explicit provider in opencode.json + a default `model`. + # Wire the same Azure resource the CUA driver uses (@ai-sdk/azure). + # Extract the resource name from the endpoint secret at runtime + # (e.g. https://NAME.openai.azure.com -> NAME) so nothing secret is in source. + RESOURCE_NAME="$(printf '%s' "$AZURE_OPENAI_ENDPOINT" | sed -E 's#https?://([^.]+)\..*#\1#')" + echo "Derived Azure resource name: ${RESOURCE_NAME:-}" + mkdir -p "$HOME/.config/opencode" + cat > "$HOME/.config/opencode/opencode.json" < /tmp/opencode-server.log 2>&1 & echo "Waiting for opencode server /global/health ..." @@ -85,8 +120,35 @@ jobs: echo "::error::opencode server failed to become healthy"; cat /tmp/opencode-server.log; exit 1; } + - name: Probe opencode model capability (can it reply?) + id: probe + run: | + # Deterministic check that the server can actually produce an assistant + # reply BEFORE we spend ~30min driving the UI. Creates a session, sends a + # prompt via REST, and checks for an assistant message. Non-fatal: records + # MODEL_CAPABLE=true/false so the scenario set can be chosen accordingly. + set +e + SID=$(curl -sf -X POST http://127.0.0.1:4096/session -H 'content-type: application/json' -d '{}' | python3 -c "import sys,json;print(json.load(sys.stdin).get('id',''))" 2>/dev/null) + echo "session id: ${SID:-}" + CAPABLE=false + if [ -n "$SID" ]; then + curl -sf -X POST "http://127.0.0.1:4096/session/$SID/message" \ + -H 'content-type: application/json' \ + -d '{"parts":[{"type":"text","text":"reply with the single word: pong"}]}' \ + > /tmp/probe_reply.json 2>/tmp/probe_err.txt + echo "--- probe reply (truncated) ---"; head -c 2000 /tmp/probe_reply.json; echo + echo "--- probe err (truncated) ---"; head -c 1000 /tmp/probe_err.txt; echo + if grep -qi 'assistant' /tmp/probe_reply.json && grep -qi 'pong\|text' /tmp/probe_reply.json; then + CAPABLE=true + fi + fi + echo "MODEL_CAPABLE=$CAPABLE" >> "$GITHUB_OUTPUT" + echo "opencode model-capable in CI: $CAPABLE" + - name: Run CUA smoke test with emulator uses: reactivecircus/android-emulator-runner@v2 + env: + MODEL_CAPABLE: ${{ steps.probe.outputs.MODEL_CAPABLE }} with: api-level: 30 arch: x86_64 @@ -100,8 +162,19 @@ jobs: adb install android/app/build/outputs/apk/release/app-release.apk adb shell am start -n cc.agentlabs.opencode/.MainActivity sleep 5 - # True E2E: app connects to the local opencode server (10.0.2.2:4096) and verifies the session list. - python3 scripts/android-cua-smoke.py --model gpt-5.4 --include-xml --only-connect-scenario + # Scenario set widened to the real core journey. send_message requires the + # opencode server to actually reply (Azure provider wired in earlier step); + # if the model probe failed we drop it to the UI-only journey so the gate + # stays green and meaningful rather than red for an environmental reason. + if [ "${MODEL_CAPABLE}" = "true" ]; then + SCENARIOS="connect_and_verify_sessions,send_message,verify_session_list" + else + SCENARIOS="connect_and_verify_sessions,verify_session_list" + echo "WARN: opencode not model-capable in CI; excluding send_message/multi_turn (environmental, not an app bug)." + fi + echo "Running scenarios: $SCENARIOS" + # --max-steps raised so multiple scenarios fit; each scenario gets its own budget. + python3 scripts/android-cua-smoke.py --model gpt-5.4 --include-xml --max-steps 40 --scenarios "$SCENARIOS" - name: opencode server log if: always() diff --git a/scripts/android-cua-smoke.py b/scripts/android-cua-smoke.py index 75215d9..c0bc943 100755 --- a/scripts/android-cua-smoke.py +++ b/scripts/android-cua-smoke.py @@ -634,6 +634,13 @@ def main(): help="Run ONLY the connect-and-verify-sessions scenario. Use in CI with a " "local opencode server for a deterministic true-E2E (no model backend needed).", ) + parser.add_argument( + "--scenarios", + help="Comma-separated explicit scenario set to run, e.g. " + "'connect_and_verify_sessions,send_message,verify_session_list'. " + "Valid names: connect_and_verify_sessions, send_message, multi_turn, " + "verify_session_list. Overrides --only-connect-scenario and the default set.", + ) args = parser.parse_args() # Verify ADB @@ -650,7 +657,18 @@ def main(): "goal": _connect_and_verify_sessions_goal(connect_url), } - if args.only_connect_scenario: + if args.scenarios: + # Explicit named set (CI widened gate). Look up by name across the full catalog. + catalog = {connect_scenario["name"]: connect_scenario} + for s in SMOKE_SCENARIOS: + catalog[s["name"]] = s + requested = [n.strip() for n in args.scenarios.split(",") if n.strip()] + unknown = [n for n in requested if n not in catalog] + if unknown: + sys.exit(f"Unknown scenario(s): {', '.join(unknown)}. " + f"Valid: {', '.join(catalog.keys())}") + scenarios = [catalog[n] for n in requested] + elif args.only_connect_scenario: # CI true-E2E: just connect to the local opencode server and verify the list. scenarios = [connect_scenario] else: