test(ci): widen cua-smoke gate to real core journey (connect→send→reply→list)
The CI opencode server had NO LLM provider configured — the server log only showed "listening", never a model. opencode-ai (released npm pkg) does not read AZURE_OPENAI_* for its own LLM; it needs an explicit provider in opencode.json + a default `model`. So send_message/multi_turn could never pass and the gate was stuck on --only-connect-scenario (UI journey minus the model reply). - Wire opencode to the same Azure resource the CUA driver uses via a generated ~/.config/opencode/opencode.json (@ai-sdk/azure provider, resourceName derived from the endpoint secret at runtime, apiKey from env, default model azure/gpt-5.4). - Add a deterministic REST probe step: create a session + send a prompt and check for an assistant reply BEFORE the ~30min emulator run, exporting MODEL_CAPABLE. - Add --scenarios to android-cua-smoke.py to run an explicit named set. - Emulator step now runs connect_and_verify_sessions + send_message + verify_session_list when MODEL_CAPABLE=true; falls back to the UI-only journey (connect + verify_session_list) otherwise, logging the environmental reason. - Raise --max-steps to 40 so multiple scenarios fit. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
79
.github/workflows/cua-smoke.yml
vendored
79
.github/workflows/cua-smoke.yml
vendored
@@ -69,9 +69,44 @@ jobs:
|
|||||||
- name: Install Python deps
|
- name: Install Python deps
|
||||||
run: pip install openai
|
run: pip install openai
|
||||||
|
|
||||||
- name: Install & start opencode server on runner host
|
- name: Configure opencode Azure provider
|
||||||
|
env:
|
||||||
|
AZURE_OPENAI_MODEL: "gpt-5.4"
|
||||||
run: |
|
run: |
|
||||||
npm install -g opencode-ai
|
npm install -g opencode-ai
|
||||||
|
# opencode (released npm pkg) does NOT read AZURE_OPENAI_* for its own LLM.
|
||||||
|
# It needs an explicit provider in opencode.json + a default `model`.
|
||||||
|
# Wire the same Azure resource the CUA driver uses (@ai-sdk/azure).
|
||||||
|
# Extract the resource name from the endpoint secret at runtime
|
||||||
|
# (e.g. https://NAME.openai.azure.com -> NAME) so nothing secret is in source.
|
||||||
|
RESOURCE_NAME="$(printf '%s' "$AZURE_OPENAI_ENDPOINT" | sed -E 's#https?://([^.]+)\..*#\1#')"
|
||||||
|
echo "Derived Azure resource name: ${RESOURCE_NAME:-<empty>}"
|
||||||
|
mkdir -p "$HOME/.config/opencode"
|
||||||
|
cat > "$HOME/.config/opencode/opencode.json" <<EOF
|
||||||
|
{
|
||||||
|
"\$schema": "https://opencode.ai/config.json",
|
||||||
|
"provider": {
|
||||||
|
"azure": {
|
||||||
|
"npm": "@ai-sdk/azure",
|
||||||
|
"name": "Azure OpenAI",
|
||||||
|
"options": {
|
||||||
|
"resourceName": "${RESOURCE_NAME}",
|
||||||
|
"apiKey": "{env:AZURE_OPENAI_API_KEY}",
|
||||||
|
"apiVersion": "${AZURE_OPENAI_API_VERSION}"
|
||||||
|
},
|
||||||
|
"models": {
|
||||||
|
"${AZURE_OPENAI_MODEL}": { "name": "Azure ${AZURE_OPENAI_MODEL}" }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"model": "azure/${AZURE_OPENAI_MODEL}",
|
||||||
|
"small_model": "azure/${AZURE_OPENAI_MODEL}"
|
||||||
|
}
|
||||||
|
EOF
|
||||||
|
echo "opencode.json written:"; cat "$HOME/.config/opencode/opencode.json"
|
||||||
|
|
||||||
|
- name: Start opencode server on runner host
|
||||||
|
run: |
|
||||||
# Bind all interfaces so the emulator can reach it via 10.0.2.2.
|
# Bind all interfaces so the emulator can reach it via 10.0.2.2.
|
||||||
nohup opencode serve --hostname 0.0.0.0 --port 4096 > /tmp/opencode-server.log 2>&1 &
|
nohup opencode serve --hostname 0.0.0.0 --port 4096 > /tmp/opencode-server.log 2>&1 &
|
||||||
echo "Waiting for opencode server /global/health ..."
|
echo "Waiting for opencode server /global/health ..."
|
||||||
@@ -85,8 +120,35 @@ jobs:
|
|||||||
echo "::error::opencode server failed to become healthy"; cat /tmp/opencode-server.log; exit 1;
|
echo "::error::opencode server failed to become healthy"; cat /tmp/opencode-server.log; exit 1;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
- name: Probe opencode model capability (can it reply?)
|
||||||
|
id: probe
|
||||||
|
run: |
|
||||||
|
# Deterministic check that the server can actually produce an assistant
|
||||||
|
# reply BEFORE we spend ~30min driving the UI. Creates a session, sends a
|
||||||
|
# prompt via REST, and checks for an assistant message. Non-fatal: records
|
||||||
|
# MODEL_CAPABLE=true/false so the scenario set can be chosen accordingly.
|
||||||
|
set +e
|
||||||
|
SID=$(curl -sf -X POST http://127.0.0.1:4096/session -H 'content-type: application/json' -d '{}' | python3 -c "import sys,json;print(json.load(sys.stdin).get('id',''))" 2>/dev/null)
|
||||||
|
echo "session id: ${SID:-<none>}"
|
||||||
|
CAPABLE=false
|
||||||
|
if [ -n "$SID" ]; then
|
||||||
|
curl -sf -X POST "http://127.0.0.1:4096/session/$SID/message" \
|
||||||
|
-H 'content-type: application/json' \
|
||||||
|
-d '{"parts":[{"type":"text","text":"reply with the single word: pong"}]}' \
|
||||||
|
> /tmp/probe_reply.json 2>/tmp/probe_err.txt
|
||||||
|
echo "--- probe reply (truncated) ---"; head -c 2000 /tmp/probe_reply.json; echo
|
||||||
|
echo "--- probe err (truncated) ---"; head -c 1000 /tmp/probe_err.txt; echo
|
||||||
|
if grep -qi 'assistant' /tmp/probe_reply.json && grep -qi 'pong\|text' /tmp/probe_reply.json; then
|
||||||
|
CAPABLE=true
|
||||||
|
fi
|
||||||
|
fi
|
||||||
|
echo "MODEL_CAPABLE=$CAPABLE" >> "$GITHUB_OUTPUT"
|
||||||
|
echo "opencode model-capable in CI: $CAPABLE"
|
||||||
|
|
||||||
- name: Run CUA smoke test with emulator
|
- name: Run CUA smoke test with emulator
|
||||||
uses: reactivecircus/android-emulator-runner@v2
|
uses: reactivecircus/android-emulator-runner@v2
|
||||||
|
env:
|
||||||
|
MODEL_CAPABLE: ${{ steps.probe.outputs.MODEL_CAPABLE }}
|
||||||
with:
|
with:
|
||||||
api-level: 30
|
api-level: 30
|
||||||
arch: x86_64
|
arch: x86_64
|
||||||
@@ -100,8 +162,19 @@ jobs:
|
|||||||
adb install android/app/build/outputs/apk/release/app-release.apk
|
adb install android/app/build/outputs/apk/release/app-release.apk
|
||||||
adb shell am start -n cc.agentlabs.opencode/.MainActivity
|
adb shell am start -n cc.agentlabs.opencode/.MainActivity
|
||||||
sleep 5
|
sleep 5
|
||||||
# True E2E: app connects to the local opencode server (10.0.2.2:4096) and verifies the session list.
|
# Scenario set widened to the real core journey. send_message requires the
|
||||||
python3 scripts/android-cua-smoke.py --model gpt-5.4 --include-xml --only-connect-scenario
|
# opencode server to actually reply (Azure provider wired in earlier step);
|
||||||
|
# if the model probe failed we drop it to the UI-only journey so the gate
|
||||||
|
# stays green and meaningful rather than red for an environmental reason.
|
||||||
|
if [ "${MODEL_CAPABLE}" = "true" ]; then
|
||||||
|
SCENARIOS="connect_and_verify_sessions,send_message,verify_session_list"
|
||||||
|
else
|
||||||
|
SCENARIOS="connect_and_verify_sessions,verify_session_list"
|
||||||
|
echo "WARN: opencode not model-capable in CI; excluding send_message/multi_turn (environmental, not an app bug)."
|
||||||
|
fi
|
||||||
|
echo "Running scenarios: $SCENARIOS"
|
||||||
|
# --max-steps raised so multiple scenarios fit; each scenario gets its own budget.
|
||||||
|
python3 scripts/android-cua-smoke.py --model gpt-5.4 --include-xml --max-steps 40 --scenarios "$SCENARIOS"
|
||||||
|
|
||||||
- name: opencode server log
|
- name: opencode server log
|
||||||
if: always()
|
if: always()
|
||||||
|
|||||||
@@ -634,6 +634,13 @@ def main():
|
|||||||
help="Run ONLY the connect-and-verify-sessions scenario. Use in CI with a "
|
help="Run ONLY the connect-and-verify-sessions scenario. Use in CI with a "
|
||||||
"local opencode server for a deterministic true-E2E (no model backend needed).",
|
"local opencode server for a deterministic true-E2E (no model backend needed).",
|
||||||
)
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--scenarios",
|
||||||
|
help="Comma-separated explicit scenario set to run, e.g. "
|
||||||
|
"'connect_and_verify_sessions,send_message,verify_session_list'. "
|
||||||
|
"Valid names: connect_and_verify_sessions, send_message, multi_turn, "
|
||||||
|
"verify_session_list. Overrides --only-connect-scenario and the default set.",
|
||||||
|
)
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
|
|
||||||
# Verify ADB
|
# Verify ADB
|
||||||
@@ -650,7 +657,18 @@ def main():
|
|||||||
"goal": _connect_and_verify_sessions_goal(connect_url),
|
"goal": _connect_and_verify_sessions_goal(connect_url),
|
||||||
}
|
}
|
||||||
|
|
||||||
if args.only_connect_scenario:
|
if args.scenarios:
|
||||||
|
# Explicit named set (CI widened gate). Look up by name across the full catalog.
|
||||||
|
catalog = {connect_scenario["name"]: connect_scenario}
|
||||||
|
for s in SMOKE_SCENARIOS:
|
||||||
|
catalog[s["name"]] = s
|
||||||
|
requested = [n.strip() for n in args.scenarios.split(",") if n.strip()]
|
||||||
|
unknown = [n for n in requested if n not in catalog]
|
||||||
|
if unknown:
|
||||||
|
sys.exit(f"Unknown scenario(s): {', '.join(unknown)}. "
|
||||||
|
f"Valid: {', '.join(catalog.keys())}")
|
||||||
|
scenarios = [catalog[n] for n in requested]
|
||||||
|
elif args.only_connect_scenario:
|
||||||
# CI true-E2E: just connect to the local opencode server and verify the list.
|
# CI true-E2E: just connect to the local opencode server and verify the list.
|
||||||
scenarios = [connect_scenario]
|
scenarios = [connect_scenario]
|
||||||
else:
|
else:
|
||||||
|
|||||||
Reference in New Issue
Block a user