name: CUA Smoke Test on: workflow_dispatch: push: branches: [main] tags: ["v*"] paths: - "scripts/android-cua-smoke.py" - "src/**" - "app/**" - ".github/workflows/cua-smoke.yml" jobs: cua-test: runs-on: ubuntu-latest timeout-minutes: 60 env: AZURE_OPENAI_API_KEY: ${{ secrets.AZURE_OPENAI_API_KEY }} AZURE_OPENAI_ENDPOINT: ${{ secrets.AZURE_OPENAI_ENDPOINT }} # CUA driver uses chat-completions; 2024-08-01-preview is sufficient. AZURE_OPENAI_API_VERSION: "2024-08-01-preview" # opencode's @ai-sdk/azure provider hits the new /openai/v1/responses # endpoint. That endpoint accepts ONLY api-version=preview or v1 — every # date-based value (2024-*-preview, 2025-*-preview, 2025-04-01-preview) # returns 400 "API version not supported" and forces MODEL_CAPABLE=false, # which makes send_message/multi_turn scenarios get skipped (#22). # `preview` is also the @ai-sdk/azure 3.x default. OPENCODE_AZURE_API_VERSION: "preview" # Android emulator reaches the runner host loopback via 10.0.2.2. # A real `opencode serve` runs on the host (see steps below), making this a true E2E. OPENCODE_URL: "http://10.0.2.2:4096" ARCHIVEBOX_URL: ${{ secrets.ARCHIVEBOX_URL }} ARCHIVEBOX_API_KEY: ${{ secrets.ARCHIVEBOX_API_KEY }} steps: - uses: actions/checkout@v6 - uses: actions/setup-node@v6 with: node-version: 20 cache: npm - uses: actions/setup-java@v5 with: distribution: temurin java-version: 17 - name: Setup Android SDK uses: android-actions/setup-android@v4 - name: Add emulator to PATH run: echo "$ANDROID_HOME/emulator" >> $GITHUB_PATH - name: Enable KVM run: | echo 'KERNEL=="kvm", GROUP="kvm", MODE="0666", OPTIONS+="static_node=kvm"' | sudo tee /etc/udev/rules.d/99-kvm4all.rules sudo udevadm control --reload-rules sudo udevadm trigger --name-match=kvm - name: Cache Gradle uses: actions/cache@v5 with: path: | ~/.gradle/caches ~/.gradle/wrapper android/.gradle key: ${{ runner.os }}-gradle-${{ hashFiles('android/**/*.gradle*', 'android/gradle/wrapper/gradle-wrapper.properties') }} restore-keys: | ${{ runner.os }}-gradle- - name: Purge stale generated sources # The Gradle cache (restore-keys prefix fallback) can restore a # generated autolinking tree from a previous package id. Gradle then # reuses ReactNativeApplicationEntryPoint.java referencing the OLD # package (ai.opencode.mobile.BuildConfig) and compileReleaseJavaWithJavac # fails. Delete generated sources so prebuild + Gradle regenerate them # for the current package (cc.agentlabs.opencode). run: rm -rf android/app/build/generated android/build/generated android/app/build/intermediates - name: Install dependencies & build APK env: SENTRY_DISABLE_AUTO_UPLOAD: "true" run: | npm install --legacy-peer-deps npx expo prebuild --platform android --no-install keytool -genkey -v -keystore android/app/debug.keystore -storepass android -alias androiddebugkey -keypass android -keyalg RSA -keysize 2048 -validity 10000 -dname "CN=Android Debug,O=Android,C=US" cd android && ./gradlew assembleRelease - name: Install Python deps run: pip install openai - name: Configure opencode Azure provider env: AZURE_OPENAI_MODEL: "gpt-5.4" run: | npm install -g opencode-ai # opencode (released npm pkg) does NOT read AZURE_OPENAI_* for its own LLM. # It needs an explicit provider in opencode.json + a default `model`. # Wire the same Azure resource the CUA driver uses (@ai-sdk/azure). # Extract the resource name from the endpoint secret at runtime # (e.g. https://NAME.openai.azure.com -> NAME) so nothing secret is in source. RESOURCE_NAME="$(printf '%s' "$AZURE_OPENAI_ENDPOINT" | sed -E 's#https?://([^.]+)\..*#\1#')" echo "Derived Azure resource name: ${RESOURCE_NAME:-}" mkdir -p "$HOME/.config/opencode" cat > "$HOME/.config/opencode/opencode.json" < /tmp/opencode-server.log 2>&1 & SRV_PID=$! echo "opencode pid=$SRV_PID" echo "Waiting for opencode server /global/health ..." HEALTHY=0 for i in $(seq 1 60); do if curl -sf --connect-timeout 2 -m 5 http://127.0.0.1:4096/global/health > /dev/null; then echo "opencode server healthy after ${i}s" HEALTHY=1 break fi # Periodic state dump every 10s while waiting if [ $((i % 10)) -eq 0 ]; then echo "--- @${i}s: server log so far ---" tail -20 /tmp/opencode-server.log || true echo "--- listening ports ---" ss -tlnp 2>/dev/null | grep -E ':4096|opencode' || echo "no listener on 4096" echo "--- pid alive? ---" kill -0 $SRV_PID 2>/dev/null && echo "pid $SRV_PID alive" || echo "pid $SRV_PID DEAD" fi sleep 1 done if [ "$HEALTHY" != "1" ]; then echo "::error::opencode server failed to become healthy in 60s" echo "--- final server log ---" cat /tmp/opencode-server.log || true echo "--- ss listing ---" ss -tlnp 2>/dev/null || true echo "--- ps tree ---" ps -ef | grep -E 'opencode|node|npm' | head -20 || true exit 1 fi - name: Probe opencode model capability (can it reply?) id: probe env: AZURE_OPENAI_MODEL: "gpt-5.4" run: | # Deterministic check that the server can actually produce an assistant # reply BEFORE we spend ~30min driving the UI. Creates a session, sends a # prompt via REST, and checks for an assistant message. Non-fatal: records # MODEL_CAPABLE=true/false so the scenario set can be chosen accordingly. set +e # Use -s (not -sf): we WANT the body even on non-2xx so failures are visible. SID=$(curl -s -X POST http://127.0.0.1:4096/session -H 'content-type: application/json' -d '{}' | python3 -c "import sys,json;print(json.load(sys.stdin).get('id',''))" 2>/dev/null) echo "session id: ${SID:-}" CAPABLE=false if [ -n "$SID" ]; then # The /session/{id}/message endpoint (what the app SDK's session.prompt # hits) REQUIRES an explicit model {providerID, modelID}; without it opencode # has nothing to run. Mirror exactly what the app sends: azure/gpt-5.4. HTTP=$(curl -s -o /tmp/probe_reply.json -w '%{http_code}' \ -X POST "http://127.0.0.1:4096/session/$SID/message" \ -H 'content-type: application/json' \ -d "{\"model\":{\"providerID\":\"azure\",\"modelID\":\"$AZURE_OPENAI_MODEL\"},\"parts\":[{\"type\":\"text\",\"text\":\"reply with the single word: pong\"}]}" \ 2>/tmp/probe_err.txt) echo "probe HTTP status: ${HTTP:-}" echo "--- probe reply (truncated) ---"; head -c 3000 /tmp/probe_reply.json; echo echo "--- probe err (truncated) ---"; head -c 1000 /tmp/probe_err.txt; echo # Success = an assistant message part with non-empty text came back. if [ "$HTTP" = "200" ] && grep -qi 'assistant' /tmp/probe_reply.json && grep -qi 'pong\|"text"' /tmp/probe_reply.json; then CAPABLE=true fi fi echo "MODEL_CAPABLE=$CAPABLE" >> "$GITHUB_OUTPUT" echo "opencode model-capable in CI: $CAPABLE" # Choose the scenario set HERE (bash) and export it, so the emulator step's # script (run by dash, which mangles multi-line if/fi blocks) stays single-line. if [ "$CAPABLE" = "true" ]; then SCENARIOS="connect_and_verify_sessions,send_message,verify_session_list" else SCENARIOS="connect_and_verify_sessions,verify_session_list" echo "WARN: opencode not model-capable in CI; excluding send_message/multi_turn (environmental, not an app bug)." fi echo "SCENARIOS=$SCENARIOS" >> "$GITHUB_OUTPUT" echo "chosen scenarios: $SCENARIOS" - name: Run CUA smoke test with emulator uses: reactivecircus/android-emulator-runner@v2 env: MODEL_CAPABLE: ${{ steps.probe.outputs.MODEL_CAPABLE }} SCENARIOS: ${{ steps.probe.outputs.SCENARIOS }} with: api-level: 30 arch: x86_64 target: google_apis emulator-options: -no-window -no-audio -no-boot-anim -gpu swiftshader_indirect -no-snapshot # NOTE: android-emulator-runner runs this script with /usr/bin/sh (dash). # Avoid bash-only constructs like multi-line `|| { ... }` brace groups. script: | # Non-fatal re-check; server health was already gated in the prior step. curl -sf http://127.0.0.1:4096/global/health || echo "WARN: opencode health re-check failed (started in prior step)" adb install android/app/build/outputs/apk/release/app-release.apk adb shell am start -n cc.agentlabs.opencode/.MainActivity sleep 5 # SCENARIOS is computed in the probe step (bash) and passed in via env, so # this script stays single-line — dash (used by android-emulator-runner) # mangles multi-line if/fi blocks. The widened core journey is # connect -> create session -> send -> reply -> list; send_message is only # included when the model probe confirmed opencode can reply. echo "Running scenarios: ${SCENARIOS}" # --max-steps raised so multiple scenarios fit; each scenario gets its own budget. python3 scripts/android-cua-smoke.py --model gpt-5.4 --include-xml --max-steps 40 --scenarios "${SCENARIOS}" - name: opencode server log if: always() run: cat /tmp/opencode-server.log || true - name: Upload test artifacts if: always() uses: actions/upload-artifact@v7 with: name: cua-artifacts-${{ github.run_number }} path: | /tmp/cua_*.png /tmp/cua_*.mp4 if-no-files-found: ignore