320 lines
14 KiB
YAML
320 lines
14 KiB
YAML
name: CUA Smoke Test
|
|
|
|
on:
|
|
workflow_dispatch:
|
|
inputs:
|
|
opencode_url:
|
|
description: "OpenCode server URL (overrides default). Use Tailscale URL for a live server."
|
|
required: false
|
|
default: ""
|
|
scenario:
|
|
description: "Test scenario to run (default: showcase)"
|
|
required: false
|
|
default: "showcase"
|
|
type: choice
|
|
options:
|
|
- showcase
|
|
- e2e
|
|
e2e_project_dir:
|
|
description: "[e2e] Project directory on the server (default: ~/workspace/opencode-mobile)"
|
|
required: false
|
|
default: "~/workspace/opencode-mobile"
|
|
e2e_model_hint:
|
|
description: "[e2e] Model name substring to select in picker (default: deepseek)"
|
|
required: false
|
|
default: "deepseek"
|
|
e2e_task:
|
|
description: "[e2e] Coding task to submit to the AI agent"
|
|
required: false
|
|
default: "Write a hello_world.py file that prints 'Hello World' to stdout."
|
|
e2e_filename:
|
|
description: "[e2e] Expected output filename to validate"
|
|
required: false
|
|
default: "hello_world.py"
|
|
query:
|
|
description: "Natural-language test query (enables --query mode, overrides scenario)"
|
|
required: false
|
|
default: ""
|
|
push:
|
|
branches: [main]
|
|
tags: ["v*"]
|
|
paths:
|
|
- "scripts/android-cua-smoke.py"
|
|
- "src/**"
|
|
- "app/**"
|
|
- ".github/workflows/cua-smoke.yml"
|
|
|
|
jobs:
|
|
cua-test:
|
|
runs-on: ubuntu-latest
|
|
timeout-minutes: 60
|
|
env:
|
|
AZURE_OPENAI_API_KEY: ${{ secrets.AZURE_OPENAI_API_KEY }}
|
|
AZURE_OPENAI_ENDPOINT: ${{ secrets.AZURE_OPENAI_ENDPOINT }}
|
|
# CUA driver uses chat-completions; 2024-08-01-preview is sufficient.
|
|
AZURE_OPENAI_API_VERSION: "2024-08-01-preview"
|
|
# opencode's @ai-sdk/azure provider hits the new /openai/v1/responses
|
|
# endpoint. That endpoint accepts ONLY api-version=preview or v1 — every
|
|
# date-based value (2024-*-preview, 2025-*-preview, 2025-04-01-preview)
|
|
# returns 400 "API version not supported" and forces MODEL_CAPABLE=false,
|
|
# which makes send_message/multi_turn scenarios get skipped (#22).
|
|
# `preview` is also the @ai-sdk/azure 3.x default.
|
|
OPENCODE_AZURE_API_VERSION: "preview"
|
|
# Default: emulator reaches runner host via 10.0.2.2. Override with dispatch input for live server.
|
|
OPENCODE_URL: ${{ inputs.opencode_url != '' && inputs.opencode_url || 'http://10.0.2.2:4096' }}
|
|
ARCHIVEBOX_URL: ${{ secrets.ARCHIVEBOX_URL }}
|
|
ARCHIVEBOX_API_KEY: ${{ secrets.ARCHIVEBOX_API_KEY }}
|
|
steps:
|
|
- uses: actions/checkout@v6
|
|
|
|
- uses: actions/setup-node@v6
|
|
with:
|
|
node-version: 20
|
|
cache: npm
|
|
|
|
- uses: actions/setup-java@v5
|
|
with:
|
|
distribution: temurin
|
|
java-version: 17
|
|
|
|
- name: Setup Android SDK
|
|
uses: android-actions/setup-android@v4
|
|
|
|
- name: Add emulator to PATH
|
|
run: echo "$ANDROID_HOME/emulator" >> $GITHUB_PATH
|
|
|
|
- name: Enable KVM
|
|
run: |
|
|
echo 'KERNEL=="kvm", GROUP="kvm", MODE="0666", OPTIONS+="static_node=kvm"' | sudo tee /etc/udev/rules.d/99-kvm4all.rules
|
|
sudo udevadm control --reload-rules
|
|
sudo udevadm trigger --name-match=kvm
|
|
|
|
- name: Cache Gradle
|
|
uses: actions/cache@v5
|
|
with:
|
|
path: |
|
|
~/.gradle/caches
|
|
~/.gradle/wrapper
|
|
android/.gradle
|
|
key: ${{ runner.os }}-gradle-${{ hashFiles('android/**/*.gradle*', 'android/gradle/wrapper/gradle-wrapper.properties') }}
|
|
restore-keys: |
|
|
${{ runner.os }}-gradle-
|
|
|
|
- name: Purge stale generated sources
|
|
# The Gradle cache (restore-keys prefix fallback) can restore a
|
|
# generated autolinking tree from a previous package id. Gradle then
|
|
# reuses ReactNativeApplicationEntryPoint.java referencing the OLD
|
|
# package (ai.opencode.mobile.BuildConfig) and compileReleaseJavaWithJavac
|
|
# fails. Delete generated sources so prebuild + Gradle regenerate them
|
|
# for the current package (cc.agentlabs.opencode).
|
|
run: rm -rf android/app/build/generated android/build/generated android/app/build/intermediates
|
|
|
|
- name: Install dependencies & build APK
|
|
env:
|
|
SENTRY_DISABLE_AUTO_UPLOAD: "true"
|
|
run: |
|
|
npm install --legacy-peer-deps
|
|
npx expo prebuild --platform android --no-install
|
|
keytool -genkey -v -keystore android/app/debug.keystore -storepass android -alias androiddebugkey -keypass android -keyalg RSA -keysize 2048 -validity 10000 -dname "CN=Android Debug,O=Android,C=US"
|
|
cd android && ./gradlew assembleRelease
|
|
|
|
- name: Install Python deps
|
|
run: pip install openai
|
|
|
|
- name: Configure opencode Azure provider
|
|
env:
|
|
AZURE_OPENAI_MODEL: "gpt-5.4"
|
|
run: |
|
|
npm install -g opencode-ai
|
|
# opencode (released npm pkg) does NOT read AZURE_OPENAI_* for its own LLM.
|
|
# It needs an explicit provider in opencode.json + a default `model`.
|
|
# Wire the same Azure resource the CUA driver uses (@ai-sdk/azure).
|
|
# Extract the resource name from the endpoint secret at runtime
|
|
# (e.g. https://NAME.openai.azure.com -> NAME) so nothing secret is in source.
|
|
RESOURCE_NAME="$(printf '%s' "$AZURE_OPENAI_ENDPOINT" | sed -E 's#https?://([^.]+)\..*#\1#')"
|
|
echo "Derived Azure resource name: ${RESOURCE_NAME:-<empty>}"
|
|
mkdir -p "$HOME/.config/opencode"
|
|
cat > "$HOME/.config/opencode/opencode.json" <<EOF
|
|
{
|
|
"\$schema": "https://opencode.ai/config.json",
|
|
"provider": {
|
|
"azure": {
|
|
"npm": "@ai-sdk/azure",
|
|
"name": "Azure OpenAI",
|
|
"options": {
|
|
"resourceName": "${RESOURCE_NAME}",
|
|
"apiKey": "{env:AZURE_OPENAI_API_KEY}",
|
|
"apiVersion": "${OPENCODE_AZURE_API_VERSION}"
|
|
},
|
|
"models": {
|
|
"${AZURE_OPENAI_MODEL}": { "name": "Azure ${AZURE_OPENAI_MODEL}" }
|
|
}
|
|
}
|
|
},
|
|
"model": "azure/${AZURE_OPENAI_MODEL}",
|
|
"small_model": "azure/${AZURE_OPENAI_MODEL}"
|
|
}
|
|
EOF
|
|
echo "opencode.json written:"; cat "$HOME/.config/opencode/opencode.json"
|
|
|
|
- name: Start opencode server on runner host
|
|
if: ${{ inputs.opencode_url == '' }}
|
|
run: |
|
|
set -x
|
|
# Bind all interfaces so the emulator can reach it via 10.0.2.2.
|
|
nohup opencode serve --hostname 0.0.0.0 --port 4096 --print-logs > /tmp/opencode-server.log 2>&1 &
|
|
SRV_PID=$!
|
|
echo "opencode pid=$SRV_PID"
|
|
echo "Waiting for opencode server /global/health ..."
|
|
HEALTHY=0
|
|
for i in $(seq 1 60); do
|
|
if curl -sf --connect-timeout 2 -m 5 http://127.0.0.1:4096/global/health > /dev/null; then
|
|
echo "opencode server healthy after ${i}s"
|
|
HEALTHY=1
|
|
break
|
|
fi
|
|
# Periodic state dump every 10s while waiting
|
|
if [ $((i % 10)) -eq 0 ]; then
|
|
echo "--- @${i}s: server log so far ---"
|
|
tail -20 /tmp/opencode-server.log || true
|
|
echo "--- listening ports ---"
|
|
ss -tlnp 2>/dev/null | grep -E ':4096|opencode' || echo "no listener on 4096"
|
|
echo "--- pid alive? ---"
|
|
kill -0 $SRV_PID 2>/dev/null && echo "pid $SRV_PID alive" || echo "pid $SRV_PID DEAD"
|
|
fi
|
|
sleep 1
|
|
done
|
|
if [ "$HEALTHY" != "1" ]; then
|
|
echo "::error::opencode server failed to become healthy in 60s"
|
|
echo "--- final server log ---"
|
|
cat /tmp/opencode-server.log || true
|
|
echo "--- ss listing ---"
|
|
ss -tlnp 2>/dev/null || true
|
|
echo "--- ps tree ---"
|
|
ps -ef | grep -E 'opencode|node|npm' | head -20 || true
|
|
exit 1
|
|
fi
|
|
|
|
- name: Probe opencode model capability (can it reply?)
|
|
id: probe
|
|
if: ${{ inputs.opencode_url == '' }}
|
|
env:
|
|
AZURE_OPENAI_MODEL: "gpt-5.4"
|
|
run: |
|
|
# Deterministic check that the server can actually produce an assistant
|
|
# reply BEFORE we spend ~30min driving the UI. Creates a session, sends a
|
|
# prompt via REST, and checks for an assistant message. Non-fatal: records
|
|
# MODEL_CAPABLE=true/false as an informational signal (the emulator step
|
|
# always runs --showcase regardless of model capability).
|
|
set +e
|
|
# Use -s (not -sf): we WANT the body even on non-2xx so failures are visible.
|
|
SID=$(curl -s -X POST http://127.0.0.1:4096/session -H 'content-type: application/json' -d '{}' | python3 -c "import sys,json;print(json.load(sys.stdin).get('id',''))" 2>/dev/null)
|
|
echo "session id: ${SID:-<none>}"
|
|
CAPABLE=false
|
|
if [ -n "$SID" ]; then
|
|
# The /session/{id}/message endpoint (what the app SDK's session.prompt
|
|
# hits) REQUIRES an explicit model {providerID, modelID}; without it opencode
|
|
# has nothing to run. Mirror exactly what the app sends: azure/gpt-5.4.
|
|
HTTP=$(curl -s -o /tmp/probe_reply.json -w '%{http_code}' \
|
|
-X POST "http://127.0.0.1:4096/session/$SID/message" \
|
|
-H 'content-type: application/json' \
|
|
-d "{\"model\":{\"providerID\":\"azure\",\"modelID\":\"$AZURE_OPENAI_MODEL\"},\"parts\":[{\"type\":\"text\",\"text\":\"reply with the single word: pong\"}]}" \
|
|
2>/tmp/probe_err.txt)
|
|
echo "probe HTTP status: ${HTTP:-<none>}"
|
|
echo "--- probe reply (truncated) ---"; head -c 3000 /tmp/probe_reply.json; echo
|
|
echo "--- probe err (truncated) ---"; head -c 1000 /tmp/probe_err.txt; echo
|
|
# Success = an assistant message part with non-empty text came back.
|
|
if [ "$HTTP" = "200" ] && grep -qi 'assistant' /tmp/probe_reply.json && grep -qi 'pong\|"text"' /tmp/probe_reply.json; then
|
|
CAPABLE=true
|
|
fi
|
|
|
|
# Also test prompt_async (the endpoint the app actually uses).
|
|
# Fire-and-forget, then poll /session/$SID/message for the response.
|
|
if [ "$CAPABLE" = "true" ]; then
|
|
echo "--- testing prompt_async endpoint (app's actual code path) ---"
|
|
ASYNC_HTTP=$(curl -s -o /tmp/async_resp.txt -w '%{http_code}' \
|
|
-X POST "http://127.0.0.1:4096/session/$SID/prompt_async" \
|
|
-H 'content-type: application/json' \
|
|
-d "{\"model\":{\"providerID\":\"azure\",\"modelID\":\"$AZURE_OPENAI_MODEL\"},\"parts\":[{\"type\":\"text\",\"text\":\"reply with the single word: async_pong\"}]}" \
|
|
2>/dev/null)
|
|
echo "prompt_async HTTP: $ASYNC_HTTP"
|
|
echo "prompt_async body: $(cat /tmp/async_resp.txt)"
|
|
# Poll for response (up to 15s)
|
|
ASYNC_OK=false
|
|
for i in $(seq 1 15); do
|
|
sleep 1
|
|
MSGS=$(curl -sf "http://127.0.0.1:4096/session/$SID/message" 2>/dev/null)
|
|
if echo "$MSGS" | grep -qi 'async_pong'; then
|
|
echo "prompt_async reply confirmed after ${i}s"
|
|
ASYNC_OK=true
|
|
break
|
|
fi
|
|
done
|
|
if [ "$ASYNC_OK" = "false" ]; then
|
|
echo "::warning::prompt_async did NOT produce a reply within 15s (sync probe worked). This may explain CUA send_message failures."
|
|
echo "--- messages after async probe ---"
|
|
curl -sf "http://127.0.0.1:4096/session/$SID/message" 2>/dev/null | python3 -c "import sys,json; msgs=json.load(sys.stdin); [print(f'{m[\"role\"]}: {[p.get(\"text\",\"\")[:80] for p in m.get(\"parts\",[])]}') for m in msgs[-4:]]" 2>/dev/null || true
|
|
fi
|
|
fi
|
|
fi
|
|
echo "MODEL_CAPABLE=$CAPABLE" >> "$GITHUB_OUTPUT"
|
|
echo "opencode model-capable in CI: $CAPABLE"
|
|
|
|
- name: Write CUA runner script
|
|
run: |
|
|
cat > /tmp/cua-runner.sh << 'RUNNEREOF'
|
|
#!/bin/sh
|
|
# Generated by CI workflow — dispatches to the right CUA mode.
|
|
set -e
|
|
if [ "${CUA_SCENARIO}" = "e2e" ]; then
|
|
exec python3 scripts/android-cua-smoke.py --e2e --model gpt-5.4 --include-xml --opencode-url "${OPENCODE_URL}"
|
|
elif [ -n "${CUA_QUERY}" ]; then
|
|
exec python3 scripts/android-cua-smoke.py --query "${CUA_QUERY}" --model gpt-5.4 --include-xml --opencode-url "${OPENCODE_URL}" --eval-output /tmp/cua_eval_report.json
|
|
else
|
|
exec python3 scripts/android-cua-smoke.py --showcase --model gpt-5.4 --include-xml --opencode-url "${OPENCODE_URL}"
|
|
fi
|
|
RUNNEREOF
|
|
chmod +x /tmp/cua-runner.sh
|
|
|
|
- name: Run CUA smoke test with emulator
|
|
uses: reactivecircus/android-emulator-runner@v2
|
|
env:
|
|
OPENCODE_URL: ${{ env.OPENCODE_URL }}
|
|
CUA_SCENARIO: ${{ inputs.scenario || 'showcase' }}
|
|
CUA_QUERY: ${{ inputs.query || '' }}
|
|
CUA_E2E_PROJECT_DIR: ${{ inputs.e2e_project_dir || '~/workspace/opencode-mobile' }}
|
|
CUA_E2E_MODEL_HINT: ${{ inputs.e2e_model_hint || 'deepseek' }}
|
|
CUA_E2E_TASK: ${{ inputs.e2e_task || 'Write a hello_world.py that prints Hello World' }}
|
|
CUA_E2E_FILENAME: ${{ inputs.e2e_filename || 'hello_world.py' }}
|
|
with:
|
|
api-level: 28
|
|
arch: x86_64
|
|
target: default
|
|
disable-animations: true
|
|
emulator-boot-timeout: 600
|
|
emulator-options: -no-window -no-audio -no-boot-anim -gpu swiftshader_indirect -no-snapshot
|
|
# NOTE: android-emulator-runner action wraps script content in
|
|
# sh -c "..." which breaks when inner " chars appear. Use a
|
|
# temp shell script to avoid quoting conflicts entirely.
|
|
script: |
|
|
adb shell pm clear cc.agentlabs.opencode || true
|
|
adb install android/app/build/outputs/apk/release/app-release.apk
|
|
adb shell am start -n cc.agentlabs.opencode/.MainActivity
|
|
sleep 5
|
|
sh /tmp/cua-runner.sh
|
|
|
|
- name: opencode server log
|
|
if: always()
|
|
run: cat /tmp/opencode-server.log || true
|
|
|
|
- name: Upload test artifacts
|
|
if: always()
|
|
uses: actions/upload-artifact@v7
|
|
with:
|
|
name: cua-artifacts-${{ github.run_number }}
|
|
path: |
|
|
/tmp/cua_*.png
|
|
/tmp/cua_*.mp4
|
|
/tmp/cua_eval_report.json
|
|
if-no-files-found: ignore
|