From d413d5f927be162ee810dd38f7307a2af361b680 Mon Sep 17 00:00:00 2001 From: Dennis V <2119348+dzianisv@users.noreply.github.com> Date: Wed, 24 Jun 2026 04:57:03 +0000 Subject: [PATCH] docs: add --e2e and --query to CI dispatch + AGENTS.md cua-smoke.yml: - Add scenario/query/e2e_* workflow_dispatch inputs - Runner step dispatches to --query / --e2e / --showcase based on inputs - Upload /tmp/cua_eval_report.json as artifact (--query output) AGENTS.md: - Document all 3 run modes: --showcase, --e2e, --query - List available models on dev server (deepseek-v4-flash-free etc.) - Add dispatch inputs reference for CI - Add --e2e / --query to 'when to run' guidance Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> --- .github/workflows/cua-smoke.yml | 73 ++++++++++++++++++++++++++++++--- AGENTS.md | 42 ++++++++++++++++++- 2 files changed, 108 insertions(+), 7 deletions(-) diff --git a/.github/workflows/cua-smoke.yml b/.github/workflows/cua-smoke.yml index 1d159b7..f96f040 100644 --- a/.github/workflows/cua-smoke.yml +++ b/.github/workflows/cua-smoke.yml @@ -7,6 +7,34 @@ on: description: "OpenCode server URL (overrides default). Use Tailscale URL for a live server." required: false default: "" + scenario: + description: "Test scenario to run (default: showcase)" + required: false + default: "showcase" + type: choice + options: + - showcase + - e2e + e2e_project_dir: + description: "[e2e] Project directory on the server (default: ~/workspace/opencode-mobile)" + required: false + default: "~/workspace/opencode-mobile" + e2e_model_hint: + description: "[e2e] Model name substring to select in picker (default: deepseek)" + required: false + default: "deepseek" + e2e_task: + description: "[e2e] Coding task to submit to the AI agent" + required: false + default: "Write a hello_world.py file that prints 'Hello World' to stdout." + e2e_filename: + description: "[e2e] Expected output filename to validate" + required: false + default: "hello_world.py" + query: + description: "Natural-language test query (enables --query mode, overrides scenario)" + required: false + default: "" push: branches: [main] tags: ["v*"] @@ -236,6 +264,12 @@ jobs: uses: reactivecircus/android-emulator-runner@v2 env: OPENCODE_URL: ${{ env.OPENCODE_URL }} + CUA_SCENARIO: ${{ inputs.scenario || 'showcase' }} + CUA_QUERY: ${{ inputs.query || '' }} + CUA_E2E_PROJECT_DIR: ${{ inputs.e2e_project_dir || '~/workspace/opencode-mobile' }} + CUA_E2E_MODEL_HINT: ${{ inputs.e2e_model_hint || 'deepseek' }} + CUA_E2E_TASK: ${{ inputs.e2e_task || "Write a hello_world.py file that prints 'Hello World' to stdout." }} + CUA_E2E_FILENAME: ${{ inputs.e2e_filename || 'hello_world.py' }} with: api-level: 28 arch: x86_64 @@ -245,16 +279,42 @@ jobs: emulator-options: -no-window -no-audio -no-boot-anim -gpu swiftshader_indirect -no-snapshot # NOTE: android-emulator-runner runs this script with /usr/bin/sh (dash). script: | - # Clear any stale app state so the showcase starts with a fresh connection screen. + # Clear any stale app state so the test starts with a fresh connection screen. adb shell pm clear cc.agentlabs.opencode || true adb install android/app/build/outputs/apk/release/app-release.apk adb shell am start -n cc.agentlabs.opencode/.MainActivity sleep 5 - # --showcase runs the structured onboarding flow which includes: - # session_list: pre-created session MUST appear (not empty-list-ok) - # sessions_reload: navigate back from session, list must be non-empty - # These are the regression guards for the "sessions not loading" bug. - python3 scripts/android-cua-smoke.py --showcase --model gpt-5.4 --include-xml --opencode-url "${OPENCODE_URL}" + + # Dispatch to the requested scenario. + # --query overrides scenario when provided. + if [ -n "${CUA_QUERY}" ]; then + echo "Running --query mode" + python3 scripts/android-cua-smoke.py \ + --query "${CUA_QUERY}" \ + --model gpt-5.4 --include-xml \ + --opencode-url "${OPENCODE_URL}" \ + --eval-output /tmp/cua_eval_report.json + elif [ "${CUA_SCENARIO}" = "e2e" ]; then + echo "Running --e2e mode" + python3 scripts/android-cua-smoke.py \ + --e2e \ + --model gpt-5.4 --include-xml \ + --opencode-url "${OPENCODE_URL}" \ + --e2e-project-dir "${CUA_E2E_PROJECT_DIR}" \ + --e2e-model-hint "${CUA_E2E_MODEL_HINT}" \ + --e2e-task "${CUA_E2E_TASK}" \ + --e2e-filename "${CUA_E2E_FILENAME}" + else + echo "Running --showcase mode (default regression guard)" + # --showcase runs the structured onboarding flow which includes: + # session_list: pre-created session MUST appear (not empty-list-ok) + # sessions_reload: navigate back from session, list must be non-empty + # These are the regression guards for the "sessions not loading" bug. + python3 scripts/android-cua-smoke.py \ + --showcase \ + --model gpt-5.4 --include-xml \ + --opencode-url "${OPENCODE_URL}" + fi - name: opencode server log if: always() @@ -268,4 +328,5 @@ jobs: path: | /tmp/cua_*.png /tmp/cua_*.mp4 + /tmp/cua_eval_report.json if-no-files-found: ignore diff --git a/AGENTS.md b/AGENTS.md index 70f461f..758fc9b 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -112,12 +112,45 @@ Gradle cache: `/Volumes/Dzianis-3/macbook2020/gradle-cache` (or symlink from `/V The script `scripts/android-cua-smoke.py` drives the emulator via ADB using a vision model loop (screenshot → LLM → action → repeat). -### Running locally +### Run modes +**`--showcase` (default, regression guard)** ```bash source ~/.env.d/azure-openai.env python3 scripts/android-cua-smoke.py --model gpt-5.4 --include-xml ``` +Runs: connect → sessions load (pre-created session MUST appear) → new session → sessions reload → TypeScript task → settings. + +**`--e2e` (full coding task, project dir, model picker)** +```bash +source ~/.env.d/azure-openai.env +python3 scripts/android-cua-smoke.py --e2e \ + --opencode-url http://100.108.64.76:4096 \ + --e2e-project-dir ~/workspace/opencode-mobile \ + --e2e-model-hint deepseek \ + --e2e-task "Write a hello_world.py file that prints 'Hello World' to stdout." \ + --e2e-filename hello_world.py +``` +Phases: connect → long-press FAB → enter project dir → select deepseek model → submit task → **DETERMINISTIC** API poll for idle → **DETERMINISTIC** API scan for filename → **DETERMINISTIC** ADB uiautomator check. + +**`--query` (natural-language test description)** +```bash +source ~/.env.d/azure-openai.env +python3 scripts/android-cua-smoke.py \ + --opencode-url http://100.108.64.76:4096 \ + --query "Open android app. Setup against remote opencode server. Go to sessions. \ + Open a new project inside ~/workspace/opencode-mobile. \ + Choose opencode/deepseek model. Start a new session. \ + Ask to write hello_world.py. Validate that agent completed task." \ + --eval-output /tmp/eval-report.json +``` +The LLM plans phases from the query, executes them, runs deterministic checks, and prints a scored evaluation report (overall/score/phases/recommendations). + +### Available models on dev server (100.108.64.76:4096) + +Key models: `deepseek-v4-flash-free` (opencode), `claude-fable-5` (anthropic), `gpt-5.4` (github-copilot), `gemini-3.5-flash` (github-copilot) + +Use `--e2e-model-hint deepseek` to select `deepseek-v4-flash-free` (free quota, good for coding tasks). ### Azure OpenAI credentials @@ -141,12 +174,19 @@ Secrets required: `AZURE_OPENAI_API_KEY`, `AZURE_OPENAI_ENDPOINT` (already set o **Triggers**: Runs on push to `main` (with path filters) AND on `v*` tags (releases). +**Dispatch inputs** (workflow_dispatch): +- `scenario`: `showcase` (default) | `e2e` +- `query`: natural-language test description (enables `--query` mode, overrides scenario) +- `opencode_url`: override server URL (use Tailscale URL for live server) +- `e2e_project_dir`, `e2e_model_hint`, `e2e_task`, `e2e_filename`: e2e mode params + ### When to run CUA test **MANDATORY**: Run the CUA smoke test before any merge to `main` or release: 1. Before merging a PR that touches `src/**`, `app/**`, or `scripts/android-cua-smoke.py` 2. After creating a release tag — CI runs it automatically 3. When debugging UI issues — run locally with `--include-xml` for richer context +4. When validating a specific AI coding task — use `--e2e` or `--query` If the CUA test fails, do NOT merge or release until fixed.