From 19b656aae89e7123699ea593c3f13713993fd8c3 Mon Sep 17 00:00:00 2001 From: Dennis V <2119348+dzianisv@users.noreply.github.com> Date: Mon, 22 Jun 2026 19:44:29 +0000 Subject: [PATCH] cua: replace send_message pong with real coding task (helloworld.py + helloworld_test.py) --- .github/workflows/cua-smoke.yml | 11 ++++--- scripts/android-cua-smoke.py | 57 +++++++++++++++++---------------- 2 files changed, 35 insertions(+), 33 deletions(-) diff --git a/.github/workflows/cua-smoke.yml b/.github/workflows/cua-smoke.yml index c9fd437..a9742e1 100644 --- a/.github/workflows/cua-smoke.yml +++ b/.github/workflows/cua-smoke.yml @@ -227,10 +227,10 @@ jobs: # Choose the scenario set HERE (bash) and export it, so the emulator step's # script (run by dash, which mangles multi-line if/fi blocks) stays single-line. if [ "$CAPABLE" = "true" ]; then - SCENARIOS="connect_and_verify_sessions,send_message,verify_session_list" + SCENARIOS="connect_and_verify_sessions,coding_task,verify_session_list" else SCENARIOS="connect_and_verify_sessions,verify_session_list" - echo "WARN: opencode not model-capable in CI; excluding send_message/multi_turn (environmental, not an app bug)." + echo "WARN: opencode not model-capable in CI; excluding coding_task (environmental, not an app bug)." fi echo "SCENARIOS=$SCENARIOS" >> "$GITHUB_OUTPUT" echo "chosen scenarios: $SCENARIOS" @@ -256,11 +256,12 @@ jobs: # SCENARIOS is computed in the probe step (bash) and passed in via env, so # this script stays single-line — dash (used by android-emulator-runner) # mangles multi-line if/fi blocks. The widened core journey is - # connect -> create session -> send -> reply -> list; send_message is only + # connect -> create session -> code -> verify -> list; coding_task is only # included when the model probe confirmed opencode can reply. echo "Running scenarios: ${SCENARIOS}" - # --max-steps raised so multiple scenarios fit; each scenario gets its own budget. - python3 scripts/android-cua-smoke.py --model gpt-5.4 --include-xml --max-steps 40 --scenarios "${SCENARIOS}" + # --max-steps raised so each scenario (especially coding_task with its + # long wait-for-completion loop) has enough budget. + python3 scripts/android-cua-smoke.py --model gpt-5.4 --include-xml --max-steps 60 --scenarios "${SCENARIOS}" - name: opencode server log if: always() diff --git a/scripts/android-cua-smoke.py b/scripts/android-cua-smoke.py index 143bb37..51771a1 100755 --- a/scripts/android-cua-smoke.py +++ b/scripts/android-cua-smoke.py @@ -6,15 +6,15 @@ Full onboarding showcase — drives an Android emulator via ADB using an LLM vis screenshot → vision model → action → repeat Demonstrates the complete first-run journey: - 1. App opens on connection screen (no saved connections) - 2. Configure opencode server URL - 3. Connect — session list loads - 4. Create new AI coding session - 5. Submit a TypeScript "hello world" task - 6. Watch opencode work (tool calls, file writes), wait for idle - 7. Verify output / success response - 8. Navigate to Settings — show model selection - 9. Screenshot settings screen + 1. App opens on connection screen (no saved connections) + 2. Configure opencode server URL + 3. Connect — session list loads + 4. Create new AI coding session + 5. Submit a real Python coding task (helloworld.py + helloworld_test.py) + 6. Watch opencode work (tool calls, file writes), wait for idle + 7. Verify output / success response + 8. Navigate to Settings — show model selection + 9. Screenshot settings screen Requirements: pip install openai @@ -68,12 +68,19 @@ APP_PACKAGE = "cc.agentlabs.opencode" # Default opencode Tailscale dev server DEFAULT_OPENCODE_URL = "http://100.108.64.76:4096" -# TypeScript task prompt sent to the AI coding session +# Coding task prompts sent to the AI coding session TYPESCRIPT_TASK = ( "Write a TypeScript hello world app. " "Create a file hello.ts that prints 'Hello, World!' to the console." ) +PYTHON_CODING_TASK = ( + "Write a Python hello world program. " + "Create helloworld.py that prints 'Hello, World!' and a function greet(name) that returns a greeting string. " + "Also create helloworld_test.py with pytest tests covering both print output and greet(). " + "Make sure both files are well-formed and the tests pass." +) + # --------------------------------------------------------------------------- # Global state # --------------------------------------------------------------------------- @@ -777,26 +784,20 @@ def run_onboarding_showcase( SMOKE_SCENARIOS = [ { - "name": "send_message", + "name": "coding_task", "goal": ( "You see the OpenCode mobile app. Tap the '+' button (top-right) to create a new session. " - "Tap the text input at the bottom. Type 'ping'. Press back to dismiss keyboard. " - "Use the send action. Wait 5 seconds, then take another screenshot. " - "If you don't yet see an assistant reply, wait another 10 seconds and re-check. " - "If still no assistant bubble, wait another 15 seconds and re-check one more time. " - "Report success if you see both a 'You' bubble and an 'Assistant' bubble. " - "Report failure only after at least 30 seconds of total waiting with no assistant bubble." - ), - }, - { - "name": "multi_turn", - "goal": ( - "You see the OpenCode mobile app. Tap '+' (top-right) to create a new session. " - "Tap the text input. Type 'what is 2+2'. Press back. Use send action. " - "Wait up to 30 seconds for an assistant reply (re-check every 10 seconds). " - "Then tap the text input again, type 'and 3+3?'. Press back. Use send action. " - "Wait up to 30 seconds for the second assistant reply (re-check every 10 seconds). " - "Report success if you see two assistant reply bubbles." + "Tap the text input at the bottom. " + f"Type this exact task: {PYTHON_CODING_TASK!r} " + "Press back to dismiss the keyboard. " + "Use the send action to submit the task. " + "After sending, wait and watch — opencode will think and then produce code. " + "Wait up to 120 seconds total for the session to complete " + "(look for file creation messages, a summary from assistant, or 'idle' status). " + "Re-check every 15 seconds by looking at the screen. " + "Take a screenshot showing the final result (the completed code or success summary). " + "Report success if you see evidence of both helloworld.py and helloworld_test.py being created. " + "Report failure only if the screen clearly shows an error with no recovery." ), }, {