feat(cua): full onboarding showcase test - connect, session, TypeScript task, settings

Rewrites android-cua-smoke.py to demonstrate the complete first-run journey
instead of the previous "ping" smoke test. The new structured multi-phase
flow covers: server connection setup, session list, new session creation,
TypeScript hello-world task submission (watching tool calls/file writes),
output verification, and Settings/model-selection screenshot.

Key changes:
- run_onboarding_showcase() orchestrates 6 sequential CUA phases with
  per-phase goals, step budgets, and PASS/FAIL phase tracking
- run_cua_step() replaces run_cua() — accepts step_label, action_delay,
  saves labeled screenshots (/tmp/cua_<phase>_<step>.png) for debugging
- Global --speed-multiplier flag scales all _sleep() calls (0.5 = 2x faster)
- Showcase is now the default mode; legacy --goal / --scenarios flags retained
  for backwards compat and CI regression scenarios
- Tighter action_delay (0.7s) and trimmed history window (14 turns) vs
  previous 1.0s / 12 turns
- Phase banner log lines ("STEP N: ...") narrate the video in real time

Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01No3k1AEioE4PNUZg12TxQo
This commit is contained in:
Dennis V
2026-06-21 01:20:44 +00:00
parent 09b034eaf3
commit 3c498972ce

View File

@@ -2,43 +2,41 @@
""" """
Android Computer-Use Agent (CUA) smoke test for OpenCode Mobile. Android Computer-Use Agent (CUA) smoke test for OpenCode Mobile.
Drives an Android emulator via ADB using an LLM vision loop: Full onboarding showcase — drives an Android emulator via ADB using an LLM vision loop:
screenshot → vision model → action → repeat screenshot → vision model → action → repeat
Inspired by: Demonstrates the complete first-run journey:
- openai/openai-cua-sample-app (browser CUA pattern) 1. App opens on connection screen (no saved connections)
- X-PLUG/MobileAgent (ADB + VLM loop) 2. Configure opencode server URL
- TencentQQGYLab/AppAgent (multimodal smartphone agent) 3. Connect — session list loads
4. Create new AI coding session
5. Submit a TypeScript "hello world" task
6. Watch opencode work (tool calls, file writes), wait for idle
7. Verify output / success response
8. Navigate to Settings — show model selection
9. Screenshot settings screen
Requirements: Requirements:
pip install openai pip install openai
ADB in PATH with a connected device/emulator. ADB in PATH with a connected device/emulator.
Usage: Usage:
# Azure OpenAI (recommended — already configured via ~/.env.d/azure-openai.env)
source ~/.env.d/azure-openai.env
python scripts/android-cua-smoke.py --model gpt-5.4 --include-xml
# OpenAI # OpenAI
export OPENAI_API_KEY=sk-... export OPENAI_API_KEY=sk-...
python scripts/android-cua-smoke.py python scripts/android-cua-smoke.py --model gpt-4o --include-xml
# Azure OpenAI (with deployment) # Run ONLY the onboarding showcase (default and primary flow):
export OPENAI_API_KEY=<key> python scripts/android-cua-smoke.py --showcase
export OPENAI_BASE_URL=https://<resource>.openai.azure.com/openai/deployments/<deployment>/
python scripts/android-cua-smoke.py --model gpt-4o
# Google Gemini (via OpenAI-compat) # Custom goal (legacy / quick debugging):
export GEMINI_API_KEY=AIza...
python scripts/android-cua-smoke.py
# xAI Grok
export XAI_API_KEY=xai-...
python scripts/android-cua-smoke.py
# Any OpenAI-compatible endpoint (LiteLLM, Ollama, etc.)
export OPENAI_API_KEY=dummy
export OPENAI_BASE_URL=http://localhost:4000/v1
python scripts/android-cua-smoke.py --model gpt-4o
# Custom goal
python scripts/android-cua-smoke.py --goal "Open settings and toggle dark mode" python scripts/android-cua-smoke.py --goal "Open settings and toggle dark mode"
# Speed up for a demo video (tighter waits, fewer retries):
python scripts/android-cua-smoke.py --speed-multiplier 0.5
""" """
import argparse import argparse
@@ -61,6 +59,32 @@ except ImportError:
sys.exit("openai package required: pip install openai") sys.exit("openai package required: pip install openai")
# ---------------------------------------------------------------------------
# Constants
# ---------------------------------------------------------------------------
APP_PACKAGE = "cc.agentlabs.opencode"
# Default opencode Tailscale dev server
DEFAULT_OPENCODE_URL = "http://100.108.64.76:4096"
# TypeScript task prompt sent to the AI coding session
TYPESCRIPT_TASK = (
"Write a TypeScript hello world app. "
"Create a file hello.ts that prints 'Hello, World!' to the console."
)
# ---------------------------------------------------------------------------
# Global state
# ---------------------------------------------------------------------------
_step_counter = 0
_speed_multiplier = 1.0 # Set via --speed-multiplier; <1.0 = faster
def _sleep(seconds: float) -> None:
"""Interruptible sleep that respects the global speed multiplier."""
time.sleep(max(0.2, seconds * _speed_multiplier))
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
@@ -78,10 +102,6 @@ def adb(*args: str) -> str:
return result.stdout.strip() return result.stdout.strip()
_step_counter = 0
APP_PACKAGE = "cc.agentlabs.opencode"
def _bounds_center(bounds: str) -> tuple[int, int] | None: def _bounds_center(bounds: str) -> tuple[int, int] | None:
match = re.match(r"\[(\d+),(\d+)\]\[(\d+),(\d+)\]", bounds or "") match = re.match(r"\[(\d+),(\d+)\]\[(\d+),(\d+)\]", bounds or "")
if not match: if not match:
@@ -111,7 +131,7 @@ def ensure_app_foreground(package: str = APP_PACKAGE, retries: int = 3,
return True return True
adb("shell", "monkey", "-p", package, "-c", "android.intent.category.LAUNCHER", "1") adb("shell", "monkey", "-p", package, "-c", "android.intent.category.LAUNCHER", "1")
time.sleep(2.0) _sleep(2.0)
if verbose: if verbose:
seen = current or "unknown" seen = current or "unknown"
@@ -172,7 +192,7 @@ def maybe_dismiss_telemetry_consent(package: str = APP_PACKAGE,
for label, (x, y) in candidates: for label, (x, y) in candidates:
if any(marker in label for marker in dismiss_markers): if any(marker in label for marker in dismiss_markers):
adb("shell", "input", "tap", str(x), str(y)) adb("shell", "input", "tap", str(x), str(y))
time.sleep(1.0) _sleep(1.0)
if verbose: if verbose:
print(f" [prep] dismissed telemetry consent via '{label or 'button'}' at ({x}, {y})") print(f" [prep] dismissed telemetry consent via '{label or 'button'}' at ({x}, {y})")
return True return True
@@ -182,10 +202,13 @@ def maybe_dismiss_telemetry_consent(package: str = APP_PACKAGE,
return False return False
def screenshot_b64() -> str: def screenshot_b64(label: str = "") -> str:
"""Capture emulator screenshot and return as base64 PNG. Retries on timeout.""" """Capture emulator screenshot and return as base64 PNG. Retries on timeout."""
global _step_counter global _step_counter
_step_counter += 1 _step_counter += 1
suffix = f"_{label}" if label else ""
debug_path = f"/tmp/cua_step_{_step_counter:03d}{suffix}.png"
for attempt in range(3): for attempt in range(3):
try: try:
result = subprocess.run( result = subprocess.run(
@@ -193,15 +216,14 @@ def screenshot_b64() -> str:
capture_output=True, timeout=30, capture_output=True, timeout=30,
) )
if result.returncode == 0 and len(result.stdout) > 100: if result.returncode == 0 and len(result.stdout) > 100:
# Save for debugging
debug_path = f"/tmp/cua_step_{_step_counter:03d}.png"
Path(debug_path).write_bytes(result.stdout) Path(debug_path).write_bytes(result.stdout)
return base64.b64encode(result.stdout).decode() return base64.b64encode(result.stdout).decode()
except subprocess.TimeoutExpired: except subprocess.TimeoutExpired:
if attempt < 2: if attempt < 2:
time.sleep(3) _sleep(3)
continue continue
raise raise
# Fallback: screencap on device then pull # Fallback: screencap on device then pull
with tempfile.NamedTemporaryFile(suffix=".png", delete=False) as f: with tempfile.NamedTemporaryFile(suffix=".png", delete=False) as f:
path = f.name path = f.name
@@ -211,7 +233,7 @@ def screenshot_b64() -> str:
subprocess.run(["adb", "pull", "/sdcard/_cua_screen.png", path], subprocess.run(["adb", "pull", "/sdcard/_cua_screen.png", path],
capture_output=True, timeout=10) capture_output=True, timeout=10)
data = Path(path).read_bytes() data = Path(path).read_bytes()
Path(f"/tmp/cua_step_{_step_counter:03d}.png").write_bytes(data) Path(debug_path).write_bytes(data)
return base64.b64encode(data).decode() return base64.b64encode(data).decode()
finally: finally:
Path(path).unlink(missing_ok=True) Path(path).unlink(missing_ok=True)
@@ -227,7 +249,6 @@ def start_screen_recording(scenario_name: str) -> tuple:
stop_event = threading.Event() stop_event = threading.Event()
def _record(): def _record():
# --time-limit 180 caps at 3 min; we stop it early via SIGTERM via adb
try: try:
subprocess.run( subprocess.run(
["adb", "shell", f"screenrecord --time-limit 180 {remote_path}"], ["adb", "shell", f"screenrecord --time-limit 180 {remote_path}"],
@@ -238,19 +259,18 @@ def start_screen_recording(scenario_name: str) -> tuple:
thread = threading.Thread(target=_record, daemon=True) thread = threading.Thread(target=_record, daemon=True)
thread.start() thread.start()
time.sleep(1.0) # let recorder spin up _sleep(1.0)
return thread, stop_event, remote_path return thread, stop_event, remote_path
def stop_screen_recording(thread: threading.Thread, remote_path: str, def stop_screen_recording(thread: threading.Thread, remote_path: str,
local_path: str) -> bool: local_path: str) -> bool:
"""Stop recorder, pull video to local_path. Returns True on success.""" """Stop recorder, pull video to local_path. Returns True on success."""
# Stop screenrecord via pkill (SIGINT flushes MP4 moov atom)
subprocess.run( subprocess.run(
["adb", "shell", "pkill", "-2", "screenrecord"], ["adb", "shell", "pkill", "-2", "screenrecord"],
capture_output=True, timeout=10, capture_output=True, timeout=10,
) )
time.sleep(2.0) # let MP4 finalise _sleep(2.0)
thread.join(timeout=5) thread.join(timeout=5)
result = subprocess.run( result = subprocess.run(
@@ -269,11 +289,7 @@ def stop_screen_recording(thread: threading.Thread, remote_path: str,
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
def upload_to_archivebox(video_path: str, scenario_name: str) -> bool: def upload_to_archivebox(video_path: str, scenario_name: str) -> bool:
"""Upload video to ArchiveBox if ARCHIVEBOX_URL is configured. """Upload video to ArchiveBox if ARCHIVEBOX_URL is configured."""
Reads ARCHIVEBOX_URL and ARCHIVEBOX_API_KEY from environment.
Returns True if uploaded, False/skipped otherwise.
"""
url = os.environ.get("ARCHIVEBOX_URL", "").rstrip("/") url = os.environ.get("ARCHIVEBOX_URL", "").rstrip("/")
api_key = os.environ.get("ARCHIVEBOX_API_KEY", "") api_key = os.environ.get("ARCHIVEBOX_API_KEY", "")
if not url: if not url:
@@ -282,7 +298,6 @@ def upload_to_archivebox(video_path: str, scenario_name: str) -> bool:
try: try:
import urllib.request import urllib.request
import urllib.parse
video_data = Path(video_path).read_bytes() video_data = Path(video_path).read_bytes()
boundary = "----CUAUploadBoundary" boundary = "----CUAUploadBoundary"
@@ -304,7 +319,6 @@ def upload_to_archivebox(video_path: str, scenario_name: str) -> bool:
req = urllib.request.Request(f"{url}/api/v1/add", data=body, headers=headers, method="POST") req = urllib.request.Request(f"{url}/api/v1/add", data=body, headers=headers, method="POST")
with urllib.request.urlopen(req, timeout=60) as resp: with urllib.request.urlopen(req, timeout=60) as resp:
resp_data = resp.read().decode(errors="replace")
print(f" [archivebox] uploaded {scenario_name}.mp4 → {url} ({resp.status})") print(f" [archivebox] uploaded {scenario_name}.mp4 → {url} ({resp.status})")
return True return True
except Exception as exc: except Exception as exc:
@@ -313,7 +327,7 @@ def upload_to_archivebox(video_path: str, scenario_name: str) -> bool:
def ui_dump() -> str: def ui_dump() -> str:
"""Dump UI hierarchy XML and return as string (optional context for LLM).""" """Dump UI hierarchy XML and return as string."""
adb("shell", "uiautomator", "dump", "/sdcard/_cua_ui.xml") adb("shell", "uiautomator", "dump", "/sdcard/_cua_ui.xml")
result = subprocess.run( result = subprocess.run(
["adb", "pull", "/sdcard/_cua_ui.xml", "/tmp/_cua_ui.xml"], ["adb", "pull", "/sdcard/_cua_ui.xml", "/tmp/_cua_ui.xml"],
@@ -335,7 +349,6 @@ def execute_action(action: dict) -> str:
elif act == "type": elif act == "type":
text = action.get("text", "") text = action.get("text", "")
# ADB input text needs escaping
escaped = text.replace(" ", "%s").replace("&", "\\&").replace(";", "\\;") escaped = text.replace(" ", "%s").replace("&", "\\&").replace(";", "\\;")
adb("shell", "input", "text", escaped) adb("shell", "input", "text", escaped)
return f"typed '{text}'" return f"typed '{text}'"
@@ -358,23 +371,18 @@ def execute_action(action: dict) -> str:
return f"swiped ({x1},{y1})->({x2},{y2})" return f"swiped ({x1},{y1})->({x2},{y2})"
elif act == "send": elif act == "send":
# Auto-locate send button: rightmost clickable ViewGroup in the bottom input bar. # Auto-locate send button: rightmost clickable button in the bottom input bar.
# Threshold is screen-relative (bottom 25%) so it works on any emulator
# resolution — API 30 default profile is 1080x1920, not the 2400-tall pixel
# we previously hardcoded against.
_, screen_h = get_screen_size() _, screen_h = get_screen_size()
bottom_threshold = int(screen_h * 0.75) bottom_threshold = int(screen_h * 0.75)
xml = ui_dump() xml = ui_dump()
matches = re.findall(r'clickable="true"[^>]*bounds="\[(\d+),(\d+)\]\[(\d+),(\d+)\]"', xml) matches = re.findall(r'clickable="true"[^>]*bounds="\[(\d+),(\d+)\]\[(\d+),(\d+)\]"', xml)
bottom_buttons = [(int(x1), int(y1), int(x2), int(y2)) for x1, y1, x2, y2 in matches if int(y1) > bottom_threshold] bottom_buttons = [(int(x1), int(y1), int(x2), int(y2)) for x1, y1, x2, y2 in matches if int(y1) > bottom_threshold]
if bottom_buttons: if bottom_buttons:
# Rightmost = highest center-x (not left-edge x1, which misidentifies wide buttons)
send_btn = max(bottom_buttons, key=lambda b: (b[0] + b[2]) // 2) send_btn = max(bottom_buttons, key=lambda b: (b[0] + b[2]) // 2)
cx = (send_btn[0] + send_btn[2]) // 2 cx = (send_btn[0] + send_btn[2]) // 2
cy = (send_btn[1] + send_btn[3]) // 2 cy = (send_btn[1] + send_btn[3]) // 2
adb("shell", "input", "tap", str(cx), str(cy)) adb("shell", "input", "tap", str(cx), str(cy))
return f"send button tapped ({cx}, {cy})" return f"send button tapped ({cx}, {cy})"
# Fallback: tap bottom-right corner of the screen, offset slightly inward
screen_w, _ = get_screen_size() screen_w, _ = get_screen_size()
fx = screen_w - 80 fx = screen_w - 80
fy = screen_h - 120 fy = screen_h - 120
@@ -383,9 +391,15 @@ def execute_action(action: dict) -> str:
elif act == "wait": elif act == "wait":
secs = float(action.get("seconds", 2)) secs = float(action.get("seconds", 2))
time.sleep(secs) _sleep(secs)
return f"waited {secs}s" return f"waited {secs}s"
elif act == "screenshot":
# Explicit screenshot action — agent wants to observe current state
label = action.get("label", "observe")
screenshot_b64(label)
return f"screenshot taken ({label})"
elif act == "done": elif act == "done":
return "DONE" return "DONE"
@@ -403,7 +417,7 @@ def execute_action(action: dict) -> str:
SYSTEM_PROMPT = """\ SYSTEM_PROMPT = """\
You are an Android phone automation agent. You control the device by issuing actions. You are an Android phone automation agent. You control the device by issuing actions.
On each turn you receive a screenshot of the current screen. On each turn you receive a screenshot of the current Android screen.
Respond with a JSON object for ONE action to take next. Respond with a JSON object for ONE action to take next.
Available actions: Available actions:
@@ -411,19 +425,21 @@ Available actions:
{"type": "type", "text": "<string>"} {"type": "type", "text": "<string>"}
{"type": "key", "key": "enter|back|home|delete|tab"} {"type": "key", "key": "enter|back|home|delete|tab"}
{"type": "swipe", "x1": <int>, "y1": <int>, "x2": <int>, "y2": <int>, "duration": <ms>} {"type": "swipe", "x1": <int>, "y1": <int>, "x2": <int>, "y2": <int>, "duration": <ms>}
{"type": "send"} -- tap the send button (auto-locates via UI hierarchy) {"type": "send"} -- tap the send/submit button (auto-locates via UI hierarchy)
{"type": "wait", "seconds": <float>} {"type": "wait", "seconds": <float>}
{"type": "screenshot", "label": "<tag>"} -- observe current state without acting
{"type": "done", "summary": "<what was accomplished>"} {"type": "done", "summary": "<what was accomplished>"}
{"type": "fail", "reason": "<why the goal cannot be achieved>"} {"type": "fail", "reason": "<why the goal cannot be achieved>"}
Rules: Rules:
- Issue exactly ONE action per turn as a JSON object. No markdown, no explanation outside JSON. - Issue exactly ONE action per turn as a JSON object. No markdown, no explanation outside JSON.
- Coordinates are in pixels relative to the screenshot dimensions. - Coordinates are in pixels relative to the screenshot dimensions.
- IMPORTANT: In this app, pressing "enter" inserts a newline, it does NOT send the message. - IMPORTANT: In this app, pressing "enter" inserts a newline — it does NOT send the message.
To send a message, use the {"type": "send"} action which auto-locates and taps the send button. To send a message use {"type": "send"} which auto-locates and taps the send/arrow button.
After typing, dismiss the keyboard by pressing "back", then use {"type": "send"}. After typing your message, press "back" to dismiss the keyboard, then use {"type": "send"}.
- When the goal is fully achieved, respond with {"type": "done", ...}. - Be efficient: skip unnecessary waits, tap directly on visible targets.
- If stuck after several attempts, respond with {"type": "fail", ...}. - When the goal is fully achieved respond with {"type": "done", "summary": "..."}.
- If genuinely stuck after 5+ attempts on the same element respond with {"type": "fail", ...}.
""" """
@@ -456,7 +472,6 @@ def make_client(model: str):
azure_endpoint=os.environ["AZURE_OPENAI_ENDPOINT"], azure_endpoint=os.environ["AZURE_OPENAI_ENDPOINT"],
api_version=os.environ.get("AZURE_OPENAI_API_VERSION", "2024-08-01-preview"), api_version=os.environ.get("AZURE_OPENAI_API_VERSION", "2024-08-01-preview"),
), azure_model ), azure_model
# Azure AI Foundry endpoint (cognitiveservices.azure.com/openai/v1) — use OpenAI client
if os.environ.get("AZURE_DEV_AI_API_KEY"): if os.environ.get("AZURE_DEV_AI_API_KEY"):
base_url = os.environ.get("AZURE_DEV_AI_BASE_URL", "https://vibe-dev-ai.cognitiveservices.azure.com/openai/v1") base_url = os.environ.get("AZURE_DEV_AI_BASE_URL", "https://vibe-dev-ai.cognitiveservices.azure.com/openai/v1")
azure_model = os.environ.get("AZURE_DEV_AI_MODEL", "gpt-4o-2024-11-20") azure_model = os.environ.get("AZURE_DEV_AI_MODEL", "gpt-4o-2024-11-20")
@@ -479,15 +494,9 @@ def make_client(model: str):
@lru_cache(maxsize=1) @lru_cache(maxsize=1)
def get_screen_size() -> tuple[int, int]: def get_screen_size() -> tuple[int, int]:
"""Return (width, height) of the connected device screen. """Return (width, height) of the connected device screen. Cached."""
Cached after the first call — screen dimensions are stable for the
lifetime of a test run, and caching avoids a redundant ADB round-trip
on every `send` action.
"""
try: try:
out = adb("shell", "wm", "size") out = adb("shell", "wm", "size")
# "Physical size: 1080x1920" or "Override size: 1080x1920"
for line in out.splitlines(): for line in out.splitlines():
if "size:" in line.lower(): if "size:" in line.lower():
dims = line.split(":")[-1].strip() dims = line.split(":")[-1].strip()
@@ -498,75 +507,272 @@ def get_screen_size() -> tuple[int, int]:
return 1080, 1920 return 1080, 1920
def run_cua(goal: str, max_steps: int = 30, model: str = "gpt-4o", def run_cua_step(goal: str, max_steps: int = 30, model: str = "gpt-4o",
include_ui_xml: bool = False, verbose: bool = True) -> dict: include_ui_xml: bool = False, verbose: bool = True,
"""Run the CUA loop until done/fail/max_steps.""" step_label: str = "", action_delay: float = 0.8) -> dict:
"""Run the CUA loop for a single goal until done/fail/max_steps.
Args:
goal: Natural-language instruction for this step.
max_steps: Hard cap on LLM turns.
model: Vision model deployment name.
include_ui_xml: Append UI hierarchy XML to each prompt turn.
verbose: Print action log.
step_label: Short name shown in logs/screenshot filenames.
action_delay: Seconds to pause after each action (scaled by speed_multiplier).
"""
client, model = make_client(model) client, model = make_client(model)
history = [] history = []
screen_w, screen_h = get_screen_size() screen_w, screen_h = get_screen_size()
label_prefix = f"[{step_label}] " if step_label else ""
for step in range(1, max_steps + 1): for step in range(1, max_steps + 1):
# Capture screenshot img_b64 = screenshot_b64(label=f"{step_label}_{step:02d}" if step_label else f"{step:03d}")
img_b64 = screenshot_b64()
# Build user message with screenshot content: list = [
content = [ {
{"type": "text", "text": f"Step {step}. Screen is {screen_w}x{screen_h} pixels. Goal: {goal}\nWhat action should I take next?"}, "type": "text",
"text": (
f"{label_prefix}Step {step}/{max_steps}. "
f"Screen: {screen_w}x{screen_h}px. "
f"Goal: {goal}\n"
"What action should I take next?"
),
},
{"type": "image_url", "image_url": {"url": f"data:image/png;base64,{img_b64}", "detail": "high"}}, {"type": "image_url", "image_url": {"url": f"data:image/png;base64,{img_b64}", "detail": "high"}},
] ]
# Optionally include UI XML for better element identification
if include_ui_xml: if include_ui_xml:
xml = ui_dump() xml = ui_dump()
if xml: if xml:
# Truncate to avoid token explosion content.append({"type": "text", "text": f"UI hierarchy (truncated to 4000 chars):\n{xml[:4000]}"})
content.append({"type": "text", "text": f"UI hierarchy (truncated):\n{xml[:4000]}"})
history.append({"role": "user", "content": content}) history.append({"role": "user", "content": content})
# Call LLM
reply = call_llm(client, model, SYSTEM_PROMPT, history) reply = call_llm(client, model, SYSTEM_PROMPT, history)
history.append({"role": "assistant", "content": reply}) history.append({"role": "assistant", "content": reply})
# Parse action # Parse action — tolerate markdown fences and multi-object responses
try: try:
clean = reply.strip() clean = reply.strip()
# Strip markdown code fences
if clean.startswith("```"): if clean.startswith("```"):
clean = clean.split("\n", 1)[1].rsplit("```", 1)[0].strip() clean = clean.split("\n", 1)[1].rsplit("```", 1)[0].strip()
# If model returned multiple JSON objects, take the first
m = re.search(r'\{[^{}]*\}', clean) m = re.search(r'\{[^{}]*\}', clean)
if m: action = json.loads(m.group(0)) if m else json.loads(clean)
action = json.loads(m.group(0))
else:
action = json.loads(clean)
except json.JSONDecodeError: except json.JSONDecodeError:
if verbose: if verbose:
print(f" [step {step}] Failed to parse: {reply[:100]}") print(f" {label_prefix}[step {step}] Failed to parse: {reply[:120]}")
continue continue
# Execute
result = execute_action(action) result = execute_action(action)
if verbose: if verbose:
print(f" [step {step}] {action.get('type', '?')} -> {result}") print(f" {label_prefix}[step {step}] {action.get('type', '?')} -> {result}")
if result == "DONE": if result == "DONE":
return {"status": "success", "steps": step, "summary": action.get("summary", "")} return {"status": "success", "steps": step, "summary": action.get("summary", "")}
if result.startswith("FAIL"): if result.startswith("FAIL"):
return {"status": "fail", "steps": step, "reason": action.get("reason", "")} return {"status": "fail", "steps": step, "reason": action.get("reason", "")}
# Brief pause between actions for UI to settle # Trim history to keep context manageable
time.sleep(1.0) if len(history) > 14:
history = history[-14:]
# Keep history manageable: only last 6 turns (12 messages) + system _sleep(action_delay)
if len(history) > 12:
history = history[-12:]
return {"status": "timeout", "steps": max_steps} return {"status": "timeout", "steps": max_steps}
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
# Smoke test scenarios # Onboarding showcase — structured multi-phase flow
# ---------------------------------------------------------------------------
# Banner printed before each named phase so the video is narrated by log output
PHASE_BANNERS = {
"connect": "STEP 1-2: Opening app — configuring server connection",
"session_list": "STEP 3: Connected — viewing session list",
"new_session": "STEP 4: Creating a new AI coding session",
"typescript": "STEP 5-6: Submitting TypeScript task — watching opencode work",
"verify": "STEP 7: Verifying task output / success response",
"settings": "STEP 8-9: Navigating to Settings — showing model selection",
}
def _banner(key: str) -> None:
line = "=" * 64
msg = PHASE_BANNERS.get(key, key)
print(f"\n{line}")
print(f" {msg}")
print(f"{line}\n")
def run_onboarding_showcase(
opencode_url: str = DEFAULT_OPENCODE_URL,
model: str = "gpt-5.4",
include_ui_xml: bool = False,
verbose: bool = True,
max_steps_per_phase: int = 20,
) -> dict:
"""Execute the full first-run onboarding journey.
Each phase is a focused CUA sub-goal. Phases are run sequentially.
Returns a summary dict with per-phase results.
"""
results: dict[str, dict] = {}
def _run(key: str, goal: str, max_steps: int | None = None) -> bool:
"""Run one phase. Returns True if succeeded."""
_banner(key)
steps = max_steps or max_steps_per_phase
r = run_cua_step(
goal=goal,
max_steps=steps,
model=model,
include_ui_xml=include_ui_xml,
verbose=verbose,
step_label=key,
action_delay=0.7,
)
results[key] = r
ok = r["status"] == "success"
icon = "OK" if ok else "FAIL"
print(f"\n [{icon}] Phase '{key}': {r['status']} in {r['steps']} steps")
if r.get("summary"):
print(f" {r['summary']}")
if r.get("reason"):
print(f" reason: {r['reason']}")
return ok
# -----------------------------------------------------------------------
# Phase 1-2: Open app, configure server connection
# -----------------------------------------------------------------------
ok = _run(
"connect",
goal=(
f"You are on the OpenCode mobile app. "
"The screen shows either a connection screen (first launch) or an empty connections list. "
"Your goal: add a new connection to the opencode server. "
"Look for an 'Add Connection', '+', or 'New Connection' button and tap it. "
f"In the URL / Host field type '{opencode_url}'. "
"Leave username and password blank. "
"Tap 'Save', 'Connect', or 'Done' to save the connection. "
"Report done when you can see the connection has been saved or the app navigated away from the add-connection form."
),
max_steps=max_steps_per_phase,
)
if not ok:
return {"status": "fail", "phase": "connect", "results": results}
_sleep(2.0)
# -----------------------------------------------------------------------
# Phase 3: Connect to server — view session list
# -----------------------------------------------------------------------
ok = _run(
"session_list",
goal=(
"The connection has been saved. "
"Now tap on the saved connection entry to connect to the server. "
"Wait up to 10 seconds for the session list screen to appear. "
"The session list may be empty (no sessions yet) — that is fine. "
"Report done when you can see the session list screen (even if empty)."
),
max_steps=15,
)
if not ok:
return {"status": "fail", "phase": "session_list", "results": results}
_sleep(1.5)
# -----------------------------------------------------------------------
# Phase 4: Create new session
# -----------------------------------------------------------------------
ok = _run(
"new_session",
goal=(
"You are on the sessions list screen. "
"Tap the '+' button (usually top-right) to create a new AI coding session. "
"Wait up to 5 seconds for the new session / chat screen to open. "
"Report done once you see a text input field at the bottom of the screen "
"(the session chat/input view is open)."
),
max_steps=12,
)
if not ok:
return {"status": "fail", "phase": "new_session", "results": results}
_sleep(1.0)
# -----------------------------------------------------------------------
# Phase 5-6: Type TypeScript task and wait for opencode to complete
# -----------------------------------------------------------------------
ok = _run(
"typescript",
goal=(
f"You are inside a new OpenCode session (chat view with a text input at the bottom). "
f"Tap the text input field. "
f"Type this exact message: {TYPESCRIPT_TASK!r} "
"Press back to dismiss the keyboard. "
"Use the send action to submit. "
"After sending, wait and watch — opencode will show tool calls and file writes as it works. "
"Wait up to 90 seconds total for the session to go idle/complete "
"(no new activity for at least 5 seconds, or a completion indicator appears). "
"Re-check every 15 seconds by looking at the screen. "
"Report done when opencode appears to have finished (idle, no spinners, last message is a summary or file was created)."
),
max_steps=25,
)
if not ok:
return {"status": "fail", "phase": "typescript", "results": results}
_sleep(2.0)
# -----------------------------------------------------------------------
# Phase 7: Verify output / success
# -----------------------------------------------------------------------
ok = _run(
"verify",
goal=(
"The opencode session has finished. "
"Look at the chat to confirm the TypeScript hello world task succeeded. "
"You should see: a mention of 'hello.ts', 'Hello, World!', a file creation tool call, "
"or a success summary from the assistant. "
"Take a clear screenshot showing the result. "
"Report done with a brief summary of what you see as evidence of success. "
"Report fail only if the screen clearly shows an error with no recovery."
),
max_steps=8,
)
# Verify phase is informational — continue even on uncertain result
_sleep(1.5)
# -----------------------------------------------------------------------
# Phase 8-9: Navigate to Settings, show model selection
# -----------------------------------------------------------------------
_run(
"settings",
goal=(
"Navigate to the Settings screen of the OpenCode mobile app. "
"Look for a gear icon, 'Settings' tab in the bottom navigation bar, "
"or a hamburger menu that contains Settings. Tap it. "
"Once on the Settings screen, look for a 'Model' or 'AI Model' option and tap it "
"to show the model selection list. "
"Take a screenshot showing the model list or model setting. "
"You do NOT need to change the model — just show it is accessible. "
"Report done when the settings/model screen is visible in a screenshot."
),
max_steps=15,
)
# Overall status: success if connect + session + typescript all succeeded
critical = ["connect", "session_list", "new_session", "typescript"]
failed_critical = [k for k in critical if results.get(k, {}).get("status") != "success"]
overall = "success" if not failed_critical else "partial"
return {"status": overall, "phase_results": results}
# ---------------------------------------------------------------------------
# Legacy smoke scenarios (kept for backwards compat / --scenario flag)
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
SMOKE_SCENARIOS = [ SMOKE_SCENARIOS = [
@@ -576,7 +782,7 @@ SMOKE_SCENARIOS = [
"You see the OpenCode mobile app. Tap the '+' button (top-right) to create a new session. " "You see the OpenCode mobile app. Tap the '+' button (top-right) to create a new session. "
"Tap the text input at the bottom. Type 'ping'. Press back to dismiss keyboard. " "Tap the text input at the bottom. Type 'ping'. Press back to dismiss keyboard. "
"Use the send action. Wait 5 seconds, then take another screenshot. " "Use the send action. Wait 5 seconds, then take another screenshot. "
"If you don't yet see an assistant reply, wait another 10 seconds and re-check (assistant replies can take 15+ seconds). " "If you don't yet see an assistant reply, wait another 10 seconds and re-check. "
"If still no assistant bubble, wait another 15 seconds and re-check one more time. " "If still no assistant bubble, wait another 15 seconds and re-check one more time. "
"Report success if you see both a 'You' bubble and an 'Assistant' bubble. " "Report success if you see both a 'You' bubble and an 'Assistant' bubble. "
"Report failure only after at least 30 seconds of total waiting with no assistant bubble." "Report failure only after at least 30 seconds of total waiting with no assistant bubble."
@@ -598,7 +804,7 @@ SMOKE_SCENARIOS = [
"goal": ( "goal": (
"You see the OpenCode mobile app. Tap the '+' button (top-right) to create a new session. " "You see the OpenCode mobile app. Tap the '+' button (top-right) to create a new session. "
"Wait 2 seconds for the session to be created. " "Wait 2 seconds for the session to be created. "
"Navigate back to the sessions list by tapping the bottom-left 'Sessions' tab or pressing the back button. " "Navigate back to the sessions list by tapping the 'Sessions' tab or pressing back. "
"Wait 3 seconds for the session list to load. " "Wait 3 seconds for the session list to load. "
"Report success if you can see at least one session entry in the list. " "Report success if you can see at least one session entry in the list. "
"Report failure if the sessions list appears empty or shows an error message." "Report failure if the sessions list appears empty or shows an error message."
@@ -606,8 +812,7 @@ SMOKE_SCENARIOS = [
}, },
] ]
# Extended scenarios requiring an external OpenCode server.
# Run with: python scripts/android-cua-smoke.py --opencode-url http://<host>:<port>
def _connect_and_verify_sessions_goal(url: str) -> str: def _connect_and_verify_sessions_goal(url: str) -> str:
return ( return (
f"You see the OpenCode mobile app. " f"You see the OpenCode mobile app. "
@@ -626,38 +831,83 @@ def _connect_and_verify_sessions_goal(url: str) -> str:
) )
# ---------------------------------------------------------------------------
# CLI entry point
# ---------------------------------------------------------------------------
def main(): def main():
parser = argparse.ArgumentParser(description="Android CUA smoke test") parser = argparse.ArgumentParser(
parser.add_argument("--goal", help="Custom goal (overrides built-in scenarios)") description="OpenCode Mobile Android CUA smoke test — full onboarding showcase",
parser.add_argument("--model", default="gpt-4o", help="Vision model to use") formatter_class=argparse.RawDescriptionHelpFormatter,
parser.add_argument("--max-steps", type=int, default=30) epilog="""
parser.add_argument("--include-xml", action="store_true", help="Include UI XML in context") Examples:
parser.add_argument("--quiet", action="store_true") # Full onboarding showcase (default, recommended for demo video):
source ~/.env.d/azure-openai.env
python scripts/android-cua-smoke.py --model gpt-5.4 --include-xml
# Speed up for a faster demo (0.5 = half the wait times):
python scripts/android-cua-smoke.py --speed-multiplier 0.5
# Legacy single-goal mode:
python scripts/android-cua-smoke.py --goal "Open settings"
# Legacy named scenario:
python scripts/android-cua-smoke.py --scenarios send_message,verify_session_list
""",
)
# Showcase mode (new default)
parser.add_argument(
"--showcase",
action="store_true",
default=True,
help="Run the full onboarding showcase (default). Demonstrates connect → session → TypeScript task → settings.",
)
parser.add_argument( parser.add_argument(
"--opencode-url", "--opencode-url",
help="OpenCode server URL (e.g. http://100.108.64.76:4096). " default=None,
"Used by the default connect-and-verify regression scenario.", help=f"OpenCode server URL (default: {DEFAULT_OPENCODE_URL}).",
) )
# Speed control
parser.add_argument( parser.add_argument(
"--skip-connect-scenario", "--speed-multiplier",
action="store_true", type=float,
help="Skip the default connect-and-verify regression scenario.", default=1.0,
) metavar="FACTOR",
parser.add_argument( help="Scale all wait/sleep durations. 0.5 = twice as fast, 2.0 = twice as slow. Default: 1.0",
"--only-connect-scenario",
action="store_true",
help="Run ONLY the connect-and-verify-sessions scenario. Use in CI with a "
"local opencode server for a deterministic true-E2E (no model backend needed).",
) )
# Model / verbosity
parser.add_argument("--model", default="gpt-4o", help="Vision model deployment name.")
parser.add_argument("--max-steps", type=int, default=20, help="Max LLM steps per phase (showcase) or total (legacy).")
parser.add_argument("--include-xml", action="store_true", help="Include UI hierarchy XML in LLM context (more accurate, more tokens).")
parser.add_argument("--quiet", action="store_true")
# Legacy / compat flags
parser.add_argument("--goal", help="Legacy: single custom goal (disables showcase).")
parser.add_argument( parser.add_argument(
"--scenarios", "--scenarios",
help="Comma-separated explicit scenario set to run, e.g. " help="Legacy: comma-separated scenario names to run (disables showcase). "
"'connect_and_verify_sessions,send_message,verify_session_list'. " "Valid: connect_and_verify_sessions, send_message, multi_turn, verify_session_list.",
"Valid names: connect_and_verify_sessions, send_message, multi_turn, "
"verify_session_list. Overrides --only-connect-scenario and the default set.",
) )
parser.add_argument(
"--skip-connect-scenario", action="store_true",
help="Legacy: skip the connect-and-verify regression scenario.",
)
parser.add_argument(
"--only-connect-scenario", action="store_true",
help="Legacy: run ONLY the connect-and-verify-sessions scenario.",
)
args = parser.parse_args() args = parser.parse_args()
# Apply speed multiplier globally
global _speed_multiplier
_speed_multiplier = args.speed_multiplier
if args.speed_multiplier != 1.0:
print(f"[speed] multiplier={args.speed_multiplier} — all waits scaled accordingly")
# Verify ADB # Verify ADB
try: try:
devices = adb("devices") devices = adb("devices")
@@ -666,29 +916,81 @@ def main():
except FileNotFoundError: except FileNotFoundError:
sys.exit("adb not found in PATH") sys.exit("adb not found in PATH")
connect_url = args.opencode_url or os.environ.get("OPENCODE_URL") or "http://100.108.64.76:4096" connect_url = args.opencode_url or os.environ.get("OPENCODE_URL") or DEFAULT_OPENCODE_URL
# -----------------------------------------------------------------------
# Determine run mode: showcase vs. legacy scenarios
# -----------------------------------------------------------------------
use_legacy = bool(args.goal or args.scenarios or args.only_connect_scenario)
if not use_legacy:
# ----------------------------------------------------------------
# NEW DEFAULT: Full onboarding showcase
# ----------------------------------------------------------------
print("\n" + "=" * 64)
print(" OpenCode Mobile — Full Onboarding Showcase")
print(f" Server: {connect_url}")
print(f" Model: {args.model}")
print(f" Speed: {_speed_multiplier}x")
print("=" * 64)
rec_thread, _stop_ev, remote_path = start_screen_recording("onboarding_showcase")
local_video = "/tmp/cua_onboarding_showcase.mp4"
try:
if not ensure_app_foreground(verbose=not args.quiet):
print("[prep] warning: could not confirm app in foreground")
maybe_dismiss_telemetry_consent(verbose=not args.quiet)
ensure_app_foreground(verbose=not args.quiet)
result = run_onboarding_showcase(
opencode_url=connect_url,
model=args.model,
include_ui_xml=args.include_xml,
verbose=not args.quiet,
max_steps_per_phase=args.max_steps,
)
finally:
stop_screen_recording(rec_thread, remote_path, local_video)
upload_to_archivebox(local_video, "onboarding_showcase")
print("\n" + "=" * 64)
print(f" Showcase result: {result['status'].upper()}")
if local_video and Path(local_video).exists():
print(f" Video: {local_video}")
print("=" * 64)
# Print per-phase summary table
phase_results = result.get("phase_results", {})
if phase_results:
print("\n Phase breakdown:")
for phase, pr in phase_results.items():
icon = "PASS" if pr["status"] == "success" else "FAIL"
print(f" [{icon}] {phase:20s} {pr['status']:8s} {pr['steps']} steps")
sys.exit(0 if result["status"] == "success" else 1)
# -----------------------------------------------------------------------
# LEGACY MODE: named/custom scenarios
# -----------------------------------------------------------------------
connect_scenario = { connect_scenario = {
"name": "connect_and_verify_sessions", "name": "connect_and_verify_sessions",
"goal": _connect_and_verify_sessions_goal(connect_url), "goal": _connect_and_verify_sessions_goal(connect_url),
} }
if args.scenarios: if args.scenarios:
# Explicit named set (CI widened gate). Look up by name across the full catalog.
catalog = {connect_scenario["name"]: connect_scenario} catalog = {connect_scenario["name"]: connect_scenario}
for s in SMOKE_SCENARIOS: for s in SMOKE_SCENARIOS:
catalog[s["name"]] = s catalog[s["name"]] = s
requested = [n.strip() for n in args.scenarios.split(",") if n.strip()] requested = [n.strip() for n in args.scenarios.split(",") if n.strip()]
unknown = [n for n in requested if n not in catalog] unknown = [n for n in requested if n not in catalog]
if unknown: if unknown:
sys.exit(f"Unknown scenario(s): {', '.join(unknown)}. " sys.exit(f"Unknown scenario(s): {', '.join(unknown)}. Valid: {', '.join(catalog.keys())}")
f"Valid: {', '.join(catalog.keys())}")
scenarios = [catalog[n] for n in requested] scenarios = [catalog[n] for n in requested]
elif args.only_connect_scenario: elif args.only_connect_scenario:
# CI true-E2E: just connect to the local opencode server and verify the list.
scenarios = [connect_scenario] scenarios = [connect_scenario]
else: else:
scenarios = [{"name": "custom", "goal": args.goal}] if args.goal else list(SMOKE_SCENARIOS) scenarios = [{"name": "custom", "goal": args.goal}] if args.goal else list(SMOKE_SCENARIOS)
# Keep connect-and-verify in the default smoke path so regressions are exercised.
if not args.goal and not args.skip_connect_scenario: if not args.goal and not args.skip_connect_scenario:
scenarios.append(connect_scenario) scenarios.append(connect_scenario)
@@ -700,7 +1002,6 @@ def main():
print(f"Goal: {scenario['goal'][:80]}...") print(f"Goal: {scenario['goal'][:80]}...")
print(f"{'='*60}") print(f"{'='*60}")
# Start screen recording
rec_thread, _stop_ev, remote_path = start_screen_recording(scenario["name"]) rec_thread, _stop_ev, remote_path = start_screen_recording(scenario["name"])
local_video = f"/tmp/cua_{scenario['name']}.mp4" local_video = f"/tmp/cua_{scenario['name']}.mp4"
@@ -710,15 +1011,15 @@ def main():
maybe_dismiss_telemetry_consent(verbose=not args.quiet) maybe_dismiss_telemetry_consent(verbose=not args.quiet)
ensure_app_foreground(verbose=not args.quiet) ensure_app_foreground(verbose=not args.quiet)
result = run_cua( result = run_cua_step(
goal=scenario["goal"], goal=scenario["goal"],
max_steps=args.max_steps, max_steps=args.max_steps,
model=args.model, model=args.model,
include_ui_xml=args.include_xml, include_ui_xml=args.include_xml,
verbose=not args.quiet, verbose=not args.quiet,
step_label=scenario["name"],
) )
finally: finally:
# Always stop and pull the recording
stop_screen_recording(rec_thread, remote_path, local_video) stop_screen_recording(rec_thread, remote_path, local_video)
upload_to_archivebox(local_video, scenario["name"]) upload_to_archivebox(local_video, scenario["name"])
@@ -735,7 +1036,6 @@ def main():
if result.get("reason"): if result.get("reason"):
print(f"Reason: {result['reason']}") print(f"Reason: {result['reason']}")
# Exit code: 0 if all passed
failed = [r for r in results if r["status"] != "success"] failed = [r for r in results if r["status"] != "success"]
if failed: if failed:
print(f"\n{'!'*60}") print(f"\n{'!'*60}")