From 3c498972ce35c156bf83f5dcfd088d510ed15430 Mon Sep 17 00:00:00 2001 From: Dennis V <2119348+dzianisv@users.noreply.github.com> Date: Sun, 21 Jun 2026 01:20:44 +0000 Subject: [PATCH] feat(cua): full onboarding showcase test - connect, session, TypeScript task, settings MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Rewrites android-cua-smoke.py to demonstrate the complete first-run journey instead of the previous "ping" smoke test. The new structured multi-phase flow covers: server connection setup, session list, new session creation, TypeScript hello-world task submission (watching tool calls/file writes), output verification, and Settings/model-selection screenshot. Key changes: - run_onboarding_showcase() orchestrates 6 sequential CUA phases with per-phase goals, step budgets, and PASS/FAIL phase tracking - run_cua_step() replaces run_cua() — accepts step_label, action_delay, saves labeled screenshots (/tmp/cua__.png) for debugging - Global --speed-multiplier flag scales all _sleep() calls (0.5 = 2x faster) - Showcase is now the default mode; legacy --goal / --scenarios flags retained for backwards compat and CI regression scenarios - Tighter action_delay (0.7s) and trimmed history window (14 turns) vs previous 1.0s / 12 turns - Phase banner log lines ("STEP N: ...") narrate the video in real time Co-Authored-By: Claude Sonnet 4.6 Claude-Session: https://claude.ai/code/session_01No3k1AEioE4PNUZg12TxQo --- scripts/android-cua-smoke.py | 566 +++++++++++++++++++++++++++-------- 1 file changed, 433 insertions(+), 133 deletions(-) diff --git a/scripts/android-cua-smoke.py b/scripts/android-cua-smoke.py index 75b46f7..143bb37 100755 --- a/scripts/android-cua-smoke.py +++ b/scripts/android-cua-smoke.py @@ -2,43 +2,41 @@ """ Android Computer-Use Agent (CUA) smoke test for OpenCode Mobile. -Drives an Android emulator via ADB using an LLM vision loop: +Full onboarding showcase — drives an Android emulator via ADB using an LLM vision loop: screenshot → vision model → action → repeat -Inspired by: -- openai/openai-cua-sample-app (browser CUA pattern) -- X-PLUG/MobileAgent (ADB + VLM loop) -- TencentQQGYLab/AppAgent (multimodal smartphone agent) +Demonstrates the complete first-run journey: + 1. App opens on connection screen (no saved connections) + 2. Configure opencode server URL + 3. Connect — session list loads + 4. Create new AI coding session + 5. Submit a TypeScript "hello world" task + 6. Watch opencode work (tool calls, file writes), wait for idle + 7. Verify output / success response + 8. Navigate to Settings — show model selection + 9. Screenshot settings screen Requirements: pip install openai ADB in PATH with a connected device/emulator. Usage: + # Azure OpenAI (recommended — already configured via ~/.env.d/azure-openai.env) + source ~/.env.d/azure-openai.env + python scripts/android-cua-smoke.py --model gpt-5.4 --include-xml + # OpenAI export OPENAI_API_KEY=sk-... - python scripts/android-cua-smoke.py + python scripts/android-cua-smoke.py --model gpt-4o --include-xml - # Azure OpenAI (with deployment) - export OPENAI_API_KEY= - export OPENAI_BASE_URL=https://.openai.azure.com/openai/deployments// - python scripts/android-cua-smoke.py --model gpt-4o + # Run ONLY the onboarding showcase (default and primary flow): + python scripts/android-cua-smoke.py --showcase - # Google Gemini (via OpenAI-compat) - export GEMINI_API_KEY=AIza... - python scripts/android-cua-smoke.py - - # xAI Grok - export XAI_API_KEY=xai-... - python scripts/android-cua-smoke.py - - # Any OpenAI-compatible endpoint (LiteLLM, Ollama, etc.) - export OPENAI_API_KEY=dummy - export OPENAI_BASE_URL=http://localhost:4000/v1 - python scripts/android-cua-smoke.py --model gpt-4o - - # Custom goal + # Custom goal (legacy / quick debugging): python scripts/android-cua-smoke.py --goal "Open settings and toggle dark mode" + + # Speed up for a demo video (tighter waits, fewer retries): + python scripts/android-cua-smoke.py --speed-multiplier 0.5 """ import argparse @@ -61,6 +59,32 @@ except ImportError: sys.exit("openai package required: pip install openai") +# --------------------------------------------------------------------------- +# Constants +# --------------------------------------------------------------------------- + +APP_PACKAGE = "cc.agentlabs.opencode" + +# Default opencode Tailscale dev server +DEFAULT_OPENCODE_URL = "http://100.108.64.76:4096" + +# TypeScript task prompt sent to the AI coding session +TYPESCRIPT_TASK = ( + "Write a TypeScript hello world app. " + "Create a file hello.ts that prints 'Hello, World!' to the console." +) + +# --------------------------------------------------------------------------- +# Global state +# --------------------------------------------------------------------------- + +_step_counter = 0 +_speed_multiplier = 1.0 # Set via --speed-multiplier; <1.0 = faster + + +def _sleep(seconds: float) -> None: + """Interruptible sleep that respects the global speed multiplier.""" + time.sleep(max(0.2, seconds * _speed_multiplier)) # --------------------------------------------------------------------------- @@ -78,10 +102,6 @@ def adb(*args: str) -> str: return result.stdout.strip() -_step_counter = 0 -APP_PACKAGE = "cc.agentlabs.opencode" - - def _bounds_center(bounds: str) -> tuple[int, int] | None: match = re.match(r"\[(\d+),(\d+)\]\[(\d+),(\d+)\]", bounds or "") if not match: @@ -111,7 +131,7 @@ def ensure_app_foreground(package: str = APP_PACKAGE, retries: int = 3, return True adb("shell", "monkey", "-p", package, "-c", "android.intent.category.LAUNCHER", "1") - time.sleep(2.0) + _sleep(2.0) if verbose: seen = current or "unknown" @@ -172,7 +192,7 @@ def maybe_dismiss_telemetry_consent(package: str = APP_PACKAGE, for label, (x, y) in candidates: if any(marker in label for marker in dismiss_markers): adb("shell", "input", "tap", str(x), str(y)) - time.sleep(1.0) + _sleep(1.0) if verbose: print(f" [prep] dismissed telemetry consent via '{label or 'button'}' at ({x}, {y})") return True @@ -182,10 +202,13 @@ def maybe_dismiss_telemetry_consent(package: str = APP_PACKAGE, return False -def screenshot_b64() -> str: +def screenshot_b64(label: str = "") -> str: """Capture emulator screenshot and return as base64 PNG. Retries on timeout.""" global _step_counter _step_counter += 1 + suffix = f"_{label}" if label else "" + debug_path = f"/tmp/cua_step_{_step_counter:03d}{suffix}.png" + for attempt in range(3): try: result = subprocess.run( @@ -193,15 +216,14 @@ def screenshot_b64() -> str: capture_output=True, timeout=30, ) if result.returncode == 0 and len(result.stdout) > 100: - # Save for debugging - debug_path = f"/tmp/cua_step_{_step_counter:03d}.png" Path(debug_path).write_bytes(result.stdout) return base64.b64encode(result.stdout).decode() except subprocess.TimeoutExpired: if attempt < 2: - time.sleep(3) + _sleep(3) continue raise + # Fallback: screencap on device then pull with tempfile.NamedTemporaryFile(suffix=".png", delete=False) as f: path = f.name @@ -211,7 +233,7 @@ def screenshot_b64() -> str: subprocess.run(["adb", "pull", "/sdcard/_cua_screen.png", path], capture_output=True, timeout=10) data = Path(path).read_bytes() - Path(f"/tmp/cua_step_{_step_counter:03d}.png").write_bytes(data) + Path(debug_path).write_bytes(data) return base64.b64encode(data).decode() finally: Path(path).unlink(missing_ok=True) @@ -227,7 +249,6 @@ def start_screen_recording(scenario_name: str) -> tuple: stop_event = threading.Event() def _record(): - # --time-limit 180 caps at 3 min; we stop it early via SIGTERM via adb try: subprocess.run( ["adb", "shell", f"screenrecord --time-limit 180 {remote_path}"], @@ -238,19 +259,18 @@ def start_screen_recording(scenario_name: str) -> tuple: thread = threading.Thread(target=_record, daemon=True) thread.start() - time.sleep(1.0) # let recorder spin up + _sleep(1.0) return thread, stop_event, remote_path def stop_screen_recording(thread: threading.Thread, remote_path: str, local_path: str) -> bool: """Stop recorder, pull video to local_path. Returns True on success.""" - # Stop screenrecord via pkill (SIGINT flushes MP4 moov atom) subprocess.run( ["adb", "shell", "pkill", "-2", "screenrecord"], capture_output=True, timeout=10, ) - time.sleep(2.0) # let MP4 finalise + _sleep(2.0) thread.join(timeout=5) result = subprocess.run( @@ -269,11 +289,7 @@ def stop_screen_recording(thread: threading.Thread, remote_path: str, # --------------------------------------------------------------------------- def upload_to_archivebox(video_path: str, scenario_name: str) -> bool: - """Upload video to ArchiveBox if ARCHIVEBOX_URL is configured. - - Reads ARCHIVEBOX_URL and ARCHIVEBOX_API_KEY from environment. - Returns True if uploaded, False/skipped otherwise. - """ + """Upload video to ArchiveBox if ARCHIVEBOX_URL is configured.""" url = os.environ.get("ARCHIVEBOX_URL", "").rstrip("/") api_key = os.environ.get("ARCHIVEBOX_API_KEY", "") if not url: @@ -282,7 +298,6 @@ def upload_to_archivebox(video_path: str, scenario_name: str) -> bool: try: import urllib.request - import urllib.parse video_data = Path(video_path).read_bytes() boundary = "----CUAUploadBoundary" @@ -304,7 +319,6 @@ def upload_to_archivebox(video_path: str, scenario_name: str) -> bool: req = urllib.request.Request(f"{url}/api/v1/add", data=body, headers=headers, method="POST") with urllib.request.urlopen(req, timeout=60) as resp: - resp_data = resp.read().decode(errors="replace") print(f" [archivebox] uploaded {scenario_name}.mp4 → {url} ({resp.status})") return True except Exception as exc: @@ -313,7 +327,7 @@ def upload_to_archivebox(video_path: str, scenario_name: str) -> bool: def ui_dump() -> str: - """Dump UI hierarchy XML and return as string (optional context for LLM).""" + """Dump UI hierarchy XML and return as string.""" adb("shell", "uiautomator", "dump", "/sdcard/_cua_ui.xml") result = subprocess.run( ["adb", "pull", "/sdcard/_cua_ui.xml", "/tmp/_cua_ui.xml"], @@ -335,7 +349,6 @@ def execute_action(action: dict) -> str: elif act == "type": text = action.get("text", "") - # ADB input text needs escaping escaped = text.replace(" ", "%s").replace("&", "\\&").replace(";", "\\;") adb("shell", "input", "text", escaped) return f"typed '{text}'" @@ -358,23 +371,18 @@ def execute_action(action: dict) -> str: return f"swiped ({x1},{y1})->({x2},{y2})" elif act == "send": - # Auto-locate send button: rightmost clickable ViewGroup in the bottom input bar. - # Threshold is screen-relative (bottom 25%) so it works on any emulator - # resolution — API 30 default profile is 1080x1920, not the 2400-tall pixel - # we previously hardcoded against. + # Auto-locate send button: rightmost clickable button in the bottom input bar. _, screen_h = get_screen_size() bottom_threshold = int(screen_h * 0.75) xml = ui_dump() matches = re.findall(r'clickable="true"[^>]*bounds="\[(\d+),(\d+)\]\[(\d+),(\d+)\]"', xml) bottom_buttons = [(int(x1), int(y1), int(x2), int(y2)) for x1, y1, x2, y2 in matches if int(y1) > bottom_threshold] if bottom_buttons: - # Rightmost = highest center-x (not left-edge x1, which misidentifies wide buttons) send_btn = max(bottom_buttons, key=lambda b: (b[0] + b[2]) // 2) cx = (send_btn[0] + send_btn[2]) // 2 cy = (send_btn[1] + send_btn[3]) // 2 adb("shell", "input", "tap", str(cx), str(cy)) return f"send button tapped ({cx}, {cy})" - # Fallback: tap bottom-right corner of the screen, offset slightly inward screen_w, _ = get_screen_size() fx = screen_w - 80 fy = screen_h - 120 @@ -383,9 +391,15 @@ def execute_action(action: dict) -> str: elif act == "wait": secs = float(action.get("seconds", 2)) - time.sleep(secs) + _sleep(secs) return f"waited {secs}s" + elif act == "screenshot": + # Explicit screenshot action — agent wants to observe current state + label = action.get("label", "observe") + screenshot_b64(label) + return f"screenshot taken ({label})" + elif act == "done": return "DONE" @@ -403,7 +417,7 @@ def execute_action(action: dict) -> str: SYSTEM_PROMPT = """\ You are an Android phone automation agent. You control the device by issuing actions. -On each turn you receive a screenshot of the current screen. +On each turn you receive a screenshot of the current Android screen. Respond with a JSON object for ONE action to take next. Available actions: @@ -411,19 +425,21 @@ Available actions: {"type": "type", "text": ""} {"type": "key", "key": "enter|back|home|delete|tab"} {"type": "swipe", "x1": , "y1": , "x2": , "y2": , "duration": } - {"type": "send"} -- tap the send button (auto-locates via UI hierarchy) + {"type": "send"} -- tap the send/submit button (auto-locates via UI hierarchy) {"type": "wait", "seconds": } + {"type": "screenshot", "label": ""} -- observe current state without acting {"type": "done", "summary": ""} {"type": "fail", "reason": ""} Rules: - Issue exactly ONE action per turn as a JSON object. No markdown, no explanation outside JSON. - Coordinates are in pixels relative to the screenshot dimensions. -- IMPORTANT: In this app, pressing "enter" inserts a newline, it does NOT send the message. - To send a message, use the {"type": "send"} action which auto-locates and taps the send button. - After typing, dismiss the keyboard by pressing "back", then use {"type": "send"}. -- When the goal is fully achieved, respond with {"type": "done", ...}. -- If stuck after several attempts, respond with {"type": "fail", ...}. +- IMPORTANT: In this app, pressing "enter" inserts a newline — it does NOT send the message. + To send a message use {"type": "send"} which auto-locates and taps the send/arrow button. + After typing your message, press "back" to dismiss the keyboard, then use {"type": "send"}. +- Be efficient: skip unnecessary waits, tap directly on visible targets. +- When the goal is fully achieved respond with {"type": "done", "summary": "..."}. +- If genuinely stuck after 5+ attempts on the same element respond with {"type": "fail", ...}. """ @@ -456,7 +472,6 @@ def make_client(model: str): azure_endpoint=os.environ["AZURE_OPENAI_ENDPOINT"], api_version=os.environ.get("AZURE_OPENAI_API_VERSION", "2024-08-01-preview"), ), azure_model - # Azure AI Foundry endpoint (cognitiveservices.azure.com/openai/v1) — use OpenAI client if os.environ.get("AZURE_DEV_AI_API_KEY"): base_url = os.environ.get("AZURE_DEV_AI_BASE_URL", "https://vibe-dev-ai.cognitiveservices.azure.com/openai/v1") azure_model = os.environ.get("AZURE_DEV_AI_MODEL", "gpt-4o-2024-11-20") @@ -479,15 +494,9 @@ def make_client(model: str): @lru_cache(maxsize=1) def get_screen_size() -> tuple[int, int]: - """Return (width, height) of the connected device screen. - - Cached after the first call — screen dimensions are stable for the - lifetime of a test run, and caching avoids a redundant ADB round-trip - on every `send` action. - """ + """Return (width, height) of the connected device screen. Cached.""" try: out = adb("shell", "wm", "size") - # "Physical size: 1080x1920" or "Override size: 1080x1920" for line in out.splitlines(): if "size:" in line.lower(): dims = line.split(":")[-1].strip() @@ -498,75 +507,272 @@ def get_screen_size() -> tuple[int, int]: return 1080, 1920 -def run_cua(goal: str, max_steps: int = 30, model: str = "gpt-4o", - include_ui_xml: bool = False, verbose: bool = True) -> dict: - """Run the CUA loop until done/fail/max_steps.""" +def run_cua_step(goal: str, max_steps: int = 30, model: str = "gpt-4o", + include_ui_xml: bool = False, verbose: bool = True, + step_label: str = "", action_delay: float = 0.8) -> dict: + """Run the CUA loop for a single goal until done/fail/max_steps. + + Args: + goal: Natural-language instruction for this step. + max_steps: Hard cap on LLM turns. + model: Vision model deployment name. + include_ui_xml: Append UI hierarchy XML to each prompt turn. + verbose: Print action log. + step_label: Short name shown in logs/screenshot filenames. + action_delay: Seconds to pause after each action (scaled by speed_multiplier). + """ client, model = make_client(model) history = [] screen_w, screen_h = get_screen_size() + label_prefix = f"[{step_label}] " if step_label else "" for step in range(1, max_steps + 1): - # Capture screenshot - img_b64 = screenshot_b64() + img_b64 = screenshot_b64(label=f"{step_label}_{step:02d}" if step_label else f"{step:03d}") - # Build user message with screenshot - content = [ - {"type": "text", "text": f"Step {step}. Screen is {screen_w}x{screen_h} pixels. Goal: {goal}\nWhat action should I take next?"}, + content: list = [ + { + "type": "text", + "text": ( + f"{label_prefix}Step {step}/{max_steps}. " + f"Screen: {screen_w}x{screen_h}px. " + f"Goal: {goal}\n" + "What action should I take next?" + ), + }, {"type": "image_url", "image_url": {"url": f"data:image/png;base64,{img_b64}", "detail": "high"}}, ] - # Optionally include UI XML for better element identification if include_ui_xml: xml = ui_dump() if xml: - # Truncate to avoid token explosion - content.append({"type": "text", "text": f"UI hierarchy (truncated):\n{xml[:4000]}"}) + content.append({"type": "text", "text": f"UI hierarchy (truncated to 4000 chars):\n{xml[:4000]}"}) history.append({"role": "user", "content": content}) - # Call LLM reply = call_llm(client, model, SYSTEM_PROMPT, history) history.append({"role": "assistant", "content": reply}) - # Parse action + # Parse action — tolerate markdown fences and multi-object responses try: clean = reply.strip() - # Strip markdown code fences if clean.startswith("```"): clean = clean.split("\n", 1)[1].rsplit("```", 1)[0].strip() - # If model returned multiple JSON objects, take the first m = re.search(r'\{[^{}]*\}', clean) - if m: - action = json.loads(m.group(0)) - else: - action = json.loads(clean) + action = json.loads(m.group(0)) if m else json.loads(clean) except json.JSONDecodeError: if verbose: - print(f" [step {step}] Failed to parse: {reply[:100]}") + print(f" {label_prefix}[step {step}] Failed to parse: {reply[:120]}") continue - # Execute result = execute_action(action) if verbose: - print(f" [step {step}] {action.get('type', '?')} -> {result}") + print(f" {label_prefix}[step {step}] {action.get('type', '?')} -> {result}") if result == "DONE": return {"status": "success", "steps": step, "summary": action.get("summary", "")} if result.startswith("FAIL"): return {"status": "fail", "steps": step, "reason": action.get("reason", "")} - # Brief pause between actions for UI to settle - time.sleep(1.0) + # Trim history to keep context manageable + if len(history) > 14: + history = history[-14:] - # Keep history manageable: only last 6 turns (12 messages) + system - if len(history) > 12: - history = history[-12:] + _sleep(action_delay) return {"status": "timeout", "steps": max_steps} # --------------------------------------------------------------------------- -# Smoke test scenarios +# Onboarding showcase — structured multi-phase flow +# --------------------------------------------------------------------------- + +# Banner printed before each named phase so the video is narrated by log output +PHASE_BANNERS = { + "connect": "STEP 1-2: Opening app — configuring server connection", + "session_list": "STEP 3: Connected — viewing session list", + "new_session": "STEP 4: Creating a new AI coding session", + "typescript": "STEP 5-6: Submitting TypeScript task — watching opencode work", + "verify": "STEP 7: Verifying task output / success response", + "settings": "STEP 8-9: Navigating to Settings — showing model selection", +} + + +def _banner(key: str) -> None: + line = "=" * 64 + msg = PHASE_BANNERS.get(key, key) + print(f"\n{line}") + print(f" {msg}") + print(f"{line}\n") + + +def run_onboarding_showcase( + opencode_url: str = DEFAULT_OPENCODE_URL, + model: str = "gpt-5.4", + include_ui_xml: bool = False, + verbose: bool = True, + max_steps_per_phase: int = 20, +) -> dict: + """Execute the full first-run onboarding journey. + + Each phase is a focused CUA sub-goal. Phases are run sequentially. + Returns a summary dict with per-phase results. + """ + + results: dict[str, dict] = {} + + def _run(key: str, goal: str, max_steps: int | None = None) -> bool: + """Run one phase. Returns True if succeeded.""" + _banner(key) + steps = max_steps or max_steps_per_phase + r = run_cua_step( + goal=goal, + max_steps=steps, + model=model, + include_ui_xml=include_ui_xml, + verbose=verbose, + step_label=key, + action_delay=0.7, + ) + results[key] = r + ok = r["status"] == "success" + icon = "OK" if ok else "FAIL" + print(f"\n [{icon}] Phase '{key}': {r['status']} in {r['steps']} steps") + if r.get("summary"): + print(f" {r['summary']}") + if r.get("reason"): + print(f" reason: {r['reason']}") + return ok + + # ----------------------------------------------------------------------- + # Phase 1-2: Open app, configure server connection + # ----------------------------------------------------------------------- + ok = _run( + "connect", + goal=( + f"You are on the OpenCode mobile app. " + "The screen shows either a connection screen (first launch) or an empty connections list. " + "Your goal: add a new connection to the opencode server. " + "Look for an 'Add Connection', '+', or 'New Connection' button and tap it. " + f"In the URL / Host field type '{opencode_url}'. " + "Leave username and password blank. " + "Tap 'Save', 'Connect', or 'Done' to save the connection. " + "Report done when you can see the connection has been saved or the app navigated away from the add-connection form." + ), + max_steps=max_steps_per_phase, + ) + if not ok: + return {"status": "fail", "phase": "connect", "results": results} + + _sleep(2.0) + + # ----------------------------------------------------------------------- + # Phase 3: Connect to server — view session list + # ----------------------------------------------------------------------- + ok = _run( + "session_list", + goal=( + "The connection has been saved. " + "Now tap on the saved connection entry to connect to the server. " + "Wait up to 10 seconds for the session list screen to appear. " + "The session list may be empty (no sessions yet) — that is fine. " + "Report done when you can see the session list screen (even if empty)." + ), + max_steps=15, + ) + if not ok: + return {"status": "fail", "phase": "session_list", "results": results} + + _sleep(1.5) + + # ----------------------------------------------------------------------- + # Phase 4: Create new session + # ----------------------------------------------------------------------- + ok = _run( + "new_session", + goal=( + "You are on the sessions list screen. " + "Tap the '+' button (usually top-right) to create a new AI coding session. " + "Wait up to 5 seconds for the new session / chat screen to open. " + "Report done once you see a text input field at the bottom of the screen " + "(the session chat/input view is open)." + ), + max_steps=12, + ) + if not ok: + return {"status": "fail", "phase": "new_session", "results": results} + + _sleep(1.0) + + # ----------------------------------------------------------------------- + # Phase 5-6: Type TypeScript task and wait for opencode to complete + # ----------------------------------------------------------------------- + ok = _run( + "typescript", + goal=( + f"You are inside a new OpenCode session (chat view with a text input at the bottom). " + f"Tap the text input field. " + f"Type this exact message: {TYPESCRIPT_TASK!r} " + "Press back to dismiss the keyboard. " + "Use the send action to submit. " + "After sending, wait and watch — opencode will show tool calls and file writes as it works. " + "Wait up to 90 seconds total for the session to go idle/complete " + "(no new activity for at least 5 seconds, or a completion indicator appears). " + "Re-check every 15 seconds by looking at the screen. " + "Report done when opencode appears to have finished (idle, no spinners, last message is a summary or file was created)." + ), + max_steps=25, + ) + if not ok: + return {"status": "fail", "phase": "typescript", "results": results} + + _sleep(2.0) + + # ----------------------------------------------------------------------- + # Phase 7: Verify output / success + # ----------------------------------------------------------------------- + ok = _run( + "verify", + goal=( + "The opencode session has finished. " + "Look at the chat to confirm the TypeScript hello world task succeeded. " + "You should see: a mention of 'hello.ts', 'Hello, World!', a file creation tool call, " + "or a success summary from the assistant. " + "Take a clear screenshot showing the result. " + "Report done with a brief summary of what you see as evidence of success. " + "Report fail only if the screen clearly shows an error with no recovery." + ), + max_steps=8, + ) + # Verify phase is informational — continue even on uncertain result + _sleep(1.5) + + # ----------------------------------------------------------------------- + # Phase 8-9: Navigate to Settings, show model selection + # ----------------------------------------------------------------------- + _run( + "settings", + goal=( + "Navigate to the Settings screen of the OpenCode mobile app. " + "Look for a gear icon, 'Settings' tab in the bottom navigation bar, " + "or a hamburger menu that contains Settings. Tap it. " + "Once on the Settings screen, look for a 'Model' or 'AI Model' option and tap it " + "to show the model selection list. " + "Take a screenshot showing the model list or model setting. " + "You do NOT need to change the model — just show it is accessible. " + "Report done when the settings/model screen is visible in a screenshot." + ), + max_steps=15, + ) + + # Overall status: success if connect + session + typescript all succeeded + critical = ["connect", "session_list", "new_session", "typescript"] + failed_critical = [k for k in critical if results.get(k, {}).get("status") != "success"] + overall = "success" if not failed_critical else "partial" + return {"status": overall, "phase_results": results} + + +# --------------------------------------------------------------------------- +# Legacy smoke scenarios (kept for backwards compat / --scenario flag) # --------------------------------------------------------------------------- SMOKE_SCENARIOS = [ @@ -576,7 +782,7 @@ SMOKE_SCENARIOS = [ "You see the OpenCode mobile app. Tap the '+' button (top-right) to create a new session. " "Tap the text input at the bottom. Type 'ping'. Press back to dismiss keyboard. " "Use the send action. Wait 5 seconds, then take another screenshot. " - "If you don't yet see an assistant reply, wait another 10 seconds and re-check (assistant replies can take 15+ seconds). " + "If you don't yet see an assistant reply, wait another 10 seconds and re-check. " "If still no assistant bubble, wait another 15 seconds and re-check one more time. " "Report success if you see both a 'You' bubble and an 'Assistant' bubble. " "Report failure only after at least 30 seconds of total waiting with no assistant bubble." @@ -598,7 +804,7 @@ SMOKE_SCENARIOS = [ "goal": ( "You see the OpenCode mobile app. Tap the '+' button (top-right) to create a new session. " "Wait 2 seconds for the session to be created. " - "Navigate back to the sessions list by tapping the bottom-left 'Sessions' tab or pressing the back button. " + "Navigate back to the sessions list by tapping the 'Sessions' tab or pressing back. " "Wait 3 seconds for the session list to load. " "Report success if you can see at least one session entry in the list. " "Report failure if the sessions list appears empty or shows an error message." @@ -606,8 +812,7 @@ SMOKE_SCENARIOS = [ }, ] -# Extended scenarios requiring an external OpenCode server. -# Run with: python scripts/android-cua-smoke.py --opencode-url http://: + def _connect_and_verify_sessions_goal(url: str) -> str: return ( f"You see the OpenCode mobile app. " @@ -626,38 +831,83 @@ def _connect_and_verify_sessions_goal(url: str) -> str: ) +# --------------------------------------------------------------------------- +# CLI entry point +# --------------------------------------------------------------------------- + def main(): - parser = argparse.ArgumentParser(description="Android CUA smoke test") - parser.add_argument("--goal", help="Custom goal (overrides built-in scenarios)") - parser.add_argument("--model", default="gpt-4o", help="Vision model to use") - parser.add_argument("--max-steps", type=int, default=30) - parser.add_argument("--include-xml", action="store_true", help="Include UI XML in context") - parser.add_argument("--quiet", action="store_true") + parser = argparse.ArgumentParser( + description="OpenCode Mobile Android CUA smoke test — full onboarding showcase", + formatter_class=argparse.RawDescriptionHelpFormatter, + epilog=""" +Examples: + # Full onboarding showcase (default, recommended for demo video): + source ~/.env.d/azure-openai.env + python scripts/android-cua-smoke.py --model gpt-5.4 --include-xml + + # Speed up for a faster demo (0.5 = half the wait times): + python scripts/android-cua-smoke.py --speed-multiplier 0.5 + + # Legacy single-goal mode: + python scripts/android-cua-smoke.py --goal "Open settings" + + # Legacy named scenario: + python scripts/android-cua-smoke.py --scenarios send_message,verify_session_list +""", + ) + + # Showcase mode (new default) + parser.add_argument( + "--showcase", + action="store_true", + default=True, + help="Run the full onboarding showcase (default). Demonstrates connect → session → TypeScript task → settings.", + ) parser.add_argument( "--opencode-url", - help="OpenCode server URL (e.g. http://100.108.64.76:4096). " - "Used by the default connect-and-verify regression scenario.", + default=None, + help=f"OpenCode server URL (default: {DEFAULT_OPENCODE_URL}).", ) + + # Speed control parser.add_argument( - "--skip-connect-scenario", - action="store_true", - help="Skip the default connect-and-verify regression scenario.", - ) - parser.add_argument( - "--only-connect-scenario", - action="store_true", - help="Run ONLY the connect-and-verify-sessions scenario. Use in CI with a " - "local opencode server for a deterministic true-E2E (no model backend needed).", + "--speed-multiplier", + type=float, + default=1.0, + metavar="FACTOR", + help="Scale all wait/sleep durations. 0.5 = twice as fast, 2.0 = twice as slow. Default: 1.0", ) + + # Model / verbosity + parser.add_argument("--model", default="gpt-4o", help="Vision model deployment name.") + parser.add_argument("--max-steps", type=int, default=20, help="Max LLM steps per phase (showcase) or total (legacy).") + parser.add_argument("--include-xml", action="store_true", help="Include UI hierarchy XML in LLM context (more accurate, more tokens).") + parser.add_argument("--quiet", action="store_true") + + # Legacy / compat flags + parser.add_argument("--goal", help="Legacy: single custom goal (disables showcase).") parser.add_argument( "--scenarios", - help="Comma-separated explicit scenario set to run, e.g. " - "'connect_and_verify_sessions,send_message,verify_session_list'. " - "Valid names: connect_and_verify_sessions, send_message, multi_turn, " - "verify_session_list. Overrides --only-connect-scenario and the default set.", + help="Legacy: comma-separated scenario names to run (disables showcase). " + "Valid: connect_and_verify_sessions, send_message, multi_turn, verify_session_list.", ) + parser.add_argument( + "--skip-connect-scenario", action="store_true", + help="Legacy: skip the connect-and-verify regression scenario.", + ) + parser.add_argument( + "--only-connect-scenario", action="store_true", + help="Legacy: run ONLY the connect-and-verify-sessions scenario.", + ) + args = parser.parse_args() + # Apply speed multiplier globally + global _speed_multiplier + _speed_multiplier = args.speed_multiplier + if args.speed_multiplier != 1.0: + print(f"[speed] multiplier={args.speed_multiplier} — all waits scaled accordingly") + # Verify ADB try: devices = adb("devices") @@ -666,29 +916,81 @@ def main(): except FileNotFoundError: sys.exit("adb not found in PATH") - connect_url = args.opencode_url or os.environ.get("OPENCODE_URL") or "http://100.108.64.76:4096" + connect_url = args.opencode_url or os.environ.get("OPENCODE_URL") or DEFAULT_OPENCODE_URL + + # ----------------------------------------------------------------------- + # Determine run mode: showcase vs. legacy scenarios + # ----------------------------------------------------------------------- + use_legacy = bool(args.goal or args.scenarios or args.only_connect_scenario) + + if not use_legacy: + # ---------------------------------------------------------------- + # NEW DEFAULT: Full onboarding showcase + # ---------------------------------------------------------------- + print("\n" + "=" * 64) + print(" OpenCode Mobile — Full Onboarding Showcase") + print(f" Server: {connect_url}") + print(f" Model: {args.model}") + print(f" Speed: {_speed_multiplier}x") + print("=" * 64) + + rec_thread, _stop_ev, remote_path = start_screen_recording("onboarding_showcase") + local_video = "/tmp/cua_onboarding_showcase.mp4" + + try: + if not ensure_app_foreground(verbose=not args.quiet): + print("[prep] warning: could not confirm app in foreground") + maybe_dismiss_telemetry_consent(verbose=not args.quiet) + ensure_app_foreground(verbose=not args.quiet) + + result = run_onboarding_showcase( + opencode_url=connect_url, + model=args.model, + include_ui_xml=args.include_xml, + verbose=not args.quiet, + max_steps_per_phase=args.max_steps, + ) + finally: + stop_screen_recording(rec_thread, remote_path, local_video) + upload_to_archivebox(local_video, "onboarding_showcase") + + print("\n" + "=" * 64) + print(f" Showcase result: {result['status'].upper()}") + if local_video and Path(local_video).exists(): + print(f" Video: {local_video}") + print("=" * 64) + + # Print per-phase summary table + phase_results = result.get("phase_results", {}) + if phase_results: + print("\n Phase breakdown:") + for phase, pr in phase_results.items(): + icon = "PASS" if pr["status"] == "success" else "FAIL" + print(f" [{icon}] {phase:20s} {pr['status']:8s} {pr['steps']} steps") + + sys.exit(0 if result["status"] == "success" else 1) + + # ----------------------------------------------------------------------- + # LEGACY MODE: named/custom scenarios + # ----------------------------------------------------------------------- connect_scenario = { "name": "connect_and_verify_sessions", "goal": _connect_and_verify_sessions_goal(connect_url), } if args.scenarios: - # Explicit named set (CI widened gate). Look up by name across the full catalog. catalog = {connect_scenario["name"]: connect_scenario} for s in SMOKE_SCENARIOS: catalog[s["name"]] = s requested = [n.strip() for n in args.scenarios.split(",") if n.strip()] unknown = [n for n in requested if n not in catalog] if unknown: - sys.exit(f"Unknown scenario(s): {', '.join(unknown)}. " - f"Valid: {', '.join(catalog.keys())}") + sys.exit(f"Unknown scenario(s): {', '.join(unknown)}. Valid: {', '.join(catalog.keys())}") scenarios = [catalog[n] for n in requested] elif args.only_connect_scenario: - # CI true-E2E: just connect to the local opencode server and verify the list. scenarios = [connect_scenario] else: scenarios = [{"name": "custom", "goal": args.goal}] if args.goal else list(SMOKE_SCENARIOS) - # Keep connect-and-verify in the default smoke path so regressions are exercised. if not args.goal and not args.skip_connect_scenario: scenarios.append(connect_scenario) @@ -700,7 +1002,6 @@ def main(): print(f"Goal: {scenario['goal'][:80]}...") print(f"{'='*60}") - # Start screen recording rec_thread, _stop_ev, remote_path = start_screen_recording(scenario["name"]) local_video = f"/tmp/cua_{scenario['name']}.mp4" @@ -710,15 +1011,15 @@ def main(): maybe_dismiss_telemetry_consent(verbose=not args.quiet) ensure_app_foreground(verbose=not args.quiet) - result = run_cua( + result = run_cua_step( goal=scenario["goal"], max_steps=args.max_steps, model=args.model, include_ui_xml=args.include_xml, verbose=not args.quiet, + step_label=scenario["name"], ) finally: - # Always stop and pull the recording stop_screen_recording(rec_thread, remote_path, local_video) upload_to_archivebox(local_video, scenario["name"]) @@ -735,7 +1036,6 @@ def main(): if result.get("reason"): print(f"Reason: {result['reason']}") - # Exit code: 0 if all passed failed = [r for r in results if r["status"] != "success"] if failed: print(f"\n{'!'*60}")