feat(cua): full onboarding showcase test - connect, session, TypeScript task, settings
Rewrites android-cua-smoke.py to demonstrate the complete first-run journey
instead of the previous "ping" smoke test. The new structured multi-phase
flow covers: server connection setup, session list, new session creation,
TypeScript hello-world task submission (watching tool calls/file writes),
output verification, and Settings/model-selection screenshot.
Key changes:
- run_onboarding_showcase() orchestrates 6 sequential CUA phases with
per-phase goals, step budgets, and PASS/FAIL phase tracking
- run_cua_step() replaces run_cua() — accepts step_label, action_delay,
saves labeled screenshots (/tmp/cua_<phase>_<step>.png) for debugging
- Global --speed-multiplier flag scales all _sleep() calls (0.5 = 2x faster)
- Showcase is now the default mode; legacy --goal / --scenarios flags retained
for backwards compat and CI regression scenarios
- Tighter action_delay (0.7s) and trimmed history window (14 turns) vs
previous 1.0s / 12 turns
- Phase banner log lines ("STEP N: ...") narrate the video in real time
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01No3k1AEioE4PNUZg12TxQo
This commit is contained in:
@@ -2,43 +2,41 @@
|
|||||||
"""
|
"""
|
||||||
Android Computer-Use Agent (CUA) smoke test for OpenCode Mobile.
|
Android Computer-Use Agent (CUA) smoke test for OpenCode Mobile.
|
||||||
|
|
||||||
Drives an Android emulator via ADB using an LLM vision loop:
|
Full onboarding showcase — drives an Android emulator via ADB using an LLM vision loop:
|
||||||
screenshot → vision model → action → repeat
|
screenshot → vision model → action → repeat
|
||||||
|
|
||||||
Inspired by:
|
Demonstrates the complete first-run journey:
|
||||||
- openai/openai-cua-sample-app (browser CUA pattern)
|
1. App opens on connection screen (no saved connections)
|
||||||
- X-PLUG/MobileAgent (ADB + VLM loop)
|
2. Configure opencode server URL
|
||||||
- TencentQQGYLab/AppAgent (multimodal smartphone agent)
|
3. Connect — session list loads
|
||||||
|
4. Create new AI coding session
|
||||||
|
5. Submit a TypeScript "hello world" task
|
||||||
|
6. Watch opencode work (tool calls, file writes), wait for idle
|
||||||
|
7. Verify output / success response
|
||||||
|
8. Navigate to Settings — show model selection
|
||||||
|
9. Screenshot settings screen
|
||||||
|
|
||||||
Requirements:
|
Requirements:
|
||||||
pip install openai
|
pip install openai
|
||||||
ADB in PATH with a connected device/emulator.
|
ADB in PATH with a connected device/emulator.
|
||||||
|
|
||||||
Usage:
|
Usage:
|
||||||
|
# Azure OpenAI (recommended — already configured via ~/.env.d/azure-openai.env)
|
||||||
|
source ~/.env.d/azure-openai.env
|
||||||
|
python scripts/android-cua-smoke.py --model gpt-5.4 --include-xml
|
||||||
|
|
||||||
# OpenAI
|
# OpenAI
|
||||||
export OPENAI_API_KEY=sk-...
|
export OPENAI_API_KEY=sk-...
|
||||||
python scripts/android-cua-smoke.py
|
python scripts/android-cua-smoke.py --model gpt-4o --include-xml
|
||||||
|
|
||||||
# Azure OpenAI (with deployment)
|
# Run ONLY the onboarding showcase (default and primary flow):
|
||||||
export OPENAI_API_KEY=<key>
|
python scripts/android-cua-smoke.py --showcase
|
||||||
export OPENAI_BASE_URL=https://<resource>.openai.azure.com/openai/deployments/<deployment>/
|
|
||||||
python scripts/android-cua-smoke.py --model gpt-4o
|
|
||||||
|
|
||||||
# Google Gemini (via OpenAI-compat)
|
# Custom goal (legacy / quick debugging):
|
||||||
export GEMINI_API_KEY=AIza...
|
|
||||||
python scripts/android-cua-smoke.py
|
|
||||||
|
|
||||||
# xAI Grok
|
|
||||||
export XAI_API_KEY=xai-...
|
|
||||||
python scripts/android-cua-smoke.py
|
|
||||||
|
|
||||||
# Any OpenAI-compatible endpoint (LiteLLM, Ollama, etc.)
|
|
||||||
export OPENAI_API_KEY=dummy
|
|
||||||
export OPENAI_BASE_URL=http://localhost:4000/v1
|
|
||||||
python scripts/android-cua-smoke.py --model gpt-4o
|
|
||||||
|
|
||||||
# Custom goal
|
|
||||||
python scripts/android-cua-smoke.py --goal "Open settings and toggle dark mode"
|
python scripts/android-cua-smoke.py --goal "Open settings and toggle dark mode"
|
||||||
|
|
||||||
|
# Speed up for a demo video (tighter waits, fewer retries):
|
||||||
|
python scripts/android-cua-smoke.py --speed-multiplier 0.5
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import argparse
|
import argparse
|
||||||
@@ -61,6 +59,32 @@ except ImportError:
|
|||||||
sys.exit("openai package required: pip install openai")
|
sys.exit("openai package required: pip install openai")
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Constants
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
APP_PACKAGE = "cc.agentlabs.opencode"
|
||||||
|
|
||||||
|
# Default opencode Tailscale dev server
|
||||||
|
DEFAULT_OPENCODE_URL = "http://100.108.64.76:4096"
|
||||||
|
|
||||||
|
# TypeScript task prompt sent to the AI coding session
|
||||||
|
TYPESCRIPT_TASK = (
|
||||||
|
"Write a TypeScript hello world app. "
|
||||||
|
"Create a file hello.ts that prints 'Hello, World!' to the console."
|
||||||
|
)
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Global state
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
_step_counter = 0
|
||||||
|
_speed_multiplier = 1.0 # Set via --speed-multiplier; <1.0 = faster
|
||||||
|
|
||||||
|
|
||||||
|
def _sleep(seconds: float) -> None:
|
||||||
|
"""Interruptible sleep that respects the global speed multiplier."""
|
||||||
|
time.sleep(max(0.2, seconds * _speed_multiplier))
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
@@ -78,10 +102,6 @@ def adb(*args: str) -> str:
|
|||||||
return result.stdout.strip()
|
return result.stdout.strip()
|
||||||
|
|
||||||
|
|
||||||
_step_counter = 0
|
|
||||||
APP_PACKAGE = "cc.agentlabs.opencode"
|
|
||||||
|
|
||||||
|
|
||||||
def _bounds_center(bounds: str) -> tuple[int, int] | None:
|
def _bounds_center(bounds: str) -> tuple[int, int] | None:
|
||||||
match = re.match(r"\[(\d+),(\d+)\]\[(\d+),(\d+)\]", bounds or "")
|
match = re.match(r"\[(\d+),(\d+)\]\[(\d+),(\d+)\]", bounds or "")
|
||||||
if not match:
|
if not match:
|
||||||
@@ -111,7 +131,7 @@ def ensure_app_foreground(package: str = APP_PACKAGE, retries: int = 3,
|
|||||||
return True
|
return True
|
||||||
|
|
||||||
adb("shell", "monkey", "-p", package, "-c", "android.intent.category.LAUNCHER", "1")
|
adb("shell", "monkey", "-p", package, "-c", "android.intent.category.LAUNCHER", "1")
|
||||||
time.sleep(2.0)
|
_sleep(2.0)
|
||||||
|
|
||||||
if verbose:
|
if verbose:
|
||||||
seen = current or "unknown"
|
seen = current or "unknown"
|
||||||
@@ -172,7 +192,7 @@ def maybe_dismiss_telemetry_consent(package: str = APP_PACKAGE,
|
|||||||
for label, (x, y) in candidates:
|
for label, (x, y) in candidates:
|
||||||
if any(marker in label for marker in dismiss_markers):
|
if any(marker in label for marker in dismiss_markers):
|
||||||
adb("shell", "input", "tap", str(x), str(y))
|
adb("shell", "input", "tap", str(x), str(y))
|
||||||
time.sleep(1.0)
|
_sleep(1.0)
|
||||||
if verbose:
|
if verbose:
|
||||||
print(f" [prep] dismissed telemetry consent via '{label or 'button'}' at ({x}, {y})")
|
print(f" [prep] dismissed telemetry consent via '{label or 'button'}' at ({x}, {y})")
|
||||||
return True
|
return True
|
||||||
@@ -182,10 +202,13 @@ def maybe_dismiss_telemetry_consent(package: str = APP_PACKAGE,
|
|||||||
return False
|
return False
|
||||||
|
|
||||||
|
|
||||||
def screenshot_b64() -> str:
|
def screenshot_b64(label: str = "") -> str:
|
||||||
"""Capture emulator screenshot and return as base64 PNG. Retries on timeout."""
|
"""Capture emulator screenshot and return as base64 PNG. Retries on timeout."""
|
||||||
global _step_counter
|
global _step_counter
|
||||||
_step_counter += 1
|
_step_counter += 1
|
||||||
|
suffix = f"_{label}" if label else ""
|
||||||
|
debug_path = f"/tmp/cua_step_{_step_counter:03d}{suffix}.png"
|
||||||
|
|
||||||
for attempt in range(3):
|
for attempt in range(3):
|
||||||
try:
|
try:
|
||||||
result = subprocess.run(
|
result = subprocess.run(
|
||||||
@@ -193,15 +216,14 @@ def screenshot_b64() -> str:
|
|||||||
capture_output=True, timeout=30,
|
capture_output=True, timeout=30,
|
||||||
)
|
)
|
||||||
if result.returncode == 0 and len(result.stdout) > 100:
|
if result.returncode == 0 and len(result.stdout) > 100:
|
||||||
# Save for debugging
|
|
||||||
debug_path = f"/tmp/cua_step_{_step_counter:03d}.png"
|
|
||||||
Path(debug_path).write_bytes(result.stdout)
|
Path(debug_path).write_bytes(result.stdout)
|
||||||
return base64.b64encode(result.stdout).decode()
|
return base64.b64encode(result.stdout).decode()
|
||||||
except subprocess.TimeoutExpired:
|
except subprocess.TimeoutExpired:
|
||||||
if attempt < 2:
|
if attempt < 2:
|
||||||
time.sleep(3)
|
_sleep(3)
|
||||||
continue
|
continue
|
||||||
raise
|
raise
|
||||||
|
|
||||||
# Fallback: screencap on device then pull
|
# Fallback: screencap on device then pull
|
||||||
with tempfile.NamedTemporaryFile(suffix=".png", delete=False) as f:
|
with tempfile.NamedTemporaryFile(suffix=".png", delete=False) as f:
|
||||||
path = f.name
|
path = f.name
|
||||||
@@ -211,7 +233,7 @@ def screenshot_b64() -> str:
|
|||||||
subprocess.run(["adb", "pull", "/sdcard/_cua_screen.png", path],
|
subprocess.run(["adb", "pull", "/sdcard/_cua_screen.png", path],
|
||||||
capture_output=True, timeout=10)
|
capture_output=True, timeout=10)
|
||||||
data = Path(path).read_bytes()
|
data = Path(path).read_bytes()
|
||||||
Path(f"/tmp/cua_step_{_step_counter:03d}.png").write_bytes(data)
|
Path(debug_path).write_bytes(data)
|
||||||
return base64.b64encode(data).decode()
|
return base64.b64encode(data).decode()
|
||||||
finally:
|
finally:
|
||||||
Path(path).unlink(missing_ok=True)
|
Path(path).unlink(missing_ok=True)
|
||||||
@@ -227,7 +249,6 @@ def start_screen_recording(scenario_name: str) -> tuple:
|
|||||||
stop_event = threading.Event()
|
stop_event = threading.Event()
|
||||||
|
|
||||||
def _record():
|
def _record():
|
||||||
# --time-limit 180 caps at 3 min; we stop it early via SIGTERM via adb
|
|
||||||
try:
|
try:
|
||||||
subprocess.run(
|
subprocess.run(
|
||||||
["adb", "shell", f"screenrecord --time-limit 180 {remote_path}"],
|
["adb", "shell", f"screenrecord --time-limit 180 {remote_path}"],
|
||||||
@@ -238,19 +259,18 @@ def start_screen_recording(scenario_name: str) -> tuple:
|
|||||||
|
|
||||||
thread = threading.Thread(target=_record, daemon=True)
|
thread = threading.Thread(target=_record, daemon=True)
|
||||||
thread.start()
|
thread.start()
|
||||||
time.sleep(1.0) # let recorder spin up
|
_sleep(1.0)
|
||||||
return thread, stop_event, remote_path
|
return thread, stop_event, remote_path
|
||||||
|
|
||||||
|
|
||||||
def stop_screen_recording(thread: threading.Thread, remote_path: str,
|
def stop_screen_recording(thread: threading.Thread, remote_path: str,
|
||||||
local_path: str) -> bool:
|
local_path: str) -> bool:
|
||||||
"""Stop recorder, pull video to local_path. Returns True on success."""
|
"""Stop recorder, pull video to local_path. Returns True on success."""
|
||||||
# Stop screenrecord via pkill (SIGINT flushes MP4 moov atom)
|
|
||||||
subprocess.run(
|
subprocess.run(
|
||||||
["adb", "shell", "pkill", "-2", "screenrecord"],
|
["adb", "shell", "pkill", "-2", "screenrecord"],
|
||||||
capture_output=True, timeout=10,
|
capture_output=True, timeout=10,
|
||||||
)
|
)
|
||||||
time.sleep(2.0) # let MP4 finalise
|
_sleep(2.0)
|
||||||
thread.join(timeout=5)
|
thread.join(timeout=5)
|
||||||
|
|
||||||
result = subprocess.run(
|
result = subprocess.run(
|
||||||
@@ -269,11 +289,7 @@ def stop_screen_recording(thread: threading.Thread, remote_path: str,
|
|||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
def upload_to_archivebox(video_path: str, scenario_name: str) -> bool:
|
def upload_to_archivebox(video_path: str, scenario_name: str) -> bool:
|
||||||
"""Upload video to ArchiveBox if ARCHIVEBOX_URL is configured.
|
"""Upload video to ArchiveBox if ARCHIVEBOX_URL is configured."""
|
||||||
|
|
||||||
Reads ARCHIVEBOX_URL and ARCHIVEBOX_API_KEY from environment.
|
|
||||||
Returns True if uploaded, False/skipped otherwise.
|
|
||||||
"""
|
|
||||||
url = os.environ.get("ARCHIVEBOX_URL", "").rstrip("/")
|
url = os.environ.get("ARCHIVEBOX_URL", "").rstrip("/")
|
||||||
api_key = os.environ.get("ARCHIVEBOX_API_KEY", "")
|
api_key = os.environ.get("ARCHIVEBOX_API_KEY", "")
|
||||||
if not url:
|
if not url:
|
||||||
@@ -282,7 +298,6 @@ def upload_to_archivebox(video_path: str, scenario_name: str) -> bool:
|
|||||||
|
|
||||||
try:
|
try:
|
||||||
import urllib.request
|
import urllib.request
|
||||||
import urllib.parse
|
|
||||||
|
|
||||||
video_data = Path(video_path).read_bytes()
|
video_data = Path(video_path).read_bytes()
|
||||||
boundary = "----CUAUploadBoundary"
|
boundary = "----CUAUploadBoundary"
|
||||||
@@ -304,7 +319,6 @@ def upload_to_archivebox(video_path: str, scenario_name: str) -> bool:
|
|||||||
|
|
||||||
req = urllib.request.Request(f"{url}/api/v1/add", data=body, headers=headers, method="POST")
|
req = urllib.request.Request(f"{url}/api/v1/add", data=body, headers=headers, method="POST")
|
||||||
with urllib.request.urlopen(req, timeout=60) as resp:
|
with urllib.request.urlopen(req, timeout=60) as resp:
|
||||||
resp_data = resp.read().decode(errors="replace")
|
|
||||||
print(f" [archivebox] uploaded {scenario_name}.mp4 → {url} ({resp.status})")
|
print(f" [archivebox] uploaded {scenario_name}.mp4 → {url} ({resp.status})")
|
||||||
return True
|
return True
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
@@ -313,7 +327,7 @@ def upload_to_archivebox(video_path: str, scenario_name: str) -> bool:
|
|||||||
|
|
||||||
|
|
||||||
def ui_dump() -> str:
|
def ui_dump() -> str:
|
||||||
"""Dump UI hierarchy XML and return as string (optional context for LLM)."""
|
"""Dump UI hierarchy XML and return as string."""
|
||||||
adb("shell", "uiautomator", "dump", "/sdcard/_cua_ui.xml")
|
adb("shell", "uiautomator", "dump", "/sdcard/_cua_ui.xml")
|
||||||
result = subprocess.run(
|
result = subprocess.run(
|
||||||
["adb", "pull", "/sdcard/_cua_ui.xml", "/tmp/_cua_ui.xml"],
|
["adb", "pull", "/sdcard/_cua_ui.xml", "/tmp/_cua_ui.xml"],
|
||||||
@@ -335,7 +349,6 @@ def execute_action(action: dict) -> str:
|
|||||||
|
|
||||||
elif act == "type":
|
elif act == "type":
|
||||||
text = action.get("text", "")
|
text = action.get("text", "")
|
||||||
# ADB input text needs escaping
|
|
||||||
escaped = text.replace(" ", "%s").replace("&", "\\&").replace(";", "\\;")
|
escaped = text.replace(" ", "%s").replace("&", "\\&").replace(";", "\\;")
|
||||||
adb("shell", "input", "text", escaped)
|
adb("shell", "input", "text", escaped)
|
||||||
return f"typed '{text}'"
|
return f"typed '{text}'"
|
||||||
@@ -358,23 +371,18 @@ def execute_action(action: dict) -> str:
|
|||||||
return f"swiped ({x1},{y1})->({x2},{y2})"
|
return f"swiped ({x1},{y1})->({x2},{y2})"
|
||||||
|
|
||||||
elif act == "send":
|
elif act == "send":
|
||||||
# Auto-locate send button: rightmost clickable ViewGroup in the bottom input bar.
|
# Auto-locate send button: rightmost clickable button in the bottom input bar.
|
||||||
# Threshold is screen-relative (bottom 25%) so it works on any emulator
|
|
||||||
# resolution — API 30 default profile is 1080x1920, not the 2400-tall pixel
|
|
||||||
# we previously hardcoded against.
|
|
||||||
_, screen_h = get_screen_size()
|
_, screen_h = get_screen_size()
|
||||||
bottom_threshold = int(screen_h * 0.75)
|
bottom_threshold = int(screen_h * 0.75)
|
||||||
xml = ui_dump()
|
xml = ui_dump()
|
||||||
matches = re.findall(r'clickable="true"[^>]*bounds="\[(\d+),(\d+)\]\[(\d+),(\d+)\]"', xml)
|
matches = re.findall(r'clickable="true"[^>]*bounds="\[(\d+),(\d+)\]\[(\d+),(\d+)\]"', xml)
|
||||||
bottom_buttons = [(int(x1), int(y1), int(x2), int(y2)) for x1, y1, x2, y2 in matches if int(y1) > bottom_threshold]
|
bottom_buttons = [(int(x1), int(y1), int(x2), int(y2)) for x1, y1, x2, y2 in matches if int(y1) > bottom_threshold]
|
||||||
if bottom_buttons:
|
if bottom_buttons:
|
||||||
# Rightmost = highest center-x (not left-edge x1, which misidentifies wide buttons)
|
|
||||||
send_btn = max(bottom_buttons, key=lambda b: (b[0] + b[2]) // 2)
|
send_btn = max(bottom_buttons, key=lambda b: (b[0] + b[2]) // 2)
|
||||||
cx = (send_btn[0] + send_btn[2]) // 2
|
cx = (send_btn[0] + send_btn[2]) // 2
|
||||||
cy = (send_btn[1] + send_btn[3]) // 2
|
cy = (send_btn[1] + send_btn[3]) // 2
|
||||||
adb("shell", "input", "tap", str(cx), str(cy))
|
adb("shell", "input", "tap", str(cx), str(cy))
|
||||||
return f"send button tapped ({cx}, {cy})"
|
return f"send button tapped ({cx}, {cy})"
|
||||||
# Fallback: tap bottom-right corner of the screen, offset slightly inward
|
|
||||||
screen_w, _ = get_screen_size()
|
screen_w, _ = get_screen_size()
|
||||||
fx = screen_w - 80
|
fx = screen_w - 80
|
||||||
fy = screen_h - 120
|
fy = screen_h - 120
|
||||||
@@ -383,9 +391,15 @@ def execute_action(action: dict) -> str:
|
|||||||
|
|
||||||
elif act == "wait":
|
elif act == "wait":
|
||||||
secs = float(action.get("seconds", 2))
|
secs = float(action.get("seconds", 2))
|
||||||
time.sleep(secs)
|
_sleep(secs)
|
||||||
return f"waited {secs}s"
|
return f"waited {secs}s"
|
||||||
|
|
||||||
|
elif act == "screenshot":
|
||||||
|
# Explicit screenshot action — agent wants to observe current state
|
||||||
|
label = action.get("label", "observe")
|
||||||
|
screenshot_b64(label)
|
||||||
|
return f"screenshot taken ({label})"
|
||||||
|
|
||||||
elif act == "done":
|
elif act == "done":
|
||||||
return "DONE"
|
return "DONE"
|
||||||
|
|
||||||
@@ -403,7 +417,7 @@ def execute_action(action: dict) -> str:
|
|||||||
SYSTEM_PROMPT = """\
|
SYSTEM_PROMPT = """\
|
||||||
You are an Android phone automation agent. You control the device by issuing actions.
|
You are an Android phone automation agent. You control the device by issuing actions.
|
||||||
|
|
||||||
On each turn you receive a screenshot of the current screen.
|
On each turn you receive a screenshot of the current Android screen.
|
||||||
Respond with a JSON object for ONE action to take next.
|
Respond with a JSON object for ONE action to take next.
|
||||||
|
|
||||||
Available actions:
|
Available actions:
|
||||||
@@ -411,19 +425,21 @@ Available actions:
|
|||||||
{"type": "type", "text": "<string>"}
|
{"type": "type", "text": "<string>"}
|
||||||
{"type": "key", "key": "enter|back|home|delete|tab"}
|
{"type": "key", "key": "enter|back|home|delete|tab"}
|
||||||
{"type": "swipe", "x1": <int>, "y1": <int>, "x2": <int>, "y2": <int>, "duration": <ms>}
|
{"type": "swipe", "x1": <int>, "y1": <int>, "x2": <int>, "y2": <int>, "duration": <ms>}
|
||||||
{"type": "send"} -- tap the send button (auto-locates via UI hierarchy)
|
{"type": "send"} -- tap the send/submit button (auto-locates via UI hierarchy)
|
||||||
{"type": "wait", "seconds": <float>}
|
{"type": "wait", "seconds": <float>}
|
||||||
|
{"type": "screenshot", "label": "<tag>"} -- observe current state without acting
|
||||||
{"type": "done", "summary": "<what was accomplished>"}
|
{"type": "done", "summary": "<what was accomplished>"}
|
||||||
{"type": "fail", "reason": "<why the goal cannot be achieved>"}
|
{"type": "fail", "reason": "<why the goal cannot be achieved>"}
|
||||||
|
|
||||||
Rules:
|
Rules:
|
||||||
- Issue exactly ONE action per turn as a JSON object. No markdown, no explanation outside JSON.
|
- Issue exactly ONE action per turn as a JSON object. No markdown, no explanation outside JSON.
|
||||||
- Coordinates are in pixels relative to the screenshot dimensions.
|
- Coordinates are in pixels relative to the screenshot dimensions.
|
||||||
- IMPORTANT: In this app, pressing "enter" inserts a newline, it does NOT send the message.
|
- IMPORTANT: In this app, pressing "enter" inserts a newline — it does NOT send the message.
|
||||||
To send a message, use the {"type": "send"} action which auto-locates and taps the send button.
|
To send a message use {"type": "send"} which auto-locates and taps the send/arrow button.
|
||||||
After typing, dismiss the keyboard by pressing "back", then use {"type": "send"}.
|
After typing your message, press "back" to dismiss the keyboard, then use {"type": "send"}.
|
||||||
- When the goal is fully achieved, respond with {"type": "done", ...}.
|
- Be efficient: skip unnecessary waits, tap directly on visible targets.
|
||||||
- If stuck after several attempts, respond with {"type": "fail", ...}.
|
- When the goal is fully achieved respond with {"type": "done", "summary": "..."}.
|
||||||
|
- If genuinely stuck after 5+ attempts on the same element respond with {"type": "fail", ...}.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
|
||||||
@@ -456,7 +472,6 @@ def make_client(model: str):
|
|||||||
azure_endpoint=os.environ["AZURE_OPENAI_ENDPOINT"],
|
azure_endpoint=os.environ["AZURE_OPENAI_ENDPOINT"],
|
||||||
api_version=os.environ.get("AZURE_OPENAI_API_VERSION", "2024-08-01-preview"),
|
api_version=os.environ.get("AZURE_OPENAI_API_VERSION", "2024-08-01-preview"),
|
||||||
), azure_model
|
), azure_model
|
||||||
# Azure AI Foundry endpoint (cognitiveservices.azure.com/openai/v1) — use OpenAI client
|
|
||||||
if os.environ.get("AZURE_DEV_AI_API_KEY"):
|
if os.environ.get("AZURE_DEV_AI_API_KEY"):
|
||||||
base_url = os.environ.get("AZURE_DEV_AI_BASE_URL", "https://vibe-dev-ai.cognitiveservices.azure.com/openai/v1")
|
base_url = os.environ.get("AZURE_DEV_AI_BASE_URL", "https://vibe-dev-ai.cognitiveservices.azure.com/openai/v1")
|
||||||
azure_model = os.environ.get("AZURE_DEV_AI_MODEL", "gpt-4o-2024-11-20")
|
azure_model = os.environ.get("AZURE_DEV_AI_MODEL", "gpt-4o-2024-11-20")
|
||||||
@@ -479,15 +494,9 @@ def make_client(model: str):
|
|||||||
|
|
||||||
@lru_cache(maxsize=1)
|
@lru_cache(maxsize=1)
|
||||||
def get_screen_size() -> tuple[int, int]:
|
def get_screen_size() -> tuple[int, int]:
|
||||||
"""Return (width, height) of the connected device screen.
|
"""Return (width, height) of the connected device screen. Cached."""
|
||||||
|
|
||||||
Cached after the first call — screen dimensions are stable for the
|
|
||||||
lifetime of a test run, and caching avoids a redundant ADB round-trip
|
|
||||||
on every `send` action.
|
|
||||||
"""
|
|
||||||
try:
|
try:
|
||||||
out = adb("shell", "wm", "size")
|
out = adb("shell", "wm", "size")
|
||||||
# "Physical size: 1080x1920" or "Override size: 1080x1920"
|
|
||||||
for line in out.splitlines():
|
for line in out.splitlines():
|
||||||
if "size:" in line.lower():
|
if "size:" in line.lower():
|
||||||
dims = line.split(":")[-1].strip()
|
dims = line.split(":")[-1].strip()
|
||||||
@@ -498,75 +507,272 @@ def get_screen_size() -> tuple[int, int]:
|
|||||||
return 1080, 1920
|
return 1080, 1920
|
||||||
|
|
||||||
|
|
||||||
def run_cua(goal: str, max_steps: int = 30, model: str = "gpt-4o",
|
def run_cua_step(goal: str, max_steps: int = 30, model: str = "gpt-4o",
|
||||||
include_ui_xml: bool = False, verbose: bool = True) -> dict:
|
include_ui_xml: bool = False, verbose: bool = True,
|
||||||
"""Run the CUA loop until done/fail/max_steps."""
|
step_label: str = "", action_delay: float = 0.8) -> dict:
|
||||||
|
"""Run the CUA loop for a single goal until done/fail/max_steps.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
goal: Natural-language instruction for this step.
|
||||||
|
max_steps: Hard cap on LLM turns.
|
||||||
|
model: Vision model deployment name.
|
||||||
|
include_ui_xml: Append UI hierarchy XML to each prompt turn.
|
||||||
|
verbose: Print action log.
|
||||||
|
step_label: Short name shown in logs/screenshot filenames.
|
||||||
|
action_delay: Seconds to pause after each action (scaled by speed_multiplier).
|
||||||
|
"""
|
||||||
client, model = make_client(model)
|
client, model = make_client(model)
|
||||||
history = []
|
history = []
|
||||||
screen_w, screen_h = get_screen_size()
|
screen_w, screen_h = get_screen_size()
|
||||||
|
label_prefix = f"[{step_label}] " if step_label else ""
|
||||||
|
|
||||||
for step in range(1, max_steps + 1):
|
for step in range(1, max_steps + 1):
|
||||||
# Capture screenshot
|
img_b64 = screenshot_b64(label=f"{step_label}_{step:02d}" if step_label else f"{step:03d}")
|
||||||
img_b64 = screenshot_b64()
|
|
||||||
|
|
||||||
# Build user message with screenshot
|
content: list = [
|
||||||
content = [
|
{
|
||||||
{"type": "text", "text": f"Step {step}. Screen is {screen_w}x{screen_h} pixels. Goal: {goal}\nWhat action should I take next?"},
|
"type": "text",
|
||||||
|
"text": (
|
||||||
|
f"{label_prefix}Step {step}/{max_steps}. "
|
||||||
|
f"Screen: {screen_w}x{screen_h}px. "
|
||||||
|
f"Goal: {goal}\n"
|
||||||
|
"What action should I take next?"
|
||||||
|
),
|
||||||
|
},
|
||||||
{"type": "image_url", "image_url": {"url": f"data:image/png;base64,{img_b64}", "detail": "high"}},
|
{"type": "image_url", "image_url": {"url": f"data:image/png;base64,{img_b64}", "detail": "high"}},
|
||||||
]
|
]
|
||||||
|
|
||||||
# Optionally include UI XML for better element identification
|
|
||||||
if include_ui_xml:
|
if include_ui_xml:
|
||||||
xml = ui_dump()
|
xml = ui_dump()
|
||||||
if xml:
|
if xml:
|
||||||
# Truncate to avoid token explosion
|
content.append({"type": "text", "text": f"UI hierarchy (truncated to 4000 chars):\n{xml[:4000]}"})
|
||||||
content.append({"type": "text", "text": f"UI hierarchy (truncated):\n{xml[:4000]}"})
|
|
||||||
|
|
||||||
history.append({"role": "user", "content": content})
|
history.append({"role": "user", "content": content})
|
||||||
|
|
||||||
# Call LLM
|
|
||||||
reply = call_llm(client, model, SYSTEM_PROMPT, history)
|
reply = call_llm(client, model, SYSTEM_PROMPT, history)
|
||||||
history.append({"role": "assistant", "content": reply})
|
history.append({"role": "assistant", "content": reply})
|
||||||
|
|
||||||
# Parse action
|
# Parse action — tolerate markdown fences and multi-object responses
|
||||||
try:
|
try:
|
||||||
clean = reply.strip()
|
clean = reply.strip()
|
||||||
# Strip markdown code fences
|
|
||||||
if clean.startswith("```"):
|
if clean.startswith("```"):
|
||||||
clean = clean.split("\n", 1)[1].rsplit("```", 1)[0].strip()
|
clean = clean.split("\n", 1)[1].rsplit("```", 1)[0].strip()
|
||||||
# If model returned multiple JSON objects, take the first
|
|
||||||
m = re.search(r'\{[^{}]*\}', clean)
|
m = re.search(r'\{[^{}]*\}', clean)
|
||||||
if m:
|
action = json.loads(m.group(0)) if m else json.loads(clean)
|
||||||
action = json.loads(m.group(0))
|
|
||||||
else:
|
|
||||||
action = json.loads(clean)
|
|
||||||
except json.JSONDecodeError:
|
except json.JSONDecodeError:
|
||||||
if verbose:
|
if verbose:
|
||||||
print(f" [step {step}] Failed to parse: {reply[:100]}")
|
print(f" {label_prefix}[step {step}] Failed to parse: {reply[:120]}")
|
||||||
continue
|
continue
|
||||||
|
|
||||||
# Execute
|
|
||||||
result = execute_action(action)
|
result = execute_action(action)
|
||||||
if verbose:
|
if verbose:
|
||||||
print(f" [step {step}] {action.get('type', '?')} -> {result}")
|
print(f" {label_prefix}[step {step}] {action.get('type', '?')} -> {result}")
|
||||||
|
|
||||||
if result == "DONE":
|
if result == "DONE":
|
||||||
return {"status": "success", "steps": step, "summary": action.get("summary", "")}
|
return {"status": "success", "steps": step, "summary": action.get("summary", "")}
|
||||||
if result.startswith("FAIL"):
|
if result.startswith("FAIL"):
|
||||||
return {"status": "fail", "steps": step, "reason": action.get("reason", "")}
|
return {"status": "fail", "steps": step, "reason": action.get("reason", "")}
|
||||||
|
|
||||||
# Brief pause between actions for UI to settle
|
# Trim history to keep context manageable
|
||||||
time.sleep(1.0)
|
if len(history) > 14:
|
||||||
|
history = history[-14:]
|
||||||
|
|
||||||
# Keep history manageable: only last 6 turns (12 messages) + system
|
_sleep(action_delay)
|
||||||
if len(history) > 12:
|
|
||||||
history = history[-12:]
|
|
||||||
|
|
||||||
return {"status": "timeout", "steps": max_steps}
|
return {"status": "timeout", "steps": max_steps}
|
||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Smoke test scenarios
|
# Onboarding showcase — structured multi-phase flow
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
# Banner printed before each named phase so the video is narrated by log output
|
||||||
|
PHASE_BANNERS = {
|
||||||
|
"connect": "STEP 1-2: Opening app — configuring server connection",
|
||||||
|
"session_list": "STEP 3: Connected — viewing session list",
|
||||||
|
"new_session": "STEP 4: Creating a new AI coding session",
|
||||||
|
"typescript": "STEP 5-6: Submitting TypeScript task — watching opencode work",
|
||||||
|
"verify": "STEP 7: Verifying task output / success response",
|
||||||
|
"settings": "STEP 8-9: Navigating to Settings — showing model selection",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _banner(key: str) -> None:
|
||||||
|
line = "=" * 64
|
||||||
|
msg = PHASE_BANNERS.get(key, key)
|
||||||
|
print(f"\n{line}")
|
||||||
|
print(f" {msg}")
|
||||||
|
print(f"{line}\n")
|
||||||
|
|
||||||
|
|
||||||
|
def run_onboarding_showcase(
|
||||||
|
opencode_url: str = DEFAULT_OPENCODE_URL,
|
||||||
|
model: str = "gpt-5.4",
|
||||||
|
include_ui_xml: bool = False,
|
||||||
|
verbose: bool = True,
|
||||||
|
max_steps_per_phase: int = 20,
|
||||||
|
) -> dict:
|
||||||
|
"""Execute the full first-run onboarding journey.
|
||||||
|
|
||||||
|
Each phase is a focused CUA sub-goal. Phases are run sequentially.
|
||||||
|
Returns a summary dict with per-phase results.
|
||||||
|
"""
|
||||||
|
|
||||||
|
results: dict[str, dict] = {}
|
||||||
|
|
||||||
|
def _run(key: str, goal: str, max_steps: int | None = None) -> bool:
|
||||||
|
"""Run one phase. Returns True if succeeded."""
|
||||||
|
_banner(key)
|
||||||
|
steps = max_steps or max_steps_per_phase
|
||||||
|
r = run_cua_step(
|
||||||
|
goal=goal,
|
||||||
|
max_steps=steps,
|
||||||
|
model=model,
|
||||||
|
include_ui_xml=include_ui_xml,
|
||||||
|
verbose=verbose,
|
||||||
|
step_label=key,
|
||||||
|
action_delay=0.7,
|
||||||
|
)
|
||||||
|
results[key] = r
|
||||||
|
ok = r["status"] == "success"
|
||||||
|
icon = "OK" if ok else "FAIL"
|
||||||
|
print(f"\n [{icon}] Phase '{key}': {r['status']} in {r['steps']} steps")
|
||||||
|
if r.get("summary"):
|
||||||
|
print(f" {r['summary']}")
|
||||||
|
if r.get("reason"):
|
||||||
|
print(f" reason: {r['reason']}")
|
||||||
|
return ok
|
||||||
|
|
||||||
|
# -----------------------------------------------------------------------
|
||||||
|
# Phase 1-2: Open app, configure server connection
|
||||||
|
# -----------------------------------------------------------------------
|
||||||
|
ok = _run(
|
||||||
|
"connect",
|
||||||
|
goal=(
|
||||||
|
f"You are on the OpenCode mobile app. "
|
||||||
|
"The screen shows either a connection screen (first launch) or an empty connections list. "
|
||||||
|
"Your goal: add a new connection to the opencode server. "
|
||||||
|
"Look for an 'Add Connection', '+', or 'New Connection' button and tap it. "
|
||||||
|
f"In the URL / Host field type '{opencode_url}'. "
|
||||||
|
"Leave username and password blank. "
|
||||||
|
"Tap 'Save', 'Connect', or 'Done' to save the connection. "
|
||||||
|
"Report done when you can see the connection has been saved or the app navigated away from the add-connection form."
|
||||||
|
),
|
||||||
|
max_steps=max_steps_per_phase,
|
||||||
|
)
|
||||||
|
if not ok:
|
||||||
|
return {"status": "fail", "phase": "connect", "results": results}
|
||||||
|
|
||||||
|
_sleep(2.0)
|
||||||
|
|
||||||
|
# -----------------------------------------------------------------------
|
||||||
|
# Phase 3: Connect to server — view session list
|
||||||
|
# -----------------------------------------------------------------------
|
||||||
|
ok = _run(
|
||||||
|
"session_list",
|
||||||
|
goal=(
|
||||||
|
"The connection has been saved. "
|
||||||
|
"Now tap on the saved connection entry to connect to the server. "
|
||||||
|
"Wait up to 10 seconds for the session list screen to appear. "
|
||||||
|
"The session list may be empty (no sessions yet) — that is fine. "
|
||||||
|
"Report done when you can see the session list screen (even if empty)."
|
||||||
|
),
|
||||||
|
max_steps=15,
|
||||||
|
)
|
||||||
|
if not ok:
|
||||||
|
return {"status": "fail", "phase": "session_list", "results": results}
|
||||||
|
|
||||||
|
_sleep(1.5)
|
||||||
|
|
||||||
|
# -----------------------------------------------------------------------
|
||||||
|
# Phase 4: Create new session
|
||||||
|
# -----------------------------------------------------------------------
|
||||||
|
ok = _run(
|
||||||
|
"new_session",
|
||||||
|
goal=(
|
||||||
|
"You are on the sessions list screen. "
|
||||||
|
"Tap the '+' button (usually top-right) to create a new AI coding session. "
|
||||||
|
"Wait up to 5 seconds for the new session / chat screen to open. "
|
||||||
|
"Report done once you see a text input field at the bottom of the screen "
|
||||||
|
"(the session chat/input view is open)."
|
||||||
|
),
|
||||||
|
max_steps=12,
|
||||||
|
)
|
||||||
|
if not ok:
|
||||||
|
return {"status": "fail", "phase": "new_session", "results": results}
|
||||||
|
|
||||||
|
_sleep(1.0)
|
||||||
|
|
||||||
|
# -----------------------------------------------------------------------
|
||||||
|
# Phase 5-6: Type TypeScript task and wait for opencode to complete
|
||||||
|
# -----------------------------------------------------------------------
|
||||||
|
ok = _run(
|
||||||
|
"typescript",
|
||||||
|
goal=(
|
||||||
|
f"You are inside a new OpenCode session (chat view with a text input at the bottom). "
|
||||||
|
f"Tap the text input field. "
|
||||||
|
f"Type this exact message: {TYPESCRIPT_TASK!r} "
|
||||||
|
"Press back to dismiss the keyboard. "
|
||||||
|
"Use the send action to submit. "
|
||||||
|
"After sending, wait and watch — opencode will show tool calls and file writes as it works. "
|
||||||
|
"Wait up to 90 seconds total for the session to go idle/complete "
|
||||||
|
"(no new activity for at least 5 seconds, or a completion indicator appears). "
|
||||||
|
"Re-check every 15 seconds by looking at the screen. "
|
||||||
|
"Report done when opencode appears to have finished (idle, no spinners, last message is a summary or file was created)."
|
||||||
|
),
|
||||||
|
max_steps=25,
|
||||||
|
)
|
||||||
|
if not ok:
|
||||||
|
return {"status": "fail", "phase": "typescript", "results": results}
|
||||||
|
|
||||||
|
_sleep(2.0)
|
||||||
|
|
||||||
|
# -----------------------------------------------------------------------
|
||||||
|
# Phase 7: Verify output / success
|
||||||
|
# -----------------------------------------------------------------------
|
||||||
|
ok = _run(
|
||||||
|
"verify",
|
||||||
|
goal=(
|
||||||
|
"The opencode session has finished. "
|
||||||
|
"Look at the chat to confirm the TypeScript hello world task succeeded. "
|
||||||
|
"You should see: a mention of 'hello.ts', 'Hello, World!', a file creation tool call, "
|
||||||
|
"or a success summary from the assistant. "
|
||||||
|
"Take a clear screenshot showing the result. "
|
||||||
|
"Report done with a brief summary of what you see as evidence of success. "
|
||||||
|
"Report fail only if the screen clearly shows an error with no recovery."
|
||||||
|
),
|
||||||
|
max_steps=8,
|
||||||
|
)
|
||||||
|
# Verify phase is informational — continue even on uncertain result
|
||||||
|
_sleep(1.5)
|
||||||
|
|
||||||
|
# -----------------------------------------------------------------------
|
||||||
|
# Phase 8-9: Navigate to Settings, show model selection
|
||||||
|
# -----------------------------------------------------------------------
|
||||||
|
_run(
|
||||||
|
"settings",
|
||||||
|
goal=(
|
||||||
|
"Navigate to the Settings screen of the OpenCode mobile app. "
|
||||||
|
"Look for a gear icon, 'Settings' tab in the bottom navigation bar, "
|
||||||
|
"or a hamburger menu that contains Settings. Tap it. "
|
||||||
|
"Once on the Settings screen, look for a 'Model' or 'AI Model' option and tap it "
|
||||||
|
"to show the model selection list. "
|
||||||
|
"Take a screenshot showing the model list or model setting. "
|
||||||
|
"You do NOT need to change the model — just show it is accessible. "
|
||||||
|
"Report done when the settings/model screen is visible in a screenshot."
|
||||||
|
),
|
||||||
|
max_steps=15,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Overall status: success if connect + session + typescript all succeeded
|
||||||
|
critical = ["connect", "session_list", "new_session", "typescript"]
|
||||||
|
failed_critical = [k for k in critical if results.get(k, {}).get("status") != "success"]
|
||||||
|
overall = "success" if not failed_critical else "partial"
|
||||||
|
return {"status": overall, "phase_results": results}
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Legacy smoke scenarios (kept for backwards compat / --scenario flag)
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
SMOKE_SCENARIOS = [
|
SMOKE_SCENARIOS = [
|
||||||
@@ -576,7 +782,7 @@ SMOKE_SCENARIOS = [
|
|||||||
"You see the OpenCode mobile app. Tap the '+' button (top-right) to create a new session. "
|
"You see the OpenCode mobile app. Tap the '+' button (top-right) to create a new session. "
|
||||||
"Tap the text input at the bottom. Type 'ping'. Press back to dismiss keyboard. "
|
"Tap the text input at the bottom. Type 'ping'. Press back to dismiss keyboard. "
|
||||||
"Use the send action. Wait 5 seconds, then take another screenshot. "
|
"Use the send action. Wait 5 seconds, then take another screenshot. "
|
||||||
"If you don't yet see an assistant reply, wait another 10 seconds and re-check (assistant replies can take 15+ seconds). "
|
"If you don't yet see an assistant reply, wait another 10 seconds and re-check. "
|
||||||
"If still no assistant bubble, wait another 15 seconds and re-check one more time. "
|
"If still no assistant bubble, wait another 15 seconds and re-check one more time. "
|
||||||
"Report success if you see both a 'You' bubble and an 'Assistant' bubble. "
|
"Report success if you see both a 'You' bubble and an 'Assistant' bubble. "
|
||||||
"Report failure only after at least 30 seconds of total waiting with no assistant bubble."
|
"Report failure only after at least 30 seconds of total waiting with no assistant bubble."
|
||||||
@@ -598,7 +804,7 @@ SMOKE_SCENARIOS = [
|
|||||||
"goal": (
|
"goal": (
|
||||||
"You see the OpenCode mobile app. Tap the '+' button (top-right) to create a new session. "
|
"You see the OpenCode mobile app. Tap the '+' button (top-right) to create a new session. "
|
||||||
"Wait 2 seconds for the session to be created. "
|
"Wait 2 seconds for the session to be created. "
|
||||||
"Navigate back to the sessions list by tapping the bottom-left 'Sessions' tab or pressing the back button. "
|
"Navigate back to the sessions list by tapping the 'Sessions' tab or pressing back. "
|
||||||
"Wait 3 seconds for the session list to load. "
|
"Wait 3 seconds for the session list to load. "
|
||||||
"Report success if you can see at least one session entry in the list. "
|
"Report success if you can see at least one session entry in the list. "
|
||||||
"Report failure if the sessions list appears empty or shows an error message."
|
"Report failure if the sessions list appears empty or shows an error message."
|
||||||
@@ -606,8 +812,7 @@ SMOKE_SCENARIOS = [
|
|||||||
},
|
},
|
||||||
]
|
]
|
||||||
|
|
||||||
# Extended scenarios requiring an external OpenCode server.
|
|
||||||
# Run with: python scripts/android-cua-smoke.py --opencode-url http://<host>:<port>
|
|
||||||
def _connect_and_verify_sessions_goal(url: str) -> str:
|
def _connect_and_verify_sessions_goal(url: str) -> str:
|
||||||
return (
|
return (
|
||||||
f"You see the OpenCode mobile app. "
|
f"You see the OpenCode mobile app. "
|
||||||
@@ -626,38 +831,83 @@ def _connect_and_verify_sessions_goal(url: str) -> str:
|
|||||||
)
|
)
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# CLI entry point
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
def main():
|
def main():
|
||||||
parser = argparse.ArgumentParser(description="Android CUA smoke test")
|
parser = argparse.ArgumentParser(
|
||||||
parser.add_argument("--goal", help="Custom goal (overrides built-in scenarios)")
|
description="OpenCode Mobile Android CUA smoke test — full onboarding showcase",
|
||||||
parser.add_argument("--model", default="gpt-4o", help="Vision model to use")
|
formatter_class=argparse.RawDescriptionHelpFormatter,
|
||||||
parser.add_argument("--max-steps", type=int, default=30)
|
epilog="""
|
||||||
parser.add_argument("--include-xml", action="store_true", help="Include UI XML in context")
|
Examples:
|
||||||
parser.add_argument("--quiet", action="store_true")
|
# Full onboarding showcase (default, recommended for demo video):
|
||||||
|
source ~/.env.d/azure-openai.env
|
||||||
|
python scripts/android-cua-smoke.py --model gpt-5.4 --include-xml
|
||||||
|
|
||||||
|
# Speed up for a faster demo (0.5 = half the wait times):
|
||||||
|
python scripts/android-cua-smoke.py --speed-multiplier 0.5
|
||||||
|
|
||||||
|
# Legacy single-goal mode:
|
||||||
|
python scripts/android-cua-smoke.py --goal "Open settings"
|
||||||
|
|
||||||
|
# Legacy named scenario:
|
||||||
|
python scripts/android-cua-smoke.py --scenarios send_message,verify_session_list
|
||||||
|
""",
|
||||||
|
)
|
||||||
|
|
||||||
|
# Showcase mode (new default)
|
||||||
|
parser.add_argument(
|
||||||
|
"--showcase",
|
||||||
|
action="store_true",
|
||||||
|
default=True,
|
||||||
|
help="Run the full onboarding showcase (default). Demonstrates connect → session → TypeScript task → settings.",
|
||||||
|
)
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"--opencode-url",
|
"--opencode-url",
|
||||||
help="OpenCode server URL (e.g. http://100.108.64.76:4096). "
|
default=None,
|
||||||
"Used by the default connect-and-verify regression scenario.",
|
help=f"OpenCode server URL (default: {DEFAULT_OPENCODE_URL}).",
|
||||||
)
|
)
|
||||||
|
|
||||||
|
# Speed control
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"--skip-connect-scenario",
|
"--speed-multiplier",
|
||||||
action="store_true",
|
type=float,
|
||||||
help="Skip the default connect-and-verify regression scenario.",
|
default=1.0,
|
||||||
)
|
metavar="FACTOR",
|
||||||
parser.add_argument(
|
help="Scale all wait/sleep durations. 0.5 = twice as fast, 2.0 = twice as slow. Default: 1.0",
|
||||||
"--only-connect-scenario",
|
|
||||||
action="store_true",
|
|
||||||
help="Run ONLY the connect-and-verify-sessions scenario. Use in CI with a "
|
|
||||||
"local opencode server for a deterministic true-E2E (no model backend needed).",
|
|
||||||
)
|
)
|
||||||
|
|
||||||
|
# Model / verbosity
|
||||||
|
parser.add_argument("--model", default="gpt-4o", help="Vision model deployment name.")
|
||||||
|
parser.add_argument("--max-steps", type=int, default=20, help="Max LLM steps per phase (showcase) or total (legacy).")
|
||||||
|
parser.add_argument("--include-xml", action="store_true", help="Include UI hierarchy XML in LLM context (more accurate, more tokens).")
|
||||||
|
parser.add_argument("--quiet", action="store_true")
|
||||||
|
|
||||||
|
# Legacy / compat flags
|
||||||
|
parser.add_argument("--goal", help="Legacy: single custom goal (disables showcase).")
|
||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"--scenarios",
|
"--scenarios",
|
||||||
help="Comma-separated explicit scenario set to run, e.g. "
|
help="Legacy: comma-separated scenario names to run (disables showcase). "
|
||||||
"'connect_and_verify_sessions,send_message,verify_session_list'. "
|
"Valid: connect_and_verify_sessions, send_message, multi_turn, verify_session_list.",
|
||||||
"Valid names: connect_and_verify_sessions, send_message, multi_turn, "
|
|
||||||
"verify_session_list. Overrides --only-connect-scenario and the default set.",
|
|
||||||
)
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--skip-connect-scenario", action="store_true",
|
||||||
|
help="Legacy: skip the connect-and-verify regression scenario.",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--only-connect-scenario", action="store_true",
|
||||||
|
help="Legacy: run ONLY the connect-and-verify-sessions scenario.",
|
||||||
|
)
|
||||||
|
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
# Apply speed multiplier globally
|
||||||
|
global _speed_multiplier
|
||||||
|
_speed_multiplier = args.speed_multiplier
|
||||||
|
if args.speed_multiplier != 1.0:
|
||||||
|
print(f"[speed] multiplier={args.speed_multiplier} — all waits scaled accordingly")
|
||||||
|
|
||||||
# Verify ADB
|
# Verify ADB
|
||||||
try:
|
try:
|
||||||
devices = adb("devices")
|
devices = adb("devices")
|
||||||
@@ -666,29 +916,81 @@ def main():
|
|||||||
except FileNotFoundError:
|
except FileNotFoundError:
|
||||||
sys.exit("adb not found in PATH")
|
sys.exit("adb not found in PATH")
|
||||||
|
|
||||||
connect_url = args.opencode_url or os.environ.get("OPENCODE_URL") or "http://100.108.64.76:4096"
|
connect_url = args.opencode_url or os.environ.get("OPENCODE_URL") or DEFAULT_OPENCODE_URL
|
||||||
|
|
||||||
|
# -----------------------------------------------------------------------
|
||||||
|
# Determine run mode: showcase vs. legacy scenarios
|
||||||
|
# -----------------------------------------------------------------------
|
||||||
|
use_legacy = bool(args.goal or args.scenarios or args.only_connect_scenario)
|
||||||
|
|
||||||
|
if not use_legacy:
|
||||||
|
# ----------------------------------------------------------------
|
||||||
|
# NEW DEFAULT: Full onboarding showcase
|
||||||
|
# ----------------------------------------------------------------
|
||||||
|
print("\n" + "=" * 64)
|
||||||
|
print(" OpenCode Mobile — Full Onboarding Showcase")
|
||||||
|
print(f" Server: {connect_url}")
|
||||||
|
print(f" Model: {args.model}")
|
||||||
|
print(f" Speed: {_speed_multiplier}x")
|
||||||
|
print("=" * 64)
|
||||||
|
|
||||||
|
rec_thread, _stop_ev, remote_path = start_screen_recording("onboarding_showcase")
|
||||||
|
local_video = "/tmp/cua_onboarding_showcase.mp4"
|
||||||
|
|
||||||
|
try:
|
||||||
|
if not ensure_app_foreground(verbose=not args.quiet):
|
||||||
|
print("[prep] warning: could not confirm app in foreground")
|
||||||
|
maybe_dismiss_telemetry_consent(verbose=not args.quiet)
|
||||||
|
ensure_app_foreground(verbose=not args.quiet)
|
||||||
|
|
||||||
|
result = run_onboarding_showcase(
|
||||||
|
opencode_url=connect_url,
|
||||||
|
model=args.model,
|
||||||
|
include_ui_xml=args.include_xml,
|
||||||
|
verbose=not args.quiet,
|
||||||
|
max_steps_per_phase=args.max_steps,
|
||||||
|
)
|
||||||
|
finally:
|
||||||
|
stop_screen_recording(rec_thread, remote_path, local_video)
|
||||||
|
upload_to_archivebox(local_video, "onboarding_showcase")
|
||||||
|
|
||||||
|
print("\n" + "=" * 64)
|
||||||
|
print(f" Showcase result: {result['status'].upper()}")
|
||||||
|
if local_video and Path(local_video).exists():
|
||||||
|
print(f" Video: {local_video}")
|
||||||
|
print("=" * 64)
|
||||||
|
|
||||||
|
# Print per-phase summary table
|
||||||
|
phase_results = result.get("phase_results", {})
|
||||||
|
if phase_results:
|
||||||
|
print("\n Phase breakdown:")
|
||||||
|
for phase, pr in phase_results.items():
|
||||||
|
icon = "PASS" if pr["status"] == "success" else "FAIL"
|
||||||
|
print(f" [{icon}] {phase:20s} {pr['status']:8s} {pr['steps']} steps")
|
||||||
|
|
||||||
|
sys.exit(0 if result["status"] == "success" else 1)
|
||||||
|
|
||||||
|
# -----------------------------------------------------------------------
|
||||||
|
# LEGACY MODE: named/custom scenarios
|
||||||
|
# -----------------------------------------------------------------------
|
||||||
connect_scenario = {
|
connect_scenario = {
|
||||||
"name": "connect_and_verify_sessions",
|
"name": "connect_and_verify_sessions",
|
||||||
"goal": _connect_and_verify_sessions_goal(connect_url),
|
"goal": _connect_and_verify_sessions_goal(connect_url),
|
||||||
}
|
}
|
||||||
|
|
||||||
if args.scenarios:
|
if args.scenarios:
|
||||||
# Explicit named set (CI widened gate). Look up by name across the full catalog.
|
|
||||||
catalog = {connect_scenario["name"]: connect_scenario}
|
catalog = {connect_scenario["name"]: connect_scenario}
|
||||||
for s in SMOKE_SCENARIOS:
|
for s in SMOKE_SCENARIOS:
|
||||||
catalog[s["name"]] = s
|
catalog[s["name"]] = s
|
||||||
requested = [n.strip() for n in args.scenarios.split(",") if n.strip()]
|
requested = [n.strip() for n in args.scenarios.split(",") if n.strip()]
|
||||||
unknown = [n for n in requested if n not in catalog]
|
unknown = [n for n in requested if n not in catalog]
|
||||||
if unknown:
|
if unknown:
|
||||||
sys.exit(f"Unknown scenario(s): {', '.join(unknown)}. "
|
sys.exit(f"Unknown scenario(s): {', '.join(unknown)}. Valid: {', '.join(catalog.keys())}")
|
||||||
f"Valid: {', '.join(catalog.keys())}")
|
|
||||||
scenarios = [catalog[n] for n in requested]
|
scenarios = [catalog[n] for n in requested]
|
||||||
elif args.only_connect_scenario:
|
elif args.only_connect_scenario:
|
||||||
# CI true-E2E: just connect to the local opencode server and verify the list.
|
|
||||||
scenarios = [connect_scenario]
|
scenarios = [connect_scenario]
|
||||||
else:
|
else:
|
||||||
scenarios = [{"name": "custom", "goal": args.goal}] if args.goal else list(SMOKE_SCENARIOS)
|
scenarios = [{"name": "custom", "goal": args.goal}] if args.goal else list(SMOKE_SCENARIOS)
|
||||||
# Keep connect-and-verify in the default smoke path so regressions are exercised.
|
|
||||||
if not args.goal and not args.skip_connect_scenario:
|
if not args.goal and not args.skip_connect_scenario:
|
||||||
scenarios.append(connect_scenario)
|
scenarios.append(connect_scenario)
|
||||||
|
|
||||||
@@ -700,7 +1002,6 @@ def main():
|
|||||||
print(f"Goal: {scenario['goal'][:80]}...")
|
print(f"Goal: {scenario['goal'][:80]}...")
|
||||||
print(f"{'='*60}")
|
print(f"{'='*60}")
|
||||||
|
|
||||||
# Start screen recording
|
|
||||||
rec_thread, _stop_ev, remote_path = start_screen_recording(scenario["name"])
|
rec_thread, _stop_ev, remote_path = start_screen_recording(scenario["name"])
|
||||||
local_video = f"/tmp/cua_{scenario['name']}.mp4"
|
local_video = f"/tmp/cua_{scenario['name']}.mp4"
|
||||||
|
|
||||||
@@ -710,15 +1011,15 @@ def main():
|
|||||||
maybe_dismiss_telemetry_consent(verbose=not args.quiet)
|
maybe_dismiss_telemetry_consent(verbose=not args.quiet)
|
||||||
ensure_app_foreground(verbose=not args.quiet)
|
ensure_app_foreground(verbose=not args.quiet)
|
||||||
|
|
||||||
result = run_cua(
|
result = run_cua_step(
|
||||||
goal=scenario["goal"],
|
goal=scenario["goal"],
|
||||||
max_steps=args.max_steps,
|
max_steps=args.max_steps,
|
||||||
model=args.model,
|
model=args.model,
|
||||||
include_ui_xml=args.include_xml,
|
include_ui_xml=args.include_xml,
|
||||||
verbose=not args.quiet,
|
verbose=not args.quiet,
|
||||||
|
step_label=scenario["name"],
|
||||||
)
|
)
|
||||||
finally:
|
finally:
|
||||||
# Always stop and pull the recording
|
|
||||||
stop_screen_recording(rec_thread, remote_path, local_video)
|
stop_screen_recording(rec_thread, remote_path, local_video)
|
||||||
upload_to_archivebox(local_video, scenario["name"])
|
upload_to_archivebox(local_video, scenario["name"])
|
||||||
|
|
||||||
@@ -735,7 +1036,6 @@ def main():
|
|||||||
if result.get("reason"):
|
if result.get("reason"):
|
||||||
print(f"Reason: {result['reason']}")
|
print(f"Reason: {result['reason']}")
|
||||||
|
|
||||||
# Exit code: 0 if all passed
|
|
||||||
failed = [r for r in results if r["status"] != "success"]
|
failed = [r for r in results if r["status"] != "success"]
|
||||||
if failed:
|
if failed:
|
||||||
print(f"\n{'!'*60}")
|
print(f"\n{'!'*60}")
|
||||||
|
|||||||
Reference in New Issue
Block a user