From bca521fce2cdc04da8aaeec536a2c7fc6ab342c2 Mon Sep 17 00:00:00 2001 From: johnmalek312 Date: Sat, 18 Oct 2025 02:53:57 +1100 Subject: [PATCH] doc update --- docs/.generated-files.txt | 4 - docs/docs.json | 35 +- docs/v1/concepts/portal-app.mdx | 2 +- docs/v1/quickstart.mdx | 4 +- docs/v2/concepts/portal-app.mdx | 2 +- docs/v3/concepts/android-tools.mdx | 2 +- docs/v3/sdk/adb-tools.mdx | 425 ----- docs/v3/sdk/base-tools.mdx | 191 -- docs/v3/sdk/droid-agent.mdx | 71 - docs/v3/sdk/ios-tools.mdx | 279 --- docs/v4/concepts/agent-architecture.mdx | 241 +++ docs/v4/concepts/architecture.mdx | 1793 ------------------- docs/v4/concepts/event-streaming.mdx | 975 ---------- docs/v4/concepts/events-and-workflows.mdx | 165 ++ docs/v4/concepts/overview.mdx | 107 ++ docs/v4/concepts/prompts.mdx | 332 ++++ docs/v4/concepts/scripter-agent.mdx | 165 ++ docs/v4/concepts/shared-state.mdx | 60 + docs/v4/concepts/workflow-architecture.mdx | 1160 ------------ docs/v4/guides/app-cards.mdx | 24 +- docs/v4/guides/cli.mdx | 1201 +++---------- docs/v4/guides/configuration.mdx | 1595 ----------------- docs/v4/guides/custom-tools-credentials.mdx | 1291 +++++-------- docs/v4/guides/custom-variables.mdx | 88 +- docs/v4/guides/device-setup.mdx | 1209 ++++--------- docs/v4/guides/overview.mdx | 9 +- docs/v4/guides/structured-output.mdx | 1155 +----------- docs/v4/guides/telemetry-tracing.mdx | 53 +- docs/v4/overview.mdx | 174 +- docs/v4/quickstart.mdx | 65 +- docs/v4/sdk.mdx | 49 + docs/v4/sdk/adb-tools.mdx | 14 +- docs/v4/sdk/base-tools.mdx | 12 +- docs/v4/sdk/configuration.mdx | 693 +++++++ docs/v4/sdk/droid-agent.mdx | 65 +- docs/v4/sdk/ios-tools.mdx | 10 +- gen-docs-sdk-ref.sh | 2 +- pyproject.toml | 2 +- 38 files changed, 3145 insertions(+), 10579 deletions(-) delete mode 100644 docs/.generated-files.txt delete mode 100644 docs/v3/sdk/adb-tools.mdx delete mode 100644 docs/v3/sdk/base-tools.mdx delete mode 100644 docs/v3/sdk/droid-agent.mdx delete mode 100644 docs/v3/sdk/ios-tools.mdx create mode 100644 docs/v4/concepts/agent-architecture.mdx delete mode 100644 docs/v4/concepts/architecture.mdx delete mode 100644 docs/v4/concepts/event-streaming.mdx create mode 100644 docs/v4/concepts/events-and-workflows.mdx create mode 100644 docs/v4/concepts/overview.mdx create mode 100644 docs/v4/concepts/prompts.mdx create mode 100644 docs/v4/concepts/scripter-agent.mdx create mode 100644 docs/v4/concepts/shared-state.mdx delete mode 100644 docs/v4/concepts/workflow-architecture.mdx delete mode 100644 docs/v4/guides/configuration.mdx create mode 100644 docs/v4/sdk.mdx create mode 100644 docs/v4/sdk/configuration.mdx diff --git a/docs/.generated-files.txt b/docs/.generated-files.txt deleted file mode 100644 index 6a6d032..0000000 --- a/docs/.generated-files.txt +++ /dev/null @@ -1,4 +0,0 @@ -md5 0b688f460703d59bd84fe71387e626d5 v3/sdk/droid-agent.mdx -md5 47f362d52ba26155d647efec628294cf v3/sdk/base-tools.mdx -md5 2e83b80e94101d983ed52ac5ae91314f v3/sdk/adb-tools.mdx -md5 7d779482901cc5eb62288ac9c8d85199 v3/sdk/ios-tools.mdx diff --git a/docs/docs.json b/docs/docs.json index ac451f6..d527e24 100644 --- a/docs/docs.json +++ b/docs/docs.json @@ -23,21 +23,12 @@ "v4/quickstart" ] }, - { - "group": "Concepts", - "pages": [ - "v4/concepts/architecture", - "v4/concepts/workflow-architecture", - "v4/concepts/event-streaming" - ] - }, { "group": "Guides", "pages": [ "v4/guides/overview", - "v4/guides/cli", - "v4/guides/configuration", "v4/guides/device-setup", + "v4/guides/cli", "v4/guides/custom-tools-credentials", "v4/guides/custom-variables", "v4/guides/app-cards", @@ -45,13 +36,26 @@ "v4/guides/telemetry-tracing" ] }, + { + "group": "Concepts", + "pages": [ + "v4/concepts/overview", + "v4/concepts/agent-architecture", + "v4/concepts/scripter-agent", + "v4/concepts/shared-state", + "v4/concepts/events-and-workflows", + "v4/concepts/prompts" + ] + }, { "group": "SDK Reference", "pages": [ + "v4/sdk", "v4/sdk/droid-agent", "v4/sdk/adb-tools", "v4/sdk/ios-tools", - "v4/sdk/base-tools" + "v4/sdk/base-tools", + "v4/sdk/configuration" ] } ] @@ -85,15 +89,6 @@ "v3/concepts/android-tools", "v3/concepts/portal-app" ] - }, - { - "group": "SDK Reference", - "pages": [ - "v3/sdk/droid-agent", - "v3/sdk/adb-tools", - "v3/sdk/ios-tools", - "v3/sdk/base-tools" - ] } ] }, diff --git a/docs/v1/concepts/portal-app.mdx b/docs/v1/concepts/portal-app.mdx index 3c4167f..5cd5415 100644 --- a/docs/v1/concepts/portal-app.mdx +++ b/docs/v1/concepts/portal-app.mdx @@ -48,7 +48,7 @@ The DroidRun Portal App: ## 🚀 Installation -The DroidRun Portal App is available from the [DroidRun Portal repository](https://github.com/droidrun/droidrun-portal). For installation instructions, see the [Quickstart](/quickstart) guide. +The DroidRun Portal App is available from the [DroidRun Portal repository](https://github.com/droidrun/droidrun-portal). For installation instructions, see the [Quickstart](/v1/quickstart) guide. ## 🔧 Troubleshooting diff --git a/docs/v1/quickstart.mdx b/docs/v1/quickstart.mdx index d4a5580..bb297f7 100644 --- a/docs/v1/quickstart.mdx +++ b/docs/v1/quickstart.mdx @@ -289,5 +289,5 @@ pip show droidrun Now that you've got DroidRun running, you can: -- Learn about the [ReAct agent system](/concepts/agent) -- Discover all [Android interactions](/concepts/android-control) \ No newline at end of file +- Learn about the [ReAct agent system](/v1/concepts/agent) +- Discover all [Android interactions](/v1/concepts/android-control) \ No newline at end of file diff --git a/docs/v2/concepts/portal-app.mdx b/docs/v2/concepts/portal-app.mdx index 3c4167f..b1c9c1b 100644 --- a/docs/v2/concepts/portal-app.mdx +++ b/docs/v2/concepts/portal-app.mdx @@ -48,7 +48,7 @@ The DroidRun Portal App: ## 🚀 Installation -The DroidRun Portal App is available from the [DroidRun Portal repository](https://github.com/droidrun/droidrun-portal). For installation instructions, see the [Quickstart](/quickstart) guide. +The DroidRun Portal App is available from the [DroidRun Portal repository](https://github.com/droidrun/droidrun-portal). For installation instructions, see the [Quickstart](/v2/quickstart) guide. ## 🔧 Troubleshooting diff --git a/docs/v3/concepts/android-tools.mdx b/docs/v3/concepts/android-tools.mdx index 8a39d8b..e53f738 100644 --- a/docs/v3/concepts/android-tools.mdx +++ b/docs/v3/concepts/android-tools.mdx @@ -100,4 +100,4 @@ format, image_data = await tools.take_screenshot() | `complete(success, reason)` | Finish task | None | ## Dive Deeper -You can find the SDK Reference for AdbTools [here](../sdk/adb-tools) \ No newline at end of file +SDK reference documentation is available in the v4 documentation. \ No newline at end of file diff --git a/docs/v3/sdk/adb-tools.mdx b/docs/v3/sdk/adb-tools.mdx deleted file mode 100644 index fa3ecd0..0000000 --- a/docs/v3/sdk/adb-tools.mdx +++ /dev/null @@ -1,425 +0,0 @@ ---- -title: AdbTools ---- - -UI Actions - Core UI interaction tools for Android device control. - - - -## AdbTools - -```python -class AdbTools(Tools) -``` - -Core UI interaction tools for Android device control. - - - -#### AdbTools.\_\_init\_\_ - -```python -def __init__( - serial: str | None = None, - use_tcp: bool = False, - tcp_port: int = 8080 -) -> None -``` - -Initialize the AdbTools instance. - -**Arguments**: - -- `serial` - Device serial number -- `use_tcp` - Whether to use TCP communication (default: False) -- `tcp_port` - TCP port for communication (default: 8080) - - - -#### AdbTools.setup\_tcp\_forward - -```python -def setup_tcp_forward() -> bool -``` - -Set up ADB TCP port forwarding for communication with the portal app. - -**Returns**: - -- `bool` - True if forwarding was set up successfully, False otherwise - - - -#### AdbTools.teardown\_tcp\_forward - -```python -def teardown_tcp_forward() -> bool -``` - -Remove ADB TCP port forwarding. - -**Returns**: - -- `bool` - True if forwarding was removed successfully, False otherwise - - - -#### AdbTools.\_\_del\_\_ - -```python -def __del__() -``` - -Cleanup when the object is destroyed. - - - -#### AdbTools.tap\_by\_index - -```python -def tap_by_index(index: int) -> str -``` - -Tap on a UI element by its index. - -This function uses the cached clickable elements -to find the element with the given index and tap on its center coordinates. - -**Arguments**: - -- `index` - Index of the element to tap - - -**Returns**: - - Result message - - - -#### AdbTools.tap\_by\_coordinates - -```python -def tap_by_coordinates(x: int, y: int) -> bool -``` - -Tap on the device screen at specific coordinates. - -**Arguments**: - -- `x` - X coordinate -- `y` - Y coordinate - - -**Returns**: - - Bool indicating success or failure - - - -#### AdbTools.tap - -```python -def tap(index: int) -> str -``` - -Tap on a UI element by its index. - -This function uses the cached clickable elements from the last get_clickables call -to find the element with the given index and tap on its center coordinates. - -**Arguments**: - -- `index` - Index of the element to tap - - -**Returns**: - - Result message - - - -#### AdbTools.swipe - -```python -def swipe( - start_x: int, - start_y: int, - end_x: int, - end_y: int, - duration_ms: float = 300 -) -> bool -``` - -Performs a straight-line swipe gesture on the device screen. -To perform a hold (long press), set the start and end coordinates to the same values and increase the duration as needed. - -**Arguments**: - -- `start_x` - Starting X coordinate -- `start_y` - Starting Y coordinate -- `end_x` - Ending X coordinate -- `end_y` - Ending Y coordinate -- `duration` - Duration of swipe in seconds - -**Returns**: - - Bool indicating success or failure - - - -#### AdbTools.drag - -```python -def drag( - start_x: int, - start_y: int, - end_x: int, - end_y: int, - duration: float = 3 -) -> bool -``` - -Performs a straight-line drag and drop gesture on the device screen. - -**Arguments**: - -- `start_x` - Starting X coordinate -- `start_y` - Starting Y coordinate -- `end_x` - Ending X coordinate -- `end_y` - Ending Y coordinate -- `duration` - Duration of swipe in seconds - -**Returns**: - - Bool indicating success or failure - - - -#### AdbTools.input\_text - -```python -def input_text(text: str) -> str -``` - -Input text on the device. -Always make sure that the Focused Element is not None before inputting text. - -**Arguments**: - -- `text` - Text to input. Can contain spaces, newlines, and special characters including non-ASCII. - - -**Returns**: - - Result message - - - -#### AdbTools.back - -```python -def back() -> str -``` - -Go back on the current view. -This presses the Android back button. - - - -#### AdbTools.press\_key - -```python -def press_key(keycode: int) -> str -``` - -Press a key on the Android device. - -Common keycodes: -- 3: HOME -- 4: BACK -- 66: ENTER -- 67: DELETE - -**Arguments**: - -- `keycode` - Android keycode to press - - - -#### AdbTools.start\_app - -```python -def start_app(package: str, activity: str | None = None) -> str -``` - -Start an app on the device. - -**Arguments**: - -- `package` - Package name (e.g., "com.android.settings") -- `activity` - Optional activity name - - - -#### AdbTools.install\_app - -```python -def install_app( - apk_path: str, - reinstall: bool = False, - grant_permissions: bool = True -) -> str -``` - -Install an app on the device. - -**Arguments**: - -- `apk_path` - Path to the APK file -- `reinstall` - Whether to reinstall if app exists -- `grant_permissions` - Whether to grant all permissions - - - -#### AdbTools.take\_screenshot - -```python -def take_screenshot() -> Tuple[str, bytes] -``` - -Take a screenshot of the device. -This function captures the current screen and adds the screenshot to context in the next message. -Also stores the screenshot in the screenshots list with timestamp for later GIF creation. - - - -#### AdbTools.list\_packages - -```python -def list_packages(include_system_apps: bool = False) -> List[str] -``` - -List installed packages on the device. - -**Arguments**: - -- `include_system_apps` - Whether to include system apps (default: False) - - -**Returns**: - - List of package names - - - -#### AdbTools.complete - -```python -def complete(success: bool, reason: str = "") -``` - -Mark the task as finished. - -**Arguments**: - -- `success` - Indicates if the task was successful. -- `reason` - Reason for failure/success - - - -#### AdbTools.remember - -```python -def remember(information: str) -> str -``` - -Store important information to remember for future context. - -This information will be extracted and included into your next steps to maintain context -across interactions. Use this for critical facts, observations, or user preferences -that should influence future decisions. - -**Arguments**: - -- `information` - The information to remember - - -**Returns**: - - Confirmation message - - - -#### AdbTools.get\_memory - -```python -def get_memory() -> List[str] -``` - -Retrieve all stored memory items. - -**Returns**: - - List of stored memory items - - - -#### AdbTools.get\_state - -```python -def get_state(serial: Optional[str] = None) -> Dict[str, Any] -``` - -Get both the a11y tree and phone state in a single call using the combined /state endpoint. - -**Arguments**: - -- `serial` - Optional device serial number - - -**Returns**: - - Dictionary containing both 'a11y_tree' and 'phone_state' data - - - -#### AdbTools.get\_a11y\_tree - -```python -def get_a11y_tree() -> Dict[str, Any] -``` - -Get just the accessibility tree using the /a11y_tree endpoint. - -**Returns**: - - Dictionary containing accessibility tree data - - - -#### AdbTools.get\_phone\_state - -```python -def get_phone_state() -> Dict[str, Any] -``` - -Get just the phone state using the /phone_state endpoint. - -**Returns**: - - Dictionary containing phone state data - - - -#### AdbTools.ping - -```python -def ping() -> Dict[str, Any] -``` - -Test the TCP connection using the /ping endpoint. - -**Returns**: - - Dictionary with ping result - diff --git a/docs/v3/sdk/base-tools.mdx b/docs/v3/sdk/base-tools.mdx deleted file mode 100644 index 331d875..0000000 --- a/docs/v3/sdk/base-tools.mdx +++ /dev/null @@ -1,191 +0,0 @@ ---- -title: Tools ---- - - - -## Tools - -```python -class Tools(ABC) -``` - -Abstract base class for all tools. -This class provides a common interface for all tools to implement. - - - -#### Tools.ui\_action - -```python -def ui_action(func) -``` - -" -Decorator to capture screenshots and UI states for actions that modify the UI. - - - -#### Tools.get\_state - -```python -def get_state() -> Dict[str, Any] -``` - -Get the current state of the tool. - - - -#### Tools.tap\_by\_index - -```python -def tap_by_index(index: int) -> str -``` - -Tap the element at the given index. - - - -#### Tools.swipe - -```python -def swipe( - start_x: int, - start_y: int, - end_x: int, - end_y: int, - duration_ms: int = 300 -) -> bool -``` - -Swipe from the given start coordinates to the given end coordinates. - - - -#### Tools.drag - -```python -def drag( - start_x: int, - start_y: int, - end_x: int, - end_y: int, - duration_ms: int = 3000 -) -> bool -``` - -Drag from the given start coordinates to the given end coordinates. - - - -#### Tools.input\_text - -```python -def input_text(text: str) -> str -``` - -Input the given text into a focused input field. - - - -#### Tools.back - -```python -def back() -> str -``` - -Press the back button. - - - -#### Tools.press\_key - -```python -def press_key(keycode: int) -> str -``` - -Enter the given keycode. - - - -#### Tools.start\_app - -```python -def start_app(package: str, activity: str = "") -> str -``` - -Start the given app. - - - -#### Tools.take\_screenshot - -```python -def take_screenshot() -> Tuple[str, bytes] -``` - -Take a screenshot of the device. - - - -#### Tools.list\_packages - -```python -def list_packages(include_system_apps: bool = False) -> List[str] -``` - -List all packages on the device. - - - -#### Tools.remember - -```python -def remember(information: str) -> str -``` - -Remember the given information. This is used to store information in the tool's memory. - - - -#### Tools.get\_memory - -```python -def get_memory() -> List[str] -``` - -Get the memory of the tool. - - - -#### Tools.complete - -```python -def complete(success: bool, reason: str = "") -> None -``` - -Complete the tool. This is used to indicate that the tool has completed its task. - - - -#### describe\_tools - -```python -def describe_tools( - tools: Tools, - exclude_tools: Optional[List[str]] = None -) -> Dict[str, Callable[..., Any]] -``` - -Describe the tools available for the given Tools instance. - -**Arguments**: - -- `tools` - The Tools instance to describe. -- `exclude_tools` - List of tool names to exclude from the description. - - -**Returns**: - - A dictionary mapping tool names to their descriptions. - diff --git a/docs/v3/sdk/droid-agent.mdx b/docs/v3/sdk/droid-agent.mdx deleted file mode 100644 index c3f6ca6..0000000 --- a/docs/v3/sdk/droid-agent.mdx +++ /dev/null @@ -1,71 +0,0 @@ ---- -title: DroidAgent ---- - -DroidAgent - A wrapper class that coordinates the planning and execution of tasks -to achieve a user's goal on an Android device. - - - -## DroidAgent - -```python -class DroidAgent(Workflow) -``` - -A wrapper class that coordinates between PlannerAgent (creates plans) and - CodeActAgent (executes tasks) to achieve a user's goal. - - - -#### DroidAgent.\_\_init\_\_ - -```python -def __init__( - goal: str, - llm: LLM, - tools: Tools, - personas: List[AgentPersona] = [DEFAULT], - max_steps: int = 15, - timeout: int = 1000, - vision: bool = False, - reasoning: bool = False, - reflection: bool = False, - enable_tracing: bool = False, - debug: bool = False, - save_trajectories: str = "none", - excluded_tools: List[str] = None, - *args, - **kwargs -) -``` - -Initialize the DroidAgent wrapper. - -**Arguments**: - -- `goal` - The user's goal or command to execute -- `llm` - The language model to use for both agents -- `max_steps` - Maximum number of steps for both agents -- `timeout` - Timeout for agent execution in seconds -- `reasoning` - Whether to use the PlannerAgent for complex reasoning (True) - or send tasks directly to CodeActAgent (False) -- `reflection` - Whether to reflect on steps the CodeActAgent did to give the PlannerAgent advice -- `enable_tracing` - Whether to enable Arize Phoenix tracing -- `debug` - Whether to enable verbose debug logging -- `save_trajectories` - Trajectory saving level. Can be: - - "none" (no saving) - - "step" (save per step) - - "action" (save per action) -- `**kwargs` - Additional keyword arguments to pass to the agents - - - -#### DroidAgent.run - -```python -def run(*args, **kwargs) -> WorkflowHandler -``` - -Run the DroidAgent workflow. - diff --git a/docs/v3/sdk/ios-tools.mdx b/docs/v3/sdk/ios-tools.mdx deleted file mode 100644 index 08092fe..0000000 --- a/docs/v3/sdk/ios-tools.mdx +++ /dev/null @@ -1,279 +0,0 @@ ---- -title: IOSTools ---- - -UI Actions - Core UI interaction tools for iOS device control. - - - -## IOSTools - -```python -class IOSTools(Tools) -``` - -Core UI interaction tools for iOS device control. - - - -#### IOSTools.\_\_init\_\_ - -```python -def __init__(url: str, bundle_identifiers: List[str] = []) -> None -``` - -Initialize the IOSTools instance. - -**Arguments**: - -- `url` - iOS device URL. This is the URL of the iOS device. It is used to send requests to the iOS device. -- `bundle_identifiers` - List of bundle identifiers to include in the list of packages - - - -#### IOSTools.get\_state - -```python -def get_state() -> List[Dict[str, Any]] -``` - -Get all clickable UI elements from the iOS device using accessibility API. - -**Returns**: - - List of dictionaries containing UI elements extracted from the device screen - - - -#### IOSTools.tap\_by\_index - -```python -def tap_by_index(index: int) -> str -``` - -Tap on a UI element by its index. - -This function uses the cached clickable elements -to find the element with the given index and tap on its center coordinates. - -**Arguments**: - -- `index` - Index of the element to tap - - -**Returns**: - - Result message - - - -#### IOSTools.tap - -```python -def tap(index: int) -> str -``` - -Tap on a UI element by its index. - -This function uses the cached clickable elements from the last get_clickables call -to find the element with the given index and tap on its center coordinates. - -**Arguments**: - -- `index` - Index of the element to tap - - -**Returns**: - - Result message - - - -#### IOSTools.swipe - -```python -def swipe( - start_x: int, - start_y: int, - end_x: int, - end_y: int, - duration_ms: int = 300 -) -> bool -``` - -Performs a straight-line swipe gesture on the device screen. -To perform a hold (long press), set the start and end coordinates to the same values and increase the duration as needed. - -**Arguments**: - -- `start_x` - Starting X coordinate -- `start_y` - Starting Y coordinate -- `end_x` - Ending X coordinate -- `end_y` - Ending Y coordinate -- `duration_ms` - Duration of swipe in milliseconds (not used in iOS API) - -**Returns**: - - Bool indicating success or failure - - - -#### IOSTools.drag - -```python -def drag( - start_x: int, - start_y: int, - end_x: int, - end_y: int, - duration_ms: int = 3000 -) -> bool -``` - -Drag from the given start coordinates to the given end coordinates. - -**Arguments**: - -- `start_x` - Starting X coordinate -- `start_y` - Starting Y coordinate -- `end_x` - Ending X coordinate -- `end_y` - Ending Y coordinate -- `duration_ms` - Duration of swipe in milliseconds - -**Returns**: - - Bool indicating success or failure - - - -#### IOSTools.input\_text - -```python -def input_text(text: str) -> str -``` - -Input text on the iOS device. - -**Arguments**: - -- `text` - Text to input. Can contain spaces, newlines, and special characters including non-ASCII. - - -**Returns**: - - Result message - - - -#### IOSTools.back - -```python -def back() -> str -``` - - - -#### IOSTools.press\_key - -```python -def press_key(keycode: int) -> str -``` - -Press a key on the iOS device. - -iOS Key codes: -- 0: HOME -- 4: ACTION -- 5: CAMERA - -**Arguments**: - -- `keycode` - iOS keycode to press - - - -#### IOSTools.start\_app - -```python -def start_app(package: str, activity: str = "") -> str -``` - -Start an app on the iOS device. - -**Arguments**: - -- `package` - Bundle identifier (e.g., "com.apple.MobileSMS") -- `activity` - Optional activity name (not used on iOS) - - - -#### IOSTools.take\_screenshot - -```python -def take_screenshot() -> Tuple[str, bytes] -``` - -Take a screenshot of the iOS device. -This function captures the current screen and adds the screenshot to context in the next message. -Also stores the screenshot in the screenshots list with timestamp for later GIF creation. - - - -#### IOSTools.list\_packages - -```python -def list_packages(include_system_apps: bool = True) -> List[str] -``` - - - -#### IOSTools.remember - -```python -def remember(information: str) -> str -``` - -Store important information to remember for future context. - -This information will be included in future LLM prompts to help maintain context -across interactions. Use this for critical facts, observations, or user preferences -that should influence future decisions. - -**Arguments**: - -- `information` - The information to remember - - -**Returns**: - - Confirmation message - - - -#### IOSTools.get\_memory - -```python -def get_memory() -> List[str] -``` - -Retrieve all stored memory items. - -**Returns**: - - List of stored memory items - - - -#### IOSTools.complete - -```python -def complete(success: bool, reason: str = "") -``` - -Mark the task as finished. - -**Arguments**: - -- `success` - Indicates if the task was successful. -- `reason` - Reason for failure/success - diff --git a/docs/v4/concepts/agent-architecture.mdx b/docs/v4/concepts/agent-architecture.mdx new file mode 100644 index 0000000..8dda9d3 --- /dev/null +++ b/docs/v4/concepts/agent-architecture.mdx @@ -0,0 +1,241 @@ +--- +title: 'Multi-Agent Architecture' +description: 'Droidrun v4 hierarchical agent system with specialized roles for planning, execution, and computation.' +--- + +## What is Multi-Agent Architecture? + +Droidrun v4 uses a **hierarchical multi-agent system** where specialized agents work together: + +- **DroidAgent**: Main orchestrator coordinating all agents +- **ManagerAgent**: Strategic planner creating task plans +- **ExecutorAgent**: Tactical actor executing atomic actions +- **CodeActAgent**: Direct code generator for simple tasks +- **ScripterAgent**: Off-device Python executor for API calls, file operations, and computations + +**Location**: `droidrun/agent/droid/droid_agent.py` + +## How It Works + +``` +DroidAgent (orchestrator) +├── Reasoning Mode: ManagerAgent → ExecutorAgent → ScripterAgent +└── Direct Mode: CodeActAgent +``` + +All agents share `DroidAgentState` for coordination and communicate through events. + +## DroidAgent (Orchestrator) + +Entry point for all tasks. Routes to appropriate agents based on mode. + +```python +from droidrun.agent.droid import DroidAgent +from droidrun.config_manager import DroidrunConfig + + +config = DroidrunConfig() + +# Reasoning mode (complex tasks) +agent = DroidAgent(reasoning=True, config=config) + +# Direct mode (simple tasks) +agent = DroidAgent(reasoning=False, config=config) + +result = agent.run("Send message to John") +``` + +## ManagerAgent (Planner) + +Creates strategic plans and breaks tasks into subgoals. + +**Location**: `droidrun/agent/manager/manager_agent.py:46` + +```python +class ManagerPlan(BaseModel): + current_subgoal: str # Next subgoal for Executor + reasoning: str # Why this subgoal + should_finalize: bool # Task complete? + script_block: str | None # Python for ScripterAgent + full_plan: List[str] # Complete plan +``` + +**Configuration:** +```yaml +agent: + manager: + max_steps: 10 + vision: true + +llm_profiles: + manager: + provider: Anthropic + model: claude-sonnet-4 + temperature: 0.7 +``` + +## ExecutorAgent (Actor) + +Executes atomic actions for each subgoal. + +**Location**: `droidrun/agent/executor/executor_agent.py` + +```python +class ExecutorAction(BaseModel): + action: str # "click", "type", "swipe", etc. + parameters: dict # Action parameters + reasoning: str # Why this action + +class ExecutorResult(BaseModel): + success: bool # Action succeeded? + outcome: str # What happened + error_message: str | None +``` + +**Configuration:** +```yaml +agent: + executor: + max_steps: 5 + vision: true + +llm_profiles: + executor: + provider: OpenAI + model: gpt-4o + temperature: 0.3 +``` + +## CodeActAgent (Direct Executor) + +Generates Python code using atomic actions (no planning overhead). + +**Location**: `droidrun/agent/codeact/codeact_agent.py` + +```python +# Available functions in CodeAct +click(index: int) +long_press(index: int) +type(text: str, index: int = None) +swipe(coordinate: tuple, coordinate2: tuple) +system_button(button: str) +open_app(text: str) +get_state() -> dict +take_screenshot() -> str +remember(information: str) +complete(success: bool, reason: str) +``` + +**Configuration:** +```yaml +agent: + codeact: + max_steps: 15 + vision: false + safe_execution: + enabled: true + +llm_profiles: + codeact: + provider: GoogleGenAI + model: models/gemini-2.0-flash-exp +``` + +## ScripterAgent (Python Executor) + +Executes off-device Python for API calls, file operations, data processing, and computations. + +**Location**: `droidrun/agent/scripter/` + +Triggered when Manager delegates tasks requiring off-device computation. ScripterAgent is a **ReAct agent** that iteratively generates and executes Python code, then returns a final message to Manager. + +```python +# Manager delegates with context + task +""" +User needs weather in San Francisco for clothing decision. +Task: Fetch current weather and report temperature + conditions +API: https://api.weather.com/forecast?city=San Francisco +""" + +# ScripterAgent (ReAct loop): +# 1. Generates code +import requests +response = requests.get("https://api.weather.com/forecast", + params={"city": "San Francisco"}) +print(response.json()) + +# 2. Observes output: {'temp': 62, 'description': 'Partly cloudy'} + +# 3. Returns message to Manager: +"The weather in San Francisco is 62°F with partly cloudy conditions." +``` + +**Configuration:** +```yaml +agent: + scripter: + max_steps: 10 + safe_execution: + enabled: true + allowed_modules: + - datetime + - json + - requests +``` + +## Agent Coordination + +### Shared State + +All agents read/write `DroidAgentState`: + +```python +state = DroidAgentState( + task="Book flight", + action_history=[], + visited_packages=[], + error_count=0, + scripter_results={}, + manager_plan="", + executor_feedback="", + step_count=0 +) +``` + +### Event Flow (Reasoning Mode) + +``` +StartEvent + ↓ +ManagerInputEvent → run_manager() + ↓ +ManagerPlanEvent → handle_manager_plan() + ↓ +ExecutorInputEvent → run_executor() + ↓ +ExecutorResultEvent → handle_executor_result() + ↓ +[loop or finalize] + ↓ +FinalizeEvent → finalize() + ↓ +ResultEvent (StopEvent) +``` + +## Quick Reference + +| Agent | Role | Best For | Config Key | +|-------|------|----------|------------| +| DroidAgent | Orchestrator | Entry point | `agent.*` | +| ManagerAgent | Planner | Strategy, recovery | `agent.manager.*` | +| ExecutorAgent | Actor | Action execution | `agent.executor.*` | +| CodeActAgent | Direct | Simple tasks | `agent.codeact.*` | +| ScripterAgent | Python Executor | APIs, files, data | `agent.scripter.*` | + +## Related Topics + +- [Reasoning Mode](./reasoning-mode) - Manager → Executor workflow +- [Direct Mode](./direct-mode) - CodeActAgent workflow +- [ScripterAgent](./scripter-agent) - Off-device computation +- [Shared State](./shared-state) - DroidAgentState coordination +- [Configuration](./configuration) - Per-agent LLM profiles diff --git a/docs/v4/concepts/architecture.mdx b/docs/v4/concepts/architecture.mdx deleted file mode 100644 index 667bdaf..0000000 --- a/docs/v4/concepts/architecture.mdx +++ /dev/null @@ -1,1793 +0,0 @@ ---- -title: 'Multi-Agent Architecture' -description: 'Deep dive into DroidRun agent system, workflow coordination, and execution patterns' ---- - -## Overview - -DroidRun uses a sophisticated multi-agent architecture built on [LlamaIndex workflows](https://docs.llamaindex.ai/en/stable/module_guides/workflow/). The system employs hierarchical coordination between specialized agents, each with distinct responsibilities, to achieve complex device automation goals. - -```mermaid -graph TB - User[User Goal] --> DA[DroidAgent
Coordinator] - - DA -->|reasoning=false| CA[CodeActAgent
Direct Execution] - DA -->|reasoning=true| MA[ManagerAgent
Planning] - - MA --> EA[ExecutorAgent
Action Execution] - MA -->|" -3. [ ] Save formatted vCard to file - - - - - -``` - -#### Error Escalation - -The Manager tracks action outcomes and detects repeated failures: - -```python -# In DroidAgentState -err_to_manager_thresh: int = 2 # Consecutive errors before escalation - -# Error detection in DroidAgent.handle_executor_result() -if len(self.shared_state.action_outcomes) >= err_thresh: - latest = self.shared_state.action_outcomes[-err_thresh:] - error_count = sum(1 for o in latest if not o) - if error_count == err_thresh: - logger.warning(f"⚠️ Error escalation: {err_thresh} consecutive errors") - self.shared_state.error_flag_plan = True -``` - -When `error_flag_plan=True`, the Manager receives error context and adjusts its strategy. - ---- - -### ExecutorAgent - Action Specialist - -**Location:** `droidrun/agent/executor/executor_agent.py` - -The Executor is responsible for selecting and executing specific atomic actions to achieve the current subgoal provided by the Manager. - -#### Responsibilities - -1. **Action Selection**: Chooses the best atomic action for the subgoal -2. **Action Execution**: Executes device interactions (tap, type, swipe, etc.) -3. **Outcome Reporting**: Reports success/failure and provides summaries -4. **Custom Tool Support**: Can execute user-defined custom tools - -#### Executor Workflow - -```mermaid -sequenceDiagram - participant DA as DroidAgent - participant E as ExecutorAgent - participant Tools as AdbTools - - DA->>E: ExecutorInputEvent(subgoal) - E->>E: think()
(Choose action) - E-->>DA: ExecutorInternalActionEvent
(streamed for UI) - E->>E: execute()
(Run action) - E->>Tools: click(index) / type(text) / etc. - Tools-->>E: Result - E-->>DA: ExecutorInternalResultEvent
(streamed for UI) - E-->>DA: ExecutorResultEvent
(coordination) - DA->>DA: Update shared state - DA->>Manager: ManagerInputEvent
(Loop back) -``` - -#### Action Format - -The Executor returns actions in JSON format: - -```json -{ - "action": "click", - "index": 5, - "thought": "The 'Alarms' tab button is at index 5", - "description": "Click the Alarms tab to navigate" -} -``` - -For text input: - -```json -{ - "action": "type", - "text": "john@example.com", - "index": 3, - "thought": "Need to input email into the focused text field", - "description": "Type email address into the input field" -} -``` - -#### Custom Tool Execution - -The Executor can execute custom tools defined by users: - -```python -def get_password(tools, credential_id: str) -> str: - """Retrieve password from credential manager.""" - return tools.credential_manager.get_secret(credential_id) - -custom_tools = { - "get_password": { - "signature": "get_password(credential_id: str) -> str", - "description": "Retrieve a password from secure storage", - "function": get_password - } -} - -agent = DroidAgent( - goal="Login to app with stored credentials", - llm=llm, - tools=tools, - custom_tools=custom_tools # Executor can now use get_password -) -``` - -Executor action: - -```json -{ - "action": "get_password", - "credential_id": "gmail_password", - "thought": "Need to retrieve stored password for login", - "description": "Get Gmail password from credential manager" -} -``` - ---- - -### CodeActAgent - Direct Executor - -**Location:** `droidrun/agent/codeact/codeact_agent.py` - -CodeActAgent implements a ReAct-style (Reasoning + Acting) cycle that generates and executes Python code to interact with the device. It's used in non-reasoning mode or when DroidAgent needs to execute code-based tasks. - -#### ReAct Cycle - -```mermaid -graph LR - A[Task Input] --> B[Think
Generate Code] - B --> C[Execute Code] - C --> D{Complete?} - D -->|No| E[Observe Result] - E --> B - D -->|Yes| F[Return Result] - - style B fill:#fbbf24,color:#000 - style C fill:#ef4444,color:#fff - style E fill:#3b82f6,color:#fff -``` - -#### Code Generation - -CodeActAgent generates Python code using available atomic actions: - -```python -# Example generated code -click(tools, 5) # Tap element at index 5 -type(tools, "hello@example.com", 3) # Type into element 3 -remember("User's email is hello@example.com") # Store in memory -complete(True, "Email entered successfully") # Mark complete -``` - -#### Execution Flow - -```mermaid -sequenceDiagram - participant DA as DroidAgent - participant CA as CodeActAgent - participant Executor as SimpleCodeExecutor - participant Tools as Device Tools - - DA->>CA: StartEvent(task) - CA->>CA: prepare_chat()
(Build prompt) - - loop Until complete or max_steps - CA->>CA: handle_llm_input()
(Get device state) - CA->>LLM: Chat with context - LLM-->>CA: Code + Thoughts - CA-->>DA: TaskThinkingEvent
(streamed) - - CA->>CA: handle_llm_output()
(Parse code) - CA->>Executor: execute_code() - Executor->>Tools: click/type/swipe/etc. - Tools-->>Executor: Result - Executor-->>CA: Output / Error - CA-->>DA: TaskExecutionResultEvent
(streamed) - - alt complete() called - CA-->>DA: TaskEndEvent - else continue - CA->>CA: Add observation to chat - end - end - - CA-->>DA: StopEvent(result) -``` - -#### Safe Execution Mode - -CodeActAgent supports restricted execution for security: - -```yaml -# config.yaml -safe_execution: - codeact: - safe_execution: true - allowed_modules: - - re - - json - - datetime - blocked_modules: - - os - - subprocess - - sys - allowed_builtins: - - len - - str - - int - - float - blocked_builtins: - - eval - - exec - - __import__ -``` - -When enabled: -- Only allowed modules can be imported -- Blocked modules raise `ImportError` -- Only allowed builtins are available -- No file system or network access - -#### Memory System - -CodeActAgent maintains episodic memory across steps: - -```python -# Using remember() function -remember("User prefers dark mode") -remember("Last search query: weather forecast") - -# Memory is injected into subsequent prompts -# Accessible via self.tools.memory in code -``` - ---- - -### ScripterAgent - Off-Device Computation - -**Location:** `droidrun/agent/scripter/scripter_agent.py` - -ScripterAgent handles Python code execution for tasks that don't require device interaction - data processing, API calls, computations, etc. - -#### Use Cases - -- **Data Transformation**: Parse, format, or transform extracted data -- **API Calls**: Fetch information from external services -- **Calculations**: Perform mathematical or logical computations -- **File Operations**: Process files on the host machine -- **Text Processing**: Parse and manipulate text with regex, etc. - -#### ScripterAgent vs CodeActAgent - -| Feature | ScripterAgent | CodeActAgent | -|---------|---------------|--------------| -| **Device Access** | No | Yes | -| **Available Tools** | Python stdlib + requests | Atomic actions (click, type, etc.) | -| **Execution Context** | Host machine | Device via ADB | -| **State Persistence** | Yes (Jupyter-style) | No (fresh each call) | -| **Completion Signal** | No code in response | `complete()` function | -| **Max Steps** | Configurable (default: 10) | Agent max_steps | - -#### Workflow - -```mermaid -sequenceDiagram - participant M as ManagerAgent - participant DA as DroidAgent - participant SA as ScripterAgent - participant Exec as SimpleCodeExecutor - - M->>DA: "ManagerPlanEvent
(with -3. [ ] Save CSV file - - - - - -``` - -ScripterAgent execution: - -```python -# Step 1: Parse and convert -import json - -contacts = [ - {"name": "Alice", "phone": "555-0100"}, - {"name": "Bob", "phone": "555-0101"} -] - -csv_lines = ["name,phone"] -for contact in contacts: - csv_lines.append(f"{contact['name']},{contact['phone']}") - -csv_output = "\n".join(csv_lines) -print(csv_output) - -# Step 2: Return result (no code, just message) -# "Here is the CSV format: -# name,phone -# Alice,555-0100 -# Bob,555-0101" -``` - -#### Configuration - -```yaml -# config.yaml -agent: - scripter: - enabled: true - max_steps: 10 - execution_timeout: 30 - safe_execution: true -``` - ---- - -## Helper Workflows - -### AppStarter - Intelligent App Launching - -**Location:** `droidrun/agent/oneflows/app_starter_workflow.py` - -AppStarter uses an LLM to match natural language app descriptions to installed package names, then launches the app. - -#### How It Works - -```python -# Called by ExecutorAgent when executing open_app action -async def open_app(tools: Tools, app_description: str) -> str: - """Open app using LLM-based package matching.""" - # Get installed apps - apps = tools.get_apps(include_system=True) - - # Create AppStarter workflow - workflow = AppStarter(tools=tools, llm=app_opener_llm) - - # Run workflow - result = await workflow.run(app_description=app_description) - return result -``` - -#### Example - -```python -# User: "Open Settings" -# AppStarter: -# 1. Gets app list: [{"label": "Settings", "package": "com.android.settings"}, ...] -# 2. Asks LLM to match "Settings" → "com.android.settings" -# 3. Calls tools.start_app("com.android.settings") -``` - ---- - -### TextManipulator - Advanced Text Editing - -**Location:** `droidrun/agent/oneflows/text_manipulator.py` - -TextManipulator handles complex text editing tasks that go beyond simple typing - edits, insertions, formatting, etc. - -#### When It's Used - -The Manager detects text manipulation mode when: -1. A text field is focused -2. The field contains existing text -3. The subgoal requires modifying (not just typing) text - -#### How It Works - -```python -# Called by ExecutorAgent when text manipulation is needed -def run_text_manipulation_agent( - instruction: str, # Overall goal - current_subgoal: str, # What to do with text - current_text: str, # Current text field content - overall_plan: str, # Full plan context - historical_plan: str, # Progress so far - llm: LLM, - max_retries: int = 4 -) -> tuple[str, str]: - """Generate code to manipulate text field content.""" - # Returns: (final_text, raw_code) -``` - -#### Constrained Execution - -TextManipulator uses a highly restricted sandbox: - -```python -# ONLY these are available: -# - ORIGINAL: str (current text content) -# - input_text(text: str): function (clear and type) - -# NO imports, NO builtins, NO file system -``` - -#### Code Generation Pattern - -```python -# Example generated code: -new_text = ORIGINAL.replace("old", "new") -input_text(new_text) - -# Or: -lines = ORIGINAL.split("\n") -lines.append("New line") -new_text = "\n".join(lines) -input_text(new_text) -``` - -#### Error Correction Loop - -If code fails, the error is sent back to the LLM: - -```python -# Attempt 1: Code fails -try: - exec(code, sandbox) -except Exception as e: - # Attempt 2: Send error back - messages.append(ChatMessage( - role="user", - content=f"Your code had this error:\n{traceback.format_exc()}\n\nFix it." - )) - # LLM generates corrected code -``` - -#### Example Usage - -Manager detects text manipulation: - -```python -# Device state shows focused text field: -# - -# Manager subgoal: -# "Fix the typo in the focused text field (change 'Wrld' to 'World')" - -# TextManipulator generates: -new_text = ORIGINAL.replace("Wrld", "World") -input_text(new_text) - -# Result: "Hello World" typed into field -``` - ---- - -### StructuredOutputAgent - Data Extraction - -**Location:** `droidrun/agent/oneflows/structured_output_agent.py` - -StructuredOutputAgent extracts structured data from the final answer using LlamaIndex's `structured_predict()`. - -#### Use Case - -When you need to extract specific fields from the agent's answer: - -```python -from pydantic import BaseModel, Field - -class WeatherInfo(BaseModel): - """Weather forecast data.""" - temperature: int = Field(description="Temperature in Fahrenheit") - condition: str = Field(description="Weather condition (sunny, rainy, etc.)") - location: str = Field(description="City name") - -agent = DroidAgent( - goal="Check the weather forecast", - llm=llm, - tools=tools, - output_model=WeatherInfo # Request structured output -) - -result = await agent.run() -# result.structured_output = WeatherInfo(temperature=72, condition="sunny", location="San Francisco") -``` - -#### How It Works - -```mermaid -sequenceDiagram - participant DA as DroidAgent - participant SO as StructuredOutputAgent - participant LLM as LLM - - DA->>DA: Task complete
(FinalizeEvent with answer) - - alt output_model is set - DA->>SO: Run extraction workflow - SO->>LLM: structured_predict()
(answer + Pydantic model) - LLM-->>SO: Structured data - SO-->>DA: Extracted object - DA->>DA: Add to result.structured_output - else output_model is None - DA->>DA: result.structured_output = None - end - - DA-->>User: Final result with structured data -``` - -#### Workflow - -```python -# Inside StructuredOutputAgent -async def extract_structured_output(self, ctx: Context, ev: StartEvent) -> StopEvent: - """Extract structured output using structured_predict().""" - - # Create prompt for extraction - prompt = PromptTemplate( - "Extract structured information from the following text:\n\n{text}" - ) - - # Use structured_predict to extract data - structured_output = self.llm.structured_predict( - self.pydantic_model, - prompt, - text=self.answer_text - ) - - return StopEvent(result={ - "structured_output": structured_output, - "success": True - }) -``` - -#### Example - -Agent answer: -``` -The weather in San Francisco is currently 72°F and sunny. -``` - -Extracted structure: -```python -WeatherInfo( - temperature=72, - condition="sunny", - location="San Francisco" -) -``` - ---- - -## Workflow Events & Coordination - -DroidRun uses two types of events: - -### Coordination Events - -**Purpose**: Route workflow execution between agents -**Location**: `droidrun/agent/droid/events.py` -**Characteristics**: -- Minimal data (only what's needed for routing) -- NOT streamed to frontend -- Used in workflow step handlers - -```python -class ManagerInputEvent(Event): - """Trigger Manager workflow for planning""" - pass - -class ManagerPlanEvent(Event): - """Coordination event from ManagerAgent to DroidAgent""" - plan: str - current_subgoal: str - thought: str - manager_answer: str = "" - -class ExecutorInputEvent(Event): - """Trigger Executor workflow for action execution""" - current_subgoal: str - -class ExecutorResultEvent(Event): - """Coordination event from ExecutorAgent to DroidAgent""" - action: Dict - outcome: bool - error: str - summary: str -``` - -### Internal Events - -**Purpose**: Stream debugging information to frontend/logs -**Location**: `droidrun/agent/manager/events.py`, `droidrun/agent/executor/events.py` -**Characteristics**: -- Rich metadata (thoughts, raw JSON, etc.) -- Streamed to frontend via `ctx.write_event_to_stream()` -- Used for debugging and UI updates - -```python -class ManagerInternalPlanEvent(Event): - """Internal Manager planning event with full state""" - plan: str - current_subgoal: str - thought: str - manager_answer: str = "" - memory_update: str = "" # Debugging: LLM's memory additions - -class ExecutorInternalActionEvent(Event): - """Internal Executor action selection event""" - action_json: str - thought: str # Debugging: LLM's reasoning - description: str - -class ExecutorInternalResultEvent(Event): - """Internal Executor result event""" - action: Dict - outcome: bool - error: str - summary: str - thought: str = "" # Debugging: LLM's thought process - action_json: str = "" # Debugging: Raw action JSON -``` - -### Event Flow Pattern - -```python -# In child agent (e.g., ManagerAgent) -@step -async def think(self, ctx: Context, ev: ManagerThinkingEvent) -> ManagerInternalPlanEvent: - # ... LLM call and planning ... - - event = ManagerInternalPlanEvent( - plan=parsed["plan"], - current_subgoal=parsed["current_subgoal"], - thought=parsed["thought"], - manager_answer=parsed["answer"], - memory_update=memory_update # Rich metadata - ) - - # Stream to frontend for debugging/UI - ctx.write_event_to_stream(event) - - return event # Propagate to finalize() - -@step -async def finalize(self, ctx: Context, ev: ManagerInternalPlanEvent) -> StopEvent: - # Return minimal data to parent workflow - return StopEvent(result={ - "plan": ev.plan, - "current_subgoal": ev.current_subgoal, - "thought": ev.thought, - "manager_answer": ev.manager_answer - # memory_update NOT included (debugging only) - }) - -# In parent (DroidAgent) -@step -async def run_manager(self, ctx: Context, ev: ManagerInputEvent) -> ManagerPlanEvent: - # Run Manager workflow - handler = self.manager_agent.run() - - # Stream nested events (including ManagerInternalPlanEvent) - async for nested_ev in handler.stream_events(): - self.handle_stream_event(nested_ev, ctx) # Forward to parent stream - - result = await handler # Get minimal StopEvent result - - # Return coordination event (minimal data for routing) - return ManagerPlanEvent( - plan=result["plan"], - current_subgoal=result["current_subgoal"], - thought=result["thought"], - manager_answer=result.get("manager_answer", "") - ) -``` - ---- - -## DroidAgentState - Shared Coordination State - -**Location**: `droidrun/agent/droid/events.py` - -`DroidAgentState` is a Pydantic model that holds coordination state shared across all agents. It's passed to child agents during initialization and updated throughout execution. - -### Key Fields - -```python -class DroidAgentState(BaseModel): - # Task context - instruction: str = "" # Original user goal - step_number: int = 0 # Current step counter - - # Device state - formatted_device_state: str = "" # Current UI hierarchy - previous_formatted_device_state: str # Previous UI (for comparison) - focused_text: str = "" # Text in focused element - a11y_tree: List[Dict] = [] # Raw accessibility tree - phone_state: Dict = {} # Device metadata - - # App tracking - current_package_name: str = "" - current_activity_name: str = "" - visited_packages: set = set() - visited_activities: set = set() - - # Screen capture - width: int = 0 - height: int = 0 - screenshot: str | bytes | None = None - - # Action history - action_pool: List[Dict] = [] # All actions (for replay) - action_history: List[Dict] = [] # Executed actions - summary_history: List[str] = [] # Action summaries - action_outcomes: List[bool] = [] # Success/failure - error_descriptions: List[str] = [] # Error messages - - # Last action info - last_action: Dict = {} - last_summary: str = "" - last_action_thought: str = "" - - # Memory - memory: str = "" # Remembered information - message_history: List[Dict] = [] # Chat history (Manager) - - # Planning state - plan: str = "" - current_subgoal: str = "" - finish_thought: str = "" - progress_status: str = "" - manager_answer: str = "" # For answer-type tasks - - # Error handling - error_flag_plan: bool = False # Error escalation flag - err_to_manager_thresh: int = 2 # Errors before escalation - - # Script execution - scripter_history: List[Dict] = [] - last_scripter_message: str = "" - last_scripter_success: bool = True - - # Custom variables (user-defined) - custom_variables: Dict = {} - - # App Cards - app_card: str = "" # Current app guidance - app_card_loading_task: asyncio.Task | None = None -``` - -### Usage Pattern - -```python -# Initialized in DroidAgent.__init__() -self.shared_state = DroidAgentState( - instruction=goal, - err_to_manager_thresh=2, - user_id=self.user_id, - runtype=self.runtype -) - -# Passed to child agents during initialization -self.manager_agent = ManagerAgent( - llm=manager_llm, - tools_instance=tools_instance, - shared_state=self.shared_state, # Same instance shared - agent_config=self.config.agent, - custom_tools=custom_tools -) - -# Child agents read and update state -# In ManagerAgent.prepare_input(): -self.shared_state.formatted_device_state = formatted_text -self.shared_state.focused_text = focused_text -self.shared_state.update_current_app(package_name, activity_name) - -# In DroidAgent.handle_executor_result(): -self.shared_state.action_history.append(result["action"]) -self.shared_state.action_outcomes.append(result["outcome"]) -self.shared_state.step_number += 1 -``` - -### Thread Safety - -`DroidAgentState` is **NOT** thread-safe. All updates happen within the async workflow context, avoiding race conditions. However, be aware: - -- Each agent receives the **same instance** (not a copy) -- Updates are immediately visible to all agents -- No explicit locking is needed due to async single-threaded execution - ---- - -## Workflow Execution Patterns - -### Pattern 1: Parent-Child Workflow Nesting - -DroidAgent runs child workflows and streams their events: - -```python -@step -async def run_manager(self, ctx: Context, ev: ManagerInputEvent) -> ManagerPlanEvent: - # Run child workflow - handler = self.manager_agent.run() - - # Stream nested events to parent context - async for nested_ev in handler.stream_events(): - self.handle_stream_event(nested_ev, ctx) - - # Wait for completion and get result - result = await handler - - # Transform to coordination event - return ManagerPlanEvent( - plan=result["plan"], - current_subgoal=result["current_subgoal"], - # ... other fields - ) -``` - -### Pattern 2: Event Filtering - -Parent filters which events to forward: - -```python -def handle_stream_event(self, ev: Event, ctx: Context): - # Special handling for specific events - if isinstance(ev, EpisodicMemoryEvent): - self.current_episodic_memory = ev.episodic_memory - return # Don't forward - - # Never forward StopEvent (internal to child) - if not isinstance(ev, StopEvent): - ctx.write_event_to_stream(ev) # Forward to parent stream - - # Track trajectory events - if isinstance(ev, ScreenshotEvent): - self.trajectory.screenshots.append(ev.screenshot) - elif isinstance(ev, MacroEvent): - self.trajectory.macro.append(ev) -``` - -### Pattern 3: Workflow Step Handlers - -Each step is decorated with `@step` and handles specific event types: - -```python -@step -async def run_manager( - self, ctx: Context, ev: ManagerInputEvent -) -> ManagerPlanEvent | FinalizeEvent: - """Handles ManagerInputEvent, returns routing event.""" - # Pre-flight checks - if self.shared_state.step_number >= self.config.agent.max_steps: - return FinalizeEvent(success=False, reason="Max steps") - - # Execute workflow - handler = self.manager_agent.run() - async for nested_ev in handler.stream_events(): - self.handle_stream_event(nested_ev, ctx) - result = await handler - - # Return routing event - return ManagerPlanEvent(plan=result["plan"], ...) - -@step -async def handle_manager_plan( - self, ctx: Context, ev: ManagerPlanEvent -) -> ExecutorInputEvent | ScripterExecutorInputEvent | FinalizeEvent: - """Routes based on Manager's plan.""" - # Check for completion - if ev.manager_answer.strip(): - return FinalizeEvent(success=True, reason=ev.manager_answer) - - # Check for script tag - if " - -# DroidAgent routes to ScripterAgent -# Scripter executes code without device tools -# Result stored in shared_state.last_scripter_message -# Manager receives result in next cycle -``` - -**Key Features:** -- **No Device Tools**: Only Python libraries -- **State Preservation**: Variables persist across code blocks -- **Completion Signal**: No code = final answer -- **Safe Execution**: Optional restrictions (same as CodeAct) - -**Location:** `/droidrun/agent/scripter/scripter_agent.py` - ---- - -## Event System - -### Event Types - -DroidRun uses two categories of events: - -1. **Coordination Events** (`droid/events.py`) - - Minimal data for workflow routing - - Used by DroidAgent to orchestrate agents - - Examples: `ManagerInputEvent`, `ExecutorResultEvent`, `CodeActExecuteEvent` - -2. **Internal Events** (per-agent `events.py`) - - Full debug metadata - - Streamed to frontend/CLI for monitoring - - Examples: `ManagerInternalPlanEvent`, `ExecutorInternalActionEvent`, `TaskThinkingEvent` - -### Event Flow Diagram - -``` -┌──────────────────────────────────────────────────────────────┐ -│ DroidAgent Events │ -├──────────────────────────────────────────────────────────────┤ -│ │ -│ StartEvent │ -│ ↓ │ -│ ManagerInputEvent (coordination) │ -│ ↓ │ -│ ┌────────────────────────────────────────────────┐ │ -│ │ ManagerAgent (nested workflow) │ │ -│ │ │ │ -│ │ ManagerThinkingEvent (internal, streamed) │ │ -│ │ ↓ │ │ -│ │ ManagerInternalPlanEvent (internal, streamed) │ │ -│ │ ↓ │ │ -│ │ StopEvent → result returned to parent │ │ -│ └────────────────────────────────────────────────┘ │ -│ ↓ │ -│ ManagerPlanEvent (coordination) │ -│ ↓ │ -│ ExecutorInputEvent (coordination) │ -│ ↓ │ -│ ┌────────────────────────────────────────────────┐ │ -│ │ ExecutorAgent (nested workflow) │ │ -│ │ │ │ -│ │ ExecutorInternalActionEvent (internal) │ │ -│ │ ↓ │ │ -│ │ ExecutorInternalResultEvent (internal) │ │ -│ │ ↓ │ │ -│ │ StopEvent → result returned to parent │ │ -│ └────────────────────────────────────────────────┘ │ -│ ↓ │ -│ ExecutorResultEvent (coordination) │ -│ ↓ │ -│ Loop back to ManagerInputEvent │ -│ │ -└──────────────────────────────────────────────────────────────┘ -``` - -### Coordination Events - -| Event | Source | Target | Purpose | -|-------|--------|--------|---------| -| `StartEvent` | Workflow start | `start_handler()` | Initialize execution | -| `ManagerInputEvent` | Various | `run_manager()` | Trigger Manager planning | -| `ManagerPlanEvent` | `run_manager()` | `handle_manager_plan()` | Route plan to execution | -| `ExecutorInputEvent` | `handle_manager_plan()` | `run_executor()` | Trigger Executor action | -| `ExecutorResultEvent` | `run_executor()` | `handle_executor_result()` | Process action result | -| `ScripterExecutorInputEvent` | `handle_manager_plan()` | `run_scripter()` | Trigger Scripter | -| `ScripterExecutorResultEvent` | `run_scripter()` | `handle_scripter_result()` | Process script result | -| `CodeActExecuteEvent` | `start_handler()` | `execute_task()` | Direct execution task | -| `CodeActResultEvent` | `execute_task()` | `handle_codeact_execute()` | CodeAct result | -| `FinalizeEvent` | Various | `finalize()` | Complete workflow | -| `StopEvent` | `finalize()` | Workflow end | Return final result | - -### Internal Events (Streamed) - -| Event | Agent | Purpose | Fields | -|-------|-------|---------|--------| -| `ManagerThinkingEvent` | Manager | Manager is thinking | - | -| `ManagerInternalPlanEvent` | Manager | Plan created | plan, subgoal, thought, answer, memory_update | -| `ExecutorInternalActionEvent` | Executor | Action selected | action_json, thought, description | -| `ExecutorInternalResultEvent` | Executor | Action completed | action, outcome, error, summary, thought | -| `TaskInputEvent` | CodeAct | LLM input ready | input (messages) | -| `TaskThinkingEvent` | CodeAct | LLM response received | thoughts, code, usage | -| `TaskExecutionEvent` | CodeAct | Code ready for execution | code | -| `TaskExecutionResultEvent` | CodeAct | Code executed | output | -| `TaskEndEvent` | CodeAct | Task complete | success, reason | -| `ScripterThinkingEvent` | Scripter | Scripter thinking | thoughts, code, full_response | -| `ScripterExecutionEvent` | Scripter | Script executing | code | -| `ScripterExecutionResultEvent` | Scripter | Script completed | output | -| `ScripterEndEvent` | Scripter | Scripter done | message, success, code_executions | - ---- - -## Shared State Management - -### DroidAgentState - -Central coordination state shared across all agents. - -**Location:** `/droidrun/agent/droid/events.py` - -**Key Fields:** - -```python -class DroidAgentState(BaseModel): - # Task Context - instruction: str # User goal - step_number: int # Current step - - # Device State - formatted_device_state: str # UI tree + phone state (formatted) - a11y_tree: List[Dict] # Raw accessibility tree - phone_state: Dict # Raw phone state - focused_text: str # Currently focused text element - width: int, height: int # Screen dimensions - screenshot: str | bytes | None # Screenshot data - - # App Context - app_card: str # App-specific instructions - current_package_name: str # Current app package - current_activity_name: str # Current activity - visited_packages: set # Unique packages visited - visited_activities: set # Unique activities visited - - # Action History - action_history: List[Dict] # All actions taken - summary_history: List[str] # Action summaries - action_outcomes: List[bool] # Success/failure - error_descriptions: List[str] # Error messages - last_action: Dict # Most recent action - last_summary: str # Most recent summary - last_action_thought: str # LLM reasoning - - # Planning - plan: str # Current plan - current_subgoal: str # Active subgoal - progress_status: str # Progress description - manager_answer: str # Answer for answer-type tasks - - # Error Handling - error_flag_plan: bool # Error escalation flag - err_to_manager_thresh: int # Consecutive errors before escalation - - # Memory - memory: str # Accumulated agent memory - message_history: List[Dict] # Chat history (Manager) - - # Script Execution - scripter_history: List[Dict] # Script execution records - last_scripter_message: str # Most recent script result - last_scripter_success: bool # Script success status - - # Custom Extensions - custom_variables: Dict # User-defined variables -``` - -**State Update Patterns:** - -```python -# Update device state (Manager/CodeAct) -state = tools.get_state() -formatted_text, focused_text, a11y_tree, phone_state = format_device_state(state) -shared_state.formatted_device_state = formatted_text -shared_state.a11y_tree = a11y_tree -shared_state.phone_state = phone_state - -# Update current app (unified method) -shared_state.update_current_app( - package_name="com.example.app", - activity_name="MainActivity" -) - -# Update action history (Executor) -shared_state.action_history.append(action) -shared_state.action_outcomes.append(outcome) -shared_state.error_descriptions.append(error) - -# Update memory (Manager) -if memory_update: - shared_state.memory += "\n" + memory_update - -# Update scripter result (Scripter) -shared_state.last_scripter_message = result["message"] -shared_state.last_scripter_success = result["success"] -``` - ---- - -## Workflow Step Patterns - -### The `@step` Decorator - -LlamaIndex workflows use the `@step` decorator to define workflow handlers. - -**Basic Pattern:** -```python -from llama_index.core.workflow import Context, step - -@step -async def step_name(self, ctx: Context, ev: InputEvent) -> OutputEvent: - """Step handler that processes InputEvent and returns OutputEvent.""" - # 1. Extract data from event - data = ev.some_field - - # 2. Perform operations - result = await some_operation(data) - - # 3. Update context (optional) - await ctx.store.set("key", value) - - # 4. Stream events to frontend (optional) - ctx.write_event_to_stream(some_event) - - # 5. Return next event - return OutputEvent(result=result) -``` - -**Event Routing:** -- Workflow automatically routes events to matching step handlers -- Handler signature determines which events it receives: `ev: EventType` -- Return value determines next step(s) to execute - ---- - -### Nested Workflow Pattern - -DroidAgent runs child agents as nested workflows. - -**Pattern:** -```python -@step -async def run_manager(self, ctx: Context, ev: ManagerInputEvent) -> ManagerPlanEvent: - """Run Manager as nested workflow and stream its events.""" - - # 1. Start child workflow - handler = self.manager_agent.run() - - # 2. Stream all nested events to parent context - async for nested_ev in handler.stream_events(): - self.handle_stream_event(nested_ev, ctx) - - # 3. Await final result - result = await handler - - # 4. Return coordination event to parent - return ManagerPlanEvent( - plan=result["plan"], - current_subgoal=result["current_subgoal"] - ) -``` - -**Why Nested Workflows?** -- **Modularity**: Each agent is self-contained -- **Event Isolation**: Internal events don't pollute parent -- **Streaming**: All child events forwarded to frontend -- **Error Handling**: Child failures can be caught and handled - ---- - -### Loop Pattern (Manager/Executor Cycle) - -Manager and Executor continuously loop until task completion. - -```python -# Manager plans → Executor acts → Manager re-plans → Executor acts → ... - -@step -async def run_manager(self, ctx: Context, ev: ManagerInputEvent) -> ManagerPlanEvent | FinalizeEvent: - """Pre-flight check, then run Manager.""" - - # Pre-flight: Check max steps - if self.shared_state.step_number >= self.config.agent.max_steps: - return FinalizeEvent(success=False, reason="Max steps reached") - - # Run Manager workflow (nested) - handler = self.manager_agent.run() - async for nested_ev in handler.stream_events(): - self.handle_stream_event(nested_ev, ctx) - result = await handler - - return ManagerPlanEvent(plan=result["plan"], current_subgoal=result["current_subgoal"]) - -@step -async def handle_manager_plan(self, ctx: Context, ev: ManagerPlanEvent) -> ExecutorInputEvent | FinalizeEvent: - """Route Manager plan to Executor or finish.""" - - # Check if Manager provided answer (task complete) - if ev.manager_answer.strip(): - return FinalizeEvent(success=True, reason=ev.manager_answer) - - # Check if -``` - ---- - -## Safe Execution - -Safe execution restricts code execution to prevent dangerous operations. It applies to **CodeAct** and **Scripter** agents. - -### Overview - -When `safe_execution: true`: -- Imports are restricted to allowed modules -- Builtins are restricted to safe operations -- Dangerous operations (file I/O, subprocess, eval) are blocked - -### Configuration - -```yaml -safe_execution: - # === Import Control === - allow_all_imports: false # Allow all imports (dangerous!) - allowed_modules: - - json - - requests - - re - - datetime - - math - - collections - blocked_modules: - - os - - sys - - subprocess - - shutil - - socket - - # === Builtin Control === - allow_all_builtins: false # Allow all builtins (dangerous!) - allowed_builtins: [] # Empty = use safe defaults - blocked_builtins: - - open - - exec - - eval - - __import__ - - compile - - breakpoint -``` - -### Enabling Safe Execution - -**Per agent:** - -```yaml -agent: - codeact: - safe_execution: true # Enable for CodeAct - - scripter: - safe_execution: true # Enable for Scripter -``` - -### Safe Defaults - -When `allow_all_builtins: false` and `allowed_builtins: []`, DroidRun uses safe defaults: - -**Safe builtins include:** -- Type constructors: `int`, `float`, `str`, `bool`, `list`, `dict`, `tuple`, `set` -- Iteration: `range`, `enumerate`, `zip`, `map`, `filter`, `sorted` -- Math: `abs`, `round`, `pow`, `sum`, `min`, `max` -- Type checking: `type`, `isinstance`, `callable` -- Output: `print`, `repr`, `format` -- Exceptions: `Exception`, `ValueError`, `TypeError`, etc. - -**Blocked by default:** -- File I/O: `open`, `input` -- Code execution: `exec`, `eval`, `compile`, `__import__` -- System: `exit`, `quit`, `breakpoint` - -### Custom Safe Execution - -Allow specific modules for trusted use cases: - -```yaml -safe_execution: - allow_all_imports: false - allowed_modules: - - json # Parse JSON - - requests # HTTP requests - - re # Regex - - datetime # Date/time - - math # Math operations - - collections # Data structures - - itertools # Iterator tools - - functools # Functional programming - blocked_modules: - - os # Prevent file system access - - subprocess # Prevent process execution - - socket # Prevent network sockets - - sys # Prevent system access -``` - -### Dangerous: Allow All - - -**Only use this in trusted environments!** Allows arbitrary code execution. - - -```yaml -safe_execution: - allow_all_imports: true # Allow ANY import - allow_all_builtins: true # Allow ANY builtin - blocked_modules: # Still block these (recommended) - - os - - subprocess -``` - ---- - -## Prompt Customization - -DroidRun uses Jinja2 templates for agent prompts. You can customize prompts to change agent behavior. - -### Prompt Directory Structure - -``` -config/prompts/ -├── codeact/ -│ ├── system.jinja2 -│ └── user.jinja2 -├── manager/ -│ ├── system.jinja2 -│ └── rev1.jinja2 -├── executor/ -│ ├── system.jinja2 -│ └── rev1.jinja2 -└── scripter/ - └── system.jinja2 -``` - -### Prompt Configuration - -Specify prompt files in config: - -```yaml -agent: - prompts_dir: config/prompts # Base directory - - codeact: - system_prompt: system.jinja2 # codeact/system.jinja2 - user_prompt: user.jinja2 # codeact/user.jinja2 - - manager: - system_prompt: system.jinja2 # manager/system.jinja2 - - executor: - system_prompt: system.jinja2 # executor/system.jinja2 - - scripter: - system_prompt_path: system.jinja2 # scripter/system.jinja2 -``` - -### Custom Prompts - -Create custom prompt files: - -```sh -mkdir -p config/prompts/manager -touch config/prompts/manager/custom.jinja2 -``` - -**config.yaml:** - -```yaml -agent: - manager: - system_prompt: custom.jinja2 # Use custom prompt -``` - -### Prompt Variables - -Prompts support Jinja2 template variables: - -**Example prompt:** - -```jinja2 -You are an agent operating an Android phone. - - -{{ instruction }} - - -{% if device_date %} - -{{ device_date }} - -{% endif %} - -{% if app_card %} - -{{ app_card }} - -{% endif %} -``` - -**Available variables:** -- `instruction` - User's goal/command -- `device_date` - Current device date/time -- `app_card` - App-specific guidance (if available) -- `state` - Current device state (accessibility tree) -- `history` - Action history - ---- - -## Path Resolution - -DroidRun uses a unified path resolution system for all file operations. - -### Resolution Order - -For relative paths, DroidRun searches: - -1. **Working directory**: `./path/to/file` -2. **Package directory**: `/path/to/file` - -Absolute paths are used as-is. - -### Examples - -```yaml -agent: - # Relative paths (checks working dir, then package dir) - prompts_dir: config/prompts - - app_cards: - app_cards_dir: config/app_cards - -credentials: - file_path: credentials.yaml - -# Absolute paths (used as-is) -agent: - prompts_dir: /home/user/my_prompts - app_cards: - app_cards_dir: /opt/droidrun/app_cards -``` - -### Custom Paths - -**Working directory structure:** - -``` -my_project/ -├── config.yaml -├── config/ -│ ├── prompts/ -│ │ ├── manager/ -│ │ └── executor/ -│ └── app_cards/ -│ ├── app_cards.json -│ └── gmail.md -└── credentials.yaml -``` - -All paths resolve from working directory first, falling back to package directory. - ---- - -## CLI Override Patterns - -CLI flags take precedence over config file settings. - -### Common Overrides - -```sh -# Override max steps -droidrun run "Open settings" --steps 30 - -# Enable reasoning mode -droidrun run "Complex task" --reasoning - -# Enable vision for all agents -droidrun run "Find the cat" --vision - -# Enable debug logging -droidrun run "Test command" --debug - -# Save action-level trajectory -droidrun run "Perform task" --save-trajectory action - -# Enable tracing -droidrun run "Debug issue" --tracing - -# Override device -droidrun run "Open settings" --device 192.168.1.100:5555 - -# Use TCP communication -droidrun run "Open settings" --tcp - -# Custom config file -droidrun run "Open settings" --config /path/to/config.yaml -``` - -### LLM Overrides - -```sh -# Override provider and model (applies to ALL agents) -droidrun run "Open settings" \ - --provider GoogleGenAI \ - --model models/gemini-2.5-flash - -# Override temperature -droidrun run "Open settings" --temperature 0.5 - -# Use Ollama locally -droidrun run "Open settings" \ - --provider Ollama \ - --model llama3.3:70b \ - --base_url http://localhost:11434 - -# Use OpenRouter -droidrun run "Open settings" \ - --provider OpenAILike \ - --model anthropic/claude-3.7-sonnet \ - --api_base https://openrouter.ai/api/v1 -``` - -### SDK Override Pattern (Recommended) - -For SDK users, override configuration by modifying DroidRunConfig: - -```python -from droidrun import DroidAgent -from droidrun.config_manager.config_manager import DroidRunConfig - -# Load config -config = DroidRunConfig.from_yaml("config.yaml") - -# Override settings before passing to DroidAgent -config.agent.manager.vision = True -config.agent.executor.vision = True -config.agent.codeact.vision = True -config.agent.reasoning = True -config.agent.max_steps = 30 -config.tracing.enabled = True - -# Pass modified config to DroidAgent -agent = DroidAgent(goal="Complex task", config=config) -handler = agent.run() -result: ResultEvent = await handler -``` - -**Alternative: Override via DroidAgent parameters:** - -```python -from droidrun import DroidAgent -from droidrun.config_manager.config_manager import DroidRunConfig, AgentConfig - -config = DroidRunConfig.from_yaml("config.yaml") - -# Override specific configs via DroidAgent parameters -agent = DroidAgent( - goal="Complex task", - config=config, # Base config - agent_config=AgentConfig(max_steps=30, reasoning=True), # Override -) -``` - -### CLI Override Pattern (Internal Use Only) - - -**For DroidRun maintainers only**: CLI overrides use `ConfigManager` singleton internally. SDK users should NOT use ConfigManager - use DroidRunConfig as shown above. - - -CLI overrides work via direct mutation of ConfigManager: - -```python -# INTERNAL CLI USE ONLY - DO NOT USE IN YOUR CODE -from droidrun.config_manager import ConfigManager - -# ConfigManager is a singleton used by CLI -config_manager = ConfigManager() -config = config_manager.config # Returns DroidRunConfig - -# CLI mutates config directly (bypasses thread safety after lock release) -if vision is not None: - config.agent.manager.vision = vision - config.agent.executor.vision = vision - config.agent.codeact.vision = vision -``` - ---- - -## Best Practices - -### 1. Version Control Config - -Store `config.yaml` in version control: - -```sh -git add config.yaml -git commit -m "Update LLM profiles" -``` - -**Exclude sensitive data:** - -```gitignore -# .gitignore -credentials.yaml -.env -``` - -### 2. Use Environment Variables for API Keys - -Never hardcode API keys in config: - -```yaml -# DON'T do this -llm_profiles: - manager: - kwargs: - api_key: sk-abc123 # BAD! - -# DO this instead -# Set environment variable: -# export GOOGLE_API_KEY=your-key-here -``` - -LlamaIndex automatically reads API keys from environment variables. - -### 3. Profile Naming - -Use consistent profile names: - -```yaml -llm_profiles: - manager: # Planning agent - executor: # Action agent - codeact: # Direct execution - text_manipulator: # Text editing - app_opener: # App launching - scripter: # Off-device code - structured_output: # Data extraction -``` - -These names are used throughout DroidRun and must match exactly. - -### 4. Test Configuration Changes - -Validate config after changes: - -```python -from droidrun.config_manager.config_manager import DroidRunConfig - -config = DroidRunConfig.from_yaml("config.yaml") - -# Check values -print(f"Max steps: {config.agent.max_steps}") -print(f"Manager LLM: {config.llm_profiles['manager'].model}") - -# Test LLM loading -from droidrun.agent.utils.llm_picker import load_llms_from_profiles - -llms = load_llms_from_profiles( - config.llm_profiles, - profile_names=["manager"] -) -print(f"Manager LLM loaded: {llms['manager']}") -``` - -### 5. Validate Config - -Validate configuration before using: - -```python -from droidrun.config_manager.config_manager import DroidRunConfig - -def validate_config(config): - """Custom validation logic.""" - if config.agent.max_steps < 5: - raise ValueError("max_steps must be at least 5") - - if config.agent.reasoning and not config.agent.manager.vision: - print("Warning: Reasoning mode works best with manager vision enabled") - -# Load and validate -config = DroidRunConfig.from_yaml("config.yaml") -validate_config(config) - -# Config is validated, safe to use -from droidrun import DroidAgent -agent = DroidAgent(goal="Task", config=config) -``` - -### 6. Separate Environments - -Use different configs for different environments: - -``` -project/ -├── config.dev.yaml # Development -├── config.staging.yaml # Staging -└── config.prod.yaml # Production -``` - -**Load specific config:** - -```sh -droidrun run "Open settings" --config config.dev.yaml -``` - -Or use environment variable: - -```sh -export DROIDRUN_CONFIG=config.prod.yaml -droidrun run "Open settings" -``` - -### 7. Document Custom Settings - -Add comments to custom configurations: - -```yaml -agent: - max_steps: 30 # Increased for complex workflows - - manager: - # Using Claude for best reasoning performance - # Cost: ~$0.02 per task - vision: true - -llm_profiles: - manager: - provider: Anthropic - model: claude-3-7-sonnet-latest - # Temperature tuned for consistent planning - temperature: 0.15 -``` - ---- - -## Troubleshooting - -### Config not found - -**Problem:** `FileNotFoundError: config.yaml not found` - -**Solution for SDK users:** -```python -from droidrun.config_manager.config_manager import DroidRunConfig -import yaml - -# Create default config -config = DroidRunConfig() - -# Save to file -with open("config.yaml", "w") as f: - yaml.dump(config.to_dict(), f) -``` - -**Solution for CLI users:** -1. Run `droidrun devices` to generate default config -2. Or specify path: `droidrun run --config /path/to/config.yaml` - -### CLI overrides not working - -**Problem:** CLI flags don't seem to override config. - -**Solution:** -- Ensure you're using the correct flag name -- Check CLI flag is before or after the command: - ```sh - # Correct - droidrun run "Open settings" --vision - - # Also correct - droidrun run --vision "Open settings" - ``` - -### LLM not loading - -**Problem:** `ModuleNotFoundError: llama_index.llms.google_genai` - -**Solution:** -Install the required LlamaIndex integration: - -```sh -pip install llama-index-llms-google-genai -# Or -pip install 'droidrun[google]' -``` - -### Invalid API key - -**Problem:** Authentication errors when using LLMs. - -**Solution:** -Set the correct environment variable: - -```sh -# Google Gemini -export GOOGLE_API_KEY=your-key - -# OpenAI -export OPENAI_API_KEY=your-key - -# Anthropic -export ANTHROPIC_API_KEY=your-key - -# DeepSeek -export DEEPSEEK_API_KEY=your-key -``` - -### Prompt file not found - -**Problem:** `FileNotFoundError: Prompt file not found` - -**Solution:** -1. Check `prompts_dir` path in config -2. Verify prompt file exists: `ls config/prompts/manager/system.jinja2` -3. Ensure correct filename in config (case-sensitive) - -### Safe execution blocking needed imports - -**Problem:** `ImportError: Module 'requests' is not allowed` - -**Solution:** -Add module to allowed list: - -```yaml -safe_execution: - allowed_modules: - - requests - - json - - re -``` - -Or disable safe execution (not recommended): - -```yaml -agent: - codeact: - safe_execution: false -``` - ---- - -## Related Documentation - -- [CLI Usage](/docs/v4/guides/cli) - DroidRun CLI command reference -- [App Cards](/docs/v4/guides/app-cards) - App-specific instruction cards -- [Agent Architecture](/docs/v3/concepts/agent) - How agents use configuration -- [LLM Integration](/docs/v3/concepts/models) - Supported LLM providers - ---- - -**Master DroidRun configuration for complete control over agent behavior!** diff --git a/docs/v4/guides/custom-tools-credentials.mdx b/docs/v4/guides/custom-tools-credentials.mdx index 1feffdb..4e0726d 100644 --- a/docs/v4/guides/custom-tools-credentials.mdx +++ b/docs/v4/guides/custom-tools-credentials.mdx @@ -1,43 +1,55 @@ --- title: 'Custom Tools & Credential Management' -description: 'Extend DroidRun agents with custom tools and secure credential storage' +description: 'Extend Droidrun with custom Python functions and secure credential management' --- -# Custom Tools & Credential Management + + -Extend DroidRun agents with custom tools and secure credential management. This guide is focused on practical implementation for developers. +## Overview + +Custom tools are Python functions that extend agent capabilities beyond built-in atomic actions (click, type, swipe). + +**Use cases:** +- External API calls (webhooks, REST services) +- Data processing and calculations +- Database operations +- Domain-specific logic --- ## Quick Start -### Custom Tools +### Basic Example + +Simple custom tool without device access: + ```python import asyncio -from droidrun import AdbTools, DroidAgent -from llama_index.llms.google_genai import GoogleGenAI +from droidrun import DroidAgent +from droidrun.config_manager import DroidrunConfig -def my_custom_tool(tool_instance, message: str) -> str: - """Custom tool that processes a message.""" - return f"Processed: {message}" +def calculate_tax(amount: float, rate: float, **kwargs) -> str: + """Calculate tax for a given amount.""" + tax = amount * rate + total = amount + tax + return f"Tax: ${tax:.2f}, Total: ${total:.2f}" custom_tools = { - "my_custom_tool": { - "arguments": ["message"], # List arguments (excluding tool_instance) - "description": 'Process a message. Usage: {"action": "my_custom_tool", "message": "hello"}', - "function": my_custom_tool, + "calculate_tax": { + "arguments": ["amount", "rate"], + "description": "Calculate tax for a given amount and rate", + "function": calculate_tax } } async def main(): - tools = AdbTools() - llm = GoogleGenAI(model="models/gemini-2.5-flash") + config = DroidrunConfig() agent = DroidAgent( - goal="Use my custom tool to process 'Hello World'", - llm=llm, - tools=tools, - custom_tools=custom_tools # Pass custom tools here + goal="Calculate tax for $100 at 8% rate", + config=config, + custom_tools=custom_tools ) result = await agent.run() @@ -46,170 +58,145 @@ async def main(): asyncio.run(main()) ``` -### Credentials -```python -# Method 1: Direct dict (in-memory) -credentials = { - "MY_PASSWORD": "secret123", - "API_KEY": "sk-1234567890" -} - -agent = DroidAgent( - goal="Login to my app", - llm=llm, - tools=tools, - credentials=credentials # Pass credentials directly -) - -# Method 2: YAML file (config.yaml) -# credentials: -# enabled: true -# file_path: credentials.yaml -``` - --- -## Overview +## Tool Structure -### Custom Tools -User-defined Python functions that extend agent capabilities beyond built-in atomic actions (click, type, swipe, etc.). - -**Use cases:** -- Integrate with external APIs (webhooks, REST services) -- Perform complex computations or data processing -- Access third-party services (databases, cloud APIs) -- Implement domain-specific actions - -**Key features:** -- Passed via `custom_tools` parameter to `DroidAgent` -- Automatically merged with atomic actions and credential tools -- Available to all agents (Manager, Executor, CodeAct) -- Support both sync and async functions - -### Credential Management -Secure storage and retrieval of sensitive data like passwords, API keys, and tokens. - -**Key features:** -- Stored in YAML files or passed as in-memory dicts -- Never logged or exposed in output -- Automatically injected as `type_secret` custom tool -- Accessible to all agents via credential manager - -**Security Notes:** -- Credentials are NOT encrypted by DroidRun -- Always add `credentials.yaml` to `.gitignore` -- Use external encryption (GPG) or environment variables for production -- Secret values are never logged, only secret IDs - ---- - -## Custom Tools - -### Architecture - -Custom tools are defined as Python dictionaries following this structure: +All custom tools follow this format: ```python custom_tools = { "tool_name": { - "arguments": ["arg1", "arg2"], # List of parameter names (required) - "description": "Tool description with usage example", # For LLM prompt (required) - "function": callable_function # Python function to execute (required) + "arguments": ["arg1", "arg2"], # Parameter names + "description": "Tool description...", # For LLM prompt + "function": callable_function # Python function } } ``` +**Function signature:** +```python +def tool_name(arg1: type, arg2: type, *, tools=None, shared_state=None, **kwargs) -> str: + """ + Args: + arg1: Your parameter + arg2: Another parameter + tools: Tools instance (optional, injected automatically) + shared_state: DroidAgentState (optional, injected automatically) + """ + # Implementation + return "result" +``` + **Key points:** -- `arguments`: List of parameter names (excluding the required `tool_instance` first parameter) -- `description`: Clear description with JSON usage example for the LLM -- `function`: Python callable (sync or async) +- List only user arguments in `"arguments"` (not `tools` or `shared_state`) +- `tools` and `shared_state` are injected automatically as keyword arguments +- Use `**kwargs` for forward compatibility +- Return type should be `str` -The custom tool system merges seamlessly with atomic actions: -- **Atomic actions** (click, type, swipe, etc.) are always available -- **Custom tools** are added on top via the `custom_tools` parameter -- **Credential tools** (`type_secret`) are auto-injected when credentials are provided -- All agents (Manager, Executor, CodeAct) can use custom tools +--- -### Creating Custom Tools +## Using Tools Instance -#### Basic Example: Webhook Integration +Access device via the `tools` parameter: ```python -#!/usr/bin/env python3 -import asyncio -import requests -from droidrun import AdbTools, DroidAgent -from llama_index.llms.google_genai import GoogleGenAI +def screenshot_and_count(*, tools=None, shared_state=None, **kwargs) -> str: + """Take screenshot and count UI elements.""" + if not tools: + return "Error: tools instance required" -def send_webhook(tool_instance, url: str, data: str) -> str: - """ - Send data to a webhook URL. + # Take screenshot + screenshot_path, screenshot_bytes = tools.take_screenshot() - Args: - tool_instance: Tools instance (required by DroidRun) - url: Webhook URL - data: Data to send (JSON string or plain text) + # Get UI state + state = tools.get_state() + element_count = len(state.get("ui_elements", [])) - Returns: - Result message - """ - try: - response = requests.post(url, json={"data": data}, timeout=10) - response.raise_for_status() - return f"Webhook sent successfully. Status: {response.status_code}" - except Exception as e: - return f"Error sending webhook: {str(e)}" + return f"Screenshot saved. Found {element_count} UI elements" -# Define custom tool custom_tools = { - "send_webhook": { - "arguments": ["url", "data"], - "description": 'Send data to a webhook URL. Usage: {"action": "send_webhook", "url": "https://webhook.site/abc123", "data": "your data here"}', - "function": send_webhook, + "screenshot_and_count": { + "arguments": [], + "description": "Take screenshot and count UI elements on screen", + "function": screenshot_and_count } } - -async def main(): - tools = AdbTools() - llm = GoogleGenAI(model="models/gemini-2.5-flash") - - agent = DroidAgent( - goal="Check my Gmail unread count and send it to webhook https://webhook.site/abc123", - llm=llm, - tools=tools, - custom_tools=custom_tools # Pass custom tools here - ) - - result = await agent.run() - print(f"Success: {result.success}") - print(f"Reason: {result.reason}") - -if __name__ == "__main__": - asyncio.run(main()) ``` -#### Advanced Example: API Integration +**Available via `tools`:** +- `tools.take_screenshot()` - Capture screen +- `tools.get_state()` - Get UI hierarchy +- `tools.tap_by_index(index)` - Tap element +- `tools.input_text(text, index)` - Type text +- `tools.swipe(x1, y1, x2, y2)` - Swipe gesture +- All methods from AdbTools/IOSTools + +--- + +## Using Shared State + +Access agent state via `shared_state`: + +```python +def check_action_history(action_name: str, *, tools=None, shared_state=None, **kwargs) -> str: + """Check if action was recently performed.""" + if not shared_state: + return "Error: shared_state required" + + # Check recent actions + recent_actions = shared_state.action_history[-5:] + already_done = any(a.get("action") == action_name for a in recent_actions) + + if already_done: + return f"Action '{action_name}' was already performed recently" + + # Check step count + if shared_state.step_number > 10: + return "Warning: Task taking too many steps" + + # Access memory + if "skip_validation" in shared_state.memory: + return "Validation skipped per memory" + + return f"Action '{action_name}' not yet performed" + +custom_tools = { + "check_action_history": { + "arguments": ["action_name"], + "description": "Check if a specific action was recently performed in agent history", + "function": check_action_history + } +} +``` + +**DroidAgentState fields:** +- `step_number` - Current execution step +- `action_history` - List of executed actions +- `action_outcomes` - Success/failure per action +- `memory` - Agent memory dict +- `custom_variables` - User-provided variables +- `visited_packages` - Apps visited +- `current_package_name` - Current app package +- `plan` - Current Manager plan +- More in `droidrun/agent/droid/events.py` + +--- + +## Common Patterns + +### API Integration ```python -import json import requests -from typing import Dict -def fetch_weather(tool_instance, city: str) -> str: - """ - Fetch weather data for a city. - - Args: - tool_instance: Tools instance - city: City name - - Returns: - Weather information as string - """ +def fetch_weather(city: str, **kwargs) -> str: + """Fetch weather data from API.""" try: - # Example using OpenWeatherMap API - api_key = "your_api_key_here" # Or use credential manager + # Using OpenWeatherMap API example + api_key = "your_api_key" url = f"https://api.openweathermap.org/data/2.5/weather?q={city}&appid={api_key}" + response = requests.get(url, timeout=10) response.raise_for_status() @@ -219,250 +206,236 @@ def fetch_weather(tool_instance, city: str) -> str: return f"Weather in {city}: {weather}, {temp:.1f}°C" except Exception as e: - return f"Error fetching weather: {str(e)}" - -def save_to_database(tool_instance, table: str, data: str) -> str: - """ - Save data to a database. - - Args: - tool_instance: Tools instance - table: Table name - data: JSON data to save - - Returns: - Result message - """ - try: - # Parse JSON data - json_data = json.loads(data) - - # Example database operation (pseudo-code) - # db.insert(table, json_data) - - return f"Data saved to {table}: {len(json_data)} records" - except json.JSONDecodeError: - return "Error: Invalid JSON data" - except Exception as e: - return f"Error saving to database: {str(e)}" + return f"Error: {str(e)}" custom_tools = { "fetch_weather": { "arguments": ["city"], - "description": 'Fetch current weather for a city. Usage: {"action": "fetch_weather", "city": "London"}', - "function": fetch_weather, - }, - "save_to_database": { - "arguments": ["table", "data"], - "description": 'Save JSON data to database table. Usage: {"action": "save_to_database", "table": "users", "data": "{\\"name\\": \\"John\\", \\"age\\": 30}"}', - "function": save_to_database, + "description": "Fetch current weather data for a given city", + "function": fetch_weather } } ``` -### Custom Tool Function Signature - -All custom tool functions **must** follow this signature: +### Database Query ```python -def tool_function(tool_instance: Tools, arg1: type, arg2: type, ...) -> str: - """ - Tool description. +import sqlite3 - Args: - tool_instance: Tools instance (REQUIRED as first parameter) - arg1: Description of argument 1 - arg2: Description of argument 2 +def query_database(query: str, **kwargs) -> str: + """Query local database.""" + try: + conn = sqlite3.connect("app.db") + cursor = conn.execute(query) + results = cursor.fetchall() + conn.close() - Returns: - Result message (string) - """ - # Implementation - return "result" -``` - -**Critical Requirements:** - -1. **First parameter MUST be `tool_instance`**: This is the `Tools` instance (AdbTools or IOSTools) that DroidRun passes automatically. Even if you don't use it, it must be the first parameter. - -2. **List only user arguments in `arguments`**: The `arguments` list in your custom tool definition should NOT include `tool_instance`. Only list the arguments that the LLM will provide (arg1, arg2, etc.). - -3. **Return type should be `str`**: Agents expect string responses for all tool calls. - -4. **Async functions are supported**: Use `async def` for async operations. - -5. **Handle exceptions**: Always catch exceptions and return error messages as strings. - -**Example:** -```python -def my_tool(tool_instance, message: str) -> str: - # tool_instance is passed by DroidRun automatically - return f"Processed: {message}" + return f"Found {len(results)} results" + except Exception as e: + return f"Database error: {str(e)}" custom_tools = { - "my_tool": { - "arguments": ["message"], # Only list user arguments, NOT tool_instance - "description": 'Process a message. Usage: {"action": "my_tool", "message": "hello"}', - "function": my_tool, + "query_database": { + "arguments": ["query"], + "description": "Execute SQL query on local database and return results", + "function": query_database } } ``` -### Async Custom Tools - -For async operations, use `async def`: +### Async Operations ```python -import asyncio import aiohttp -async def fetch_async(tool_instance, url: str) -> str: +async def fetch_async(url: str, **kwargs) -> str: """Fetch data asynchronously.""" try: async with aiohttp.ClientSession() as session: async with session.get(url, timeout=10) as response: data = await response.text() - return f"Fetched {len(data)} bytes" + return f"Fetched {len(data)} bytes from {url}" except Exception as e: return f"Error: {str(e)}" custom_tools = { "fetch_async": { "arguments": ["url"], - "description": 'Fetch data from URL asynchronously. Usage: {"action": "fetch_async", "url": "https://api.example.com/data"}', - "function": fetch_async, + "description": "Asynchronously fetch data from a URL", + "function": fetch_async } } ``` -### How Custom Tools Work +--- -1. **Tool Registration**: Custom tools are passed to `DroidAgent` via the `custom_tools` parameter -2. **Tool Merging**: Custom tools are merged with: - - Atomic actions (click, type, swipe, etc.) - - Auto-generated credential tools (type_secret) if credentials are provided - - Built-in helper tools (open_app) -3. **Prompt Injection**: Tool descriptions are injected into agent prompts -4. **Execution**: When an agent selects a custom tool, DroidRun calls the function with: - - `tool_instance` (injected automatically by DroidRun) - - User arguments from the LLM's action +## Best Practices + +### 1. Clear Descriptions +Write descriptive, specific descriptions: -**Internal Flow Example:** ```python -# 1. Agent receives tool description in prompt: -# "send_webhook(url, data): Send data to webhook URL. Usage: {...}" +# Good +"description": "Send POST request to webhook URL with JSON data payload" -# 2. Agent decides to use tool and outputs: -# {"action": "send_webhook", "url": "https://api.example.com", "data": "hello"} - -# 3. DroidRun internally calls your function: -# result = send_webhook( -# tool_instance=tools_instance, # Injected by DroidRun -# url="https://api.example.com", # From agent output -# data="hello" # From agent output -# ) - -# 4. Result is returned to agent for next decision +# Bad +"description": "Send webhook" ``` -**Key Point:** The `tool_instance` parameter is NOT visible to the LLM. It's automatically injected by DroidRun's execution layer. +### 2. Error Handling +Always catch exceptions: -### Best Practices for Custom Tools +```python +def robust_tool(url: str, **kwargs) -> str: + try: + response = requests.get(url, timeout=10) + response.raise_for_status() + return f"Success: {response.status_code}" + except requests.Timeout: + return "Error: Request timed out" + except requests.RequestException as e: + return f"Error: {str(e)}" + except Exception as e: + return f"Unexpected error: {str(e)}" +``` - - - Write detailed descriptions with usage examples for the LLM: +### 3. Argument Validation +Validate inputs before processing: - ```python - # Good - "description": 'Send POST request to webhook URL with JSON payload. Usage: {"action": "send_webhook", "url": "https://webhook.site/abc", "data": "{\\"key\\": \\"value\\"}"}', +```python +def validated_tool(count: int, **kwargs) -> str: + if not isinstance(count, int): + return "Error: count must be integer" + if count < 0 or count > 100: + return "Error: count must be 0-100" - # Bad - "description": "Send webhook", - ``` - + return f"Processed {count} items" +``` - - Always catch exceptions and return meaningful error messages: +### 4. Logging +Use Python logging for debugging: - ```python - def robust_tool(tool_instance, url: str) -> str: - try: - response = requests.get(url, timeout=10) - response.raise_for_status() - return f"Success: {response.status_code}" - except requests.Timeout: - return "Error: Request timed out after 10 seconds" - except requests.RequestException as e: - return f"Error: HTTP request failed - {str(e)}" - except Exception as e: - return f"Error: Unexpected error - {str(e)}" - ``` - +```python +import logging +logger = logging.getLogger("droidrun") - - Validate inputs before processing: - - ```python - def validated_tool(tool_instance, count: int) -> str: - if not isinstance(count, int): - return "Error: count must be an integer" - if count < 0 or count > 100: - return "Error: count must be between 0 and 100" - - # Process valid input - return f"Processed {count} items" - ``` - - - - Add timeouts to network operations: - - ```python - def api_call(tool_instance, endpoint: str) -> str: - try: - response = requests.get(endpoint, timeout=10) # Always set timeout - return response.text - except requests.Timeout: - return "Error: API call timed out" - ``` - - - - Use Python logging for debugging (logs appear in DroidRun output): - - ```python - import logging - logger = logging.getLogger("droidrun") - - def logged_tool(tool_instance, data: str) -> str: - logger.info(f"Processing data: {data[:50]}...") # Log first 50 chars - # Process data - logger.info("Processing complete") - return "Success" - ``` - - +def logged_tool(data: str, **kwargs) -> str: + logger.info(f"Processing: {data[:50]}...") + # Process data + logger.info("Complete") + return "Success" +``` --- -## Credential Management +## Advanced Example -### Overview +Combining tools instance, shared state, and credentials: -The credential manager provides secure storage for sensitive data like passwords, API keys, and authentication tokens. Credentials are: -- **Stored in plain text** (YAML or in-memory dict - not encrypted) -- **Never logged** or exposed in output -- **Automatically available** as the `type_secret` custom tool -- **Access-logged** for audit purposes (logs secret ID, not value) +```python +import requests -**Important:** DroidRun does NOT encrypt credentials. Use external encryption (GPG), environment variables, or secret management systems for production deployments. +def send_authenticated_request( + url: str, + data: str, + *, + tools=None, + shared_state=None, + **kwargs +) -> str: + """Send authenticated API request with credential.""" + try: + # Access credentials via tools instance + if not tools or not hasattr(tools, 'credential_manager'): + return "Error: Credential manager not available" -### Setting Up Credentials + api_key = tools.credential_manager.get_credential("API_KEY") -#### Method 1: Configuration File (Recommended) + # Check if we've made too many requests + if shared_state and shared_state.step_number > 15: + return "Error: Too many API calls" -1. **Create credentials file** (`credentials.yaml`): + # Send authenticated request + headers = {"Authorization": f"Bearer {api_key}"} + response = requests.post(url, json={"data": data}, headers=headers, timeout=10) + response.raise_for_status() + + return f"Request successful: {response.status_code}" + except Exception as e: + return f"Error: {str(e)}" + +custom_tools = { + "send_authenticated_request": { + "arguments": ["url", "data"], + "description": "Send authenticated API request using stored credentials", + "function": send_authenticated_request + } +} + +# Usage with credentials +credentials = {"API_KEY": "sk-1234567890"} + +agent = DroidAgent( + goal="Send data to API", + config=config, + custom_tools=custom_tools, + credentials=credentials +) +``` + +--- + +## Related + +See [Agent Architecture](/v4/concepts/agent-architecture) for understanding shared state and custom tools integration. + + + + + +## Overview + +Secure storage for passwords, API keys, and tokens. + +**Features:** +- Stored in YAML files or in-memory dicts +- Never logged or exposed +- Auto-injected as `type_secret` tool +- Simple string or dict format + +## Quick Start + +### Method 1: In-Memory (Recommended for SDK) + +```python +import asyncio +from droidrun import DroidAgent +from droidrun.config_manager import DroidrunConfig + +async def main(): + # Define credentials directly + credentials = { + "MY_PASSWORD": "secret123", + "API_KEY": "sk-1234567890" + } + + config = DroidrunConfig() + + agent = DroidAgent( + goal="Login to my app", + config=config, + credentials=credentials # Pass directly + ) + + result = await agent.run() + print(result.success) + +asyncio.run(main()) +``` + +### Method 2: YAML File + +1. **Create credentials file:** ```yaml # credentials.yaml @@ -476,532 +449,166 @@ secrets: value: "gmail_pass_123" enabled: true - API_KEY: - value: "sk-1234567890abcdef" - enabled: true - # Simple string format (auto-enabled) - WEBHOOK_TOKEN: "webhook_secret_token" + API_KEY: "sk-1234567890abcdef" - # Disabled secret (not loaded) + # Disabled secret OLD_PASSWORD: value: "old_pass" - enabled: false # This secret will NOT be available + enabled: false # Not loaded ``` -2. **Enable in config.yaml**: +2. **Enable in config.yaml:** ```yaml # config.yaml credentials: enabled: true - file_path: credentials.yaml # Path relative to working directory + file_path: credentials.yaml ``` -3. **Use in your script**: +3. **Use in code:** ```python -#!/usr/bin/env python3 -import asyncio -from droidrun import AdbTools, DroidAgent -from droidrun.config_manager.config import DroidRunConfig -from llama_index.llms.google_genai import GoogleGenAI +from droidrun import DroidAgent +from droidrun.config_manager import DroidrunConfig -async def main(): - # Initialize config and enable credentials from file - config = DroidRunConfig() - config.credentials.enabled = True - config.credentials.file_path = "credentials.yaml" +# Config loads credentials from file +config = DroidrunConfig.from_yaml("config.yaml") - tools = AdbTools() - llm = GoogleGenAI(model="models/gemini-2.5-flash") - - agent = DroidAgent( - goal="Login to my Gmail account", - llm=llm, - tools=tools, - config=config # Credentials loaded from config - ) - - result = await agent.run() - print(f"Success: {result.success}") - -if __name__ == "__main__": - asyncio.run(main()) -``` - -#### Method 2: In-Memory Credentials (Programmatic) - -Pass credentials directly as a dictionary: - -```python -#!/usr/bin/env python3 -import asyncio -from droidrun import AdbTools, DroidAgent -from llama_index.llms.google_genai import GoogleGenAI - -async def main(): - # Define credentials in-memory - credentials = { - "MY_PASSWORD": "secret123", - "API_KEY": "sk-1234567890abcdef" - } - - tools = AdbTools() - llm = GoogleGenAI(model="models/gemini-2.5-flash") - - agent = DroidAgent( - goal="Login to my account", - llm=llm, - tools=tools, - credentials=credentials # Pass directly - ) - - result = await agent.run() - print(f"Success: {result.success}") - -if __name__ == "__main__": - asyncio.run(main()) -``` - -### Using Credentials in Agents - -When credentials are provided (via config or parameter), the `type_secret` custom tool is **automatically injected** by DroidRun. You don't need to define it manually. - -#### How Agents Use Credentials - -The agent receives available secret IDs in the system prompt: - -``` -## Available Secrets: -The credential manager has the following secret IDs available for use with the `type_secret` function: -- MY_PASSWORD -- GMAIL_PASSWORD -- API_KEY - -Use `type_secret(secret_id, index)` to type these secrets into input fields without exposing their values. -``` - -The agent then uses `type_secret` to input credentials: - -```python -# Agent's generated code (CodeAct mode) -type_secret("MY_PASSWORD", index=5) -``` - -**Note:** The `type_secret` tool is built into DroidRun (see `droidrun/agent/utils/tools.py` → `build_credential_tools()`). It's automatically added when you pass credentials to `DroidAgent`. - -#### Example: Login Automation - -```python -#!/usr/bin/env python3 -import asyncio -from droidrun import AdbTools, DroidAgent -from llama_index.llms.google_genai import GoogleGenAI - -async def main(): - # Define credentials - credentials = { - "EMAIL_USER": "user@example.com", - "EMAIL_PASS": "secret_password_123" - } - - tools = AdbTools() - llm = GoogleGenAI(model="models/gemini-2.5-flash") - - agent = DroidAgent( - goal="Open Gmail app and login with my credentials", - llm=llm, - tools=tools, - credentials=credentials - ) - - result = await agent.run() - print(f"Success: {result.success}") - print(f"Result: {result['reason']}") - -if __name__ == "__main__": - asyncio.run(main()) -``` - -**What the agent does internally:** -1. Opens Gmail app using `open_app("Gmail")` -2. Clicks on email input field: `click(index=3)` -3. Types email using regular type: `type("user@example.com", index=3)` -4. Clicks on password field: `click(index=5)` -5. Types password securely: `type_secret("EMAIL_PASS", index=5)` -6. Clicks login button: `click(index=7)` - -### Credential Manager API - -If you need direct access to credentials (e.g., for custom tools): - -```python -from droidrun.credential_manager import CredentialManager - -# Load from file -cm = CredentialManager(credentials_path="credentials.yaml") - -# Load from dict -cm = CredentialManager(credentials_dict={"PASSWORD": "secret123"}) - -# Get secret value -password = cm.get_credential("PASSWORD") - -# List available secrets -secret_ids = cm.list_available_secrets() # ["PASSWORD", "API_KEY", ...] - -# Check if secret exists -if cm.has_credential("API_KEY"): - api_key = cm.get_credential("API_KEY") -``` - -### Security Features - - - - Secret values are **never** logged or displayed: - - ``` - ✅ Logged: "🔑 Accessing secret: 'MY_PASSWORD'" - ❌ Never logged: "MY_PASSWORD value is: secret123" - ``` - - - - Error messages show secret IDs but never values: - - ``` - ✅ Error message: "Secret 'WRONG_ID' not found. Available: ['MY_PASSWORD', 'API_KEY']" - ❌ Never shown: "Secret value 'secret123' is invalid" - ``` - - - - DroidRun does NOT encrypt credentials. For production, use external encryption: - - ```bash - # Option 1: GPG encryption - gpg --encrypt --recipient you@example.com credentials.yaml - # Decrypt before running - gpg --decrypt credentials.yaml.gpg > credentials.yaml - droidrun run "your command" - rm credentials.yaml # Clean up after run - - # Option 2: Use a secret manager (AWS Secrets Manager, HashiCorp Vault, etc.) - # Option 3: Environment variables (see Accordion 5) - ``` - - **Why external encryption?** Credentials are stored in plain text YAML. Always encrypt sensitive files or use environment variables. - - - - For production deployments, consider these security measures: - - ```python - import os - from droidrun import DroidAgent - - # Option 1: Environment variables (recommended) - credentials = { - "PASSWORD": os.environ.get("APP_PASSWORD"), - "API_KEY": os.environ.get("API_KEY") - } - - # Option 2: Secret management services - # Use AWS Secrets Manager, HashiCorp Vault, etc. - # from your_secret_manager import get_secret - # credentials = {"PASSWORD": get_secret("app_password")} - - agent = DroidAgent(goal="...", credentials=credentials) - ``` - - **Always** add `credentials.yaml` to `.gitignore`: - - ```bash - # .gitignore - credentials.yaml - credentials_*.yaml - *.credentials.yaml - ``` - - - - **Always** add credential files to `.gitignore`: - - ```bash - # .gitignore - credentials.yaml - credentials_*.yaml - *.credentials.yaml - .env - secrets/ - ``` - - Never commit secrets to version control! - - - ---- - -## Combining Custom Tools and Credentials - -Custom tools can access credentials via the credential manager attached to the `tool_instance`: - -```python -#!/usr/bin/env python3 -import asyncio -import requests -from droidrun import AdbTools, DroidAgent -from llama_index.llms.google_genai import GoogleGenAI - -def send_authenticated_webhook(tool_instance, url: str, data: str) -> str: - """ - Send data to webhook with API key authentication. - - Args: - tool_instance: Tools instance (required first parameter) - url: Webhook URL - data: Data to send - - Returns: - Result message - """ - try: - # Access credential manager from tools instance - if not hasattr(tool_instance, 'credential_manager') or tool_instance.credential_manager is None: - return "Error: Credential manager not available" - - # Get API key from credential manager - api_key = tool_instance.credential_manager.get_credential("WEBHOOK_API_KEY") - - # Send authenticated request - headers = {"Authorization": f"Bearer {api_key}"} - response = requests.post(url, json={"data": data}, headers=headers, timeout=10) - response.raise_for_status() - - return f"Webhook sent successfully. Status: {response.status_code}" - except Exception as e: - return f"Error: {str(e)}" - -async def main(): - # Define credentials - credentials = { - "WEBHOOK_API_KEY": "sk-webhook-secret-key", - "EMAIL_PASSWORD": "email_pass_123" - } - - # Define custom tool - custom_tools = { - "send_authenticated_webhook": { - "arguments": ["url", "data"], - "description": 'Send authenticated webhook. Usage: {"action": "send_authenticated_webhook", "url": "https://api.example.com/webhook", "data": "message"}', - "function": send_authenticated_webhook, - } - } - - tools = AdbTools() - llm = GoogleGenAI(model="models/gemini-2.5-flash") - - agent = DroidAgent( - goal="Check my unread Gmail count and send to webhook https://api.example.com/webhook", - llm=llm, - tools=tools, - credentials=credentials, - custom_tools=custom_tools - ) - - result = await agent.run() - print(f"Success: {result.success}") - -if __name__ == "__main__": - asyncio.run(main()) -``` - ---- - -## Custom Variables vs Credentials - -DroidRun supports both **credentials** (sensitive) and **variables** (non-sensitive): - -| Feature | Credentials | Variables | -|---------|------------|-----------| -| **Purpose** | Passwords, API keys, secrets | Non-sensitive data (emails, names, config) | -| **Storage** | YAML file or in-memory dict | In-memory dict only | -| **Logging** | Never logged (only secret IDs) | May appear in logs/prompts | -| **Access** | Via `type_secret` tool | Directly in shared state | -| **Security** | Protected, sanitized errors | No special protection | - -**Example: Using Variables** -```python -import asyncio -from droidrun import AdbTools, DroidAgent -from llama_index.llms.google_genai import GoogleGenAI - -async def main(): - # Define custom variables (non-sensitive data) - variables = { - "target_email": "john@example.com", - "subject_line": "Monthly Report", - "cc_recipients": ["alice@example.com", "bob@example.com"] - } - - tools = AdbTools() - llm = GoogleGenAI(model="models/gemini-2.5-flash") - - agent = DroidAgent( - goal="Compose email to {{target_email}} with subject {{subject_line}}", - llm=llm, - tools=tools, - variables=variables # Pass non-sensitive variables - ) - - result = await agent.run() - print(result.success, result.reason) - -asyncio.run(main()) -``` - -**When to use:** -- Use **credentials** for passwords, API keys, tokens (sensitive) -- Use **variables** for emails, names, configuration (non-sensitive) - ---- - -## CLI Usage - -### With Configuration File - -```bash -# Enable credentials in config.yaml -# credentials: -# enabled: true -# file_path: credentials.yaml - -droidrun run "Login to my Gmail account" --provider GoogleGenAI --model gemini-2.5-flash -``` - -### With Custom Tools (SDK Only) - -Custom tools and programmatic credentials are only available via the SDK (not CLI). - ---- - -## Troubleshooting - - - - **Cause**: Credentials not enabled or file not found - - **Solution**: - ```yaml - # config.yaml - credentials: - enabled: true # Must be true - file_path: credentials.yaml # Check path is correct - ``` - - Or pass credentials programmatically: - ```python - agent = DroidAgent(..., credentials={"PASSWORD": "secret"}) - ``` - - - - **Cause**: Secret ID doesn't exist or is disabled - - **Solution**: - ```python - from droidrun.credential_manager import CredentialManager - - cm = CredentialManager(credentials_path="credentials.yaml") - print(cm.list_available_secrets()) # Check available secrets - ``` - - Check `credentials.yaml`: - ```yaml - secrets: - X: - value: "your_value" - enabled: true # Must be true - ``` - - - - **Cause**: Tool not registered or incorrect signature - - **Solution**: - 1. Check tool signature has `tool_instance` as first param: - ```python - def my_tool(tool_instance, arg1: str) -> str: # ✅ Correct - pass - - def my_tool(arg1: str) -> str: # ❌ Wrong - pass - ``` - - 2. Verify custom_tools format: - ```python - custom_tools = { - "tool_name": { - "arguments": ["arg1"], - "description": "...", - "function": my_tool # Function reference, not call - } - } - ``` - - 3. Check agent received custom tools: - ```python - agent = DroidAgent(..., custom_tools=custom_tools) # Pass here - ``` - - - - **Cause**: LLM doesn't understand tool description - - **Solution**: Improve description with clear usage example: - ```python - # Bad - "description": "Send webhook" - - # Good - "description": 'Send POST request to webhook. Usage: {"action": "send_webhook", "url": "https://webhook.site/abc123", "data": "your message"}' - ``` - - - ---- - -## API Reference - -For more technical details: - -### Custom Tools -- Implementation: `droidrun/agent/utils/tools.py` → `build_custom_tools()` -- Format: `{"tool_name": {"arguments": [...], "description": "...", "function": callable}}` -- Used in: `DroidAgent`, `CodeActAgent`, `ExecutorAgent`, `ManagerAgent` - -### Credential Manager -- Implementation: `droidrun/credential_manager/credential_manager.py` -- Loader: `droidrun/credential_manager/credential_loader.py` → `load_credential_manager()` -- Type Secret Tool: `droidrun/agent/utils/tools.py` → `type_secret()`, `build_credential_tools()` - -### DroidAgent Parameters -```python -DroidAgent( - goal: str, - custom_tools: dict = None, # Custom tool definitions - credentials: CredentialsConfig | dict | None = None, # Credentials config or direct dict - variables: dict | None = None, # Non-sensitive variables - # ... other parameters +agent = DroidAgent( + goal="Login to Gmail", + config=config # Credentials loaded automatically ) ``` --- -## Next Steps +## How Agents Use Credentials -- Learn about [Agent Architecture](/v4/concepts/architecture) -- Explore [DroidAgent SDK Reference](/v4/sdk/droid-agent) -- See [Configuration Guide](/v4/guides/configuration) +When credentials are provided, the `type_secret` action is **automatically available**: + +### Executor/Manager Mode +```json +{ + "action": "type_secret", + "secret_id": "MY_PASSWORD", + "index": 5 +} +``` + +### CodeAct Mode +```python +type_secret("MY_PASSWORD", index=5) +``` + +The agent never sees the actual value - only the secret ID. + +--- + +## Example: Login Automation + +```python +import asyncio +from droidrun import DroidAgent +from droidrun.config_manager import DroidrunConfig + +async def main(): + credentials = { + "EMAIL_USER": "user@example.com", + "EMAIL_PASS": "secret_password" + } + + config = DroidrunConfig() + + agent = DroidAgent( + goal="Open Gmail and login with my credentials", + config=config, + credentials=credentials + ) + + result = await agent.run() + print(f"Success: {result.success}") + +asyncio.run(main()) +``` + +**What the agent does:** +1. Opens Gmail: `open_app("Gmail")` +2. Clicks email field: `click(index=3)` +3. Types email: `type("user@example.com", index=3)` +4. Clicks password field: `click(index=5)` +5. Types password securely: `type_secret("EMAIL_PASS", index=5)` +6. Clicks login: `click(index=7)` + +## Credentials vs Variables + +| Feature | Credentials | Variables | +|---------|------------|-----------| +| **Purpose** | Passwords, API keys | Non-sensitive data | +| **Storage** | YAML or in-memory | In-memory only | +| **Logging** | Never logged | May appear in logs | +| **Access** | Via `type_secret` tool | In shared state | +| **Security** | Protected | No protection | + +**Example: Using Variables** +```python +variables = { + "target_email": "john@example.com", + "subject_line": "Monthly Report" +} + +agent = DroidAgent( + goal="Compose email to {{target_email}}", + config=config, + variables=variables # Non-sensitive +) +``` + +--- + +## Troubleshooting + +### Error: Credential manager not initialized + +**Solution:** +```yaml +# config.yaml +credentials: + enabled: true # Must be true + file_path: credentials.yaml +``` + +Or: +```python +agent = DroidAgent(..., credentials={"PASSWORD": "secret"}) +``` + +### Error: Secret 'X' not found + +**Check available secrets:** +```python +from droidrun.credential_manager import CredentialManager + +cm = CredentialManager(credentials_path="credentials.yaml") +print(cm.list_available_secrets()) +``` + +**Verify in YAML:** +```yaml +secrets: + X: + value: "your_value" + enabled: true # Must be true +``` + +--- + +## Related + +See [Configuration Guide](/v4/sdk/configuration) for credential setup. + +See [Custom Variables](/v4/guides/custom-variables) for non-sensitive data. + + + diff --git a/docs/v4/guides/custom-variables.mdx b/docs/v4/guides/custom-variables.mdx index 7a20491..423dcd6 100644 --- a/docs/v4/guides/custom-variables.mdx +++ b/docs/v4/guides/custom-variables.mdx @@ -1,15 +1,15 @@ --- title: 'Custom Variables' -description: 'Pass dynamic data and configuration to your DroidRun agents' +description: 'Pass dynamic data and configuration to your Droidrun agents' --- # Custom Variables -Pass dynamic data to your DroidRun agents using the `variables` parameter. Variables enable parameterized workflows and reusable automation. +Pass dynamic data to your Droidrun agents using the `variables` parameter. Variables enable parameterized workflows and reusable automation. - -**Important Limitation**: Custom variables are **only accessible via agent prompts**, not directly to custom tools. Tools don't have access to `shared_state.custom_variables`. See [workaround below](#accessing-variables-in-custom-tools). - +Custom variables are accessible in: +- **Agent prompts** via custom Jinja2 templates +- **Custom tools** via `shared_state.custom_variables` --- @@ -17,8 +17,7 @@ Pass dynamic data to your DroidRun agents using the `variables` parameter. Varia ```python from droidrun.agent.droid import DroidAgent -from droidrun.tools import AdbTools -from droidrun.config_manager.config_manager import DroidRunConfig +from droidrun.config_manager.config_manager import DroidrunConfig # Define custom prompts that render variables custom_prompts = { @@ -35,12 +34,11 @@ Available variables: } # Create agent with variables -config = DroidRunConfig() +config = DroidrunConfig() agent = DroidAgent( goal="Send email to recipient with subject", config=config, - tools=AdbTools(), variables={"recipient": "john@example.com", "subject": "Update"}, prompts=custom_prompts # Required to see variables ) @@ -64,7 +62,7 @@ When you pass `variables` to `DroidAgent`: **Access Summary:** - ✅ Agent prompts (via custom Jinja2 templates) - ✅ Agents can read and pass to tools as arguments -- ❌ NOT directly accessible to custom tools +- ✅ Custom tools (via `shared_state.custom_variables`) --- @@ -72,13 +70,12 @@ When you pass `variables` to `DroidAgent`: ```python from droidrun.agent.droid import DroidAgent -from droidrun.tools import AdbTools -from droidrun.config_manager.config_manager import DroidRunConfig +from droidrun.config_manager.config_manager import DroidrunConfig # Define variables variables = { "recipient": "alice@example.com", - "message": "Hello from DroidRun!" + "message": "Hello from Droidrun!" } # Custom prompt to render variables @@ -98,12 +95,11 @@ Use these variables when executing tasks. } # Create agent -config = DroidRunConfig() +config = DroidrunConfig() agent = DroidAgent( goal="Send message to recipient", config=config, - tools=AdbTools(), variables=variables, prompts=custom_prompts ) @@ -125,51 +121,45 @@ Customize these prompts to render variables: ## Accessing Variables in Custom Tools -Custom tools **cannot** directly access `DroidAgentState.custom_variables` because they don't have access to the shared state. - -### Workaround: Pass as Tool Arguments (Recommended) - -Design tools to accept parameters, then let the agent pass variable values: +Custom tools can directly access variables via the `shared_state` keyword argument: ```python from droidrun.agent.droid import DroidAgent -from droidrun.tools import AdbTools -from droidrun.config_manager.config_manager import DroidRunConfig +from droidrun.config_manager.config_manager import DroidrunConfig -def send_notification(tools_instance, title: str, channel: str): - """Send a notification to a specific channel.""" +def send_notification(title: str, *, tools=None, shared_state=None, **kwargs): + """Send a notification using channel from custom variables. + + Args: + title: Notification title + tools: Tools instance (optional, injected automatically) + shared_state: DroidAgentState (optional, injected automatically) + """ + if not shared_state: + return "Error: shared_state required" + + # Access custom variables + channel = shared_state.custom_variables.get("notification_channel", "default") return f"Sent '{title}' to {channel}" custom_tools = { "send_notification": { - "arguments": ["title", "channel"], - "description": "Send a notification with title and channel", + "arguments": ["title"], + "description": "Send a notification with title. Usage: {\"action\": \"send_notification\", \"title\": \"Alert\"}", "function": send_notification } } -custom_prompts = { - "codeact_system": """ -{% if variables %} -Variables: {% for k, v in variables.items() %}{{ k }}={{ v }} {% endfor %} -{% endif %} - """ -} - -config = DroidRunConfig() +config = DroidrunConfig() agent = DroidAgent( - goal="Send notification to alerts channel", + goal="Send notification with title 'Alert'", config=config, - tools=AdbTools(), custom_tools=custom_tools, - variables={"notification_channel": "alerts"}, - prompts=custom_prompts + variables={"notification_channel": "alerts"} ) ``` -The agent sees `notification_channel="alerts"` in the prompt and calls `send_notification(title="...", channel="alerts")`. - --- ## Use Cases @@ -208,14 +198,14 @@ agent = DroidAgent( ## Key Points -1. **Custom prompts required** - Default prompts don't render variables -2. **Access via prompts only** - Tools cannot access `shared_state.custom_variables` directly -3. **Workaround** - Agents read variables from prompts → pass as tool arguments -4. **Available to all agents** - Manager, Executor, CodeAct, Scripter all receive variables -5. **Jinja2 templates** - Use `{% if variables %}` blocks in custom prompts +1. **Custom prompts required** - Default prompts don't render variables in agent context +2. **Direct access in tools** - Custom tools access `shared_state.custom_variables` via keyword argument +3. **Available to all agents** - Manager, Executor, CodeAct, Scripter all receive variables +4. **Jinja2 templates** - Use `{% if variables %}` blocks in custom prompts +5. **Auto-injection** - `tools` and `shared_state` are injected automatically by Droidrun ## Related Documentation -- [Custom Prompts](/docs/v4/guides/prompts) - How to customize agent prompts -- [Custom Tools](/docs/v4/guides/custom-tools) - Creating custom tool functions -- [DroidAgent SDK](/docs/v4/sdk/droid-agent) - Complete API reference +- [Custom Prompts](/v4/concepts/prompts) - How to customize agent prompts +- [Custom Tools and Credentials](/v4/guides/custom-tools-credentials) - Creating custom tool functions +- [DroidAgent SDK](/v4/sdk/droid-agent) - Complete API reference diff --git a/docs/v4/guides/device-setup.mdx b/docs/v4/guides/device-setup.mdx index b18f36b..f4f42c4 100644 --- a/docs/v4/guides/device-setup.mdx +++ b/docs/v4/guides/device-setup.mdx @@ -1,284 +1,103 @@ --- title: 'Device Setup & Communication' -description: 'Complete guide to setting up Android and iOS devices, understanding the Portal app, and troubleshooting device connections' +description: 'Setting up Android and iOS devices for Droidrun automation' --- ## Overview -DroidRun controls devices through a specialized Portal app that acts as a bridge between your computer and the device. This guide covers everything from initial setup to advanced troubleshooting. - ---- - -## Understanding the Portal App - -### What is the DroidRun Portal? - -The DroidRun Portal is a companion app that runs on your device and provides: - -1. **Accessibility Tree Extraction** - Captures UI elements and their properties -2. **Device State Monitoring** - Tracks current activity, keyboard visibility, and device info -3. **Command Execution** - Executes tap, swipe, text input, and other actions -4. **Visual Feedback** - Displays overlay to show detected UI elements (optional) -5. **Dual Communication Modes** - Supports both TCP and content provider communication - -### How It Works - - - - The Portal app uses Android's [Accessibility Services](https://developer.android.com/reference/android/accessibilityservice/AccessibilityService) to: - - Monitor UI changes in real-time - - Identify interactive elements (buttons, text fields, etc.) - - Extract element positions, text content, and metadata - - Detect focused elements for text input - - - - DroidRun uses a unified `PortalClient` that automatically selects the best communication method: - - **TCP Mode** (faster): HTTP server on port 8080 with ADB port forwarding - - **Content Provider Mode** (fallback): Android content provider via ADB shell commands - - The `PortalClient` handles both modes transparently via the `prefer_tcp` flag in `AdbTools`. - - - - The Portal includes a custom IME (Input Method Editor) that: - - Supports all characters including special Unicode - - Handles base64-encoded text for reliability - - Bypasses Android's text input restrictions - - Automatically enabled when creating `AdbTools` instances - - - -### Portal App Permissions - -The Portal app requires the following permissions: - -| Permission | Purpose | Required | -|------------|---------|----------| -| Accessibility Service | Read UI state and perform actions | ✅ Yes | -| Display over other apps | Show element overlay (optional) | ❌ No | -| Query all packages | List installed apps | ✅ Yes | - - -**Privacy Note**: The Portal app only communicates locally via ADB. No data is sent to external servers. The app can be disabled when not in use through Android settings. - - ---- - -## Android Device Setup - -### Prerequisites - - - - **Option 1: Platform Tools (Recommended)** - - Download from [Android Developer Site](https://developer.android.com/studio/releases/platform-tools): - - **Windows**: Download ZIP, extract, add to PATH - - **macOS**: `brew install android-platform-tools` - - **Linux**: `sudo apt install adb` or download platform-tools - - **Verify Installation:** - ```bash - adb version - # Should show: Android Debug Bridge version 1.0.41 or higher - ``` - - - - 1. Open **Settings** on your Android device - 2. Navigate to **About phone** (or **About device**) - 3. Find **Build number** (may be under "Software information") - 4. Tap **Build number** 7 times - 5. You'll see a message: "You are now a developer!" - 6. Go back to main Settings - 7. Find **Developer options** (usually under System or Advanced) - - - - 1. Open **Settings** > **Developer options** - 2. Toggle on **USB debugging** - 3. If prompted, check "Always allow from this computer" and tap **OK** - - **Verify Connection:** - ```bash - adb devices - # Should show your device serial number - ``` - - - -### Install the Portal App - -DroidRun provides a simple command to download and install the latest Portal APK: - -```bash -droidrun setup -``` - -This command will: -1. Download the latest Portal APK from GitHub releases -2. Uninstall any existing Portal app version (using `uninstall=True` flag) -3. Install the new APK with all permissions granted (using `-g` flag) -4. Install silently without showing progress (using `silent=True` unless `--debug` is used) -5. Attempt to enable the accessibility service automatically +Droidrun controls devices through a specialized Portal app that bridges your computer and the device. - + + +## Prerequisites + + + + **macOS**: `brew install android-platform-tools` + + **Linux**: `sudo apt install adb` + + **Windows**: Download from [Android Developer Site](https://developer.android.com/studio/releases/platform-tools) + + Verify: `adb version` + + + + 1. Go to **Settings** > **About phone** + 2. Tap **Build number** 7 times (enables Developer options) + 3. Go to **Settings** > **Developer options** + 4. Enable **USB debugging** + 5. Connect device and tap **Always allow** + + Verify: `adb devices` + + + ```bash - # Using the first connected device + # Automatic setup (downloads latest Portal APK) droidrun setup - # Using a specific device + # Or specify device droidrun setup --device SERIAL_NUMBER ``` - If automatic accessibility enablement fails, you'll see instructions to enable it manually. - - - - If you have a custom APK build: - - ```bash - # Install custom APK - droidrun setup --path /path/to/portal.apk --device SERIAL_NUMBER - - # Enable debug logging for troubleshooting - droidrun setup --debug - ``` - - - - **Installation fails with "INSTALL_FAILED_UPDATE_INCOMPATIBLE":** - - The setup command automatically uninstalls existing versions (`uninstall=True`) - - If it still fails, manually uninstall: `adb uninstall com.droidrun.portal` - - **Installation fails with "Permission denied":** - - Ensure USB debugging is enabled - - Check that the device is authorized (check `adb devices`) - - **Installation succeeds but app doesn't appear:** - - Check if the APK is for the correct architecture (arm64-v8a, armeabi-v7a, x86) - - Try installing manually: `adb install -r -g portal.apk` - - - -### Enable Accessibility Service - -The Portal app requires accessibility permissions to read UI state and perform actions. - - - - The `droidrun setup` command attempts automatic enablement via ADB using the `enable_portal_accessibility()` function: - - ```bash - # This is done automatically during setup - # Service name: com.droidrun.portal/com.droidrun.portal.DroidrunAccessibilityService - adb shell settings put secure enabled_accessibility_services com.droidrun.portal/com.droidrun.portal.DroidrunAccessibilityService - adb shell settings put secure accessibility_enabled 1 - ``` - - - Automatic enablement may fail on some devices due to security restrictions. The setup command will automatically open accessibility settings if automatic enablement fails. - + This will: + - Download the latest Portal APK + - Install with all permissions granted + - Enable accessibility service automatically - - If automatic enablement fails, the setup command will automatically open accessibility settings: - - 1. The device will show **Accessibility Settings** - 2. Find **DroidRun Portal** (package: `com.droidrun.portal`) in the services list - 3. Tap on **DroidRun Portal** - 4. Toggle the switch to **ON** - 5. Tap **Allow** on the permission dialog - - - The setup command automatically runs `am start -a android.settings.ACCESSIBILITY_SETTINGS` to open settings if automatic enablement fails. - - - - - Check if the service is enabled: - + ```bash droidrun ping - ``` - - You should see: - ``` - Portal is installed and accessible. You're good to go! + # Output: Portal is installed and accessible. You're good to go! ``` -### Test Your Setup +--- -Verify everything is working: +## Portal App -```bash -# List connected devices -droidrun devices +The Droidrun Portal (`com.droidrun.portal`) provides: -# Test Portal connection (tries content provider mode by default) -droidrun ping +- **Accessibility Tree** - Extracts UI elements and their properties +- **Device State** - Tracks current activity, keyboard visibility +- **Action Execution** - Tap, swipe, text input, and other actions +- **Dual Communication** - TCP (faster) or Content Provider (fallback) -# Test Portal connection (TCP mode) -droidrun ping --tcp - -# Test with specific device -droidrun ping --device SERIAL_NUMBER - -# Run a simple command -droidrun run "Open the settings app" -``` + +The Portal only communicates locally via ADB. No data is sent to external servers. + --- ## Communication Modes -DroidRun supports two communication modes with automatic fallback: - -### TCP Mode (Recommended) - -**How it works:** -1. Portal app runs an HTTP server on device port 8080 -2. ADB forwards a local port to the device's port 8080 -3. DroidRun sends HTTP requests to `localhost:FORWARDED_PORT` - -**Advantages:** -- ⚡ **Faster**: Direct HTTP communication (~2-5x faster) -- 📦 **Efficient**: Binary data transfer for screenshots -- 🔄 **Reliable**: Automatic port reuse, no cleanup needed - -**Enable TCP mode:** - -```bash -# CLI - use --tcp flag -droidrun run "your command" --tcp - -# Or set in config.yaml -# device: -# use_tcp: true - -# Python API -tools = AdbTools(serial="DEVICE_SERIAL", use_tcp=True) -``` - -**Port forwarding details:** -- Portal listens on device port `8080` (configurable via `remote_tcp_port`) -- ADB automatically allocates a free local port (uses `tcp:0` for auto-allocation) -- Existing forwards are reused (no duplicates) -- Forwards persist until device disconnect or manual removal - -**Troubleshooting TCP mode:** - - + + **How it works:** + - Portal runs HTTP server on device port 8080 + - ADB forwards local port → device port 8080 + - Droidrun sends HTTP requests to `localhost:PORT` + + **Enable:** ```bash - # Check if port forwarding exists + # CLI + droidrun run "your command" --tcp + + # Python + tools = DeviceConfig(serial="DEVICE_SERIAL", use_tcp=True) + ``` + + **Troubleshooting:** + ```bash + # Check port forwarding adb forward --list - # Should show: SERIAL tcp:LOCAL_PORT tcp:8080 + # Test Portal server + adb shell netstat -an | grep 8080 # Remove all forwards and retry adb forward --remove-all @@ -286,465 +105,187 @@ tools = AdbTools(serial="DEVICE_SERIAL", use_tcp=True) ``` - - - Ensure Portal app is installed and accessibility service is enabled - - Check if the Portal HTTP server is running: `adb shell netstat -an | grep 8080` - - Restart the Portal app - - Try content provider mode as fallback - - - - DroidRun automatically finds free ports. If you see this error: + + **How it works:** + - Portal exposes content provider at `content://com.droidrun.portal/` + - Commands sent via ADB shell: `content query --uri ...` + - JSON responses parsed from shell output + **Usage:** ```bash - # List what's using the port - lsof -i :PORT_NUMBER # macOS/Linux - netstat -ano | findstr :PORT_NUMBER # Windows + # Default mode (no flag needed) + droidrun ping - # Remove specific forward - adb forward --remove tcp:PORT_NUMBER - ``` - - - -### Content Provider Mode (Fallback) - -**How it works:** -1. Portal app exposes a content provider at `content://com.droidrun.portal/` -2. `PortalClient` sends ADB shell commands like `content query --uri content://com.droidrun.portal/state` -3. Portal returns data as JSON in the shell output, parsed by `PortalClient` - -**Advantages:** -- 🔒 **Always Available**: No network/port requirements -- 🛡️ **Secure**: Uses Android's built-in IPC mechanism -- 🔧 **Reliable**: Works even if TCP server fails - -**Limitations:** -- 🐌 **Slower**: ADB shell overhead (~2-5x slower than TCP) -- 📏 **Data size limits**: Shell output may be truncated for large responses - -**Automatic fallback:** - -The `PortalClient` (used internally by `AdbTools`) automatically falls back to content provider mode when: -- TCP connection fails -- HTTP server is not responding -- Port forwarding setup fails - -**Manual content provider mode:** - -```bash -# Test content provider connection (default behavior) -droidrun ping - -# The --no-tcp flag is not needed as content provider is the default -# To explicitly disable TCP, use use_tcp=False in Python API -``` - -**Troubleshooting content provider mode:** - - - - The content provider returns data in this format: - ``` - Row: 0 result={"data": "{\"a11y_tree\": [...]}"} + # Python + tools = DeviceConfig(serial="DEVICE_SERIAL", use_tcp=False) ``` - If parsing fails: - - Ensure Portal app is up to date - - Check accessibility service is enabled - - Try reinstalling: `droidrun setup` - - - - The Portal content provider is accessed via: + **Troubleshooting:** ```bash - content://com.droidrun.portal/state - content://com.droidrun.portal/keyboard/input - content://com.droidrun.portal/packages - content://com.droidrun.portal/overlay_offset - ``` + # Test content provider directly + adb shell content query --uri content://com.droidrun.portal/state - If you see "Unknown URI" errors: - - Ensure Portal app is installed (`com.droidrun.portal`) - - Verify accessibility service is enabled - - Update to the latest Portal app version: `droidrun setup` - - - -### Performance Comparison - -| Operation | TCP Mode | Content Provider | -|-----------|----------|------------------| -| Get UI State | ~100-200ms | ~300-500ms | -| Input Text | ~50-100ms | ~200-300ms | -| Take Screenshot | ~150-250ms | ~400-600ms | -| Execute Action | ~50ms | ~100-200ms | - ---- - -## Wireless Debugging Setup - -Control devices over WiFi without USB cables. - -### Android 11+ (Wireless Debugging) - - - - 1. Open **Settings** > **Developer options** - 2. Enable **Wireless debugging** - 3. Tap **Wireless debugging** to open settings - 4. Note your device's IP address and port (e.g., `192.168.1.100:37757`) - - - - **Method 1: QR Code Pairing (Easiest)** - ```bash - # Android 11+ has QR code pairing in Wireless debugging settings - # Scan QR with this command: - adb pair - ``` - - **Method 2: Pairing Code** - 1. In Wireless debugging settings, tap **Pair device with pairing code** - 2. Note the pairing code and IP:port - 3. Run: `adb pair IP:PORT` - 4. Enter the pairing code when prompted - - - - ```bash - # Connect to device - adb connect IP:PORT - - # Verify connection - adb devices - # Should show: IP:PORT device - - # Use with DroidRun - droidrun ping --device IP:PORT - ``` - - - -### Android 10 and Below (TCP/IP Mode) - - - - ```bash - # Connect device via USB first - adb devices - - # Enable TCP/IP mode on port 5555 - adb tcpip 5555 - - # Device will restart ADB in network mode - ``` - - - - **On Device:** - - Settings > About phone > Status > IP address - - **Via ADB:** - ```bash - adb shell ip addr show wlan0 | grep inet - ``` - - - - ```bash - # Disconnect USB cable - # Connect to device - adb connect DEVICE_IP:5555 - - # Verify - adb devices - ``` - - - -### Managing Wireless Connections - -```bash -# List all connected devices -droidrun devices - -# Connect to device -droidrun connect 192.168.1.100:5555 - -# Disconnect from device -droidrun disconnect 192.168.1.100:5555 - -# Use specific wireless device -droidrun run "your command" --device 192.168.1.100:5555 -``` - -**Troubleshooting wireless connection:** - - - - - Ensure device and computer are on the same WiFi network - - Check firewall settings on both device and computer - - Verify port 5555 (or your port) is not blocked - - Try restarting wireless debugging - - - - - Move device closer to WiFi router - - Disable battery optimization for ADB - - Use 5GHz WiFi if available (more stable) - - Consider USB connection for critical operations - - - - ```bash - # Get IP via USB connection - adb shell ip addr show wlan0 | grep "inet " | awk '{print $2}' | cut -d/ -f1 + # Should show: Row: 0 result={"data": "{...}"} ``` --- -## Multiple Device Management - -Control multiple Android devices simultaneously. - -### List Connected Devices - -```bash -droidrun devices -# Output: -# Found 2 connected device(s): -# • emulator-5554 -# • 192.168.1.100:5555 -``` - -### Target Specific Device - -```bash -# CLI - using --device flag -droidrun run "your command" --device emulator-5554 - -# Python API -tools = AdbTools(serial="emulator-5554") -agent = DroidAgent(goal="your task", tools=tools) -``` - -### Working with Multiple Devices - -```python -import asyncio -from droidrun import AdbTools, DroidAgent -from adbutils import adb - -async def control_device(serial: str, command: str): - """Control a specific device""" - tools = AdbTools(serial=serial) - agent = DroidAgent(goal=command, tools=tools) - result = await agent.run() - return result - -async def main(): - # Get all connected devices - devices = adb.list() - - # Run tasks on multiple devices in parallel - tasks = [ - control_device(devices[0].serial, "Open settings"), - control_device(devices[1].serial, "Check battery level"), - ] - - results = await asyncio.gather(*tasks) - print(results) - -asyncio.run(main()) -``` - ---- - -## iOS Device Support - - -iOS support is currently in **beta** and requires a separate iOS Portal app. - - -### Prerequisites - -- **macOS** with Xcode installed (for iOS device communication) -- **iOS device** with Developer mode enabled -- **iOS Portal app** installed on device - -### Setup iOS Device - -The iOS Portal uses a different architecture than Android: - -1. **HTTP Server**: iOS Portal runs an HTTP server on the device -2. **Direct Communication**: No ADB required - direct HTTP requests -3. **Accessibility API**: Uses iOS accessibility APIs for UI state - -**Python API:** - -```python -from droidrun import IOSTools, DroidAgent - -# Initialize iOS tools with device URL -tools = IOSTools( - url="http://DEVICE_IP:8080", - bundle_identifiers=["com.example.app"] # Optional: list of app bundle IDs -) - -# Create agent -agent = DroidAgent( - goal="Open Settings and check WiFi", - tools=tools, - platform="ios" -) - -result = await agent.run() -``` - -**CLI:** - -```bash -# Use --ios flag to target iOS devices -droidrun run "your command" --ios - -# Or set platform in config.yaml: -# device: -# platform: ios -``` - -### iOS Limitations - -Current limitations in iOS support: - -- ❌ No `get_date()` - Returns "Not implemented for iOS" -- ❌ No `back()` - Raises `NotImplementedError` -- ❌ No `drag()` - Not implemented, returns False -- ❌ No `get_apps()` - Not available (use `list_packages()` instead) -- ⚠️ `input_text()` does not support `index` or `clear` parameters -- ⚠️ `_extract_element_coordinates_by_index()` is implemented but works differently than Android -- ⚠️ Requires manual Portal app installation - ---- - -## Troubleshooting Guide - -### Common Issues and Solutions +## Advanced Setup - - **Symptoms:** `adb devices` shows no devices or unauthorized + + ### Setup + + + + 1. **Settings** > **Developer options** > **Wireless debugging** + 2. Note IP address and port (e.g., `192.168.1.100:37757`) + + + + **QR Code Method:** + ```bash + adb pair + ``` + + **Pairing Code Method:** + 1. Tap **Pair device with pairing code** + 2. Note pairing code and IP:port + 3. Run: `adb pair IP:PORT` + 4. Enter pairing code + + + + ```bash + adb connect IP:PORT + droidrun ping --device IP:PORT + ``` + + + + ### Common Issues + + - Connection refused → Check same WiFi network and firewall + - Frequent drops → Use 5GHz WiFi or stay near router + - Can't find IP → Run `adb shell ip addr show wlan0 | grep "inet "` via USB + + + + + + ```bash + # Connect via USB first + adb tcpip 5555 + ``` + + + + ```bash + adb shell ip addr show wlan0 | grep inet + ``` + + + + ```bash + # Disconnect USB cable + adb connect DEVICE_IP:5555 + droidrun ping --device DEVICE_IP:5555 + ``` + + + + + + ### List Devices + + ```bash + droidrun devices + # Found 2 connected device(s): + # • emulator-5554 + # • 192.168.1.100:5555 + ``` + + ### Target Specific Device + + ```bash + # CLI + droidrun run "your command" --device emulator-5554 + + # Python + tools = DeviceConfig(serial="emulator-5554") + agent = DroidAgent(goal="your task", tools=tools) + ``` + + ### Parallel Control + + ```python + import asyncio + from droidrun import DeviceConfig, DroidAgent + from adbutils import adb + + async def control_device(serial: str, command: str): + device_config = DeviceConfig(serial=serial) + agent = DroidAgent(goal=command, device_config=tools) + return await agent.run() + + async def main(): + devices = adb.list() + + tasks = [ + control_device(devices[0].serial, "Open settings"), + control_device(devices[1].serial, "Check battery"), + ] + + results = await asyncio.gather(*tasks) + print(results) + + asyncio.run(main()) + ``` + + + +--- + +## Troubleshooting + + + + **Symptoms:** `adb devices` shows no devices or `unauthorized` **Solutions:** - 1. **Check USB connection:** - ```bash - # Unplug and replug USB cable - # Try a different USB port or cable - ``` - - 2. **Authorize device:** - - Unplug device - - Revoke USB debugging authorizations (Developer options > Revoke USB debugging authorizations) - - Reconnect device - - Tap "Always allow" on authorization prompt - - 3. **Restart ADB server:** - ```bash - adb kill-server - adb start-server - adb devices - ``` - - 4. **Check drivers (Windows):** - - Install [Google USB Driver](https://developer.android.com/studio/run/win-usb) - - Or manufacturer-specific drivers (Samsung, Xiaomi, etc.) + 1. Unplug/replug USB cable, try different port + 2. Revoke USB debugging authorizations (Developer options) + 3. Reconnect and tap "Always allow" + 4. Restart ADB: `adb kill-server && adb start-server` + 5. **Windows**: Install [Google USB Driver](https://developer.android.com/studio/run/win-usb) - + **Symptoms:** `droidrun ping` fails with "Portal is not installed" **Solutions:** - 1. **Reinstall Portal:** - ```bash - droidrun setup --device SERIAL - ``` - - 2. **Check installation manually:** - ```bash - adb shell pm list packages | grep droidrun - # Should show: package:com.droidrun.portal - ``` - - 3. **Verify APK architecture:** - - Portal APK must match device architecture - - Most devices: arm64-v8a - - Older devices: armeabi-v7a - - Emulators: x86 or x86_64 + 1. Reinstall: `droidrun setup` + 2. Check: `adb shell pm list packages | grep droidrun` + 3. Verify APK architecture matches device (arm64-v8a for most devices) - **Symptoms:** `droidrun ping` fails with "Portal is not enabled as an accessibility service" + **Symptoms:** `droidrun ping` fails with "accessibility service not enabled" **Solutions:** - 1. **Enable automatically:** + 1. Auto-enable: ```bash - adb shell settings put secure enabled_accessibility_services com.droidrun.portal/com.droidrun.portal.DroidrunAccessibilityService + adb shell settings put secure enabled_accessibility_services \ + com.droidrun.portal/com.droidrun.portal.DroidrunAccessibilityService adb shell settings put secure accessibility_enabled 1 ``` - - 2. **Enable manually:** - - The error will auto-open accessibility settings - - Find "DroidRun Portal" and toggle ON - - Allow permissions - - 3. **Verify service status:** + 2. Manual: Settings > Accessibility > Droidrun Portal > Toggle ON + 3. Verify: ```bash adb shell settings get secure enabled_accessibility_services - # Should contain: com.droidrun.portal/com.droidrun.portal.DroidrunAccessibilityService - ``` - - 4. **Force stop and restart:** - ```bash - adb shell am force-stop com.droidrun.portal - adb shell am start-service com.droidrun.portal/.DroidrunAccessibilityService - ``` - - - - **Symptoms:** TCP mode not working, falls back to content provider - - **Solutions:** - 1. **Check port forwarding:** - ```bash - adb forward --list - # Should show: SERIAL tcp:PORT tcp:8080 - - # If missing, reconnect: - adb forward tcp:0 tcp:8080 - ``` - - 2. **Test Portal HTTP server:** - ```bash - # Check if server is running - adb shell netstat -an | grep 8080 - - # Test endpoint - adb forward tcp:12345 tcp:8080 - curl http://localhost:12345/ping - ``` - - 3. **Restart Portal app:** - ```bash - adb shell am force-stop com.droidrun.portal - # Wait 2 seconds - adb shell am start-service com.droidrun.portal/.DroidrunAccessibilityService - ``` - - 4. **Use content provider fallback:** - ```bash - droidrun run "your command" --no-tcp + # Should contain: com.droidrun.portal/... ``` @@ -752,252 +293,144 @@ Current limitations in iOS support: **Symptoms:** `input_text()` fails or types gibberish **Solutions:** - 1. **Check DroidRun keyboard (automatically set up by AdbTools):** - ```bash - # List IMEs - adb shell ime list -a - # Should show: com.droidrun.portal/.DroidrunKeyboardIME - - # Enable keyboard (done automatically in AdbTools.__init__) - adb shell ime enable com.droidrun.portal/.DroidrunKeyboardIME - adb shell ime set com.droidrun.portal/.DroidrunKeyboardIME - ``` - - 2. **Verify keyboard is active:** + 1. Keyboard auto-enabled by `AdbTools.__init__()`: ```bash + # Verify adb shell settings get secure default_input_method # Should show: com.droidrun.portal/.DroidrunKeyboardIME ``` - - 3. **Focus element before input:** - ```python - # Tap text field first or use index parameter - tools.input_text("your text", index=5) # Taps element 5 then inputs text - - # Or manually tap first - tools.tap_by_index(5) - time.sleep(0.5) - tools.input_text("your text") - ``` - - 4. **Manually switch keyboard:** - - Long press space bar on device - - Select "DroidRun Keyboard" - - - The DroidRun keyboard is automatically enabled when you create an `AdbTools` instance via `setup_keyboard()` call in `__init__`. - + 2. Manual switch: Long press space bar → Select "Droidrun Keyboard" + 3. Focus element first: `tools.tap_by_index(5)` then `tools.input_text("text")` - + **Symptoms:** `get_state()` returns empty or incomplete UI tree **Solutions:** - 1. **Verify accessibility service:** - ```bash - droidrun ping - ``` - - 2. **Check app compatibility:** - - Some apps block accessibility services - - WebView content may not be accessible - - Games and apps with custom UI may have limited accessibility - - 3. **Wait for UI to load:** - ```python - import time - tools.tap_by_index(5) - time.sleep(1) # Wait for animation/transition - state = tools.get_state() - ``` - - 4. **Enable visual overlay:** - - Check Portal app settings - - Enable overlay to see what elements are detected - - - - **Symptoms:** Works on physical device but not on emulator - - **Solutions:** - 1. **Check emulator architecture:** - ```bash - # Most emulators are x86/x86_64 - adb shell getprop ro.product.cpu.abi - ``` - - 2. **Enable Google APIs:** - - Use emulator image with Google APIs - - Required for some services - - 3. **Increase emulator resources:** - - AVD Manager > Edit > Advanced Settings - - Increase RAM to 4GB+ - - Enable hardware acceleration - - 4. **Use `-writable-system` flag:** - ```bash - emulator -avd YOUR_AVD -writable-system - ``` - - - - **Symptoms:** ADB commands fail with permission errors - - **Solutions:** - 1. **Run ADB as root (if device is rooted):** - ```bash - adb root - adb remount - ``` - - 2. **Grant all permissions to Portal:** - ```bash - adb install -r -g portal.apk - # -g flag grants all runtime permissions - ``` - - 3. **Manually grant permissions:** - ```bash - adb shell pm grant com.droidrun.portal android.permission.QUERY_ALL_PACKAGES - ``` - - 4. **Check SELinux (advanced):** - ```bash - # Temporarily set to permissive (rooted devices only) - adb shell setenforce 0 - ``` + 1. Verify accessibility: `droidrun ping` + 2. Some apps block accessibility services (WebViews, games, custom UI) + 3. Wait for UI: `time.sleep(1)` after tap/swipe + 4. Enable Portal overlay to see detected elements -### Debug Mode + -Enable verbose logging to troubleshoot issues: + -```bash -# CLI -droidrun run "your command" --debug + +iOS support is currently in **beta**. Functionality is limited compared to Android. + -# Python API -import logging -logging.basicConfig(level=logging.DEBUG) +--- + +## Prerequisites + +- **macOS** with Xcode installed +- **iOS device** with Developer mode enabled +- **iOS Portal app** (separate from Android Portal) + +--- + +## Setup + + + + The iOS Portal app must be manually installed on your device. + + + Installation instructions and download link coming soon. + + + + + Find your iOS device's IP address in Settings > WiFi > (i) icon + + + + ```python + from droidrun import IOSTools + + tools = IOSTools(url="http://DEVICE_IP:8080") + result = tools.ping() + print(result) + ``` + + + +--- + +## Architecture + +iOS Portal uses a different architecture than Android: + +| Feature | Android | iOS | +|---------|---------|-----| +| Communication | ADB + TCP/Content Provider | HTTP server only | +| Setup Tool | `droidrun setup` | Manual installation | +| Accessibility | Android Accessibility API | iOS Accessibility API | +| Text Input | Custom keyboard IME | Direct text input | + +--- + +## Usage + +### Python API + +```python +from droidrun import IOSTools, DroidAgent + +# Initialize iOS tools with device URL +tools = IOSTools( + url="http://192.168.1.100:8080", + bundle_identifiers=["com.example.app"] # Optional +) + +# Create agent +agent = DroidAgent( + goal="Open Settings and check WiFi", + tools=tools +) + +result = await agent.run() ``` -Debug mode shows: -- ADB communication details -- Portal requests/responses -- Port forwarding status -- UI state parsing -- Error stack traces - -### Diagnostic Commands +### CLI ```bash -# Check ADB version -adb version +# Use --ios flag +droidrun run "your command" --ios -# Check device info -adb shell getprop | grep -E "ro.build|ro.product" - -# List installed packages -adb shell pm list packages | grep droidrun - -# Check accessibility services -adb shell settings get secure enabled_accessibility_services - -# Check default keyboard -adb shell settings get secure default_input_method - -# View Portal app logs -adb logcat | grep DroidRun - -# Test content provider directly -adb shell content query --uri content://com.droidrun.portal/state +# Or set platform in config.yaml: +# device: +# platform: ios ``` --- -## Best Practices +## Limitations -### Performance Optimization +Current iOS support has these limitations: -1. **Use TCP mode for production:** - ```python - tools = AdbTools(use_tcp=True) - ``` +| Feature | Status | Notes | +|---------|--------|-------| +| `get_date()` | ❌ | Returns "Not implemented for iOS" | +| `back()` | ❌ | Raises `NotImplementedError` | +| `drag()` | ❌ | Not implemented, returns False | +| `get_apps()` | ⚠️ | Use `list_packages()` instead | +| `input_text()` | ⚠️ | No `index` or `clear` parameters | +| `tap()`, `swipe()` | ✅ | Fully supported | +| `get_state()` | ✅ | Returns accessibility tree | +| `take_screenshot()` | ✅ | Fully supported | -2. **Minimize screenshot captures:** - ```python - # Use vision only when needed - agent = DroidAgent(goal="task", tools=tools, vision=True) - ``` - -3. **Cache UI state when possible:** - ```python - # Avoid repeated get_state() calls in loops - state = tools.get_state() - # Work with cached state - ``` - -4. **Use wireless debugging for development only:** - - USB is faster and more reliable - - Reserve wireless for testing/demos - -### Security Considerations - -1. **Disable Portal when not in use:** - ```bash - # Disable accessibility service - adb shell settings put secure enabled_accessibility_services "" - ``` - -2. **Revoke debugging authorization after use:** - - Settings > Developer options > Revoke USB debugging authorizations - -3. **Use TCP mode on trusted networks only:** - - Port forwarding exposes local server - - Content provider is more secure for untrusted environments - -4. **Never commit device serials or IPs:** - ```python - # Use environment variables - import os - device_serial = os.getenv("DEVICE_SERIAL") - ``` - -### Maintenance - -1. **Keep Portal app updated:** - ```bash - droidrun setup # Reinstalls latest version - ``` - -2. **Clean up port forwards periodically:** - ```bash - adb forward --remove-all - ``` - -3. **Restart ADB server if issues persist:** - ```bash - adb kill-server && adb start-server - ``` - -4. **DroidRun keyboard cleanup:** - - The CLI automatically disables the DroidRun keyboard after each run - - This prevents the keyboard from interfering with normal device usage - - The keyboard is re-enabled automatically on the next run - -5. **Monitor device battery during long sessions:** - - Keep device plugged in - - Disable sleep/screen timeout + + --- ## Next Steps -Now that your device is set up: - -- Learn about the [Agent System](/v4/concepts/agent) -- Explore [Configuration Options](/v4/guides/configuration) +- Learn about the [Agent System](/v4/concepts/agent-architecture) +- Explore [Configuration Options](/v4/sdk/configuration) - Try [Custom Tools](/v4/guides/custom-tools-credentials) - Implement [Structured Output](/v4/guides/structured-output) diff --git a/docs/v4/guides/overview.mdx b/docs/v4/guides/overview.mdx index 430f2cc..10e05e0 100644 --- a/docs/v4/guides/overview.mdx +++ b/docs/v4/guides/overview.mdx @@ -2,7 +2,7 @@ title: Guides Overview --- -Welcome to the DroidRun v4 Guides! This section provides step-by-step instructions and best practices for using DroidRun. Each guide focuses on a specific aspect of the framework, from device setup to advanced automation patterns. +Welcome to the Droidrun v4 Guides! This section provides step-by-step instructions and best practices for using Droidrun. Each guide focuses on a specific aspect of the framework, from device setup to advanced automation patterns. --- @@ -70,7 +70,7 @@ Welcome to the DroidRun v4 Guides! This section provides step-by-step instructio ## Quick Start Paths -### New to DroidRun? +### New to Droidrun? 1. [CLI Reference](./cli) - Learn CLI commands and usage 2. [Device Setup](./device-setup) - Get your device ready 3. [Configuration System](./configuration) - Configure agents and LLMs @@ -81,7 +81,7 @@ Welcome to the DroidRun v4 Guides! This section provides step-by-step instructio 2. [Custom Variables](./custom-variables) - Make workflows reusable 3. [App Cards](./app-cards) - Improve success rates for specific apps -### Extending DroidRun? +### Extending Droidrun? 1. [Custom Tools & Credentials](./custom-tools-credentials) - Add new capabilities 2. [Configuration System](./configuration) - Advanced customization 3. Check [SDK Reference](../sdk/droid-agent) for programmatic usage @@ -92,8 +92,7 @@ Welcome to the DroidRun v4 Guides! This section provides step-by-step instructio **Core Concepts:** - [Architecture](../concepts/architecture) - Multi-agent system overview -- [Workflow Architecture](../concepts/workflow-architecture) - Event-driven coordination -- [Event Streaming](../concepts/event-streaming) - Real-time monitoring +- [Events and Workflows](../concepts/events-and-workflows) - Event-driven coordination and real-time monitoring **SDK Reference:** - [DroidAgent](../sdk/droid-agent) - Main agent class diff --git a/docs/v4/guides/structured-output.mdx b/docs/v4/guides/structured-output.mdx index 2c51b81..a54d84c 100644 --- a/docs/v4/guides/structured-output.mdx +++ b/docs/v4/guides/structured-output.mdx @@ -5,698 +5,147 @@ description: 'Extract structured data from device interactions using Pydantic mo # Structured Output Extraction -DroidRun's structured output system allows you to extract structured, typed data from device automation tasks. Pass a Pydantic model to the `output_model` parameter in `DroidAgent`, and the agent will automatically collect and extract the data into a type-safe object. - -## Overview - -The structured output feature uses a simple workflow: - -1. **Define a Pydantic model** with the data you want to collect -2. **Pass it to `DroidAgent`** via the `output_model` parameter -3. **Run the agent** - it collects data during normal task execution -4. **Access typed results** from `result.structured_output` - -Benefits: -- Type-safe data extraction with Pydantic validation -- Automatic schema injection into agent prompts -- No changes needed to your task goals -- Falls back gracefully if extraction fails +Extract typed, structured data from device automation tasks by passing a Pydantic model to `DroidAgent`. The agent collects data during execution and returns a validated Python object. --- ## Quick Start -### Basic Example - ```python import asyncio from pydantic import BaseModel, Field from droidrun import DroidAgent -from droidrun.tools import AdbTools -from droidrun.config_manager import DroidRunConfig +from droidrun.config_manager import DroidrunConfig -# 1. Define your output structure +# 1. Define output structure class ContactInfo(BaseModel): - """Contact information extracted from device.""" + """Contact information from device.""" name: str = Field(description="Full name of the contact") phone: str = Field(description="Phone number") email: str = Field(description="Email address", default="Not provided") # 2. Create agent with output_model async def main(): - config = DroidRunConfig() - tools = AdbTools() + config = DroidrunConfig() agent = DroidAgent( goal="Find John Smith's contact information", config=config, - tools=tools, - output_model=ContactInfo, # Specify the output structure + output_model=ContactInfo, ) - # 3. Run agent - returns ResultEvent + # 3. Run and access structured output result = await agent.run() - # 4. Access structured data via attributes if result.success and result.structured_output: contact: ContactInfo = result.structured_output print(f"Name: {contact.name}") print(f"Phone: {contact.phone}") print(f"Email: {contact.email}") - else: - print(f"Task failed: {result.reason}") asyncio.run(main()) ``` -**Output:** -``` -Name: John Smith -Phone: +1-555-0123 -Email: john.smith@example.com -``` - --- ## How It Works -### Architecture - -The structured output system uses a two-stage approach: - -``` -User Goal → DroidAgent → Task Completion → StructuredOutputAgent → Typed Result -``` +### Two-Stage Process **Stage 1: Task Execution** -- DroidAgent performs device actions (Manager/Executor or CodeAct mode) -- Agent's system prompt is automatically injected with your Pydantic model schema -- Agent collects the required information during task execution -- Agent completes with a natural language answer containing the data +- DroidAgent performs device actions while collecting required information +- System prompt is automatically injected with your Pydantic schema +- Agent completes with natural language answer containing the data **Stage 2: Extraction (Post-Completion)** -- StructuredOutputAgent receives the final answer text from DroidAgent -- Uses LLM's `astructured_predict()` to extract data into your Pydantic model -- Validates the extracted data against your schema -- Returns typed Pydantic object or `None` if extraction fails - -### Workflow Integration - -Structured output extraction happens in the `finalize()` step of DroidAgent: - -```python -@step -async def finalize(self, ctx: Context, ev: FinalizeEvent) -> ResultEvent: - # Base result with answer - result = ResultEvent( - success=ev.success, - reason=ev.reason, - steps=self.shared_state.step_number, - structured_output=None, - ) - - # Extract structured output if model was provided - if self.output_model is not None and ev.reason: - structured_agent = StructuredOutputAgent( - llm=self.structured_output_llm, - pydantic_model=self.output_model, - answer_text=ev.reason, - ) - - handler = await structured_agent.run() - extraction_result = await handler - - if extraction_result["success"]: - result.structured_output = extraction_result["structured_output"] - - return result -``` +- `StructuredOutputAgent` receives the final answer text +- Uses LLM's `astructured_predict()` to extract data into your model +- Validates against schema and returns typed object or `None` --- -## Defining Output Models - -### Basic Model Structure - -Use Pydantic `BaseModel` with descriptive field definitions: - -```python -from pydantic import BaseModel, Field -from typing import List, Optional - -class AppInfo(BaseModel): - """Information about an installed application.""" - name: str = Field(description="Application name as shown in the app drawer") - package: str = Field(description="Android package identifier (e.g., com.example.app)") - version: str = Field(description="Version number (e.g., 1.2.3)") - size_mb: float = Field(description="Storage size in megabytes") - last_updated: str = Field(description="Last update date in YYYY-MM-DD format") -``` - -**Key Guidelines:** -- Add detailed descriptions to every field -- Use type hints (str, int, float, bool, List, Optional) -- Provide default values for optional fields -- Include a class docstring explaining the model's purpose - -### Nested Models - -For complex data structures, use nested Pydantic models: - -```python -class Address(BaseModel): - """Physical address information.""" - street: str = Field(description="Street address with number") - city: str = Field(description="City name") - state: str = Field(description="State or province") - zip_code: str = Field(description="Postal/ZIP code") - -class PersonProfile(BaseModel): - """Complete person profile with contact and address.""" - full_name: str = Field(description="First and last name") - phone: str = Field(description="Primary phone number") - email: str = Field(description="Email address") - home_address: Address = Field(description="Home address details") - work_address: Optional[Address] = Field( - description="Work address details if available", - default=None - ) -``` - -### Lists and Collections - -Extract multiple items using `List`: - -```python -from typing import List - -class Message(BaseModel): - """A single message from conversation.""" - sender: str = Field(description="Name of message sender") - content: str = Field(description="Message text content") - timestamp: str = Field(description="When message was sent (e.g., '2:30 PM')") - -class Conversation(BaseModel): - """Conversation thread with multiple messages.""" - chat_name: str = Field(description="Name of the chat or contact") - message_count: int = Field(description="Total number of messages") - messages: List[Message] = Field( - description="List of messages in chronological order", - default=[] - ) -``` - -### Optional Fields - -Use `Optional` and `default` for fields that may not be available: - -```python -from typing import Optional - -class Product(BaseModel): - """Product information from shopping app.""" - name: str = Field(description="Product name") - price: float = Field(description="Current price in dollars") - original_price: Optional[float] = Field( - description="Original price if item is on sale", - default=None - ) - discount_percent: Optional[int] = Field( - description="Discount percentage if on sale", - default=None - ) - in_stock: bool = Field( - description="Whether product is currently available", - default=True - ) - rating: Optional[float] = Field( - description="Customer rating out of 5.0", - default=None - ) -``` - ---- - -## Real-World Examples - -### Example 1: Contact Extraction - -Extract contact details from phone's contact list: - -```python -from pydantic import BaseModel, Field -from typing import Optional - -class ContactDetails(BaseModel): - """Contact information from phone's contacts app.""" - full_name: str = Field(description="Contact's full name") - phone_numbers: str = Field( - description="All phone numbers separated by commas" - ) - email_addresses: str = Field( - description="All email addresses separated by commas", - default="None" - ) - company: Optional[str] = Field( - description="Company or organization name", - default=None - ) - job_title: Optional[str] = Field( - description="Job title or position", - default=None - ) - notes: Optional[str] = Field( - description="Any notes or additional information", - default=None - ) - -# Usage -config = DroidRunConfig() - -agent = DroidAgent( - goal="Open contacts app and find Sarah Johnson's contact information", - config=config, - tools=AdbTools(), - output_model=ContactDetails, -) - -result = await agent.run() -contact = result.structured_output -# contact.full_name → "Sarah Johnson" -# contact.phone_numbers → "+1-555-0199, +1-555-0200" -# contact.email_addresses → "sarah.j@company.com" -``` - -### Example 2: Invoice Parsing - -Extract invoice details from a document or email: +## Example: Invoice Extraction ```python from pydantic import BaseModel, Field from typing import List -class InvoiceLineItem(BaseModel): - """Single line item on an invoice.""" - description: str = Field(description="Item or service description") - quantity: int = Field(description="Number of items") - unit_price: float = Field(description="Price per unit in dollars") - total: float = Field(description="Total for this line (quantity × unit_price)") - class Invoice(BaseModel): - """Complete invoice information.""" - invoice_number: str = Field(description="Unique invoice ID/number") - date: str = Field(description="Invoice date (YYYY-MM-DD format)") - due_date: str = Field(description="Payment due date (YYYY-MM-DD format)") - vendor_name: str = Field(description="Name of vendor/company issuing invoice") - customer_name: str = Field(description="Name of customer/recipient") - line_items: List[InvoiceLineItem] = Field( - description="List of items/services on invoice" - ) - subtotal: float = Field(description="Subtotal before tax in dollars") - tax_amount: float = Field(description="Tax amount in dollars") - total_due: float = Field(description="Final amount due in dollars") - -# Usage -config = DroidRunConfig() + """Invoice information.""" + invoice_number: str = Field(description="Invoice ID") + vendor_name: str = Field(description="Vendor name") + total_due: float = Field(description="Total amount in dollars") agent = DroidAgent( - goal="Open the Gmail app and find the invoice from Acme Corp, extract all invoice details", - config=config, - tools=AdbTools(), + goal="Open Gmail and extract invoice from Acme Corp email", + config=DroidrunConfig(), output_model=Invoice, ) result = await agent.run() invoice = result.structured_output -# invoice.invoice_number → "INV-2024-00123" -# invoice.total_due → 1234.56 -# invoice.line_items[0].description → "Web Development Services" -``` - -### Example 3: App Settings Audit - -Extract current settings configuration from an app: - -```python -from pydantic import BaseModel, Field - -class AppSettings(BaseModel): - """Current settings configuration for an application.""" - notifications_enabled: bool = Field( - description="Whether notifications are turned on" - ) - dark_mode: bool = Field( - description="Whether dark mode is enabled" - ) - auto_sync: bool = Field( - description="Whether automatic sync is enabled" - ) - sync_frequency: str = Field( - description="How often sync occurs (e.g., 'Every 15 minutes', 'Hourly', 'Daily')" - ) - data_saver: bool = Field( - description="Whether data saver mode is active" - ) - account_email: str = Field( - description="Email of logged-in account" - ) - storage_used_mb: float = Field( - description="Storage used by app in megabytes" - ) - -# Usage -config = DroidRunConfig() - -agent = DroidAgent( - goal="Open Spotify app, go to settings, and check all current settings", - config=config, - tools=AdbTools(), - output_model=AppSettings, -) - -result = await agent.run() -settings = result.structured_output -# settings.notifications_enabled → True -# settings.dark_mode → False -# settings.sync_frequency → "Every 30 minutes" -``` - -### Example 4: Restaurant Review Summary - -Extract review details from a restaurant app: - -```python -from pydantic import BaseModel, Field -from typing import List, Optional - -class Review(BaseModel): - """Single customer review.""" - reviewer_name: str = Field(description="Name of the person who wrote the review") - rating: float = Field(description="Star rating from 1.0 to 5.0") - date: str = Field(description="Review date (e.g., '3 days ago', '2024-01-15')") - review_text: str = Field(description="Full text of the review") - helpful_count: Optional[int] = Field( - description="Number of people who found this helpful", - default=None - ) - -class RestaurantInfo(BaseModel): - """Restaurant details with reviews.""" - name: str = Field(description="Restaurant name") - address: str = Field(description="Full address") - phone: str = Field(description="Phone number") - cuisine_type: str = Field(description="Type of cuisine (e.g., 'Italian', 'Thai')") - average_rating: float = Field(description="Overall average rating from 1.0 to 5.0") - price_range: str = Field(description="Price range (e.g., '$', '$$', '$$$')") - hours: str = Field(description="Operating hours (e.g., '11 AM - 10 PM')") - recent_reviews: List[Review] = Field( - description="List of 3-5 most recent customer reviews" - ) - -# Usage -config = DroidRunConfig() - -agent = DroidAgent( - goal="Open Yelp app and find 'Blue Plate Restaurant', get all details and recent reviews", - config=config, - tools=AdbTools(), - output_model=RestaurantInfo, -) - -result = await agent.run() -restaurant = result.structured_output -# restaurant.name → "Blue Plate Restaurant" -# restaurant.average_rating → 4.5 -# len(restaurant.recent_reviews) → 5 -``` - -### Example 5: Calendar Event Details - -Extract event information from calendar app: - -```python -from pydantic import BaseModel, Field -from typing import List, Optional - -class Attendee(BaseModel): - """Event attendee information.""" - name: str = Field(description="Attendee name") - email: str = Field(description="Attendee email address") - status: str = Field( - description="Response status (e.g., 'Accepted', 'Declined', 'Maybe', 'No response')" - ) - -class CalendarEvent(BaseModel): - """Calendar event details.""" - title: str = Field(description="Event title/name") - date: str = Field(description="Event date (YYYY-MM-DD format)") - start_time: str = Field(description="Start time (e.g., '2:00 PM')") - end_time: str = Field(description="End time (e.g., '3:30 PM')") - location: Optional[str] = Field( - description="Event location or address", - default=None - ) - description: Optional[str] = Field( - description="Event description or notes", - default=None - ) - attendees: List[Attendee] = Field( - description="List of event attendees", - default=[] - ) - is_recurring: bool = Field( - description="Whether event repeats", - default=False - ) - recurrence_pattern: Optional[str] = Field( - description="How event repeats (e.g., 'Weekly', 'Every Monday')", - default=None - ) - reminder_minutes: Optional[int] = Field( - description="Minutes before event to show reminder", - default=None - ) - -# Usage -config = DroidRunConfig() - -agent = DroidAgent( - goal="Open Google Calendar and find details for tomorrow's 'Team Standup' meeting", - config=config, - tools=AdbTools(), - output_model=CalendarEvent, -) - -result = await agent.run() -event = result.structured_output -# event.title → "Team Standup" -# event.start_time → "9:00 AM" -# len(event.attendees) → 5 -``` - -### Example 6: Form Validation After Fill - -Verify a form was filled correctly: - -```python -from pydantic import BaseModel, Field - -class FormData(BaseModel): - """Filled form data verification.""" - first_name: str = Field(description="First name entered in form") - last_name: str = Field(description="Last name entered in form") - email: str = Field(description="Email address entered in form") - phone: str = Field(description="Phone number entered in form") - address: str = Field(description="Street address entered in form") - city: str = Field(description="City entered in form") - state: str = Field(description="State entered in form") - zip_code: str = Field(description="ZIP code entered in form") - form_submitted: bool = Field( - description="Whether form was successfully submitted" - ) - confirmation_message: str = Field( - description="Confirmation message shown after submission" - ) - -# Usage -config = DroidRunConfig() - -agent = DroidAgent( - goal="Fill out the registration form with my details and verify it was submitted correctly", - config=config, - tools=AdbTools(), - output_model=FormData, -) - -result = await agent.run() -form_data = result.structured_output - -# Verify form was filled correctly -assert form_data.form_submitted, "Form was not submitted" -assert "@" in form_data.email, "Invalid email format" -print(f"Confirmation: {form_data.confirmation_message}") +print(f"Invoice {invoice.invoice_number}: ${invoice.total_due}") ``` --- ## Working with Results -### Accessing Structured Data - -The `ResultEvent` object returned by `agent.run()` contains: -- `success`: Boolean indicating task completion -- `reason`: Natural language answer from agent -- `structured_output`: Extracted Pydantic model instance (or `None`) -- `steps`: Number of execution steps taken +### Accessing Data ```python result = await agent.run() -# Check if task succeeded if result.success: - print(f"Task completed: {result.reason}") - - # Access structured data if result.structured_output: - data = result.structured_output - # Now you have a typed Pydantic object - print(f"Extracted data: {data}") + data = result.structured_output # Typed Pydantic object + print(f"Extracted: {data}") else: - print("Warning: Extraction failed, but task succeeded") + print(f"Extraction failed, text answer: {result.reason}") else: print(f"Task failed: {result.reason}") ``` -### Handling Extraction Failures - -Extraction can fail if: -- The agent's answer doesn't contain the required information -- The LLM can't parse the answer into the specified format -- Network/API errors occur during extraction - -**Graceful Handling:** - -```python -result = await agent.run() - -if not result.success: - # Task execution failed - print(f"Task failed: {result.reason}") - return None - -if result.structured_output is None: - # Extraction failed, but we have the text answer - print("Extraction failed, falling back to text answer:") - print(result.reason) - - # Optionally parse manually or retry - return None - -# Success - use structured data -data = result.structured_output -return data -``` - -### Validation and Post-Processing - -Pydantic automatically validates data types. Add custom validation: - -```python -from pydantic import BaseModel, Field, field_validator - -class PriceInfo(BaseModel): - """Product price information.""" - product_name: str = Field(description="Name of the product") - current_price: float = Field(description="Current price in dollars") - original_price: float = Field(description="Original price in dollars") - - @field_validator('current_price', 'original_price') - @classmethod - def validate_price(cls, v): - if v < 0: - raise ValueError("Price cannot be negative") - return round(v, 2) # Round to 2 decimal places - - @field_validator('current_price') - @classmethod - def validate_discount(cls, v, info): - # Ensure current price isn't higher than original - if 'original_price' in info.data: - if v > info.data['original_price']: - raise ValueError("Current price cannot exceed original price") - return v - -# Usage -result = await agent.run() -if result.structured_output: - # Validation happens automatically - price_info = result.structured_output - - # Calculate discount - if price_info.original_price > price_info.current_price: - discount = ((price_info.original_price - price_info.current_price) - / price_info.original_price * 100) - print(f"Discount: {discount:.1f}%") -``` - ### Exporting to JSON -Convert structured output to JSON for storage or APIs: - ```python result = await agent.run() if result.structured_output: - data = result.structured_output - - # Convert to dict - data_dict = data.model_dump() - - # Convert to JSON string - import json - json_str = data.model_dump_json(indent=2) - - # Save to file + # Convert to JSON and save + json_str = result.structured_output.model_dump_json(indent=2) with open("output.json", "w") as f: f.write(json_str) - - print("Data saved to output.json") ``` --- ## Configuration -### Specifying the Extraction LLM +### Custom Extraction LLM -By default, structured output extraction uses the `codeact` LLM from your config. You can specify a dedicated `structured_output` LLM profile for better control over extraction: - -**In config.yaml:** +By default, extraction uses the `codeact` LLM. Specify a dedicated `structured_output` profile: +**config.yaml:** ```yaml llm_profiles: - # Main execution LLMs codeact: provider: GoogleGenAI model: models/gemini-2.0-flash temperature: 0.3 - # Dedicated extraction LLM structured_output: provider: OpenAI model: gpt-4o-mini - temperature: 0.0 # Low temperature for consistent extraction + temperature: 0.0 # Low temp for consistent extraction ``` **Programmatically:** - ```python from droidrun.llm_utils import load_llm -from droidrun.config_manager.config_manager import DroidRunConfig -config = DroidRunConfig() +config = DroidrunConfig() -# Load LLMs llms = { "codeact": load_llm("GoogleGenAI", "models/gemini-2.0-flash"), "structured_output": load_llm("OpenAI", "gpt-4o-mini"), @@ -706,555 +155,127 @@ agent = DroidAgent( goal="Extract contact info for Alice", llms=llms, config=config, - tools=AdbTools(), output_model=ContactInfo, ) ``` -### Reasoning Mode Considerations +### Reasoning Mode -Structured output works with both reasoning modes: +Works in both direct and reasoning modes: -**Non-Reasoning Mode (Direct Execution):** ```python -config = DroidRunConfig() +# Direct mode +config = DroidrunConfig() config.agent.reasoning = False agent = DroidAgent( - goal="Find the weather for San Francisco", + goal="Find weather for SF", config=config, - tools=AdbTools(), - output_model=WeatherInfo, # Schema guides CodeActAgent + output_model=WeatherInfo, ) -``` -**Reasoning Mode (Manager/Executor):** -```python -config = DroidRunConfig() +# Reasoning mode config.agent.reasoning = True agent = DroidAgent( - goal="Find the weather for San Francisco", + goal="Find weather for SF", config=config, - tools=AdbTools(), - output_model=WeatherInfo, # Schema guides ManagerAgent planning + output_model=WeatherInfo, ) ``` -In both modes: -- The output schema is injected into system prompts -- Agents are instructed to collect the required data -- Extraction happens post-completion regardless of mode - --- ## Best Practices -### 1. Write Clear Field Descriptions - -The LLM uses field descriptions to understand what to extract. Be specific: - -**Bad:** +**1. Add clear field descriptions** - The LLM uses these to understand what to extract: ```python -class Data(BaseModel): - name: str # No description - value: float = Field(description="Value") # Too vague +name: str = Field(description="Full name of customer who placed order") ``` -**Good:** +**2. Provide defaults for optional fields** - Prevents extraction failures: ```python -class Data(BaseModel): - """Customer order information.""" - customer_name: str = Field( - description="Full name of the customer who placed the order" - ) - order_total: float = Field( - description="Total order amount in US dollars including tax and shipping" - ) +rating: Optional[float] = Field(description="Customer rating (1-5)", default=None) ``` -### 2. Use Appropriate Types - -Choose types that match your data: - -```python -class Event(BaseModel): - # String for dates if format varies - date: str = Field(description="Event date in YYYY-MM-DD format") - - # Int for counts - attendee_count: int = Field(description="Number of attendees") - - # Float for prices/measurements - ticket_price: float = Field(description="Ticket price in dollars") - - # Bool for yes/no - is_virtual: bool = Field(description="Whether event is online/virtual") - - # List for multiple items - speakers: List[str] = Field(description="Names of all speakers") -``` - -### 3. Provide Default Values - -Use defaults for optional fields to prevent extraction failures: - -```python -class Product(BaseModel): - name: str = Field(description="Product name") - - # Required field - no default - price: float = Field(description="Current price") - - # Optional fields - with defaults - description: str = Field( - description="Product description", - default="No description available" - ) - rating: Optional[float] = Field( - description="Customer rating out of 5.0", - default=None - ) - in_stock: bool = Field( - description="Availability status", - default=True - ) -``` - -### 4. Keep Models Focused - -Create targeted models for specific tasks instead of one large model: - -**Bad - Too broad:** -```python -class Everything(BaseModel): - """All possible data.""" - contact_name: Optional[str] = None - contact_phone: Optional[str] = None - email_subject: Optional[str] = None - email_body: Optional[str] = None - calendar_event: Optional[str] = None - # ... too many optional fields -``` - -**Good - Focused:** -```python -class ContactInfo(BaseModel): - """Contact details only.""" - name: str = Field(description="Contact name") - phone: str = Field(description="Phone number") - -class EmailSummary(BaseModel): - """Email summary only.""" - subject: str = Field(description="Email subject") - sender: str = Field(description="Sender name/email") - summary: str = Field(description="Brief summary of email body") -``` - -### 5. Test Your Models - -Validate models before production use: - -```python -# Test model validation -def test_model(): - # Test valid data - valid_data = ContactInfo( - name="John Doe", - phone="+1-555-0123", - email="john@example.com" - ) - assert valid_data.name == "John Doe" - - # Test validation (should raise error) - try: - invalid_data = ContactInfo( - name="", # Empty name - phone="invalid", - email="not-an-email" - ) - except ValidationError as e: - print(f"Validation caught errors: {e}") - -test_model() -``` - -### 6. Guide Collection in Goal - -Mention what data to collect in your goal: - +**3. Guide data collection in your goal**: ```python agent = DroidAgent( - goal="Open the Notes app and find the recipe for chocolate cake. " - "Extract the recipe name, ingredients list, cooking time, and instructions.", - output_model=Recipe, - # ... other params + goal="Find contact and get their phone number, email, and full name", + config=config, + output_model=ContactInfo, ) ``` -This helps the agent understand what to look for during execution. - -### 7. Handle Missing Data Gracefully - -Not all information may be available. Design for partial data: - -```python -class ProductInfo(BaseModel): - """Product information (some fields may be unavailable).""" - name: str = Field(description="Product name") - price: float = Field(description="Current price in dollars") - - # Optional fields with clear defaults - description: str = Field( - description="Product description", - default="Description not available" - ) - reviews_count: int = Field( - description="Number of customer reviews", - default=0 - ) - availability: str = Field( - description="Stock availability status", - default="Unknown" - ) - -# Use with fallbacks -result = await agent.run() -if result.structured_output: - product = result.structured_output - - # Always available - print(f"{product.name}: ${product.price}") - - # May be defaults - if product.reviews_count > 0: - print(f"Reviews: {product.reviews_count}") - if product.description != "Description not available": - print(f"Description: {product.description}") -``` - --- ## Troubleshooting -### Extraction Always Returns None +**Extraction returns None:** +- Verify `output_model` is passed to `DroidAgent` +- Check if task succeeded: `result.success` +- Enable debug logging: `config.logging.debug = True` -**Problem:** `structured_output` is always `None` even when task succeeds. +**Partial or incorrect data:** +- Add more specific field descriptions +- Mention required fields explicitly in the goal -**Solutions:** - -1. **Check if task actually completed:** - ```python - result = await agent.run() - if not result.success: - print("Task failed - no extraction attempted") - else: - print(f"Task succeeded with answer: {result.reason}") - # If reason is empty, extraction won't run - ``` - -2. **Verify output_model is provided:** - ```python - # Wrong - no output_model - agent = DroidAgent(goal="...", config=config, tools=tools) - - # Correct - agent = DroidAgent( - goal="...", - config=config, - tools=tools, - output_model=YourModel # Must be provided - ) - ``` - -3. **Check extraction LLM logs:** - ```python - config = DroidRunConfig() - config.logging.debug = True # Enable debug logging - - # Look for extraction logs: - # "🔄 Running structured output extraction..." - # "✅ Successfully extracted structured output" - # OR - # "⚠️ Structured extraction failed: [error]" - ``` - -### Partial or Incorrect Data - -**Problem:** Some fields are missing or have wrong values. - -**Solutions:** - -1. **Improve field descriptions:** - ```python - # Vague - LLM might guess wrong - phone: str = Field(description="Phone") - - # Specific - LLM knows exactly what to extract - phone: str = Field( - description="Primary phone number in international format (e.g., +1-555-0123)" - ) - ``` - -2. **Mention required fields in goal:** - ```python - agent = DroidAgent( - goal="Find contact info and make sure to get the phone number, " - "email, and full name", - output_model=ContactInfo, - # ... - ) - ``` - -3. **Use required vs optional strategically:** - ```python - class Info(BaseModel): - # Critical fields - no default (will fail if missing) - name: str = Field(description="...") - - # Nice-to-have - with default (won't fail if missing) - notes: Optional[str] = Field(description="...", default=None) - ``` - -### Validation Errors - -**Problem:** Pydantic validation fails with type errors. - -**Solutions:** - -1. **Use string types for unpredictable formats:** - ```python - # Instead of strict types - date: datetime = Field(description="Event date") - - # Use strings with format guidance - date: str = Field( - description="Event date in YYYY-MM-DD format (e.g., 2024-01-15)" - ) - ``` - -2. **Add custom validators:** - ```python - from pydantic import field_validator - - class Data(BaseModel): - price: float = Field(description="Price in dollars") - - @field_validator('price') - @classmethod - def parse_price(cls, v): - # Handle string prices like "$19.99" - if isinstance(v, str): - v = v.replace('$', '').replace(',', '') - return float(v) - return v - ``` - -3. **Use `Optional` for uncertain fields:** - ```python - # If field might not exist or have varied types - rating: Optional[float] = Field( - description="Rating from 1.0 to 5.0", - default=None - ) - ``` - -### Slow Extraction - -**Problem:** Extraction takes too long after task completion. - -**Solutions:** - -1. **Use a faster LLM for extraction:** - ```yaml - llm_profiles: - structured_output: - provider: OpenAI - model: gpt-4o-mini # Faster than gpt-4o - temperature: 0.0 - ``` - -2. **Simplify your model:** - ```python - # Complex model with many nested objects - class Complex(BaseModel): - # Many fields, nested models, long descriptions... - - # Simplified model with essential fields only - class Simple(BaseModel): - name: str = Field(description="Name") - value: float = Field(description="Value") - ``` - -3. **Pre-filter data in agent goal:** - ```python - # Instead of asking agent to extract everything - goal = "Get all information about the product" - - # Ask for specific fields only - goal = "Get the product name and price only" - ``` +**Validation errors:** +- Add `Optional` and defaults for uncertain fields --- -## Advanced Usage +## Advanced -### Streaming Support +### Multiple Items -Note: Structured output extraction happens post-completion and doesn't support streaming. However, you can stream the main task execution: +Extract lists of data using a model with `List` fields: ```python -from droidrun.config_manager.config_manager import DroidRunConfig - -config = DroidRunConfig() - -agent = DroidAgent( - goal="Extract contact for Jane Doe", - config=config, - tools=AdbTools(), - output_model=ContactInfo, -) - -handler = agent.run() - -# Stream task execution events -async for event in handler.stream_events(): - if isinstance(event, TaskThinkingEvent): - print(f"Agent thinking: {event.thoughts}") - elif isinstance(event, ManagerPlanEvent): - print(f"Plan: {event.plan}") - -# Get final result with structured output -result = await handler -contact = result.structured_output -``` - -### Multiple Extractions - -For tasks requiring multiple structured outputs, run separate agents: - -```python -from droidrun.config_manager.config_manager import DroidRunConfig - -config = DroidRunConfig() - -# First extraction -agent1 = DroidAgent( - goal="Find contact for John Smith", - config=config, - tools=tools, - output_model=ContactInfo, -) -result1 = await agent1.run() -contact1 = result1.structured_output - -# Second extraction -agent2 = DroidAgent( - goal="Find contact for Jane Doe", - config=config, - tools=tools, - output_model=ContactInfo, -) -result2 = await agent2.run() -contact2 = result2.structured_output - -# Process together -contacts = [contact1, contact2] -``` - -Alternatively, use a list model: - -```python -from droidrun.config_manager.config_manager import DroidRunConfig - -config = DroidRunConfig() - class ContactList(BaseModel): """Multiple contacts.""" - contacts: List[ContactInfo] = Field(description="List of contacts found") + contacts: List[ContactInfo] = Field(description="List of contacts") agent = DroidAgent( goal="Find contacts for John Smith and Jane Doe", config=config, - tools=tools, output_model=ContactList, ) - -result = await agent.run() -contacts = result.structured_output.contacts ``` -### Custom Extraction Prompts +### Workflow Integration -The default extraction prompt is: - -``` -Extract structured information from the following text: - -{text} -``` - -The `StructuredOutputAgent` uses LlamaIndex's `astructured_predict()` which automatically handles the extraction. If you need custom extraction logic, you can use `StructuredOutputAgent` directly: +Extraction happens automatically in `DroidAgent.finalize()`: ```python -from droidrun.agent.oneflows.structured_output_agent import StructuredOutputAgent -from droidrun.config_manager.config_manager import DroidRunConfig +@step +async def finalize(self, ctx: Context, ev: FinalizeEvent) -> ResultEvent: + result = ResultEvent( + success=ev.success, + reason=ev.reason, + steps=self.shared_state.step_number, + structured_output=None, + ) -config = DroidRunConfig() + # Extract if model was provided + if self.output_model is not None and ev.reason: + structured_agent = StructuredOutputAgent( + llm=self.structured_output_llm, + pydantic_model=self.output_model, + answer_text=ev.reason, + ) + extraction_result = await (await structured_agent.run()) + if extraction_result["success"]: + result.structured_output = extraction_result["structured_output"] -# Run agent task first -agent = DroidAgent( - goal="Find contact for John Smith", - config=config, - tools=AdbTools(), -) -result = await agent.run() - -# Then run custom extraction -extraction_agent = StructuredOutputAgent( - llm=llm, - pydantic_model=ContactInfo, - answer_text=result.reason, -) - -extraction_result = await extraction_agent.run() -if extraction_result["success"]: - contact = extraction_result.structured_output -``` - -**Note:** The extraction prompt is hardcoded in `StructuredOutputAgent`. For most use cases, improving your Pydantic model's field descriptions is more effective than customizing the extraction prompt. - -### Combining with Variables - -Use custom variables alongside structured output: - -```python -from droidrun.config_manager.config_manager import DroidRunConfig - -config = DroidRunConfig() - -agent = DroidAgent( - goal="Find recipe for {{ recipe_name }} and extract details", - config=config, - tools=tools, - variables={"recipe_name": "Chocolate Cake"}, - output_model=Recipe, -) - -result = await agent.run() -recipe = result.structured_output + return result ``` --- ## Related Documentation -- [DroidAgent API](/docs/v4/sdk/droid-agent) - Main agent documentation -- [Pydantic Documentation](https://docs.pydantic.dev/) - Learn more about Pydantic models -- [LlamaIndex structured_predict()](https://docs.llamaindex.ai/) - Underlying extraction mechanism -- [Configuration Guide](/docs/v4/concepts/configuration) - LLM profile configuration -- [Custom Variables](/docs/v4/guides/variables) - Using variables with structured output - ---- - -**Extract structured data from your automation tasks with type-safe, validated models!** +- [DroidAgent API](/v4/sdk/droid-agent) +- [Pydantic Documentation](https://docs.pydantic.dev/) +- [Configuration Guide](/v4/sdk/configuration) +- [Custom Variables](/v4/guides/custom-variables) diff --git a/docs/v4/guides/telemetry-tracing.mdx b/docs/v4/guides/telemetry-tracing.mdx index 64c1f51..2320255 100644 --- a/docs/v4/guides/telemetry-tracing.mdx +++ b/docs/v4/guides/telemetry-tracing.mdx @@ -5,7 +5,7 @@ description: 'Configure anonymous telemetry, Phoenix tracing, and trajectory rec # Telemetry & Tracing -DroidRun provides three monitoring capabilities: +Droidrun provides three monitoring capabilities: 1. **Anonymous Telemetry** - Usage analytics sent to PostHog (enabled by default) 2. **Arize Phoenix Tracing** - Real-time LLM and agent execution tracing (opt-in) @@ -46,7 +46,7 @@ phoenix serve The server starts at `http://localhost:6006` and provides a web UI for viewing traces. -**3. Enable tracing in DroidRun:** +**3. Enable tracing in Droidrun:** **Via CLI:** @@ -65,9 +65,9 @@ tracing: ```python from droidrun import DroidAgent -from droidrun.config_manager.config_manager import DroidRunConfig +from droidrun.config_manager.config_manager import DroidrunConfig -config = DroidRunConfig() +config = DroidrunConfig() config.tracing.enabled = True agent = DroidAgent(goal="Open settings", config=config) @@ -104,37 +104,22 @@ Environment variable names are lowercase: `phoenix_url` and `phoenix_project_nam ## Anonymous Telemetry -DroidRun collects anonymous usage analytics via PostHog to help improve the framework. This is completely separate from Phoenix tracing and trajectory recording. - -**Telemetry is enabled by default.** The framework tracks agent initialization, task completion, and usage patterns to help prioritize features and improve reliability. +Droidrun collects anonymous usage analytics via PostHog to help improve the framework. Telemetry is enabled by default. ### What's Collected -**Agent initialization:** -- LLM provider and model names (e.g., "GoogleGenAI", "models/gemini-2.5-pro") -- Configuration settings (max steps, timeout, vision/reasoning mode, debug flags) -- Tool platform (e.g., "AdbTools", "IOSTools") -- Trajectory save level ("none", "step", "action") -- Run type (CLI, developer, web) - -**During execution:** -- App packages and activities visited (e.g., "com.android.settings") -- Step number when visiting new packages - -**Task completion:** -- Success/failure status and reason -- Total steps taken -- Count of unique packages and activities visited - -**User identifier:** +- Task goals and completion status +- LLM providers and configuration settings (max steps, timeout, vision/reasoning mode, debug flags) +- Available tools and packages/activities visited - Anonymous UUID stored in `~/.droidrun/user_id` -- No personal information attached -**What's NOT collected:** +### What's NOT Collected + - Screenshots or screen content -- User input or task goals (the actual text you type) - API keys or credentials -- Device serial numbers or personal identifiers +- LLM responses +- Device serial numbers +- Detailed action history ### Disable Telemetry @@ -190,9 +175,9 @@ logging: ```python from droidrun import DroidAgent -from droidrun.config_manager.config_manager import DroidRunConfig +from droidrun.config_manager.config_manager import DroidrunConfig -config = DroidRunConfig() +config = DroidrunConfig() config.logging.save_trajectory = "action" agent = DroidAgent(goal="Open settings", config=config) @@ -232,7 +217,7 @@ Use these files to: | Feature | Data Location | Default | Purpose | Opt-In/Out | |---------|--------------|---------|---------|------------| -| **Anonymous Telemetry** | PostHog (cloud) | Enabled | Help improve DroidRun | `DROIDRUN_TELEMETRY_ENABLED=false` | +| **Anonymous Telemetry** | PostHog (cloud) | Enabled | Help improve Droidrun | `DROIDRUN_TELEMETRY_ENABLED=false` | | **Phoenix Tracing** | Phoenix server (local/cloud) | Disabled | Debug LLM calls and agent flow | `--tracing` flag or config | | **Trajectory Recording** | Local disk (`trajectories/`) | Disabled | Offline debugging with screenshots | `--save-trajectory step/action` | @@ -240,6 +225,6 @@ Use these files to: ## Related Documentation -- [Configuration System](/docs/v4/guides/configuration) - Configure tracing and telemetry settings -- [Event Streaming](/docs/v4/concepts/event-streaming) - Build custom monitoring integrations -- [CLI Usage](/docs/v4/guides/cli) - Command-line flags for monitoring +- [Configuration System](/v4/sdk/configuration) - Configure tracing and telemetry settings +- [Events and Workflows](/v4/concepts/events-and-workflows) - Build custom monitoring integrations +- [CLI Usage](/v4/guides/cli) - Command-line flags for monitoring diff --git a/docs/v4/overview.mdx b/docs/v4/overview.mdx index c6ad043..f37c5d3 100644 --- a/docs/v4/overview.mdx +++ b/docs/v4/overview.mdx @@ -1,22 +1,22 @@ --- title: 'Overview' -description: 'DroidRun is a powerful framework that enables you to control Android and iOS devices through intelligent LLM agents. Build sophisticated mobile automation workflows with natural language commands.' +description: 'Droidrun is a powerful framework that enables you to control Android and iOS devices through intelligent LLM agents. Build sophisticated mobile automation workflows with natural language commands.' --- -## What is DroidRun? +## What is Droidrun? -DroidRun empowers developers to automate mobile device interactions using AI-powered agents. Whether you're building testing frameworks, automating data collection, or creating intelligent mobile workflows, DroidRun provides the flexibility and power you need. +Droidrun empowers developers to automate mobile device interactions using AI-powered agents. Whether you're building testing frameworks, automating data collection, or creating intelligent mobile workflows, Droidrun provides the flexibility and power you need. -Built on [LlamaIndex workflows](https://docs.llamaindex.ai/en/stable/understanding/workflows/), DroidRun features a sophisticated multi-agent architecture that can handle everything from simple atomic tasks to complex multi-step workflows requiring strategic planning and error recovery. +Built on [LlamaIndex workflows](https://docs.llamaindex.ai/en/stable/understanding/workflows/), Droidrun features a sophisticated multi-agent architecture that can handle everything from simple atomic tasks to complex multi-step workflows requiring strategic planning and error recovery. - Get up and running with DroidRun in minutes + Get up and running with Droidrun in minutes - + Understand the hierarchical agent system - + Monitor and debug with real-time events @@ -30,7 +30,7 @@ Built on [LlamaIndex workflows](https://docs.llamaindex.ai/en/stable/understandi ### Multi-Agent Architecture -DroidRun v4 introduces a hierarchical multi-agent system with specialized agents for different responsibilities: +Droidrun v4 introduces a hierarchical multi-agent system with specialized agents for different responsibilities: - **DroidAgent** - Main coordinator orchestrating execution flow - **ManagerAgent** - High-level planning and strategic decision making @@ -42,21 +42,21 @@ DroidRun v4 introduces a hierarchical multi-agent system with specialized agents - **Direct Mode** (`reasoning=False`) - Fast, single-agent execution for simple tasks - **Reasoning Mode** (`reasoning=True`) - Strategic planning with Manager/Executor workflow for complex tasks - + Deep dive into the agent architecture ### Event Streaming System -Monitor and debug agent execution in real-time with DroidRun's comprehensive event system: +Monitor and debug agent execution in real-time with Droidrun's comprehensive event system: ```python from droidrun import DroidAgent, ResultEvent from droidrun.agent.droid.events import ManagerPlanEvent, ExecutorResultEvent -from droidrun.config_manager.config_manager import DroidRunConfig +from droidrun.config_manager.config_manager import DroidrunConfig # Create config with defaults -config = DroidRunConfig() +config = DroidrunConfig() agent = DroidAgent(goal="Open Settings and enable WiFi", config=config) handler = agent.run() @@ -77,7 +77,7 @@ Events provide rich metadata for: - Debugging and trajectory analysis - Integration with external systems - + Master the event system @@ -116,7 +116,7 @@ llm_profiles: - Ollama - DeepSeek - + Configure your agents @@ -127,7 +127,7 @@ Extract type-safe, validated data from device interactions using Pydantic models ```python from pydantic import BaseModel, Field from droidrun import DroidAgent, ResultEvent -from droidrun.config_manager.config_manager import DroidRunConfig +from droidrun.config_manager.config_manager import DroidrunConfig class ContactInfo(BaseModel): """Contact information extracted from device.""" @@ -136,7 +136,7 @@ class ContactInfo(BaseModel): email: str = Field(description="Email address") # Create config with defaults -config = DroidRunConfig() +config = DroidrunConfig() agent = DroidAgent( goal="Find John Smith's contact and extract details", @@ -162,17 +162,18 @@ Extend agent capabilities with custom tools and secure credential management: ```python from droidrun import DroidAgent -from droidrun.config_manager.config_manager import DroidRunConfig +from droidrun.config_manager.config_manager import DroidrunConfig -def search_database(query: str) -> str: +def search_database(query: str, **kwargs) -> str: """Search the local database.""" - # Your implementation - return f"Results for: {query}" + # Your database search implementation + results = db.query(query) # Example + return f"Found {len(results)} results for '{query}'" custom_tools = { "search_database": { - "signature": "search_database(query: str) -> str", - "description": "Search the local database for information", + "arguments": ["query"], + "description": "Search local database. Usage: {\"action\": \"search_database\", \"query\": \"search term\"}", "function": search_database } } @@ -184,7 +185,7 @@ credentials = { } # Create config with defaults -config = DroidRunConfig() +config = DroidrunConfig() agent = DroidAgent( goal="Search database and email results", @@ -253,7 +254,7 @@ Inject dynamic data into agent prompts: ```python from droidrun import DroidAgent -from droidrun.config_manager.config_manager import DroidRunConfig +from droidrun.config_manager.config_manager import DroidrunConfig # Define custom prompts that render variables custom_prompts = { @@ -270,7 +271,7 @@ Available variables: } # Create config with defaults -config = DroidRunConfig() +config = DroidrunConfig() agent = DroidAgent( goal="Complete task using context", @@ -287,7 +288,7 @@ agent = DroidAgent( Variables are accessible in: - ✅ Agent prompts (via custom Jinja2 templates) - ✅ Agents can read and pass to tools as arguments -- ❌ NOT directly accessible to custom tools +- ✅ Accessible to custom tools Use dynamic variables @@ -297,7 +298,7 @@ Variables are accessible in: ## Two Deployment Options -DroidRun offers flexible deployment to match your workflow: +Droidrun offers flexible deployment to match your workflow: @@ -333,128 +334,19 @@ DroidRun offers flexible deployment to match your workflow: --- -## What's New in v4 - -DroidRun v4 is a major architectural upgrade from v3, introducing powerful new capabilities: - -### New Agent System -- **ScripterAgent** - Off-device Python code execution for computations, API calls, and data processing -- **Structured Output** - Automatic extraction of typed data using Pydantic models -- **Helper Workflows** - App opener and text manipulator for specialized tasks -- **Multi-agent coordination** - Hierarchical workflow orchestration with shared state - -### Event Streaming -- **Real-time event system** - Stream execution events for monitoring and debugging -- **Rich event types** - Workflow events (Manager, Executor, Scripter), action events, and state events -- **Custom handlers** - Build custom UIs, webhooks, and integrations -- **Trajectory recording** - Capture screenshots, UI states, and action history - -### LLM Configuration -- **Per-agent LLM profiles** - Different models for Manager, Executor, CodeAct, etc. -- **Profile system** - Reusable LLM configurations in YAML -- **Provider flexibility** - Mix and match providers (OpenAI, Anthropic, Google, etc.) -- **Cost optimization** - Use powerful models for planning, fast models for actions - -### Enhanced Features -- **Custom variables** - Store and access custom data throughout execution -- **Structured output** - Type-safe data extraction with Pydantic models -- **Safe execution** - Sandboxed code execution with import/builtin restrictions -- **Credentials** - Auto-injection as custom tools with direct dict or file-based config -- **App cards** - App-specific guidance (local, server, or composite modes) -- **Improved error handling** - Error escalation and plan adjustments - -### Developer Experience -- **DroidRunConfig** - Clean configuration API with `from_yaml()` loading -- **Improved logging** - Rich console output with progress tracking -- **Tracing integration** - Arize Phoenix support for execution tracing -- **Better type hints** - Complete type annotations and IDE support -- **Event-driven architecture** - LlamaIndex workflows for easy integration - ---- - -## Quick Start - -### Installation - -```bash -# Install for cli usage. -uv tool install 'droidrun[google,anthropic,openai,deepseek,ollama,openrouter]' - -# Or for developing on top of Droidrun. -uv pip install 'droidrun[google,anthropic,openai,deepseek,ollama,openrouter]' -``` - -### Basic Usage - -```python -import asyncio -from droidrun import DroidAgent, ResultEvent -from droidrun.config_manager.config_manager import DroidRunConfig - -async def main(): - # Create config with defaults - config = DroidRunConfig() - config.agent.max_steps = 20 - config.agent.reasoning = True - - # Or load configuration from YAML - # config = DroidRunConfig.from_yaml("config.yaml") - - # Create agent (tools auto-created from device config) - agent = DroidAgent( - goal="Open Settings and enable WiFi", - config=config - ) - - # Run agent and get result - handler = agent.run() - result: ResultEvent = await handler - - if result.success: - print(f"Success: {result.reason}") - else: - print(f"Failed: {result.reason}") - -asyncio.run(main()) -``` - -### CLI Usage - -```bash -# Run a command -droidrun run "Open Settings and enable WiFi" - -# With reasoning mode -droidrun run "Set up a new alarm for 7 AM" --reasoning - -# With vision -droidrun run "Take a screenshot and describe what's on screen" --vision - -# Device management -droidrun devices # List devices -droidrun setup # Install Portal APK -droidrun ping # Test connection -``` - - - Follow the full quickstart guide - - ---- - ## Key Concepts - + Multi-agent system with specialized roles - + LlamaIndex workflow integration - + Real-time monitoring and debugging - + Agent, LLM, and system configuration @@ -499,7 +391,7 @@ droidrun ping # Test connection Follow for updates and news - See DroidRun's performance metrics + See Droidrun's performance metrics Try the managed cloud service diff --git a/docs/v4/quickstart.mdx b/docs/v4/quickstart.mdx index dfadabd..98adbdf 100644 --- a/docs/v4/quickstart.mdx +++ b/docs/v4/quickstart.mdx @@ -1,6 +1,6 @@ --- title: 'Quickstart' -description: 'Get up and running with DroidRun v4 quickly and effectively' +description: 'Get up and running with Droidrun v4 quickly and effectively' --- -This guide will help you get DroidRun v4 installed and running quickly, controlling your Android device through natural language in minutes. DroidRun v4 introduces a config-driven architecture, making it easier to customize agent behavior, LLM selection, and execution modes. +This guide will help you get Droidrun v4 installed and running quickly, controlling your Android device through natural language in minutes. Droidrun v4 introduces a config-driven architecture, making it easier to customize agent behavior, LLM selection, and execution modes. ### Prerequisites -Before installing DroidRun, ensure you have: +Before installing Droidrun, ensure you have: 1. **Python 3.11+** installed on your system 2. [Android Debug Bridge (adb)](https://developer.android.com/studio/releases/platform-tools) installed and configured @@ -27,7 +27,7 @@ Before installing DroidRun, ensure you have: ### Installation -DroidRun v4 is installed using [`uv`](https://docs.astral.sh/uv/), a fast Python package installer and resolver. +Droidrun v4 is installed using [`uv`](https://docs.astral.sh/uv/), a fast Python package installer and resolver. **Install uv (if not already installed):** @@ -57,7 +57,7 @@ If you only need specific providers, you can install just those. For example, `u ### Setup the Portal APK -DroidRun requires the Portal app to be installed on your Android device for device control. The Portal app provides accessibility services that expose the UI accessibility tree, enabling the agent to see and interact with UI elements. +Droidrun requires the Portal app to be installed on your Android device for device control. The Portal app provides accessibility services that expose the UI accessibility tree, enabling the agent to see and interact with UI elements. ```bash droidrun setup @@ -70,7 +70,7 @@ This command automatically: ### Test Connection -Verify that DroidRun can communicate with your device: +Verify that Droidrun can communicate with your device: ```bash droidrun ping @@ -85,7 +85,7 @@ If successful, you'll see: ### Configure Your LLM -DroidRun v4 uses a configuration-driven approach. On first run, DroidRun creates a `config.yaml` file with default settings. You'll need to set your API key for your chosen LLM provider. +Droidrun v4 uses a configuration-driven approach. On first run, Droidrun creates a `config.yaml` file with default settings. You'll need to set your API key for your chosen LLM provider. **Set your API key:** @@ -136,11 +136,11 @@ For complex automation or integration into your Python projects, create a script ```python import asyncio from droidrun import DroidAgent -from droidrun.config_manager.config_manager import DroidRunConfig +from droidrun.config_manager.config_manager import DroidrunConfig async def main(): # Use default configuration with built-in LLM profiles - config = DroidRunConfig() + config = DroidrunConfig() # Create agent # LLMs are automatically loaded from config.llm_profiles @@ -166,12 +166,12 @@ if __name__ == "__main__": ```python import asyncio from droidrun import DroidAgent -from droidrun.config_manager.config_manager import DroidRunConfig, AgentConfig +from droidrun.config_manager.config_manager import DroidrunConfig, AgentConfig from llama_index.llms.openai import OpenAI async def main(): # Load base configuration - config = DroidRunConfig() + config = DroidrunConfig() # Create custom agent config with overrides agent_config = AgentConfig( @@ -201,23 +201,23 @@ if __name__ == "__main__": ### Configuration Options -DroidRun v4 provides flexible configuration through: +Droidrun v4 provides flexible configuration through: **1. Default Configuration (No File Required):** ```python -config = DroidRunConfig() # Uses built-in defaults +config = DroidrunConfig() # Uses built-in defaults ``` **2. Load from YAML File:** ```python -config = DroidRunConfig.from_yaml("config.yaml") # Load custom config +config = DroidrunConfig.from_yaml("config.yaml") # Load custom config ``` **3. Programmatic Configuration:** ```python -from droidrun.config_manager.config_manager import DroidRunConfig, AgentConfig +from droidrun.config_manager.config_manager import DroidrunConfig, AgentConfig -config = DroidRunConfig() +config = DroidrunConfig() config.agent.max_steps = 25 config.agent.reasoning = True config.agent.codeact.vision = True @@ -233,18 +233,18 @@ config.agent.codeact.vision = True -For detailed configuration options, see the [Configuration Guide](/v4/guides/configuration). The CLI uses `ConfigManager` which auto-creates config.yaml on first run. +For detailed configuration options, see the [Configuration Guide](/v4/sdk/configuration). The CLI uses `ConfigManager` which auto-creates config.yaml on first run. ### Structured Output Extraction -DroidRun v4 supports structured output extraction using Pydantic models: +Droidrun v4 supports structured output extraction using Pydantic models: ```python import asyncio from pydantic import BaseModel, Field from droidrun import DroidAgent -from droidrun.config_manager.config_manager import DroidRunConfig +from droidrun.config_manager.config_manager import DroidrunConfig class BatteryInfo(BaseModel): """Battery information.""" @@ -253,7 +253,7 @@ class BatteryInfo(BaseModel): temperature: float = Field(description="Battery temperature in Celsius") async def main(): - config = DroidRunConfig() + config = DroidrunConfig() agent = DroidAgent( goal="Check battery status", @@ -276,7 +276,7 @@ if __name__ == "__main__": ### Execution Modes -DroidRun v4 supports two execution modes: +Droidrun v4 supports two execution modes: **Direct Mode (Default):** - Fast execution for simple tasks @@ -290,9 +290,9 @@ droidrun run "Open YouTube" ```python # In scripts from droidrun import DroidAgent -from droidrun.config_manager.config_manager import DroidRunConfig +from droidrun.config_manager.config_manager import DroidrunConfig -config = DroidRunConfig() +config = DroidrunConfig() agent = DroidAgent(goal="Open YouTube", config=config) ``` @@ -309,9 +309,9 @@ droidrun run "Find contacts from California and export to CSV" --reasoning ```python # In scripts - Option 1: Override config from droidrun import DroidAgent -from droidrun.config_manager.config_manager import DroidRunConfig, AgentConfig +from droidrun.config_manager.config_manager import DroidrunConfig, AgentConfig -config = DroidRunConfig() +config = DroidrunConfig() config.agent.reasoning = True agent = DroidAgent(goal="...", config=config) @@ -323,15 +323,14 @@ agent = DroidAgent(goal="...", config=config, agent_config=agent_config) ## Next Steps -Now that you've got DroidRun v4 running, explore these topics: +Now that you've got Droidrun v4 running, explore these topics: ### Core Concepts - [Architecture](/v4/concepts/architecture) - Multi-agent system architecture -- [Workflow Architecture](/v4/concepts/workflow-architecture) - Event-driven coordination -- [Event Streaming](/v4/concepts/event-streaming) - Real-time execution monitoring +- [Events and Workflows](/v4/concepts/events-and-workflows) - Event-driven coordination and real-time execution monitoring ### Configuration -- [Configuration System](/v4/guides/configuration) - Complete configuration guide +- [Configuration System](/v4/sdk/configuration) - Complete configuration guide - [Custom Tools & Credentials](/v4/guides/custom-tools-credentials) - Extend functionality - [Custom Variables](/v4/guides/custom-variables) - Pass custom data to agents - [Structured Output](/v4/guides/structured-output) - Extract structured data @@ -342,10 +341,10 @@ Now that you've got DroidRun v4 running, explore these topics: ### Advanced Topics - [Vision Mode](/v4/concepts/architecture#vision-configuration) - Screenshot processing -- [Safe Execution](/v4/guides/configuration#safe-execution) - Secure code execution -- [Tracing & Telemetry](/v4/guides/configuration#tracing-settings) - Debugging and monitoring -- [Custom Prompts](/v4/guides/configuration#prompt-customization) - Customize agent behavior +- [Safe Execution](/v4/sdk/configuration#safe-execution) - Secure code execution +- [Tracing & Telemetry](/v4/sdk/configuration#tracing-settings) - Debugging and monitoring +- [Custom Prompts](/v4/sdk/configuration#prompt-customization) - Customize agent behavior --- -**Welcome to DroidRun v4!** The config-driven architecture gives you complete control over agent behavior, making it easier than ever to build powerful device automation workflows. +**Welcome to Droidrun v4!** The config-driven architecture gives you complete control over agent behavior, making it easier than ever to build powerful device automation workflows. diff --git a/docs/v4/sdk.mdx b/docs/v4/sdk.mdx new file mode 100644 index 0000000..4e06a22 --- /dev/null +++ b/docs/v4/sdk.mdx @@ -0,0 +1,49 @@ +--- +title: 'SDK Reference' +description: 'Complete API reference for Droidrun v4 components and tools' +--- + +## Overview + +The Droidrun SDK provides a comprehensive set of APIs for building mobile automation workflows with AI agents. This reference documentation covers all major components and tools. + +--- + +## Core Components + + + + Main agent coordinator with multi-agent orchestration + + + Android device control via ADB + + + iOS device control and automation + + + Abstract base classes for tool implementations + + + +--- + +## Configuration + + + DroidrunConfig API and YAML configuration reference + + +--- + +## API Documentation + +Detailed API documentation for each component is available in the sections linked above. Each page includes: + +- Class/function signatures +- Parameter descriptions +- Return types +- Usage examples +- Best practices + +For conceptual guides and tutorials, see the [Guides](/v4/guides/overview) section. diff --git a/docs/v4/sdk/adb-tools.mdx b/docs/v4/sdk/adb-tools.mdx index 2dfcf27..90ee6f5 100644 --- a/docs/v4/sdk/adb-tools.mdx +++ b/docs/v4/sdk/adb-tools.mdx @@ -14,7 +14,7 @@ class AdbTools(Tools) Core UI interaction tools for Android device control. -AdbTools provides a comprehensive interface for interacting with Android devices through ADB (Android Debug Bridge). It supports both TCP communication and content provider modes for device communication via the DroidRun Portal app. +AdbTools provides a comprehensive interface for interacting with Android devices through ADB (Android Debug Bridge). It supports both TCP communication and content provider modes for device communication via the Droidrun Portal app. @@ -67,7 +67,7 @@ tools = AdbTools( ``` **Notes:** -- Automatically sets up the DroidRun Portal keyboard on initialization via `setup_keyboard()` +- Automatically sets up the Droidrun Portal keyboard on initialization via `setup_keyboard()` - Creates a PortalClient instance that handles TCP/content provider communication - Device serial can be emulator name, USB serial, or TCP/IP address:port @@ -293,7 +293,7 @@ result = tools.input_text("Hello\nWorld") # Multiline text **Notes:** - Always ensure a text field is focused before inputting text (use `tap_by_index()` or set `index` parameter) -- Uses the DroidRun Portal app keyboard for reliable text input via PortalClient +- Uses the Droidrun Portal app keyboard for reliable text input via PortalClient - Supports Unicode characters and special characters including non-ASCII - If `index != -1`, automatically taps the element first before inputting text - Call `get_state()` first to populate element cache if using `index` parameter @@ -803,7 +803,7 @@ tools.complete(success=False, reason="Could not find contact 'John' in contacts ## Notes -- **Portal app required**: The DroidRun Portal app must be installed and accessibility service enabled on the device +- **Portal app required**: The Droidrun Portal app must be installed and accessibility service enabled on the device - **TCP vs Content Provider**: TCP is faster but requires port forwarding (`adb forward tcp:8080 tcp:8080`). Content provider is the fallback mode using ADB shell commands. - **Element caching**: Always call `get_state()` before using `tap_by_index()` or `tap()` to populate the element cache - **Trajectory recording**: When `save_trajectories="action"`, screenshots and UI states are automatically captured for each UI action via the `@Tools.ui_action` decorator @@ -843,7 +843,7 @@ for element in state['a11y_tree']: break # Input search query -result = tools.input_text("DroidRun framework") +result = tools.input_text("Droidrun framework") print(result) # Press enter key @@ -857,10 +857,10 @@ with open("search_result.png", "wb") as f: f.write(screenshot) # Remember result for future context -tools.remember("Searched for DroidRun framework in Chrome") +tools.remember("Searched for Droidrun framework in Chrome") # Complete task -tools.complete(success=True, reason="Successfully searched for DroidRun in Chrome") +tools.complete(success=True, reason="Successfully searched for Droidrun in Chrome") # Check completion status print(f"Task finished: {tools.finished}") diff --git a/docs/v4/sdk/base-tools.mdx b/docs/v4/sdk/base-tools.mdx index bfab4ae..69445ca 100644 --- a/docs/v4/sdk/base-tools.mdx +++ b/docs/v4/sdk/base-tools.mdx @@ -350,13 +350,13 @@ Tools instances are passed to agents and provide the atomic actions for device c ```python from droidrun import DroidAgent from droidrun.tools import AdbTools -from droidrun.config_manager.config_manager import DroidRunConfig +from droidrun.config_manager.config_manager import DroidrunConfig # Create tools instance tools = AdbTools(serial="emulator-5554") # Create config -config = DroidRunConfig() +config = DroidrunConfig() # Pass to agent agent = DroidAgent( @@ -544,7 +544,7 @@ tools.swipe(100, 500, 100, 100) # Logs + captures screenshot ## See Also -- [AdbTools API](/docs/v4/sdk/adb-tools) - Android implementation -- [IOSTools API](/docs/v4/sdk/ios-tools) - iOS implementation -- [DroidAgent API](/docs/v4/sdk/droid-agent) - Agent integration -- [Custom Tools Guide](/docs/v4/guides/custom-tools-credentials) - Creating custom tools +- [AdbTools API](/v4/sdk/adb-tools) - Android implementation +- [IOSTools API](/v4/sdk/ios-tools) - iOS implementation +- [DroidAgent API](/v4/sdk/droid-agent) - Agent integration +- [Custom Tools Guide](/v4/guides/custom-tools-credentials) - Creating custom tools diff --git a/docs/v4/sdk/configuration.mdx b/docs/v4/sdk/configuration.mdx new file mode 100644 index 0000000..ddc60a9 --- /dev/null +++ b/docs/v4/sdk/configuration.mdx @@ -0,0 +1,693 @@ +--- +title: 'Configuration Reference' +description: 'Complete DroidAgent configuration guide - all parameters, minimal examples' +--- + +## Quick Start + +```python +from droidrun import DroidAgent + +# Minimal (uses defaults) +agent = DroidAgent(goal="Open settings") +result = await agent.run() +``` + +--- + +## DroidAgent Parameters + +### Required + +```python +DroidAgent( + goal="Your task", # REQUIRED: Task description +) +``` + +### Optional Parameters + +| Parameter | Type | Default | Description | +|-----------|------|---------|-------------| +| `config` | `DroidrunConfig \| None` | `None` | Full config object (loads LLMs from profiles if `llms` not provided) | +| `llms` | `dict[str, LLM] \| LLM \| None` | `None` | LLM(s) - dict for per-agent, single LLM for all, or None to load from config | +| `agent_config` | `AgentConfig \| None` | `None` | Agent behavior settings (overrides config.agent) | +| `device_config` | `DeviceConfig \| None` | `None` | Device connection settings (overrides config.device) | +| `tools` | `Tools \| ToolsConfig \| None` | `None` | Tools instance or config (overrides config.tools) | +| `logging_config` | `LoggingConfig \| None` | `None` | Logging settings (overrides config.logging) | +| `tracing_config` | `TracingConfig \| None` | `None` | Tracing settings (overrides config.tracing) | +| `telemetry_config` | `TelemetryConfig \| None` | `None` | Telemetry settings (overrides config.telemetry) | +| `custom_tools` | `dict \| None` | `None` | Custom tool definitions | +| `credentials` | `CredentialsConfig \| dict \| None` | `None` | Credentials config or dict of secrets | +| `variables` | `dict \| None` | `None` | Custom variables accessible during execution | +| `output_model` | `Type[BaseModel] \| None` | `None` | Pydantic model for structured output extraction | +| `prompts` | `dict[str, str] \| None` | `None` | Custom Jinja2 prompt templates (NOT file paths) | +| `timeout` | `int` | `1000` | Workflow timeout in seconds | + +--- + +## Configuration Classes + +### AgentConfig + +```python +from droidrun.config_manager.config_manager import AgentConfig, CodeActConfig, ManagerConfig, ExecutorConfig, ScripterConfig, AppCardConfig + +AgentConfig( + # Core settings + max_steps=15, # Max execution steps + reasoning=False, # Enable Manager/Executor workflow + after_sleep_action=1.0, # Wait after actions (seconds) + wait_for_stable_ui=0.3, # Wait for UI to stabilize (seconds) + prompts_dir="config/prompts", # Prompt templates directory + + # Sub-configs + codeact=CodeActConfig(...), + manager=ManagerConfig(...), + executor=ExecutorConfig(...), + scripter=ScripterConfig(...), + app_cards=AppCardConfig(...), +) +``` + +**CodeActConfig** +```python +CodeActConfig( + vision=False, # Enable screenshots + system_prompt="system.jinja2", # Filename in prompts_dir/codeact/ + user_prompt="user.jinja2", # Filename in prompts_dir/codeact/ + safe_execution=False, # Restrict imports/builtins +) +``` + +**ManagerConfig** +```python +ManagerConfig( + vision=False, # Enable screenshots + system_prompt="system.jinja2", # Filename in prompts_dir/manager/ +) +``` + +**ExecutorConfig** +```python +ExecutorConfig( + vision=False, # Enable screenshots + system_prompt="system.jinja2", # Filename in prompts_dir/executor/ +) +``` + +**ScripterConfig** +```python +ScripterConfig( + enabled=True, # Enable off-device Python execution + max_steps=10, # Max scripter steps + execution_timeout=30.0, # Code block timeout (seconds) + system_prompt_path="system.jinja2", # Filename in prompts_dir/scripter/ + safe_execution=False, # Restrict imports/builtins +) +``` + +**AppCardConfig** +```python +AppCardConfig( + enabled=True, # Enable app-specific instructions + mode="local", # "local" | "server" | "composite" + app_cards_dir="config/app_cards", # Directory for app card files + server_url=None, # Server URL (for server/composite modes) + server_timeout=2.0, # Server request timeout (seconds) + server_max_retries=2, # Server retry attempts +) +``` + +--- + +### DeviceConfig + +```python +from droidrun.config_manager.config_manager import DeviceConfig + +DeviceConfig( + serial=None, # Device serial/IP (None = auto-detect) + platform="android", # "android" or "ios" + use_tcp=False, # TCP vs content provider communication +) +``` + +--- + +### LoggingConfig + +```python +from droidrun.config_manager.config_manager import LoggingConfig + +LoggingConfig( + debug=False, # Enable debug logs + save_trajectory="none", # "none" | "step" | "action" + rich_text=False, # Rich text formatting in logs +) +``` + +--- + +### TracingConfig + +```python +from droidrun.config_manager.config_manager import TracingConfig + +TracingConfig( + enabled=False, # Enable Arize Phoenix tracing +) +``` + +--- + +### TelemetryConfig + +```python +from droidrun.config_manager.config_manager import TelemetryConfig + +TelemetryConfig( + enabled=True, # Enable anonymous telemetry +) +``` + +--- + +### ToolsConfig + +```python +from droidrun.config_manager.config_manager import ToolsConfig + +ToolsConfig( + allow_drag=False, # Enable drag tool +) +``` + +--- + +### CredentialsConfig + +```python +from droidrun.config_manager.config_manager import CredentialsConfig + +CredentialsConfig( + enabled=False, # Enable credential manager + file_path="credentials.yaml", # Path to credentials file +) +``` + +--- + +### SafeExecutionConfig + +```python +from droidrun.config_manager.safe_execution import SafeExecutionConfig + +SafeExecutionConfig( + # Imports + allow_all_imports=False, # Allow all imports (ignores allowed_modules) + allowed_modules=[], # Allowed module names (e.g., ["json", "requests"]) + blocked_modules=[ # Blocked modules (takes precedence) + "os", "sys", "subprocess", "shutil", "pathlib", "pty", "fcntl", + "resource", "pickle", "shelve", "marshal", "imp", "importlib", + "ctypes", "code", "codeop", "tempfile", "glob", "socket", + "socketserver", "asyncio" + ], + + # Builtins + allow_all_builtins=False, # Allow all builtins (ignores allowed_builtins) + allowed_builtins=[], # Allowed builtin names (empty = safe defaults) + blocked_builtins=[ # Blocked builtins (takes precedence) + "open", "compile", "exec", "eval", "__import__", + "breakpoint", "exit", "quit", "input" + ], +) +``` + +--- + +## LLM Configuration + +### Single LLM (All Agents) + +```python +from llama_index.llms.gemini import Gemini + +llm = Gemini(model="models/gemini-2.5-pro", temperature=0.2) +agent = DroidAgent(goal="...", llms=llm) +``` + +### Per-Agent LLMs + +```python +from llama_index.llms.openai import OpenAI +from llama_index.llms.gemini import Gemini + +agent = DroidAgent( + goal="...", + llms={ + "manager": OpenAI(model="gpt-4o"), # Planning + "executor": Gemini(model="models/gemini-2.5-flash"), # Action selection + "codeact": Gemini(model="models/gemini-2.5-pro"), # Code generation + "text_manipulator": Gemini(model="models/gemini-2.5-flash"), # Text input + "app_opener": OpenAI(model="gpt-4o-mini"), # App launching + "scripter": Gemini(model="models/gemini-2.5-flash"), # Off-device scripts + "structured_output": Gemini(model="models/gemini-2.5-flash"), # Output extraction + } +) +``` + +**LLM Keys:** +- `manager` - Planning (reasoning mode only) +- `executor` - Action selection (reasoning mode only) +- `codeact` - Code generation (direct mode) +- `scripter` - Off-device Python execution +- `text_manipulator` - Text input helper +- `app_opener` - App launching helper +- `structured_output` - Final output extraction + +--- + +## Custom Tools + +```python +def my_tool(param: str) -> str: + """Tool description.""" + return f"Result: {param}" + +agent = DroidAgent( + goal="...", + custom_tools={ + "my_tool": { + "signature": "my_tool(param: str) -> str", + "description": "Tool description", + "function": my_tool + } + } +) +``` + +--- + +## Credentials + +### Dict Format (Recommended) +```python +agent = DroidAgent( + goal="...", + credentials={ + "USERNAME": "alice@example.com", + "PASSWORD": "secret123" + } +) +# Agent can call get_username() and get_password() +``` + +### Config Format +```python +from droidrun.config_manager.config_manager import CredentialsConfig + +agent = DroidAgent( + goal="...", + credentials=CredentialsConfig( + enabled=True, + file_path="config/credentials.yaml" + ) +) +``` + +--- + +## Custom Variables + +```python +agent = DroidAgent( + goal="...", + variables={ + "api_url": "https://api.example.com", + "user_id": "12345", + "custom_data": {"key": "value"} + } +) +# Access in shared_state.custom_variables +``` + +--- + +## Structured Output + +```python +from pydantic import BaseModel + +class FlightInfo(BaseModel): + airline: str + flight_number: str + confirmation_code: str + +agent = DroidAgent( + goal="Book a flight and extract details", + output_model=FlightInfo +) +result = await agent.run() +print(result.structured_output.airline) # Typed output +``` + +--- + +## Custom Prompts + +```python +custom_prompt = """ +You are an expert mobile agent. +Goal: {{ instruction }} +Be precise and efficient. +""" + +agent = DroidAgent( + goal="...", + prompts={ + "codeact_system": custom_prompt, + "codeact_user": "...", + "manager_system": "...", + "executor_system": "...", + "scripter_system": "..." + } +) +``` + +**Template Variables:** +- `{{ instruction }}` - User's goal +- `{{ device_date }}` - Device date/time +- `{{ app_card }}` - App-specific instructions +- `{{ state }}` - Device state +- `{{ history }}` - Action history + +--- + +## Complete Example + +```python +from droidrun import DroidAgent +from droidrun.config_manager.config_manager import ( + AgentConfig, CodeActConfig, DeviceConfig, LoggingConfig, TracingConfig +) +from llama_index.llms.openai import OpenAI +from llama_index.llms.gemini import Gemini +from pydantic import BaseModel + +# Structured output +class Output(BaseModel): + name: str + value: int + +# Custom tool +def send_email(to: str, subject: str) -> str: + """Send email.""" + return f"Sent to {to}" + +agent = DroidAgent( + goal="Complex task", + + # LLMs + llms={ + "manager": OpenAI(model="gpt-4o"), + "executor": Gemini(model="models/gemini-2.5-flash"), + "codeact": Gemini(model="models/gemini-2.5-pro") + }, + + # Agent behavior + agent_config=AgentConfig( + max_steps=30, + reasoning=True, + after_sleep_action=1.5, + codeact=CodeActConfig(vision=True, safe_execution=True) + ), + + # Device + device_config=DeviceConfig( + serial="emulator-5554", + platform="android", + use_tcp=False + ), + + # Logging + logging_config=LoggingConfig( + debug=True, + save_trajectory="action" + ), + + # Tracing + tracing_config=TracingConfig(enabled=True), + + # Custom tools + custom_tools={ + "send_email": { + "signature": "send_email(to: str, subject: str) -> str", + "description": "Send email", + "function": send_email + } + }, + + # Credentials + credentials={"USERNAME": "alice", "PASSWORD": "secret"}, + + # Variables + variables={"api_url": "https://api.example.com"}, + + # Structured output + output_model=Output, + + # Timeout + timeout=600 +) + +result = await agent.run() +``` + +--- + +## YAML Config (CLI) + +For CLI usage, create `config.yaml`: + +```yaml +agent: + max_steps: 15 + reasoning: false + after_sleep_action: 1.0 + wait_for_stable_ui: 0.3 + prompts_dir: config/prompts + + codeact: + vision: false + system_prompt: system.jinja2 + user_prompt: user.jinja2 + safe_execution: false + + manager: + vision: false + system_prompt: system.jinja2 + + executor: + vision: false + system_prompt: system.jinja2 + + scripter: + enabled: true + max_steps: 10 + execution_timeout: 30.0 + system_prompt_path: system.jinja2 + safe_execution: false + + app_cards: + enabled: true + mode: local + app_cards_dir: config/app_cards + server_url: null + server_timeout: 2.0 + server_max_retries: 2 + +llm_profiles: + manager: + provider: GoogleGenAI + model: models/gemini-2.5-pro + temperature: 0.2 + kwargs: + max_tokens: 8192 + + executor: + provider: GoogleGenAI + model: models/gemini-2.5-flash + temperature: 0.1 + kwargs: + max_tokens: 4096 + + codeact: + provider: GoogleGenAI + model: models/gemini-2.5-pro + temperature: 0.2 + kwargs: + max_tokens: 8192 + + text_manipulator: + provider: GoogleGenAI + model: models/gemini-2.5-flash + temperature: 0.3 + + app_opener: + provider: OpenAI + model: gpt-4o-mini + temperature: 0.0 + + scripter: + provider: GoogleGenAI + model: models/gemini-2.5-flash + temperature: 0.1 + + structured_output: + provider: GoogleGenAI + model: models/gemini-2.5-flash + temperature: 0.0 + +device: + serial: null + platform: android + use_tcp: false + +telemetry: + enabled: true + +tracing: + enabled: false + +logging: + debug: false + save_trajectory: none + rich_text: false + +safe_execution: + allow_all_imports: false + allowed_modules: [] + blocked_modules: + - os + - sys + - subprocess + - shutil + - pathlib + - pty + - fcntl + - resource + - pickle + - shelve + - marshal + - imp + - importlib + - ctypes + - code + - codeop + - tempfile + - glob + - socket + - socketserver + - asyncio + allow_all_builtins: false + allowed_builtins: [] + blocked_builtins: + - open + - compile + - exec + - eval + - __import__ + - breakpoint + - exit + - quit + - input + +tools: + allow_drag: false + +credentials: + enabled: false + file_path: credentials.yaml +``` + +--- + +## CLI Overrides + +```bash +# Override agent settings +droidrun run "Task" --steps 30 --reasoning --vision + +# Override device +droidrun run "Task" --device emulator-5554 --tcp + +# Override LLM (applies to ALL agents) +droidrun run "Task" --provider GoogleGenAI --model models/gemini-2.5-flash + +# Override logging +droidrun run "Task" --debug --save-trajectory action --tracing + +# Custom config file +droidrun run "Task" --config /path/to/config.yaml +``` + +**All CLI Flags:** +- `--config PATH` - Custom config file +- `--device SERIAL` - Device serial/IP +- `--provider PROVIDER` - LLM provider (OpenAI, Ollama, Anthropic, GoogleGenAI, DeepSeek) +- `--model MODEL` - LLM model name +- `--temperature FLOAT` - LLM temperature +- `--steps INT` - Max steps +- `--base_url URL` - API base URL (for Ollama/OpenRouter) +- `--api_base URL` - API base URL (for OpenAI-like) +- `--vision/--no-vision` - Enable/disable vision for all agents +- `--reasoning/--no-reasoning` - Enable/disable reasoning mode +- `--tracing/--no-tracing` - Enable/disable tracing +- `--debug/--no-debug` - Enable/disable debug logs +- `--tcp/--no-tcp` - Enable/disable TCP communication +- `--save-trajectory none|step|action` - Trajectory saving level +- `--ios` - Run on iOS device + +--- + +## Configuration Priority + +When multiple sources are provided, DroidAgent uses this priority: + +1. **Direct parameters** (highest priority) + ```python + agent = DroidAgent(goal="...", agent_config=AgentConfig(max_steps=30)) + ``` + +2. **Config object** + ```python + agent = DroidAgent(goal="...", config=my_config) + ``` + +3. **Defaults** (lowest priority) + +**Example:** +```python +config = DroidrunConfig(agent=AgentConfig(max_steps=20)) + +agent = DroidAgent( + goal="...", + config=config, # max_steps=20 from config + agent_config=AgentConfig(max_steps=30) # Overrides to 30 +) +``` + +--- + +## Environment Variables + +Set API keys via environment variables: + +```bash +export GOOGLE_API_KEY=your-key +export OPENAI_API_KEY=your-key +export ANTHROPIC_API_KEY=your-key +export DEEPSEEK_API_KEY=your-key +export DROIDRUN_CONFIG=/path/to/config.yaml # Custom config path +``` diff --git a/docs/v4/sdk/droid-agent.mdx b/docs/v4/sdk/droid-agent.mdx index f2eb15f..d2f579b 100644 --- a/docs/v4/sdk/droid-agent.mdx +++ b/docs/v4/sdk/droid-agent.mdx @@ -25,7 +25,7 @@ A wrapper class that coordinates between agents to achieve a user's goal. ```python def __init__( goal: str, - config: DroidRunConfig | None = None, + config: DroidrunConfig | None = None, llms: dict[str, LLM] | LLM | None = None, agent_config: AgentConfig | None = None, device_config: DeviceConfig | None = None, @@ -47,7 +47,7 @@ Initialize the DroidAgent wrapper. **Arguments**: - `goal` _str_ - User's goal or command to execute -- `config` _DroidRunConfig | None_ - Full configuration object (required if llms not provided). Contains agent settings, LLM profiles, device config, and more. If provided, individual config overrides (agent_config, device_config, etc.) take precedence. +- `config` _DroidrunConfig | None_ - Full configuration object (required if llms not provided). Contains agent settings, LLM profiles, device config, and more. If provided, individual config overrides (agent_config, device_config, etc.) take precedence. - `llms` _dict[str, LLM] | LLM | None_ - Optional LLM configuration: - `dict[str, LLM]`: Agent-specific LLMs with keys: "manager", "executor", "codeact", "text_manipulator", "app_opener", "scripter", "structured_output" - `LLM`: Single LLM instance used for all agents @@ -75,14 +75,14 @@ Initialize the DroidAgent wrapper. ```python from droidrun import DroidAgent -from droidrun.config_manager.config_manager import DroidRunConfig +from droidrun.config_manager.config_manager import DroidrunConfig # Initialize with default config -config = DroidRunConfig() +config = DroidrunConfig() # Create agent (LLMs loaded from config.llm_profiles) agent = DroidAgent( - goal="Open Chrome and search for DroidRun", + goal="Open Chrome and search for Droidrun", config=config ) @@ -94,14 +94,14 @@ result = await agent.run() ```python from droidrun import DroidAgent -from droidrun.config_manager.config_manager import DroidRunConfig +from droidrun.config_manager.config_manager import DroidrunConfig # Load config from config.yaml -config = DroidRunConfig.from_yaml("config.yaml") +config = DroidrunConfig.from_yaml("config.yaml") # Create agent (LLMs loaded from config.llm_profiles) agent = DroidAgent( - goal="Open Chrome and search for DroidRun", + goal="Open Chrome and search for Droidrun", config=config ) @@ -113,17 +113,17 @@ result = await agent.run() ```python from droidrun import DroidAgent -from droidrun.config_manager.config_manager import DroidRunConfig +from droidrun.config_manager.config_manager import DroidrunConfig from llama_index.llms.openai import OpenAI from llama_index.llms.anthropic import Anthropic # Initialize config -config = DroidRunConfig() +config = DroidrunConfig() # Create custom LLMs llms = { - "manager": Anthropic(model="claude-sonnet-4-5", temperature=0.2), - "executor": Anthropic(model="claude-sonnet-4-5", temperature=0.1), + "manager": Anthropic(model="claude-sonnet-4-5-latest", temperature=0.2), + "executor": Anthropic(model="claude-sonnet-4-5-latest", temperature=0.1), "codeact": OpenAI(model="gpt-4o", temperature=0.2), "text_manipulator": OpenAI(model="gpt-4o-mini", temperature=0.3), "app_opener": OpenAI(model="gpt-4o-mini", temperature=0.0), @@ -145,11 +145,11 @@ result = await agent.run() ```python from droidrun import DroidAgent -from droidrun.config_manager.config_manager import DroidRunConfig +from droidrun.config_manager.config_manager import DroidrunConfig from llama_index.llms.openai import OpenAI # Initialize config -config = DroidRunConfig() +config = DroidrunConfig() # Use same LLM for all agents llm = OpenAI(model="gpt-4o", temperature=0.2) @@ -167,10 +167,10 @@ result = await agent.run() ```python from droidrun import DroidAgent -from droidrun.config_manager.config_manager import DroidRunConfig +from droidrun.config_manager.config_manager import DroidrunConfig # Initialize config -config = DroidRunConfig() +config = DroidrunConfig() # Define custom tool def search_database(query: str) -> str: @@ -206,11 +206,11 @@ result = await agent.run() ```python from droidrun import DroidAgent -from droidrun.config_manager.config_manager import DroidRunConfig +from droidrun.config_manager.config_manager import DroidrunConfig from pydantic import BaseModel, Field # Initialize config -config = DroidRunConfig() +config = DroidrunConfig() # Define output schema class WeatherInfo(BaseModel): @@ -256,10 +256,10 @@ Run the DroidAgent workflow. ```python from droidrun import DroidAgent -from droidrun.config_manager.config_manager import DroidRunConfig +from droidrun.config_manager.config_manager import DroidrunConfig # Initialize config -config = DroidRunConfig() +config = DroidrunConfig() # Create and run agent agent = DroidAgent(goal="...", config=config) @@ -274,10 +274,10 @@ print(f"Steps: {result.steps}") ```python from droidrun import DroidAgent -from droidrun.config_manager.config_manager import DroidRunConfig +from droidrun.config_manager.config_manager import DroidrunConfig # Initialize config -config = DroidRunConfig() +config = DroidrunConfig() agent = DroidAgent(goal="...", config=config) @@ -328,7 +328,7 @@ DroidAgent emits various events during execution: ## Configuration -DroidAgent uses a hierarchical configuration system. See the [Configuration Guide](/docs/v4/guides/configuration) for details. +DroidAgent uses a hierarchical configuration system. See the [Configuration Guide](/v4/sdk/configuration) for details. **Key configuration options:** @@ -365,19 +365,18 @@ tracing: **Custom Tools instance:** ```python -from droidrun import DroidAgent, AdbTools -from droidrun.config_manager.config_manager import DroidRunConfig +from droidrun import DroidAgent, DeviceConfig +from droidrun.config_manager.config_manager import DroidrunConfig # Initialize config -config = DroidRunConfig() +config = DroidrunConfig() -# Pre-configure tools -tools = AdbTools(serial="emulator-5554", use_tcp=True) +device_config = DeviceConfig(serial="emulator-5554", use_tcp=True) agent = DroidAgent( goal="Open settings", config=config, - tools=tools # Use pre-configured tools + device_config=device_config ) result = await agent.run() @@ -387,10 +386,10 @@ result = await agent.run() ```python from droidrun import DroidAgent -from droidrun.config_manager.config_manager import DroidRunConfig +from droidrun.config_manager.config_manager import DroidrunConfig # Initialize config -config = DroidRunConfig() +config = DroidrunConfig() agent = DroidAgent( goal="Complete task using context", @@ -411,10 +410,10 @@ Variables are accessible in shared_state.custom_variables throughout execution a ```python from droidrun import DroidAgent -from droidrun.config_manager.config_manager import DroidRunConfig +from droidrun.config_manager.config_manager import DroidrunConfig # Initialize config -config = DroidRunConfig() +config = DroidrunConfig() # Override default prompts with custom Jinja2 templates custom_prompts = { diff --git a/docs/v4/sdk/ios-tools.mdx b/docs/v4/sdk/ios-tools.mdx index 851a576..86b7d30 100644 --- a/docs/v4/sdk/ios-tools.mdx +++ b/docs/v4/sdk/ios-tools.mdx @@ -12,7 +12,7 @@ title: IOSTools class IOSTools(Tools) ``` -Core UI interaction tools for iOS device control via the DroidRun iOS Portal app. +Core UI interaction tools for iOS device control via the Droidrun iOS Portal app. **Status**: iOS support is in beta with limited functionality compared to AdbTools. @@ -59,7 +59,7 @@ tools = IOSTools( **Setup Requirements:** -1. Install DroidRun iOS Portal app on device +1. Install Droidrun iOS Portal app on device 2. Launch Portal app (starts HTTP server) 3. Connect device and computer to same network 4. Use displayed URL to initialize IOSTools @@ -612,7 +612,7 @@ for elem in state['a11y_tree']: if elem['type'] == 'TextField' and 'message' in elem['label'].lower(): tools.tap_by_index(elem['index']) break -tools.input_text("Hello from DroidRun!") +tools.input_text("Hello from Droidrun!") # Send state = tools.get_state() @@ -630,5 +630,5 @@ tools.complete(True, "Successfully sent iMessage") ## See Also -- [AdbTools](/docs/v4/sdk/adb-tools) - Android device control with full functionality -- [Tools Base Class](/docs/v4/sdk/base-tools) - Abstract base class reference +- [AdbTools](/v4/sdk/adb-tools) - Android device control with full functionality +- [Tools Base Class](/v4/sdk/base-tools) - Abstract base class reference diff --git a/gen-docs-sdk-ref.sh b/gen-docs-sdk-ref.sh index 3227f52..ec4ecdd 100755 --- a/gen-docs-sdk-ref.sh +++ b/gen-docs-sdk-ref.sh @@ -6,7 +6,7 @@ pydoc-markdown # rename .md to .mdx # Rename all .md files to .mdx in the docs/v3/sdk directory -find docs/v3/sdk -name "*.md" -exec sh -c 'mv "$1" "${1%.md}.mdx"' _ {} \; +find docs/v4/sdk -name "*.md" -exec sh -c 'mv "$1" "${1%.md}.mdx"' _ {} \; # update docs/v3/.generated-files.txt with the new files extension # Set sed in-place flag for macOS and Linux compatibility diff --git a/pyproject.toml b/pyproject.toml index 231f559..40c47fc 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -120,7 +120,7 @@ type = "crossref" [tool.pydoc-markdown.renderer] type = "hugo" build_directory = "docs" -content_directory = "v3/sdk" +content_directory = "v4/sdk" clean_render = true