import sys import httpx import json import base64 import pyautogui import subprocess from io import BytesIO from PIL import Image import quartz_doubleclick ANTHROPIC_API_URL = "https://api.anthropic.com/v1/messages" MODEL = "claude-3-7-sonnet-20250219" BETA_FLAG = "computer-use-2025-01-24" TOOL_VERSION = "20250124" pyautogui.PAUSE = 0.1 SCALING_FACTOR = 1.25 HEADERS = { "content-type": "application/json", "anthropic-version": "2023-06-01", "anthropic-beta": BETA_FLAG } def send_request (headers, body): with httpx.Client(timeout=60.0) as client: try: response = client.post(ANTHROPIC_API_URL, headers=headers, json=body) response.raise_for_status() return response.json() except httpx.HTTPStatusError as e: print(f"[ERROR] HTTP {e.response.status_code} - {e.response.reason_phrase}") try: print("[DETAIL] Response JSON:") print(json.dumps(e.response.json(), indent=2)) except Exception: print("[DETAIL] Raw Response Text:") print(e.response.text) sys.exit(1) def execute_computer_tool(tool_input): """Execute computer tool actions like screenshot, click, type, etc.""" action = tool_input.get("action") if action == "screenshot": # Take a screenshot screenshot = pyautogui.screenshot() # Convert to base64 buffered = BytesIO() screenshot.save(buffered, format="PNG") MAX_BINARY_SIZE = 5242880 * 3 // 4 img_data = None if buffered.tell() > MAX_BINARY_SIZE: # Reset buffer buffered.seek(0) screenshot = Image.open(buffered) # Resize to 80% repeatedly until under 5MB while True: buffered = BytesIO() new_size = (int(screenshot.width * 0.8), int(screenshot.height * 0.8)) print('new size:', new_size) screenshot = screenshot.resize(new_size, Image.Resampling.LANCZOS) screenshot.save(buffered, format="PNG", optimize=True) buffered.seek(0) img_data = buffered.read() if len(img_data) <= MAX_BINARY_SIZE: break buffered.seek(0) screenshot = Image.open(buffered) else: img_data = buffered.getvalue() # Convert to base64 img_base64 = base64.b64encode(img_data).decode('utf-8') return { "type": "image", "source": { "type": "base64", "media_type": "image/png", "data": img_base64 } } elif action == "left_click": # Get coordinates x = tool_input.get("coordinate")[0] * SCALING_FACTOR y = tool_input.get("coordinate")[1] * SCALING_FACTOR if x is None or y is None: return {"type": "text", "text": "Error: Missing coordinates for click action"} # Perform click pyautogui.click(x, y) return {"type": "text", "text": f"Clicked at coordinates ({x}, {y})"} elif action == "double_click": # Get coordinates x = tool_input.get("coordinate")[0] * SCALING_FACTOR y = tool_input.get("coordinate")[1] * SCALING_FACTOR if x is None or y is None: return {"type": "text", "text": "Error: Missing coordinates for double click action"} # Perform double click # pyautogui.doubleClick(x, y, interval=0.2) this doesn't work quartz_doubleclick.double_click(x, y) return {"type": "text", "text": f"Double-clicked at coordinates ({x}, {y})"} elif action == "type": # Get text to type text = tool_input.get("text", "") pyautogui.typewrite(text) return {"type": "text", "text": f"Typed: {text}"} elif action == "key": # Press a key or key combination text = tool_input.get("text", "") try: if '+' in text: # Handle key combinations like "command+c" keys = text.replace('super', 'command').split('+') pyautogui.hotkey(*keys, interval=0.05) # interval is required else: pyautogui.press(text) return {"type": "text", "text": f"Pressed key: {text}"} except Exception as e: return {"type": "text", "text": f"Error pressing key {text}: {str(e)}"} elif action == "scroll": # Scroll action (should we really have defaults here?) direction = tool_input.get("scroll_direction", "down") amount = tool_input.get("scroll_amount", 3) scroll_amount = -amount if direction == "down" else amount pyautogui.scroll(scroll_amount) return {"type": "text", "text": f"Scrolled {direction} by {amount}"} else: return {"type": "text", "text": f"Unknown computer action: {action}"} def execute_bash_tool(command): """Execute bash commands safely.""" try: # Execute the command result = subprocess.run( command, shell=True, capture_output=True, text=True, timeout=30 # 30 second timeout ) output = result.stdout if result.stderr: output += f"\n[STDERR]: {result.stderr}" return {"type": "text", "text": output if output else "(Command executed successfully with no output)"} except subprocess.TimeoutExpired: return {"type": "text", "text": "Error: Command timed out after 30 seconds"} except Exception as e: return {"type": "text", "text": f"Error executing command: {str(e)}"} def execute_text_editor_tool(_command, _path, **kwargs): """Execute text editor actions.""" if _command == "view": try: with open(_path, 'r') as f: content = f.read() return {"type": "text", "text": content} except Exception as e: return {"type": "text", "text": f"Error reading file: {str(e)}"} elif _command == "str_replace": old_str = kwargs.get("old_str", "") new_str = kwargs.get("new_str", "") try: with open(_path, 'r') as f: content = f.read() if old_str not in content: return {"type": "text", "text": f"Error: '{old_str}' not found in file"} # Replace the string new_content = content.replace(old_str, new_str) # Write back to file with open(_path, 'w') as f: f.write(new_content) return {"type": "text", "text": "String replaced successfully"} except Exception as e: return {"type": "text", "text": f"Error modifying file: {str(e)}"} elif _command == "create": content = kwargs.get("content", "") try: with open(_path, 'w') as f: f.write(content) return {"type": "text", "text": f"File created: {_path}"} except Exception as e: return {"type": "text", "text": f"Error creating file: {str(e)}"} else: return {"type": "text", "text": f"Unknown text editor command: {_command}"} def execute_tool(tool_name, tool_input): """Main function to execute any tool based on its name and input.""" print(f"\n[EXECUTING] Tool: {tool_name}") print(f"[INPUT] {json.dumps(tool_input, indent=2)}") if tool_name == "computer": # Pass the entire tool_input to execute_computer_tool return execute_computer_tool(tool_input) elif tool_name == "bash": command = tool_input.get("command", "") return execute_bash_tool(command) elif tool_name == "str_replace_editor": command = tool_input.get("command") path = tool_input.get("path", "") return execute_text_editor_tool(command, path, **tool_input) else: return {"type": "text", "text": f"Unknown tool: {tool_name}"} def send_prompt(api_key, messages): headers = HEADERS.copy() headers["x-api-key"] = api_key body = { "model": MODEL, "max_tokens": 2048, "tools": [ { "type": f"computer_{TOOL_VERSION}", "name": "computer", "display_width_px": 2880, "display_height_px": 1864, "display_number": 1 }, { "type": f"bash_{TOOL_VERSION}", "name": "bash" }, { "type": f"text_editor_{TOOL_VERSION}", "name": "str_replace_editor" } ], "messages": messages, "thinking": { "type": "enabled", "budget_tokens": 1024 } } return send_request(headers, body) def extract_tool_uses(response): """Extract all tool use requests from the response.""" tool_uses = [] for block in response.get("content", []): if block.get("type") == "tool_use": tool_uses.append(block) return tool_uses def run(): messages = [] if len(sys.argv) < 3: print("Usage: python claude_computer_use_cli.py ''") return api_key = sys.argv[1] prompt = sys.argv[2] messages.append({ "role": "user", "content": prompt }) print(f"Sending prompt: {prompt}") # Main conversation loop while True: response = send_prompt(api_key, messages) # Add Claude's response to messages messages.append({ 'role': 'assistant', 'content': response.get('content', []), }) # Print Claude's response print("\n[CLAUDE RESPONSE]") for block in response.get("content", []): if block["type"] == "text": print(block["text"]) elif block["type"] == "thinking": print(f"[THINKING] {block['thinking']}") elif block["type"] == "tool_use": print(f"[TOOL REQUEST] {block['name']} - {block.get('input', {})}") # Extract all tool uses tool_uses = extract_tool_uses(response) if not tool_uses: # No more tool uses, we're done break # Execute all requested tools and collect results tool_results = [] for tool_use in tool_uses: tool_name = tool_use["name"] tool_input = tool_use.get("input", {}) tool_id = tool_use["id"] # Execute the tool result_content = execute_tool(tool_name, tool_input) print('[TOOL RESULT]') for key in result_content: if key == "source": print(f"{key}: {len(result_content[key])} chars long") else: print(f"{key}: {result_content[key]}") # Add to results tool_results.append({ "type": "tool_result", "tool_use_id": tool_id, "content": [result_content] if isinstance(result_content, dict) else result_content }) # Add all tool results as a user message messages.append({ "role": "user", "content": tool_results }) print("\n[TOOL RESULTS SENT BACK TO CLAUDE]") print("\n[CONVERSATION COMPLETE]") if __name__ == "__main__": run()