2 changed files with 309 additions and 65 deletions
+263 -58
View File
@@ -1,12 +1,19 @@
import os
import sys import sys
import httpx import httpx
import json import json
import base64
import pyautogui
import subprocess
from io import BytesIO
from PIL import Image
import quartz_doubleclick
ANTHROPIC_API_URL = "https://api.anthropic.com/v1/messages" ANTHROPIC_API_URL = "https://api.anthropic.com/v1/messages"
MODEL = "claude-3-7-sonnet-20250219" MODEL = "claude-3-7-sonnet-20250219"
BETA_FLAG = "computer-use-2025-01-24" BETA_FLAG = "computer-use-2025-01-24"
TOOL_VERSION = "20250124" TOOL_VERSION = "20250124"
pyautogui.PAUSE = 0.1
SCALING_FACTOR = 1.25
HEADERS = { HEADERS = {
"content-type": "application/json", "content-type": "application/json",
@@ -32,6 +39,202 @@ def send_request (headers, body):
sys.exit(1) sys.exit(1)
def execute_computer_tool(tool_input):
"""Execute computer tool actions like screenshot, click, type, etc."""
action = tool_input.get("action")
if action == "screenshot":
# Take a screenshot
screenshot = pyautogui.screenshot()
# Convert to base64
buffered = BytesIO()
screenshot.save(buffered, format="PNG")
MAX_BINARY_SIZE = 5242880 * 3 // 4
img_data = None
if buffered.tell() > MAX_BINARY_SIZE:
# Reset buffer
buffered.seek(0)
screenshot = Image.open(buffered)
# Resize to 80% repeatedly until under 5MB
while True:
buffered = BytesIO()
new_size = (int(screenshot.width * 0.8), int(screenshot.height * 0.8))
print('new size:', new_size)
screenshot = screenshot.resize(new_size, Image.Resampling.LANCZOS)
screenshot.save(buffered, format="PNG", optimize=True)
buffered.seek(0)
img_data = buffered.read()
if len(img_data) <= MAX_BINARY_SIZE:
break
buffered.seek(0)
screenshot = Image.open(buffered)
else:
img_data = buffered.getvalue()
# Convert to base64
img_base64 = base64.b64encode(img_data).decode('utf-8')
return {
"type": "image",
"source": {
"type": "base64",
"media_type": "image/png",
"data": img_base64
}
}
elif action == "left_click":
# Get coordinates
x = tool_input.get("coordinate")[0] * SCALING_FACTOR
y = tool_input.get("coordinate")[1] * SCALING_FACTOR
if x is None or y is None:
return {"type": "text", "text": "Error: Missing coordinates for click action"}
# Perform click
pyautogui.click(x, y)
return {"type": "text", "text": f"Clicked at coordinates ({x}, {y})"}
elif action == "double_click":
# Get coordinates
x = tool_input.get("coordinate")[0] * SCALING_FACTOR
y = tool_input.get("coordinate")[1] * SCALING_FACTOR
if x is None or y is None:
return {"type": "text", "text": "Error: Missing coordinates for double click action"}
# Perform double click
# pyautogui.doubleClick(x, y, interval=0.2) this doesn't work
quartz_doubleclick.double_click(x, y)
return {"type": "text", "text": f"Double-clicked at coordinates ({x}, {y})"}
elif action == "type":
# Get text to type
text = tool_input.get("text", "")
pyautogui.typewrite(text)
return {"type": "text", "text": f"Typed: {text}"}
elif action == "key":
# Press a key or key combination
text = tool_input.get("text", "")
try:
if '+' in text:
# Handle key combinations like "command+c"
keys = text.replace('super', 'command').split('+')
pyautogui.hotkey(*keys, interval=0.05) # interval is required
else:
pyautogui.press(text)
return {"type": "text", "text": f"Pressed key: {text}"}
except Exception as e:
return {"type": "text", "text": f"Error pressing key {text}: {str(e)}"}
elif action == "scroll":
# Scroll action (should we really have defaults here?)
direction = tool_input.get("scroll_direction", "down")
amount = tool_input.get("scroll_amount", 3)
scroll_amount = -amount if direction == "down" else amount
pyautogui.scroll(scroll_amount)
return {"type": "text", "text": f"Scrolled {direction} by {amount}"}
else:
return {"type": "text", "text": f"Unknown computer action: {action}"}
def execute_bash_tool(command):
"""Execute bash commands safely."""
try:
# Execute the command
result = subprocess.run(
command,
shell=True,
capture_output=True,
text=True,
timeout=30 # 30 second timeout
)
output = result.stdout
if result.stderr:
output += f"\n[STDERR]: {result.stderr}"
return {"type": "text", "text": output if output else "(Command executed successfully with no output)"}
except subprocess.TimeoutExpired:
return {"type": "text", "text": "Error: Command timed out after 30 seconds"}
except Exception as e:
return {"type": "text", "text": f"Error executing command: {str(e)}"}
def execute_text_editor_tool(_command, _path, **kwargs):
"""Execute text editor actions."""
if _command == "view":
try:
with open(_path, 'r') as f:
content = f.read()
return {"type": "text", "text": content}
except Exception as e:
return {"type": "text", "text": f"Error reading file: {str(e)}"}
elif _command == "str_replace":
old_str = kwargs.get("old_str", "")
new_str = kwargs.get("new_str", "")
try:
with open(_path, 'r') as f:
content = f.read()
if old_str not in content:
return {"type": "text", "text": f"Error: '{old_str}' not found in file"}
# Replace the string
new_content = content.replace(old_str, new_str)
# Write back to file
with open(_path, 'w') as f:
f.write(new_content)
return {"type": "text", "text": "String replaced successfully"}
except Exception as e:
return {"type": "text", "text": f"Error modifying file: {str(e)}"}
elif _command == "create":
content = kwargs.get("content", "")
try:
with open(_path, 'w') as f:
f.write(content)
return {"type": "text", "text": f"File created: {_path}"}
except Exception as e:
return {"type": "text", "text": f"Error creating file: {str(e)}"}
else:
return {"type": "text", "text": f"Unknown text editor command: {_command}"}
def execute_tool(tool_name, tool_input):
"""Main function to execute any tool based on its name and input."""
print(f"\n[EXECUTING] Tool: {tool_name}")
print(f"[INPUT] {json.dumps(tool_input, indent=2)}")
if tool_name == "computer":
# Pass the entire tool_input to execute_computer_tool
return execute_computer_tool(tool_input)
elif tool_name == "bash":
command = tool_input.get("command", "")
return execute_bash_tool(command)
elif tool_name == "str_replace_editor":
command = tool_input.get("command")
path = tool_input.get("path", "")
return execute_text_editor_tool(command, path, **tool_input)
else:
return {"type": "text", "text": f"Unknown tool: {tool_name}"}
def send_prompt(api_key, messages): def send_prompt(api_key, messages):
headers = HEADERS.copy() headers = HEADERS.copy()
headers["x-api-key"] = api_key headers["x-api-key"] = api_key
@@ -43,8 +246,8 @@ def send_prompt(api_key, messages):
{ {
"type": f"computer_{TOOL_VERSION}", "type": f"computer_{TOOL_VERSION}",
"name": "computer", "name": "computer",
"display_width_px": 1024, "display_width_px": 2880,
"display_height_px": 768, "display_height_px": 1864,
"display_number": 1 "display_number": 1
}, },
{ {
@@ -66,11 +269,13 @@ def send_prompt(api_key, messages):
return send_request(headers, body) return send_request(headers, body)
def extract_tool_use(response): def extract_tool_uses(response):
"""Extract all tool use requests from the response."""
tool_uses = []
for block in response.get("content", []): for block in response.get("content", []):
if block.get("type") == "tool_use": if block.get("type") == "tool_use":
return block tool_uses.append(block)
return None return tool_uses
def run(): def run():
@@ -88,66 +293,66 @@ def run():
}) })
print(f"Sending prompt: {prompt}") print(f"Sending prompt: {prompt}")
first_response = send_prompt(api_key, messages)
# We can't just append the whole first response, otherwise we get 400 bad req: # Main conversation loop
# "Extra inputs are not permitted" while True:
messages.append({ response = send_prompt(api_key, messages)
'role': 'assistant',
'content': first_response.get('content', []),
})
tool_use = extract_tool_use(first_response) # Add Claude's response to messages
if not tool_use: messages.append({
print("No tool use required. Response:") 'role': 'assistant',
print(first_response["content"]) 'content': response.get('content', []),
return })
# print first response from Claude # Print Claude's response
for block in first_response.get("content", []): print("\n[CLAUDE RESPONSE]")
print(f'{block['type']}:') for block in response.get("content", []):
if block["type"] == "text": if block["type"] == "text":
print(f'\t{block["text"]}') print(block["text"])
elif block["type"] == "thinking": elif block["type"] == "thinking":
print(f'\t{block["thinking"]}') print(f"[THINKING] {block['thinking']}")
elif block["type"] == "tool_use": elif block["type"] == "tool_use":
print(f'\tTool use requested: {block["name"]}') print(f"[TOOL REQUEST] {block['name']} - {block.get('input', {})}")
print(f'\tInput: {block["input"]}')
# this is a stub for the tool execution # Extract all tool uses
fake_result_message = { tool_uses = extract_tool_uses(response)
"role": "user",
"content": [ if not tool_uses:
{ # No more tool uses, we're done
break
# Execute all requested tools and collect results
tool_results = []
for tool_use in tool_uses:
tool_name = tool_use["name"]
tool_input = tool_use.get("input", {})
tool_id = tool_use["id"]
# Execute the tool
result_content = execute_tool(tool_name, tool_input)
print('[TOOL RESULT]')
for key in result_content:
if key == "source":
print(f"{key}: {len(result_content[key])} chars long")
else:
print(f"{key}: {result_content[key]}")
# Add to results
tool_results.append({
"type": "tool_result", "type": "tool_result",
"tool_use_id": tool_use["id"], "tool_use_id": tool_id,
"content": "(Stub) Executed tool successfully" "content": [result_content] if isinstance(result_content, dict) else result_content
} })
]
}
messages.append(fake_result_message)
# send the fake result back to Claude # Add all tool results as a user message
headers = HEADERS.copy() messages.append({
headers["x-api-key"] = api_key "role": "user",
"content": tool_results
})
tool_result_msg = { print("\n[TOOL RESULTS SENT BACK TO CLAUDE]")
"model": MODEL,
"max_tokens": 2048,
"tools": [],
"messages": messages,
"thinking": {
"type": "enabled",
"budget_tokens": 1024
}
}
final_response = send_request(headers, tool_result_msg) print("\n[CONVERSATION COMPLETE]")
print("Final Claude response:")
for block in final_response.get("content", []):
if block["type"] == "text":
print(block["text"])
if __name__ == "__main__": if __name__ == "__main__":
+39
View File
@@ -0,0 +1,39 @@
'''
This script simulates a double-click at the given mouse cursor position.
'''
import Quartz
from time import sleep
def post_mouse_event(type, pos, click_state):
event = Quartz.CGEventCreateMouseEvent(
None, type, pos, Quartz.kCGMouseButtonLeft
)
Quartz.CGEventSetIntegerValueField(event, Quartz.kCGMouseEventClickState, click_state)
Quartz.CGEventPost(Quartz.kCGHIDEventTap, event)
def double_click (x, y):
pos = None
if x is None or y is None:
# Get current mouse position if no coordinates are provided
loc = Quartz.CGEventGetLocation(Quartz.CGEventCreate(None))
pos = (loc.x, loc.y)
else:
# Use provided coordinates
pos = (x, y)
# First click
post_mouse_event(Quartz.kCGEventLeftMouseDown, pos, 1)
post_mouse_event(Quartz.kCGEventLeftMouseUp, pos, 1)
sleep(0.05) # Short delay within double-click threshold
# Second click
post_mouse_event(Quartz.kCGEventLeftMouseDown, pos, 2)
post_mouse_event(Quartz.kCGEventLeftMouseUp, pos, 2)
if __name__ == "__main__":
print('test: double clicking at cursor position in 3 seconds...')
sleep(3)
double_click()