1 Commits

Author SHA1 Message Date
leopengpong 58aa434e3e add computer, bash, text editor tools 2025-06-10 09:43:35 -07:00
10 changed files with 33 additions and 327 deletions
-3
View File
@@ -1,3 +0,0 @@
[submodule "stream/thirdparty/llama.cpp"]
path = stream/thirdparty/llama.cpp
url = https://github.com/ggml-org/llama.cpp
+1
View File
@@ -0,0 +1 @@
# computeruse
-2
View File
@@ -1,2 +0,0 @@
# computeruse
This is a prototype.
-39
View File
@@ -1,39 +0,0 @@
'''
This script simulates a double-click at the given mouse cursor position.
'''
import Quartz
from time import sleep
def post_mouse_event(type, pos, click_state):
event = Quartz.CGEventCreateMouseEvent(
None, type, pos, Quartz.kCGMouseButtonLeft
)
Quartz.CGEventSetIntegerValueField(event, Quartz.kCGMouseEventClickState, click_state)
Quartz.CGEventPost(Quartz.kCGHIDEventTap, event)
def double_click (x, y):
pos = None
if x is None or y is None:
# Get current mouse position if no coordinates are provided
loc = Quartz.CGEventGetLocation(Quartz.CGEventCreate(None))
pos = (loc.x, loc.y)
else:
# Use provided coordinates
pos = (x, y)
# First click
post_mouse_event(Quartz.kCGEventLeftMouseDown, pos, 1)
post_mouse_event(Quartz.kCGEventLeftMouseUp, pos, 1)
sleep(0.05) # Short delay within double-click threshold
# Second click
post_mouse_event(Quartz.kCGEventLeftMouseDown, pos, 2)
post_mouse_event(Quartz.kCGEventLeftMouseUp, pos, 2)
if __name__ == "__main__":
print('test: double clicking at cursor position in 3 seconds...')
sleep(3)
double_click()
+32 -53
View File
@@ -6,14 +6,11 @@ import pyautogui
import subprocess
from io import BytesIO
from PIL import Image
import quartz_doubleclick
ANTHROPIC_API_URL = "https://api.anthropic.com/v1/messages"
MODEL = "claude-3-7-sonnet-20250219"
BETA_FLAG = "computer-use-2025-01-24"
TOOL_VERSION = "20250124"
pyautogui.PAUSE = 0.1
SCALING_FACTOR = 1.25
HEADERS = {
"content-type": "application/json",
@@ -51,9 +48,8 @@ def execute_computer_tool(tool_input):
buffered = BytesIO()
screenshot.save(buffered, format="PNG")
MAX_BINARY_SIZE = 5242880 * 3 // 4
img_data = None
if buffered.tell() > MAX_BINARY_SIZE:
# Check size and resize if needed (5MB = 5242880 bytes)
if buffered.tell() > 4242880:
# Reset buffer
buffered.seek(0)
screenshot = Image.open(buffered)
@@ -62,19 +58,14 @@ def execute_computer_tool(tool_input):
while True:
buffered = BytesIO()
new_size = (int(screenshot.width * 0.8), int(screenshot.height * 0.8))
print('new size:', new_size)
screenshot = screenshot.resize(new_size, Image.Resampling.LANCZOS)
screenshot.save(buffered, format="PNG", optimize=True)
buffered.seek(0)
img_data = buffered.read()
if len(img_data) <= MAX_BINARY_SIZE:
if buffered.tell() <= 4242880:
break
buffered.seek(0)
screenshot = Image.open(buffered)
else:
img_data = buffered.getvalue()
# Convert to base64
img_base64 = base64.b64encode(img_data).decode('utf-8')
buffered.seek(0)
img_base64 = base64.b64encode(buffered.read()).decode('utf-8')
return {
"type": "image",
@@ -85,10 +76,10 @@ def execute_computer_tool(tool_input):
}
}
elif action == "left_click":
elif action == "click":
# Get coordinates
x = tool_input.get("coordinate")[0] * SCALING_FACTOR
y = tool_input.get("coordinate")[1] * SCALING_FACTOR
x = tool_input.get("coordinate_x")
y = tool_input.get("coordinate_y")
if x is None or y is None:
return {"type": "text", "text": "Error: Missing coordinates for click action"}
@@ -98,14 +89,13 @@ def execute_computer_tool(tool_input):
elif action == "double_click":
# Get coordinates
x = tool_input.get("coordinate")[0] * SCALING_FACTOR
y = tool_input.get("coordinate")[1] * SCALING_FACTOR
x = tool_input.get("coordinate_x")
y = tool_input.get("coordinate_y")
if x is None or y is None:
return {"type": "text", "text": "Error: Missing coordinates for double click action"}
# Perform double click
# pyautogui.doubleClick(x, y, interval=0.2) this doesn't work
quartz_doubleclick.double_click(x, y)
pyautogui.doubleClick(x, y)
return {"type": "text", "text": f"Double-clicked at coordinates ({x}, {y})"}
elif action == "type":
@@ -116,24 +106,19 @@ def execute_computer_tool(tool_input):
elif action == "key":
# Press a key or key combination
text = tool_input.get("text", "")
key = tool_input.get("key", "")
try:
if '+' in text:
# Handle key combinations like "command+c"
keys = text.replace('super', 'command').split('+')
pyautogui.hotkey(*keys, interval=0.05) # interval is required
else:
pyautogui.press(text)
return {"type": "text", "text": f"Pressed key: {text}"}
pyautogui.press(key)
return {"type": "text", "text": f"Pressed key: {key}"}
except Exception as e:
return {"type": "text", "text": f"Error pressing key {text}: {str(e)}"}
return {"type": "text", "text": f"Error pressing key {key}: {str(e)}"}
elif action == "scroll":
# Scroll action (should we really have defaults here?)
direction = tool_input.get("scroll_direction", "down")
amount = tool_input.get("scroll_amount", 3)
# Scroll action
direction = tool_input.get("direction", "down")
amount = tool_input.get("amount", 3)
scroll_amount = -amount if direction == "down" else amount
scroll_amount = -amount if direction == "up" else amount
pyautogui.scroll(scroll_amount)
return {"type": "text", "text": f"Scrolled {direction} by {amount}"}
@@ -165,23 +150,23 @@ def execute_bash_tool(command):
return {"type": "text", "text": f"Error executing command: {str(e)}"}
def execute_text_editor_tool(_command, _path, **kwargs):
def execute_text_editor_tool(command, path, **kwargs):
"""Execute text editor actions."""
if _command == "view":
if command == "view":
try:
with open(_path, 'r') as f:
with open(path, 'r') as f:
content = f.read()
return {"type": "text", "text": content}
except Exception as e:
return {"type": "text", "text": f"Error reading file: {str(e)}"}
elif _command == "str_replace":
elif command == "str_replace":
old_str = kwargs.get("old_str", "")
new_str = kwargs.get("new_str", "")
try:
with open(_path, 'r') as f:
with open(path, 'r') as f:
content = f.read()
if old_str not in content:
@@ -191,7 +176,7 @@ def execute_text_editor_tool(_command, _path, **kwargs):
new_content = content.replace(old_str, new_str)
# Write back to file
with open(_path, 'w') as f:
with open(path, 'w') as f:
f.write(new_content)
return {"type": "text", "text": "String replaced successfully"}
@@ -199,17 +184,17 @@ def execute_text_editor_tool(_command, _path, **kwargs):
except Exception as e:
return {"type": "text", "text": f"Error modifying file: {str(e)}"}
elif _command == "create":
elif command == "create":
content = kwargs.get("content", "")
try:
with open(_path, 'w') as f:
with open(path, 'w') as f:
f.write(content)
return {"type": "text", "text": f"File created: {_path}"}
return {"type": "text", "text": f"File created: {path}"}
except Exception as e:
return {"type": "text", "text": f"Error creating file: {str(e)}"}
else:
return {"type": "text", "text": f"Unknown text editor command: {_command}"}
return {"type": "text", "text": f"Unknown text editor command: {command}"}
def execute_tool(tool_name, tool_input):
@@ -246,8 +231,8 @@ def send_prompt(api_key, messages):
{
"type": f"computer_{TOOL_VERSION}",
"name": "computer",
"display_width_px": 2880,
"display_height_px": 1864,
"display_width_px": 1024,
"display_height_px": 768,
"display_number": 1
},
{
@@ -330,12 +315,6 @@ def run():
# Execute the tool
result_content = execute_tool(tool_name, tool_input)
print('[TOOL RESULT]')
for key in result_content:
if key == "source":
print(f"{key}: {len(result_content[key])} chars long")
else:
print(f"{key}: {result_content[key]}")
# Add to results
tool_results.append({
-4
View File
@@ -1,4 +0,0 @@
.cache
stream_app
*.gguf
compile_commands.json
-11
View File
@@ -1,11 +0,0 @@
CC = gcc
CFLAGS = -Wall -g -I./thirdparty/llama.cpp/include -I./thirdparty/llama.cpp/ggml/include
TARGET = stream_app
SRC = main.c
LIBS = -L./thirdparty/llama.cpp/build/bin -lllama -lggml -lggml-base
$(TARGET): $(SRC)
$(CC) $(CFLAGS) -o $(TARGET) $(SRC) $(LIBS)
clean:
rm -f *.o $(TARGET)
-38
View File
@@ -1,38 +0,0 @@
# flowy.stream
This is a proof of concept of a showing widgets within a conversation based on intent.
## Todo
- [x] Integrate llama.cpp with local inference. This will set us up for building many parts of experience.
- [ ] Disect what llama is doing and what the Phi model is doing.
- [ ] Create conversational loop with chat and running context.
- [ ] Generate & render different types of blocks: list, email, doc, etc.
- [ ] Try different models with hugging face
- [ ] Render TUI elements based on commands
## Dependencies
### llama.cpp
```sh
#model weights
wget https://huggingface.co/microsoft/Phi-3-mini-4k-instruct-gguf/resolve/main/Phi-3-mini-4k-instruct-q4.gguf
```
```sh
cd ./thirdparty/llama.cpp
rm -rf build
mkdir -p build
cmake -S. -Bbuild -DLLAMA_CURL=OFF
cmake --build build
```
The libraries should be in `./thirdparty/llama.cpp/build/bin`.
NOTE: you may have to disable curl as a flag when configuring llama.cpp via cmake.
## Getting started
```sh
make
LD_LIBRARY_PATH=./thirdparty/llama.cpp/build/bin/ ./stream_app
```
-176
View File
@@ -1,176 +0,0 @@
#include "thirdparty/llama.cpp/ggml/include/ggml-backend.h"
#include <llama.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
typedef enum {
Unspecified,
Question,
Math,
Code,
Email,
Doc,
WebSearch,
} IntentType;
int main(int argc, char *argv[]) {
printf("Hello world\n");
char user_input[500];
printf("What's up, what can I help with?\n");
fgets(user_input, sizeof(user_input), stdin);
char prompt[1000];
snprintf(prompt, sizeof(prompt), "<|system|>You are a helpful assistant. Your goal is to take in what the user is doing and return 3 predictive actions/3 suggestions based on what the user is trying to do: e.g. change page title when in google sheets, or calculate sum, create chart.<|end|>\n<|user|>%s<|end|>\n<|assistant|>", user_input);
// number of layers to offload to the GPU
int ngl = 99;
// number of tokens to predict
int n_predict = 1000;
// load dynamic backends
ggml_backend_load_all();
// initialize the model
struct llama_model_params model_params = llama_model_default_params();
model_params.n_gpu_layers = ngl;
struct llama_model *model = llama_model_load_from_file(
"Phi-3-mini-4k-instruct-q4.gguf", model_params);
if (model == NULL) {
fprintf(stderr, "%s: error: unable to load model\n", __func__);
return 1;
}
const struct llama_vocab *vocab = llama_model_get_vocab(model);
// tokenize the prompt
// find the number of tokens in the prompt
const int n_prompt =
-llama_tokenize(vocab, prompt, strlen(prompt), NULL, 0, true, true);
// allocate space for the tokens and tokenize the prompt
llama_token *prompt_tokens = malloc(n_prompt * sizeof(llama_token));
if (prompt_tokens == NULL) {
fprintf(stderr, "%s: error: failed to allocate memory for prompt tokens\n",
__func__);
return 1;
}
if (llama_tokenize(vocab, prompt, strlen(prompt), prompt_tokens, n_prompt,
true, true) < 0) {
fprintf(stderr, "%s: error: failed to tokenize the prompt\n", __func__);
return 1;
}
// initialize the context
struct llama_context_params ctx_params = llama_context_default_params();
// n_ctx is the context size
ctx_params.n_ctx = n_prompt + n_predict - 1;
// n_batch is the maximum number of tokens that can be processed in a single
// call to llama_decode
ctx_params.n_batch = n_prompt;
// enable performance counters
ctx_params.no_perf = false;
struct llama_context *ctx = llama_init_from_model(model, ctx_params);
if (ctx == NULL) {
fprintf(stderr, "%s: error: failed to create the llama_context\n",
__func__);
return 1;
}
// initialize the sampler
struct llama_sampler_chain_params sparams =
llama_sampler_chain_default_params();
sparams.no_perf = false;
struct llama_sampler *smpl = llama_sampler_chain_init(sparams);
llama_sampler_chain_add(smpl, llama_sampler_init_greedy());
// print the prompt token-by-token
for (int i = 0; i < n_prompt; i++) {
char buf[128];
int n = llama_token_to_piece(vocab, prompt_tokens[i], buf, sizeof(buf), 0,
true);
if (n < 0) {
fprintf(stderr, "%s: error: failed to convert token to piece\n",
__func__);
return 1;
}
buf[n] = '\0';
printf("%s", buf);
}
// prepare a batch for the prompt
llama_batch batch = llama_batch_get_one(prompt_tokens, n_prompt);
// main loop
const int64_t t_main_start = ggml_time_us();
int n_decode = 0;
llama_token new_token_id;
for (int n_pos = 0; n_pos + batch.n_tokens < n_prompt + n_predict;) {
// evaluate the current batch with the transformer model
if (llama_decode(ctx, batch)) {
fprintf(stderr, "%s : failed to eval, return code %d\n", __func__, 1);
return 1;
}
n_pos += batch.n_tokens;
// sample the next token
{
new_token_id = llama_sampler_sample(smpl, ctx, -1);
// is it an end of generation?
if (llama_vocab_is_eog(vocab, new_token_id)) {
break;
}
char buf[128];
int n =
llama_token_to_piece(vocab, new_token_id, buf, sizeof(buf), 0, true);
if (n < 0) {
fprintf(stderr, "%s: error: failed to convert token to piece\n",
__func__);
return 1;
}
buf[n] = '\0';
printf("%s", buf);
fflush(stdout);
// prepare the next batch with the sampled token
batch = llama_batch_get_one(&new_token_id, 1);
n_decode += 1;
}
}
printf("\n");
const int64_t t_main_end = ggml_time_us();
fprintf(stderr, "%s: decoded %d tokens in %.2f s, speed: %.2f t/s\n",
__func__, n_decode, (t_main_end - t_main_start) / 1000000.0f,
n_decode / ((t_main_end - t_main_start) / 1000000.0f));
fprintf(stderr, "\n");
llama_perf_sampler_print(smpl);
llama_perf_context_print(ctx);
fprintf(stderr, "\n");
llama_sampler_free(smpl);
llama_free(ctx);
llama_model_free(model);
free(prompt_tokens);
}