diff --git a/ai-conversation/Makefile b/ai-conversation/Makefile index 176686d..196abb6 100644 --- a/ai-conversation/Makefile +++ b/ai-conversation/Makefile @@ -1,7 +1,7 @@ CC = gcc CFLAGS = -Wall -g -TARGET = audio_capture -SRC = audio_capture.c deepgram_client.c utils.c +TARGET = main +SRC = main.c deepgram_client.c utils.c intelligence.c audio.c LIBS = -lavdevice -lavformat -lavcodec -lavutil -lasound -lcurl $(TARGET): $(SRC) diff --git a/ai-conversation/audio_capture.c b/ai-conversation/audio.c similarity index 59% rename from ai-conversation/audio_capture.c rename to ai-conversation/audio.c index d672d20..99f8366 100644 --- a/ai-conversation/audio_capture.c +++ b/ai-conversation/audio.c @@ -1,113 +1,12 @@ -#include "deepgram_client.h" -#include "utils.h" -#include -#include -#include -#include -#include -#include +#include "audio.h" -// NOTE: find this via `arecord -l` -#define ALSA_MIC_HARDWARE_DEVICE "hw:3" -#define ALSA_SPEAKER_HARDWARE_DEVICE "default" - -void print_pcm_state(snd_pcm_t *pcm_handle) { +static void print_pcm_state(snd_pcm_t *pcm_handle) { snd_pcm_state_t state = snd_pcm_state(pcm_handle); const char *state_name = snd_pcm_state_name(state); printf("State of pcm handle: %s\n", state_name); } -char *find_ai_response(char *json_response) { - printf("Parsing response: %s \n", json_response); - char *start = strstr(json_response, "\"text\":\""); - if (!start) - return NULL; - - start += 8; // Skip past "text":" - char *end = strchr(start, '"'); - if (!end) - return NULL; - - size_t len = end - start; - char *result = malloc(len + 1); - strncpy(result, start, len); - result[len] = '\0'; - return result; -} - -char *ask_ai(char *question) { - CURL *curl; - CURLcode res; - struct curl_slist *headers = NULL; - - // Initialize response buffer - struct curl_response_data response = {0}; - response.data = malloc(1); - response.size = 0; - - // Create JSON payload - char json_payload[2048]; - snprintf(json_payload, sizeof(json_payload), - "{" - "\"model\":\"claude-3-haiku-20240307\"," - "\"max_tokens\":1024," - "\"messages\":[{\"role\":\"user\",\"content\":\"%s\"}]" - "}", - question); - - curl = curl_easy_init(); - if (curl) { - char *anthropic_api_key = getenv("ANTHROPIC_API_KEY"); - if (!anthropic_api_key) { - printf("ANTHROPIC_API_KEY not set\n"); - free(response.data); - curl_easy_cleanup(curl); - return NULL; - } - char auth_header[256]; - snprintf(auth_header, sizeof(auth_header), "x-api-key: %s", - anthropic_api_key); - headers = curl_slist_append(headers, "content-type: application/json"); - headers = curl_slist_append(headers, auth_header); - headers = curl_slist_append(headers, "anthropic-version: 2023-06-01"); - - curl_easy_setopt(curl, CURLOPT_URL, - "https://api.anthropic.com/v1/messages"); - curl_easy_setopt(curl, CURLOPT_HTTPHEADER, headers); - curl_easy_setopt(curl, CURLOPT_POST, 1L); - curl_easy_setopt(curl, CURLOPT_POSTFIELDS, json_payload); - curl_easy_setopt(curl, CURLOPT_WRITEFUNCTION, curl_write_callback_response); - curl_easy_setopt(curl, CURLOPT_WRITEDATA, &response); - - res = curl_easy_perform(curl); - if (res != CURLE_OK) { - fprintf(stderr, "curl failed: %s\n", curl_easy_strerror(res)); - free(response.data); - curl_slist_free_all(headers); - curl_easy_cleanup(curl); - return NULL; - } - - curl_slist_free_all(headers); - curl_easy_cleanup(curl); - } - - // Parse AI response from complete response - char *answer = find_ai_response(response.data); - free(response.data); - - return answer; -} - -/* count: how many bytes in the buffer - * returns int: 0 if success, -1 if failure - * NOTE: This function assumes that the audio data is using: - * - linear16 formatting (each sample is 16 bits) - * - one channel (1 sample) - * - with 48000 HZ sampling rate - * Hence, it's 2 bytes per frame. - */ -int play_pcm_data(char *buf, size_t byte_count) { +int audio_play(char *buf, size_t byte_count) { printf("going to play audio of %d bytes of data\n", (int)byte_count); snd_pcm_t *pcm_handle; @@ -192,7 +91,9 @@ int play_pcm_data(char *buf, size_t byte_count) { return 0; } -int main() { +char *audio_capture(uint seconds) { + char *output_file = "output.pcm"; + snd_pcm_t *pcm_handle = NULL; int err; @@ -200,7 +101,7 @@ int main() { SND_PCM_STREAM_CAPTURE, 0); if (err < 0) { printf("unable to open pcm device"); - return -1; + return NULL; } print_pcm_state(pcm_handle); @@ -213,14 +114,14 @@ int main() { if (err < 0) { snd_pcm_close(pcm_handle); printf("unable to get info of pcm device"); - return -1; + return NULL; } name = snd_pcm_info_get_name(pcm_info); if (err < 0) { printf("unable to get card info"); snd_pcm_close(pcm_handle); snd_pcm_info_free(pcm_info); - return -1; + return NULL; } printf("got pcm name: %s \n", name); @@ -231,14 +132,14 @@ int main() { err = snd_pcm_hw_params_malloc(¶ms); if (err < 0) { fprintf(stderr, "unable to mallow hw params"); - return -1; + return NULL; } err = snd_pcm_hw_params_any(pcm_handle, params); if (err < 0) { fprintf(stderr, "unable to find ranges for hw device"); snd_pcm_close(pcm_handle); snd_pcm_hw_params_free(params); - return -1; + return NULL; } uint period_size = 1024; @@ -259,7 +160,7 @@ int main() { fprintf(stderr, "unable to prepare pcm\n"); snd_pcm_close(pcm_handle); snd_pcm_hw_params_free(params); - return -1; + return NULL; } print_pcm_state(pcm_handle); @@ -269,12 +170,12 @@ int main() { fprintf(stderr, "unable to start pcm"); snd_pcm_close(pcm_handle); snd_pcm_hw_params_free(params); - return -1; + return NULL; } print_pcm_state(pcm_handle); - const uint seconds_to_capture = 5; + const uint seconds_to_capture = seconds; const uint periods_to_read = (samples_per_second * seconds_to_capture) / period_size; char *buf = malloc(period_size * channel_count * bytes_per_sample); @@ -285,10 +186,10 @@ int main() { * Each period is 2048 BYTES because each frame is 2 bytes. * Each frame is 2 bytes because of 1 channel, and 16 bit format */ - FILE *file = fopen("output.pcm", "wb"); + FILE *file = fopen(output_file, "wb"); if (file == NULL) { perror("Error opening file"); - return 1; + return NULL; } printf("======READING FRAMES for 5 seconds========\n"); while (snd_pcm_readi(pcm_handle, (void *)buf, period_size) > 0 && @@ -306,42 +207,9 @@ int main() { fclose(file); - // Reopen file for reading and transcribe - file = fopen("output.pcm", "rb"); - if (file != NULL) { - char *transcript = dg_transcribe(file); - if (transcript) { - printf("Transcript: %s\n", transcript); - - char *ai_answer = ask_ai(transcript); - if (ai_answer) { - printf("AI answer: %s\n", ai_answer); - - struct curl_response_data *tts_response = dg_text_to_speech(ai_answer); - int result = play_pcm_data(tts_response->data, tts_response->size); - if (result != 0) { - fprintf(stderr, "failure in playback"); - } - - free(tts_response->data); - free(tts_response); - - // we have this response data, and buffer within it. we can loop through - // it and write the pcm data within it to alsa - free(ai_answer); - } - free(transcript); - } else { - fprintf(stderr, "unable to get transcript"); - } - fclose(file); - } else { - fprintf(stderr, "unable to open file for reading\n"); - } - free(buf); snd_pcm_close(pcm_handle); snd_pcm_hw_params_free(params); - return 0; + return output_file; } diff --git a/ai-conversation/audio.h b/ai-conversation/audio.h new file mode 100644 index 0000000..4f86357 --- /dev/null +++ b/ai-conversation/audio.h @@ -0,0 +1,29 @@ +#ifndef AUDIOCAPTURE_H +#define AUDIOCAPTURE_H + +#include +#include +#include +#include +#include +#include + +// NOTE: find this via `arecord -l` +#define ALSA_MIC_HARDWARE_DEVICE "hw:3" +#define ALSA_SPEAKER_HARDWARE_DEVICE "default" + +// Captures audio for given seconds, and returns file name that we captured the +// audio into +char *audio_capture(uint seconds); + +/* count: how many bytes in the buffer + * returns int: 0 if success, -1 if failure + * NOTE: This function assumes that the audio data is using: + * - linear16 formatting (each sample is 16 bits) + * - one channel (1 sample) + * - with 48000 HZ sampling rate + * Hence, it's 2 bytes per frame. + */ +int audio_play(char *buf, size_t byte_count); + +#endif // AUDIOCAPTURE_H diff --git a/ai-conversation/audio_capture b/ai-conversation/audio_capture deleted file mode 100755 index 99d283e..0000000 Binary files a/ai-conversation/audio_capture and /dev/null differ diff --git a/ai-conversation/intelligence.c b/ai-conversation/intelligence.c new file mode 100644 index 0000000..76cf6b3 --- /dev/null +++ b/ai-conversation/intelligence.c @@ -0,0 +1,86 @@ +#include "intelligence.h" +#include "utils.h" +#include +#include + +static char *find_ai_response(char *json_response) { + printf("Parsing response: %s \n", json_response); + char *start = strstr(json_response, "\"text\":\""); + if (!start) + return NULL; + + start += 8; // Skip past "text":" + char *end = strchr(start, '"'); + if (!end) + return NULL; + + size_t len = end - start; + char *result = malloc(len + 1); + strncpy(result, start, len); + result[len] = '\0'; + return result; +} + +char *ai_ask(char *question) { + CURL *curl; + CURLcode res; + struct curl_slist *headers = NULL; + + // Initialize response buffer + struct curl_response_data response = {0}; + response.data = malloc(1); + response.size = 0; + + // Create JSON payload + char json_payload[2048]; + snprintf(json_payload, sizeof(json_payload), + "{" + "\"model\":\"claude-3-haiku-20240307\"," + "\"max_tokens\":256," + "\"messages\":[{\"role\":\"user\",\"content\":\"Make it one sentence answer: %s\"}]" + "}", + question); + + curl = curl_easy_init(); + if (curl) { + char *anthropic_api_key = getenv("ANTHROPIC_API_KEY"); + if (!anthropic_api_key) { + printf("ANTHROPIC_API_KEY not set\n"); + free(response.data); + curl_easy_cleanup(curl); + return NULL; + } + char auth_header[256]; + snprintf(auth_header, sizeof(auth_header), "x-api-key: %s", + anthropic_api_key); + headers = curl_slist_append(headers, "content-type: application/json"); + headers = curl_slist_append(headers, auth_header); + headers = curl_slist_append(headers, "anthropic-version: 2023-06-01"); + + curl_easy_setopt(curl, CURLOPT_URL, + "https://api.anthropic.com/v1/messages"); + curl_easy_setopt(curl, CURLOPT_HTTPHEADER, headers); + curl_easy_setopt(curl, CURLOPT_POST, 1L); + curl_easy_setopt(curl, CURLOPT_POSTFIELDS, json_payload); + curl_easy_setopt(curl, CURLOPT_WRITEFUNCTION, curl_write_callback_response); + curl_easy_setopt(curl, CURLOPT_WRITEDATA, &response); + + res = curl_easy_perform(curl); + if (res != CURLE_OK) { + fprintf(stderr, "curl failed: %s\n", curl_easy_strerror(res)); + free(response.data); + curl_slist_free_all(headers); + curl_easy_cleanup(curl); + return NULL; + } + + curl_slist_free_all(headers); + curl_easy_cleanup(curl); + } + + // Parse AI response from complete response + char *answer = find_ai_response(response.data); + free(response.data); + + return answer; +} diff --git a/ai-conversation/intelligence.h b/ai-conversation/intelligence.h new file mode 100644 index 0000000..7404e05 --- /dev/null +++ b/ai-conversation/intelligence.h @@ -0,0 +1,9 @@ +#ifndef INTELLIGENCE_H +#define INTELLIGENCE_H + +#include +#include + +char *ai_ask(char *question); + +#endif //INTELLIGENCE_H diff --git a/ai-conversation/main b/ai-conversation/main new file mode 100755 index 0000000..cd90b24 Binary files /dev/null and b/ai-conversation/main differ diff --git a/ai-conversation/main.c b/ai-conversation/main.c index 6c4d7f9..8fe06bb 100644 --- a/ai-conversation/main.c +++ b/ai-conversation/main.c @@ -1,5 +1,45 @@ +#include "audio.h" +#include "deepgram_client.h" +#include "intelligence.h" +#include "utils.h" #include +#include -int main() { - printf("hello world \n"); +int main(int argc, char *argv[]) { + char *capture_audio_file = audio_capture(5); + + // Reopen file for reading and transcribe + FILE *file = fopen(capture_audio_file, "rb"); + if (file != NULL) { + char *transcript = dg_transcribe(file); + if (transcript) { + printf("Transcript: %s\n", transcript); + + char *ai_answer = ai_ask(transcript); + if (ai_answer) { + printf("AI answer: %s\n", ai_answer); + + struct curl_response_data *tts_response = dg_text_to_speech(ai_answer); + int result = audio_play(tts_response->data, tts_response->size); + if (result != 0) { + fprintf(stderr, "failure in playback"); + } + + free(tts_response->data); + free(tts_response); + + // we have this response data, and buffer within it. we can loop through + // it and write the pcm data within it to alsa + free(ai_answer); + } + free(transcript); + } else { + fprintf(stderr, "unable to get transcript"); + } + fclose(file); + } else { + fprintf(stderr, "unable to open file for reading\n"); + } + + return 0; }