From c93267c9ee6f285299a900da5cfc95ec14c4ef08 Mon Sep 17 00:00:00 2001 From: talksik Date: Sun, 27 Jul 2025 14:18:00 -0700 Subject: [PATCH] create full flow with stt, anthropic, tts, and playback This organizes some components like deepgram into own module, and also allows gets the full flow to work every time we run the program. --- ai-conversation/Makefile | 2 +- ai-conversation/audio_capture.c | 595 +++++++++++++++--------------- ai-conversation/deepgram_client.c | 168 +++++++++ ai-conversation/deepgram_client.h | 21 ++ ai-conversation/utils.c | 30 ++ ai-conversation/utils.h | 24 ++ 6 files changed, 550 insertions(+), 290 deletions(-) create mode 100644 ai-conversation/deepgram_client.c create mode 100644 ai-conversation/deepgram_client.h create mode 100644 ai-conversation/utils.c create mode 100644 ai-conversation/utils.h diff --git a/ai-conversation/Makefile b/ai-conversation/Makefile index 29408bf..176686d 100644 --- a/ai-conversation/Makefile +++ b/ai-conversation/Makefile @@ -1,7 +1,7 @@ CC = gcc CFLAGS = -Wall -g TARGET = audio_capture -SRC = audio_capture.c +SRC = audio_capture.c deepgram_client.c utils.c LIBS = -lavdevice -lavformat -lavcodec -lavutil -lasound -lcurl $(TARGET): $(SRC) diff --git a/ai-conversation/audio_capture.c b/ai-conversation/audio_capture.c index 6f4eb90..d672d20 100644 --- a/ai-conversation/audio_capture.c +++ b/ai-conversation/audio_capture.c @@ -1,330 +1,347 @@ -#include -#include +#include "deepgram_client.h" +#include "utils.h" #include #include #include #include +#include +#include // NOTE: find this via `arecord -l` #define ALSA_MIC_HARDWARE_DEVICE "hw:3" +#define ALSA_SPEAKER_HARDWARE_DEVICE "default" void print_pcm_state(snd_pcm_t *pcm_handle) { - snd_pcm_state_t state = snd_pcm_state(pcm_handle); - const char *state_name = snd_pcm_state_name(state); - printf("State of pcm handle: %s\n", state_name); + snd_pcm_state_t state = snd_pcm_state(pcm_handle); + const char *state_name = snd_pcm_state_name(state); + printf("State of pcm handle: %s\n", state_name); } -struct file_data { - FILE *file; - size_t size; -}; - -struct response_data { - char *data; - size_t size; -}; - -size_t read_callback(void *ptr, size_t size, size_t nmemb, void *userdata) { - struct file_data *data = (struct file_data *)userdata; - size_t bytes_to_read = size * nmemb; - return fread(ptr, 1, bytes_to_read, data->file); -} - -// Returns string that caller must free -char* find_transcript(char* json_response) -{ +char *find_ai_response(char *json_response) { printf("Parsing response: %s \n", json_response); - char* start = strstr(json_response, "\"transcript\":\""); - if (!start) return NULL; + char *start = strstr(json_response, "\"text\":\""); + if (!start) + return NULL; + + start += 8; // Skip past "text":" + char *end = strchr(start, '"'); + if (!end) + return NULL; - start += 14; // Skip past "transcript":" - char* end = strchr(start, '"'); // Find closing quote - // Copy substring between start and end size_t len = end - start; - char* result = malloc(len + 1); + char *result = malloc(len + 1); strncpy(result, start, len); result[len] = '\0'; return result; } -size_t write_callback(void *contents, size_t size, size_t nmemb, void *userp) { - size_t total_size = size * nmemb; - struct response_data *response = (struct response_data *)userp; - - // Reallocate buffer to fit new data - response->data = realloc(response->data, response->size + total_size + 1); - if (response->data == NULL) { - return 0; // Error - } - - // Copy new data to buffer - memcpy(&response->data[response->size], contents, total_size); - response->size += total_size; - response->data[response->size] = '\0'; // Null terminate - - printf("read an extra %d bytes of data\n", (int)total_size); - - return total_size; -} - -// Returns transcript of audio -char* transcribe(FILE* audio_file) { - CURL *curl; - CURLcode res; - struct curl_slist *headers = NULL; - - // Initialize response buffer - struct response_data response = {0}; - response.data = malloc(1); - response.size = 0; - - // Get file size - fseek(audio_file, 0, SEEK_END); - long file_size = ftell(audio_file); - fseek(audio_file, 0, SEEK_SET); - - struct file_data data; - data.file = audio_file; - data.size = file_size; - - curl = curl_easy_init(); - if (curl) { - // Set headers for Deepgram API - headers = curl_slist_append(headers, "Authorization: Token YOUR_DEEPGRAM_API_KEY"); - - curl_easy_setopt(curl, CURLOPT_URL, "https://api.deepgram.com/v1/listen?encoding=linear16&sample_rate=48000&channels=1"); - curl_easy_setopt(curl, CURLOPT_HTTPHEADER, headers); - curl_easy_setopt(curl, CURLOPT_POST, 1L); - curl_easy_setopt(curl, CURLOPT_READFUNCTION, read_callback); - curl_easy_setopt(curl, CURLOPT_READDATA, &data); - curl_easy_setopt(curl, CURLOPT_POSTFIELDSIZE, file_size); - curl_easy_setopt(curl, CURLOPT_WRITEFUNCTION, write_callback); - curl_easy_setopt(curl, CURLOPT_WRITEDATA, &response); - - res = curl_easy_perform(curl); - if (res != CURLE_OK) { - fprintf(stderr, "curl failed: %s\n", curl_easy_strerror(res)); - free(response.data); - curl_slist_free_all(headers); - curl_easy_cleanup(curl); - return NULL; - } - - printf("done performing curl\n"); - - curl_slist_free_all(headers); - curl_easy_cleanup(curl); - } - - // Parse transcript from complete response - char* transcript = find_transcript(response.data); - free(response.data); - - return transcript; -} - -char* find_ai_response(char* json_response) { - printf("Parsing response: %s \n", json_response); - char* start = strstr(json_response, "\"text\":\""); - if (!start) return NULL; - - start += 8; // Skip past "text":" - char* end = strchr(start, '"'); - if (!end) return NULL; - - size_t len = end - start; - char* result = malloc(len + 1); - strncpy(result, start, len); - result[len] = '\0'; - return result; -} - char *ask_ai(char *question) { - CURL *curl; - CURLcode res; - struct curl_slist *headers = NULL; + CURL *curl; + CURLcode res; + struct curl_slist *headers = NULL; - // Initialize response buffer - struct response_data response = {0}; - response.data = malloc(1); - response.size = 0; + // Initialize response buffer + struct curl_response_data response = {0}; + response.data = malloc(1); + response.size = 0; - // Create JSON payload - char json_payload[2048]; - snprintf(json_payload, sizeof(json_payload), - "{" - "\"model\":\"claude-3-5-sonnet-20241022\"," - "\"max_tokens\":1024," - "\"messages\":[{\"role\":\"user\",\"content\":\"%s\"}]" - "}", question); + // Create JSON payload + char json_payload[2048]; + snprintf(json_payload, sizeof(json_payload), + "{" + "\"model\":\"claude-3-haiku-20240307\"," + "\"max_tokens\":1024," + "\"messages\":[{\"role\":\"user\",\"content\":\"%s\"}]" + "}", + question); - curl = curl_easy_init(); - if (curl) { - // Set headers for Anthropic API - headers = curl_slist_append(headers, "content-Type: application/json"); - headers = curl_slist_append(headers, "anthropic-version: 2023-06-01"); + curl = curl_easy_init(); + if (curl) { + char *anthropic_api_key = getenv("ANTHROPIC_API_KEY"); + if (!anthropic_api_key) { + printf("ANTHROPIC_API_KEY not set\n"); + free(response.data); + curl_easy_cleanup(curl); + return NULL; + } + char auth_header[256]; + snprintf(auth_header, sizeof(auth_header), "x-api-key: %s", + anthropic_api_key); + headers = curl_slist_append(headers, "content-type: application/json"); + headers = curl_slist_append(headers, auth_header); + headers = curl_slist_append(headers, "anthropic-version: 2023-06-01"); - curl_easy_setopt(curl, CURLOPT_URL, "https://api.anthropic.com/v1/messages"); - curl_easy_setopt(curl, CURLOPT_HTTPHEADER, headers); - curl_easy_setopt(curl, CURLOPT_POST, 1L); - curl_easy_setopt(curl, CURLOPT_POSTFIELDS, json_payload); - curl_easy_setopt(curl, CURLOPT_WRITEFUNCTION, write_callback); - curl_easy_setopt(curl, CURLOPT_WRITEDATA, &response); + curl_easy_setopt(curl, CURLOPT_URL, + "https://api.anthropic.com/v1/messages"); + curl_easy_setopt(curl, CURLOPT_HTTPHEADER, headers); + curl_easy_setopt(curl, CURLOPT_POST, 1L); + curl_easy_setopt(curl, CURLOPT_POSTFIELDS, json_payload); + curl_easy_setopt(curl, CURLOPT_WRITEFUNCTION, curl_write_callback_response); + curl_easy_setopt(curl, CURLOPT_WRITEDATA, &response); - res = curl_easy_perform(curl); - if (res != CURLE_OK) { - fprintf(stderr, "curl failed: %s\n", curl_easy_strerror(res)); - free(response.data); - curl_slist_free_all(headers); - curl_easy_cleanup(curl); - return NULL; - } - - curl_slist_free_all(headers); - curl_easy_cleanup(curl); + res = curl_easy_perform(curl); + if (res != CURLE_OK) { + fprintf(stderr, "curl failed: %s\n", curl_easy_strerror(res)); + free(response.data); + curl_slist_free_all(headers); + curl_easy_cleanup(curl); + return NULL; } - // Parse AI response from complete response - char* answer = find_ai_response(response.data); - free(response.data); + curl_slist_free_all(headers); + curl_easy_cleanup(curl); + } - return answer; + // Parse AI response from complete response + char *answer = find_ai_response(response.data); + free(response.data); + + return answer; +} + +/* count: how many bytes in the buffer + * returns int: 0 if success, -1 if failure + * NOTE: This function assumes that the audio data is using: + * - linear16 formatting (each sample is 16 bits) + * - one channel (1 sample) + * - with 48000 HZ sampling rate + * Hence, it's 2 bytes per frame. + */ +int play_pcm_data(char *buf, size_t byte_count) { + printf("going to play audio of %d bytes of data\n", (int)byte_count); + + snd_pcm_t *pcm_handle; + // The total frames in the buffer is how many bytes divided by 2 because + // of linear16 encoding, one channel + size_t total_frames = byte_count / 2; + + int err; + err = snd_pcm_open(&pcm_handle, ALSA_SPEAKER_HARDWARE_DEVICE, + SND_PCM_STREAM_PLAYBACK, 0); + if (err < 0) { + fprintf(stderr, "unable to open playback device"); + return err; + } + print_pcm_state(pcm_handle); + + const char *name = snd_pcm_name(pcm_handle); + printf("found this device: %s\n", name); + + snd_pcm_info_t *pcm_info; + snd_pcm_info_malloc(&pcm_info); + err = snd_pcm_info(pcm_handle, pcm_info); + if (err < 0) { + snd_pcm_close(pcm_handle); + printf("unable to get info of pcm device"); + return -1; + } + name = snd_pcm_info_get_name(pcm_info); + if (err < 0) { + printf("unable to get card info"); + snd_pcm_close(pcm_handle); + snd_pcm_info_free(pcm_info); + return -1; + } + printf("got pcm name: %s \n", name); + + print_pcm_state(pcm_handle); + snd_pcm_info_free(pcm_info); + if ((err = snd_pcm_set_params(pcm_handle, SND_PCM_FORMAT_S16_LE, + SND_PCM_ACCESS_RW_INTERLEAVED, 1, 48000, 1, + 500000)) < 0) { /* 0.5sec */ + printf("Playback open error: %s\n", snd_strerror(err)); + exit(EXIT_FAILURE); + } + + // we want to read in a chunk of data and writei to the pcm + size_t frames_written = 0; + size_t frames_in_each_chunk = 1024; // how many frames to read per chunk (2048 + // bytes since each frame is 2 bytes) + while (frames_written < total_frames) { + size_t frames_to_write = + frames_in_each_chunk < (total_frames - frames_written) + ? frames_in_each_chunk + : (total_frames - frames_written); + // We move the pointer x2 because each frame is 2 bytes so the next audio + // sample is 16 bits forward, not 8 + char *start_ptr = buf + (frames_written * 2); + snd_pcm_sframes_t result = + snd_pcm_writei(pcm_handle, start_ptr, frames_to_write); + if (result < 0) { + result = snd_pcm_recover(pcm_handle, result, 0); + if (result < 0) { + printf("snd_pcm_writei failed: %s\n", snd_strerror(result)); + break; + } + } + + if (result > 0 && result < frames_to_write) { + printf("Short write \n"); + } + + frames_written += result; + } + + /* pass the remaining samples, otherwise they're dropped in close */ + err = snd_pcm_drain(pcm_handle); + if (err < 0) { + printf("snd_pcm_drain failed: %s\n", snd_strerror(err)); + } + snd_pcm_close(pcm_handle); + + return 0; } int main() { - snd_pcm_t *pcm_handle = NULL; + snd_pcm_t *pcm_handle = NULL; - int err; - err = snd_pcm_open(&pcm_handle, ALSA_MIC_HARDWARE_DEVICE, SND_PCM_STREAM_CAPTURE, 0); - if (err < 0) { - printf("unable to open pcm device"); - return -1; - } - print_pcm_state(pcm_handle); + int err; + err = snd_pcm_open(&pcm_handle, ALSA_MIC_HARDWARE_DEVICE, + SND_PCM_STREAM_CAPTURE, 0); + if (err < 0) { + printf("unable to open pcm device"); + return -1; + } + print_pcm_state(pcm_handle); - const char *name = snd_pcm_name(pcm_handle); - printf("found this device: %s\n", name); + const char *name = snd_pcm_name(pcm_handle); + printf("found this device: %s\n", name); - snd_pcm_info_t *pcm_info; - snd_pcm_info_malloc(&pcm_info); - err = snd_pcm_info(pcm_handle, pcm_info); - if (err < 0) { - snd_pcm_close(pcm_handle); - printf("unable to get info of pcm device"); - return -1; - } - name = snd_pcm_info_get_name(pcm_info); - if (err < 0) { - printf("unable to get card info"); - snd_pcm_close(pcm_handle); - snd_pcm_info_free(pcm_info); - return -1; - } - printf("got pcm name: %s \n", name); - - print_pcm_state(pcm_handle); + snd_pcm_info_t *pcm_info; + snd_pcm_info_malloc(&pcm_info); + err = snd_pcm_info(pcm_handle, pcm_info); + if (err < 0) { + snd_pcm_close(pcm_handle); + printf("unable to get info of pcm device"); + return -1; + } + name = snd_pcm_info_get_name(pcm_info); + if (err < 0) { + printf("unable to get card info"); + snd_pcm_close(pcm_handle); snd_pcm_info_free(pcm_info); + return -1; + } + printf("got pcm name: %s \n", name); - snd_pcm_hw_params_t *params; - err = snd_pcm_hw_params_malloc(¶ms); - if (err < 0) { - fprintf(stderr, "unable to mallow hw params"); - return -1; - } - err = snd_pcm_hw_params_any(pcm_handle, params); - if (err < 0) { - fprintf(stderr, "unable to find ranges for hw device"); - snd_pcm_close(pcm_handle); - snd_pcm_hw_params_free(params); - return -1; - } + print_pcm_state(pcm_handle); + snd_pcm_info_free(pcm_info); - uint period_size = 1024; - uint buffer_size = 4096; - uint channel_count = 1; - uint bytes_per_sample = 2; // FORMAT_S16_LE - uint samples_per_second = 48000; - snd_pcm_hw_params_set_access(pcm_handle, params, SND_PCM_ACCESS_RW_INTERLEAVED); - snd_pcm_hw_params_set_channels(pcm_handle, params, channel_count); - snd_pcm_hw_params_set_buffer_size(pcm_handle, params, buffer_size); - snd_pcm_hw_params_set_format(pcm_handle, params, SND_PCM_FORMAT_S16_LE); - snd_pcm_hw_params_set_rate(pcm_handle, params, samples_per_second, 0); - snd_pcm_hw_params_set_periods(pcm_handle, params, period_size, 0); - - err = snd_pcm_hw_params(pcm_handle, params); - if (err < 0) { - fprintf(stderr, "unable to prepare pcm\n"); - snd_pcm_close(pcm_handle); - snd_pcm_hw_params_free(params); - return -1; - } - - print_pcm_state(pcm_handle); - - err = snd_pcm_start(pcm_handle); - if (err < 0) { - fprintf(stderr, "unable to start pcm"); - snd_pcm_close(pcm_handle); - snd_pcm_hw_params_free(params); - return -1; - } - - print_pcm_state(pcm_handle); - - const uint seconds_to_capture = 5; - const uint periods_to_read = (samples_per_second * seconds_to_capture) / period_size; - char *buf = malloc(period_size * channel_count * bytes_per_sample); - uint periods_read = 0; - /* Read one period at a time from the hardware buffer - * We set the hardware buffer to read 4096 frames. - * The period size is 1024 frames. - * Each period is 2048 BYTES because each frame is 2 bytes. - * Each frame is 2 bytes because of 1 channel, and 16 bit format - */ - FILE *file = fopen("output.pcm", "wb"); - if (file == NULL) { - perror("Error opening file"); - return 1; - } - printf("======READING FRAMES for 5 seconds========\n"); - while (snd_pcm_readi(pcm_handle, (void *)buf, period_size) > 0 && periods_read < periods_to_read) { - // Because each frame is 2 bytes (format/bit rate), cast to int16 - // int16_t *samples = (int16_t *)buf; - // for (int i = 0; i < period_size; i++) { - // printf("%d ", samples[i]); - // } - - fwrite(buf, sizeof(int16_t), period_size, file); - periods_read++; - } - printf("done capturing audio\n"); - - fclose(file); - - // Reopen file for reading and transcribe - file = fopen("output.pcm", "rb"); - if (file != NULL) { - char* transcript = transcribe(file); - if (transcript) { - printf("Transcript: %s\n", transcript); - - char *ai_answer = ask_ai(transcript); - if (ai_answer) { - printf("AI answer: %s\n", ai_answer); - free(ai_answer); - } - free(transcript); - } else { - fprintf(stderr, "unable to get transcript"); - } - fclose(file); - } else { - fprintf(stderr, "unable to open file for reading\n"); - } - - free((void *)buf); + snd_pcm_hw_params_t *params; + err = snd_pcm_hw_params_malloc(¶ms); + if (err < 0) { + fprintf(stderr, "unable to mallow hw params"); + return -1; + } + err = snd_pcm_hw_params_any(pcm_handle, params); + if (err < 0) { + fprintf(stderr, "unable to find ranges for hw device"); snd_pcm_close(pcm_handle); snd_pcm_hw_params_free(params); + return -1; + } - return 0; + uint period_size = 1024; + uint buffer_size = 4096; + uint channel_count = 1; + uint bytes_per_sample = 2; // FORMAT_S16_LE + uint samples_per_second = 48000; + snd_pcm_hw_params_set_access(pcm_handle, params, + SND_PCM_ACCESS_RW_INTERLEAVED); + snd_pcm_hw_params_set_channels(pcm_handle, params, channel_count); + snd_pcm_hw_params_set_buffer_size(pcm_handle, params, buffer_size); + snd_pcm_hw_params_set_format(pcm_handle, params, SND_PCM_FORMAT_S16_LE); + snd_pcm_hw_params_set_rate(pcm_handle, params, samples_per_second, 0); + snd_pcm_hw_params_set_periods(pcm_handle, params, period_size, 0); + + err = snd_pcm_hw_params(pcm_handle, params); + if (err < 0) { + fprintf(stderr, "unable to prepare pcm\n"); + snd_pcm_close(pcm_handle); + snd_pcm_hw_params_free(params); + return -1; + } + + print_pcm_state(pcm_handle); + + err = snd_pcm_start(pcm_handle); + if (err < 0) { + fprintf(stderr, "unable to start pcm"); + snd_pcm_close(pcm_handle); + snd_pcm_hw_params_free(params); + return -1; + } + + print_pcm_state(pcm_handle); + + const uint seconds_to_capture = 5; + const uint periods_to_read = + (samples_per_second * seconds_to_capture) / period_size; + char *buf = malloc(period_size * channel_count * bytes_per_sample); + uint periods_read = 0; + /* Read one period at a time from the hardware buffer + * We set the hardware buffer to read 4096 frames. + * The period size is 1024 frames. + * Each period is 2048 BYTES because each frame is 2 bytes. + * Each frame is 2 bytes because of 1 channel, and 16 bit format + */ + FILE *file = fopen("output.pcm", "wb"); + if (file == NULL) { + perror("Error opening file"); + return 1; + } + printf("======READING FRAMES for 5 seconds========\n"); + while (snd_pcm_readi(pcm_handle, (void *)buf, period_size) > 0 && + periods_read < periods_to_read) { + // Because each frame is 2 bytes (format/bit rate), cast to int16 + // int16_t *samples = (int16_t *)buf; + // for (int i = 0; i < period_size; i++) { + // printf("%d ", samples[i]); + // } + + fwrite(buf, sizeof(int16_t), period_size, file); + periods_read++; + } + printf("done capturing audio\n"); + + fclose(file); + + // Reopen file for reading and transcribe + file = fopen("output.pcm", "rb"); + if (file != NULL) { + char *transcript = dg_transcribe(file); + if (transcript) { + printf("Transcript: %s\n", transcript); + + char *ai_answer = ask_ai(transcript); + if (ai_answer) { + printf("AI answer: %s\n", ai_answer); + + struct curl_response_data *tts_response = dg_text_to_speech(ai_answer); + int result = play_pcm_data(tts_response->data, tts_response->size); + if (result != 0) { + fprintf(stderr, "failure in playback"); + } + + free(tts_response->data); + free(tts_response); + + // we have this response data, and buffer within it. we can loop through + // it and write the pcm data within it to alsa + free(ai_answer); + } + free(transcript); + } else { + fprintf(stderr, "unable to get transcript"); + } + fclose(file); + } else { + fprintf(stderr, "unable to open file for reading\n"); + } + + free(buf); + snd_pcm_close(pcm_handle); + snd_pcm_hw_params_free(params); + + return 0; } diff --git a/ai-conversation/deepgram_client.c b/ai-conversation/deepgram_client.c new file mode 100644 index 0000000..cb47243 --- /dev/null +++ b/ai-conversation/deepgram_client.c @@ -0,0 +1,168 @@ +#include "deepgram_client.h" +#include + +// Returns string that caller must free +static char *dg_find_transcript(char *json_response) { + printf("Parsing response: %s \n", json_response); + char *start = strstr(json_response, "\"transcript\":\""); + if (!start) + return NULL; + + start += 14; // Skip past "transcript":" + char *end = strchr(start, '"'); // Find closing quote + // Copy substring between start and end + size_t len = end - start; + char *result = malloc(len + 1); + strncpy(result, start, len); + result[len] = '\0'; + return result; +} + +char *dg_transcribe(FILE *audio_file) { + CURL *curl; + CURLcode res; + struct curl_slist *headers = NULL; + + // Initialize response buffer + struct curl_response_data response = {0}; + response.data = malloc(1); + response.size = 0; + + // Get file size + fseek(audio_file, 0, SEEK_END); + long file_size = ftell(audio_file); + fseek(audio_file, 0, SEEK_SET); + + struct curl_file_data data; + data.file = audio_file; + data.size = file_size; + + curl = curl_easy_init(); + if (curl) { + // Set headers for Deepgram API + char *deepgram_api_key = getenv("DEEPGRAM_API_KEY"); + if (!deepgram_api_key) { + fprintf(stderr, "DEEPGRAM_API_KEY environment variable not set\n"); + free(response.data); + curl_easy_cleanup(curl); + return NULL; + } + + char auth_header[256]; + snprintf(auth_header, sizeof(auth_header), "Authorization: Token %s", deepgram_api_key); + headers = curl_slist_append(headers, auth_header); + + curl_easy_setopt(curl, CURLOPT_URL, + "https://api.deepgram.com/v1/" + "listen?encoding=linear16&sample_rate=48000&channels=1"); + curl_easy_setopt(curl, CURLOPT_HTTPHEADER, headers); + curl_easy_setopt(curl, CURLOPT_POST, 1L); + curl_easy_setopt(curl, CURLOPT_READFUNCTION, curl_read_callback_file); + curl_easy_setopt(curl, CURLOPT_READDATA, &data); + curl_easy_setopt(curl, CURLOPT_POSTFIELDSIZE, file_size); + curl_easy_setopt(curl, CURLOPT_WRITEFUNCTION, curl_write_callback_response); + curl_easy_setopt(curl, CURLOPT_WRITEDATA, &response); + + res = curl_easy_perform(curl); + if (res != CURLE_OK) { + fprintf(stderr, "curl failed: %s\n", curl_easy_strerror(res)); + free(response.data); + curl_slist_free_all(headers); + curl_easy_cleanup(curl); + return NULL; + } + + printf("done performing curl\n"); + + curl_slist_free_all(headers); + curl_easy_cleanup(curl); + } + + // Parse transcript from complete response + char *transcript = dg_find_transcript(response.data); + free(response.data); + + return transcript; +} + +struct curl_response_data *dg_text_to_speech(char *text) { + CURL *curl; + CURLcode res; + struct curl_slist *headers = NULL; + + // Create JSON payload - need to escape the text properly + // Worst case: every char needs escaping, so allocate 2x + overhead + size_t text_len = strlen(text); + size_t payload_size = text_len * 2 + 100; + char *json_payload = malloc(payload_size); + + // Simple approach: replace problematic chars with spaces + char *escaped_text = malloc(text_len + 1); + strcpy(escaped_text, text); + for (int i = 0; escaped_text[i]; i++) { + if (escaped_text[i] == '"' || escaped_text[i] == '\n' || + escaped_text[i] == '\r' || escaped_text[i] == '\t') { + escaped_text[i] = ' '; + } + } + + snprintf(json_payload, payload_size, + "{" + "\"text\":\"%s\"" + "}", + escaped_text); + + free(escaped_text); + + // Initialize response buffer + struct curl_response_data *response = + malloc(sizeof(struct curl_response_data)); + response->data = malloc(1); + response->size = 0; + + curl = curl_easy_init(); + if (curl) { + char *deepgram_api_key = getenv("DEEPGRAM_API_KEY"); + if (!deepgram_api_key) { + fprintf(stderr, "DEEPGRAM_API_KEY environment variable not set\n"); + free(response->data); + curl_easy_cleanup(curl); + return NULL; + } + + char auth_header[256]; + snprintf(auth_header, sizeof(auth_header), "Authorization: Token %s", deepgram_api_key); + headers = curl_slist_append(headers, auth_header); + headers = curl_slist_append(headers, "Content-Type: application/json"); + + curl_easy_setopt(curl, CURLOPT_URL, + "https://api.deepgram.com/v1/" + "speak?encoding=linear16&sample_rate=48000"); + curl_easy_setopt(curl, CURLOPT_HTTPHEADER, headers); + curl_easy_setopt(curl, CURLOPT_POST, 1L); + curl_easy_setopt(curl, CURLOPT_POSTFIELDS, json_payload); + curl_easy_setopt(curl, CURLOPT_WRITEFUNCTION, curl_write_callback_response); + curl_easy_setopt(curl, CURLOPT_WRITEDATA, response); + + res = curl_easy_perform(curl); + if (res != CURLE_OK) { + fprintf(stderr, "curl failed: %s\n", curl_easy_strerror(res)); + free(response->data); + free(response); + curl_slist_free_all(headers); + curl_easy_cleanup(curl); + return NULL; + } + + curl_slist_free_all(headers); + curl_easy_cleanup(curl); + } + + printf("received audio of %d length for tts\n", (int)response->size); + + // Debug: print first 100 chars to see if it's audio or error message + printf("First 100 chars: %.100s\n", response->data); + + free(json_payload); + return response; +} diff --git a/ai-conversation/deepgram_client.h b/ai-conversation/deepgram_client.h new file mode 100644 index 0000000..c31e563 --- /dev/null +++ b/ai-conversation/deepgram_client.h @@ -0,0 +1,21 @@ +#ifndef DEEPGRAMCLIENT_H +#define DEEPGRAMCLIENT_H + +#include +#include +#include +#include "utils.h" + +/* + * @param audio_file: must already be opened + * @returns char *: transcript string result. Call must free. + */ +char *dg_transcribe(FILE *audio_file); + +/* +* @returns response data. Caller must free the struct AND the data within it. +* TODO: should return custom struct with details about encoding, sample rate, etc. +*/ +struct curl_response_data *dg_text_to_speech(char *text); + +#endif // DEEPGRAMCLIENT_H diff --git a/ai-conversation/utils.c b/ai-conversation/utils.c new file mode 100644 index 0000000..debcff8 --- /dev/null +++ b/ai-conversation/utils.c @@ -0,0 +1,30 @@ +#include "utils.h" +#include + +size_t curl_read_callback_file(void *ptr, size_t size, size_t nmemb, + void *userdata) { + struct curl_file_data *data = (struct curl_file_data *)userdata; + size_t bytes_to_read = size * nmemb; + return fread(ptr, 1, bytes_to_read, data->file); +} + +size_t curl_write_callback_response(void *contents, size_t size, size_t nmemb, + void *userp) { + size_t total_size = size * nmemb; + struct curl_response_data *response = (struct curl_response_data *)userp; + + // Reallocate buffer to fit new data + response->data = realloc(response->data, response->size + total_size + 1); + if (response->data == NULL) { + return 0; // Error + } + + // Copy new data to buffer + memcpy(&response->data[response->size], contents, total_size); + response->size += total_size; + response->data[response->size] = '\0'; // Null terminate + + printf("read an extra %d bytes of data\n", (int)total_size); + + return total_size; +} diff --git a/ai-conversation/utils.h b/ai-conversation/utils.h new file mode 100644 index 0000000..bc546c4 --- /dev/null +++ b/ai-conversation/utils.h @@ -0,0 +1,24 @@ +#ifndef UTILS_H +#define UTILS_H + +#include +#include + +struct curl_file_data { + FILE *file; + size_t size; +}; + +struct curl_response_data { + char *data; + size_t size; +}; + +// This callback assumes the userdata is `struct file_data` +size_t curl_read_callback_file(void *ptr, size_t size, size_t nmemb, void *userdata); + +// This callback assumes that we are aggregating callback data into an allocated response_data struct +size_t curl_write_callback_response(void *contents, size_t size, size_t nmemb, + void *userp); + +#endif // UTILS_H