This organizes some components like deepgram into own module, and also allows gets the full flow to work every time we run the program.
348 lines
10 KiB
C
348 lines
10 KiB
C
#include "deepgram_client.h"
|
|
#include "utils.h"
|
|
#include <alsa/asoundlib.h>
|
|
#include <alsa/control.h>
|
|
#include <alsa/pcm.h>
|
|
#include <curl/curl.h>
|
|
#include <stdio.h>
|
|
#include <stdlib.h>
|
|
|
|
// NOTE: find this via `arecord -l`
|
|
#define ALSA_MIC_HARDWARE_DEVICE "hw:3"
|
|
#define ALSA_SPEAKER_HARDWARE_DEVICE "default"
|
|
|
|
void print_pcm_state(snd_pcm_t *pcm_handle) {
|
|
snd_pcm_state_t state = snd_pcm_state(pcm_handle);
|
|
const char *state_name = snd_pcm_state_name(state);
|
|
printf("State of pcm handle: %s\n", state_name);
|
|
}
|
|
|
|
char *find_ai_response(char *json_response) {
|
|
printf("Parsing response: %s \n", json_response);
|
|
char *start = strstr(json_response, "\"text\":\"");
|
|
if (!start)
|
|
return NULL;
|
|
|
|
start += 8; // Skip past "text":"
|
|
char *end = strchr(start, '"');
|
|
if (!end)
|
|
return NULL;
|
|
|
|
size_t len = end - start;
|
|
char *result = malloc(len + 1);
|
|
strncpy(result, start, len);
|
|
result[len] = '\0';
|
|
return result;
|
|
}
|
|
|
|
char *ask_ai(char *question) {
|
|
CURL *curl;
|
|
CURLcode res;
|
|
struct curl_slist *headers = NULL;
|
|
|
|
// Initialize response buffer
|
|
struct curl_response_data response = {0};
|
|
response.data = malloc(1);
|
|
response.size = 0;
|
|
|
|
// Create JSON payload
|
|
char json_payload[2048];
|
|
snprintf(json_payload, sizeof(json_payload),
|
|
"{"
|
|
"\"model\":\"claude-3-haiku-20240307\","
|
|
"\"max_tokens\":1024,"
|
|
"\"messages\":[{\"role\":\"user\",\"content\":\"%s\"}]"
|
|
"}",
|
|
question);
|
|
|
|
curl = curl_easy_init();
|
|
if (curl) {
|
|
char *anthropic_api_key = getenv("ANTHROPIC_API_KEY");
|
|
if (!anthropic_api_key) {
|
|
printf("ANTHROPIC_API_KEY not set\n");
|
|
free(response.data);
|
|
curl_easy_cleanup(curl);
|
|
return NULL;
|
|
}
|
|
char auth_header[256];
|
|
snprintf(auth_header, sizeof(auth_header), "x-api-key: %s",
|
|
anthropic_api_key);
|
|
headers = curl_slist_append(headers, "content-type: application/json");
|
|
headers = curl_slist_append(headers, auth_header);
|
|
headers = curl_slist_append(headers, "anthropic-version: 2023-06-01");
|
|
|
|
curl_easy_setopt(curl, CURLOPT_URL,
|
|
"https://api.anthropic.com/v1/messages");
|
|
curl_easy_setopt(curl, CURLOPT_HTTPHEADER, headers);
|
|
curl_easy_setopt(curl, CURLOPT_POST, 1L);
|
|
curl_easy_setopt(curl, CURLOPT_POSTFIELDS, json_payload);
|
|
curl_easy_setopt(curl, CURLOPT_WRITEFUNCTION, curl_write_callback_response);
|
|
curl_easy_setopt(curl, CURLOPT_WRITEDATA, &response);
|
|
|
|
res = curl_easy_perform(curl);
|
|
if (res != CURLE_OK) {
|
|
fprintf(stderr, "curl failed: %s\n", curl_easy_strerror(res));
|
|
free(response.data);
|
|
curl_slist_free_all(headers);
|
|
curl_easy_cleanup(curl);
|
|
return NULL;
|
|
}
|
|
|
|
curl_slist_free_all(headers);
|
|
curl_easy_cleanup(curl);
|
|
}
|
|
|
|
// Parse AI response from complete response
|
|
char *answer = find_ai_response(response.data);
|
|
free(response.data);
|
|
|
|
return answer;
|
|
}
|
|
|
|
/* count: how many bytes in the buffer
|
|
* returns int: 0 if success, -1 if failure
|
|
* NOTE: This function assumes that the audio data is using:
|
|
* - linear16 formatting (each sample is 16 bits)
|
|
* - one channel (1 sample)
|
|
* - with 48000 HZ sampling rate
|
|
* Hence, it's 2 bytes per frame.
|
|
*/
|
|
int play_pcm_data(char *buf, size_t byte_count) {
|
|
printf("going to play audio of %d bytes of data\n", (int)byte_count);
|
|
|
|
snd_pcm_t *pcm_handle;
|
|
// The total frames in the buffer is how many bytes divided by 2 because
|
|
// of linear16 encoding, one channel
|
|
size_t total_frames = byte_count / 2;
|
|
|
|
int err;
|
|
err = snd_pcm_open(&pcm_handle, ALSA_SPEAKER_HARDWARE_DEVICE,
|
|
SND_PCM_STREAM_PLAYBACK, 0);
|
|
if (err < 0) {
|
|
fprintf(stderr, "unable to open playback device");
|
|
return err;
|
|
}
|
|
print_pcm_state(pcm_handle);
|
|
|
|
const char *name = snd_pcm_name(pcm_handle);
|
|
printf("found this device: %s\n", name);
|
|
|
|
snd_pcm_info_t *pcm_info;
|
|
snd_pcm_info_malloc(&pcm_info);
|
|
err = snd_pcm_info(pcm_handle, pcm_info);
|
|
if (err < 0) {
|
|
snd_pcm_close(pcm_handle);
|
|
printf("unable to get info of pcm device");
|
|
return -1;
|
|
}
|
|
name = snd_pcm_info_get_name(pcm_info);
|
|
if (err < 0) {
|
|
printf("unable to get card info");
|
|
snd_pcm_close(pcm_handle);
|
|
snd_pcm_info_free(pcm_info);
|
|
return -1;
|
|
}
|
|
printf("got pcm name: %s \n", name);
|
|
|
|
print_pcm_state(pcm_handle);
|
|
snd_pcm_info_free(pcm_info);
|
|
if ((err = snd_pcm_set_params(pcm_handle, SND_PCM_FORMAT_S16_LE,
|
|
SND_PCM_ACCESS_RW_INTERLEAVED, 1, 48000, 1,
|
|
500000)) < 0) { /* 0.5sec */
|
|
printf("Playback open error: %s\n", snd_strerror(err));
|
|
exit(EXIT_FAILURE);
|
|
}
|
|
|
|
// we want to read in a chunk of data and writei to the pcm
|
|
size_t frames_written = 0;
|
|
size_t frames_in_each_chunk = 1024; // how many frames to read per chunk (2048
|
|
// bytes since each frame is 2 bytes)
|
|
while (frames_written < total_frames) {
|
|
size_t frames_to_write =
|
|
frames_in_each_chunk < (total_frames - frames_written)
|
|
? frames_in_each_chunk
|
|
: (total_frames - frames_written);
|
|
// We move the pointer x2 because each frame is 2 bytes so the next audio
|
|
// sample is 16 bits forward, not 8
|
|
char *start_ptr = buf + (frames_written * 2);
|
|
snd_pcm_sframes_t result =
|
|
snd_pcm_writei(pcm_handle, start_ptr, frames_to_write);
|
|
if (result < 0) {
|
|
result = snd_pcm_recover(pcm_handle, result, 0);
|
|
if (result < 0) {
|
|
printf("snd_pcm_writei failed: %s\n", snd_strerror(result));
|
|
break;
|
|
}
|
|
}
|
|
|
|
if (result > 0 && result < frames_to_write) {
|
|
printf("Short write \n");
|
|
}
|
|
|
|
frames_written += result;
|
|
}
|
|
|
|
/* pass the remaining samples, otherwise they're dropped in close */
|
|
err = snd_pcm_drain(pcm_handle);
|
|
if (err < 0) {
|
|
printf("snd_pcm_drain failed: %s\n", snd_strerror(err));
|
|
}
|
|
snd_pcm_close(pcm_handle);
|
|
|
|
return 0;
|
|
}
|
|
|
|
int main() {
|
|
snd_pcm_t *pcm_handle = NULL;
|
|
|
|
int err;
|
|
err = snd_pcm_open(&pcm_handle, ALSA_MIC_HARDWARE_DEVICE,
|
|
SND_PCM_STREAM_CAPTURE, 0);
|
|
if (err < 0) {
|
|
printf("unable to open pcm device");
|
|
return -1;
|
|
}
|
|
print_pcm_state(pcm_handle);
|
|
|
|
const char *name = snd_pcm_name(pcm_handle);
|
|
printf("found this device: %s\n", name);
|
|
|
|
snd_pcm_info_t *pcm_info;
|
|
snd_pcm_info_malloc(&pcm_info);
|
|
err = snd_pcm_info(pcm_handle, pcm_info);
|
|
if (err < 0) {
|
|
snd_pcm_close(pcm_handle);
|
|
printf("unable to get info of pcm device");
|
|
return -1;
|
|
}
|
|
name = snd_pcm_info_get_name(pcm_info);
|
|
if (err < 0) {
|
|
printf("unable to get card info");
|
|
snd_pcm_close(pcm_handle);
|
|
snd_pcm_info_free(pcm_info);
|
|
return -1;
|
|
}
|
|
printf("got pcm name: %s \n", name);
|
|
|
|
print_pcm_state(pcm_handle);
|
|
snd_pcm_info_free(pcm_info);
|
|
|
|
snd_pcm_hw_params_t *params;
|
|
err = snd_pcm_hw_params_malloc(¶ms);
|
|
if (err < 0) {
|
|
fprintf(stderr, "unable to mallow hw params");
|
|
return -1;
|
|
}
|
|
err = snd_pcm_hw_params_any(pcm_handle, params);
|
|
if (err < 0) {
|
|
fprintf(stderr, "unable to find ranges for hw device");
|
|
snd_pcm_close(pcm_handle);
|
|
snd_pcm_hw_params_free(params);
|
|
return -1;
|
|
}
|
|
|
|
uint period_size = 1024;
|
|
uint buffer_size = 4096;
|
|
uint channel_count = 1;
|
|
uint bytes_per_sample = 2; // FORMAT_S16_LE
|
|
uint samples_per_second = 48000;
|
|
snd_pcm_hw_params_set_access(pcm_handle, params,
|
|
SND_PCM_ACCESS_RW_INTERLEAVED);
|
|
snd_pcm_hw_params_set_channels(pcm_handle, params, channel_count);
|
|
snd_pcm_hw_params_set_buffer_size(pcm_handle, params, buffer_size);
|
|
snd_pcm_hw_params_set_format(pcm_handle, params, SND_PCM_FORMAT_S16_LE);
|
|
snd_pcm_hw_params_set_rate(pcm_handle, params, samples_per_second, 0);
|
|
snd_pcm_hw_params_set_periods(pcm_handle, params, period_size, 0);
|
|
|
|
err = snd_pcm_hw_params(pcm_handle, params);
|
|
if (err < 0) {
|
|
fprintf(stderr, "unable to prepare pcm\n");
|
|
snd_pcm_close(pcm_handle);
|
|
snd_pcm_hw_params_free(params);
|
|
return -1;
|
|
}
|
|
|
|
print_pcm_state(pcm_handle);
|
|
|
|
err = snd_pcm_start(pcm_handle);
|
|
if (err < 0) {
|
|
fprintf(stderr, "unable to start pcm");
|
|
snd_pcm_close(pcm_handle);
|
|
snd_pcm_hw_params_free(params);
|
|
return -1;
|
|
}
|
|
|
|
print_pcm_state(pcm_handle);
|
|
|
|
const uint seconds_to_capture = 5;
|
|
const uint periods_to_read =
|
|
(samples_per_second * seconds_to_capture) / period_size;
|
|
char *buf = malloc(period_size * channel_count * bytes_per_sample);
|
|
uint periods_read = 0;
|
|
/* Read one period at a time from the hardware buffer
|
|
* We set the hardware buffer to read 4096 frames.
|
|
* The period size is 1024 frames.
|
|
* Each period is 2048 BYTES because each frame is 2 bytes.
|
|
* Each frame is 2 bytes because of 1 channel, and 16 bit format
|
|
*/
|
|
FILE *file = fopen("output.pcm", "wb");
|
|
if (file == NULL) {
|
|
perror("Error opening file");
|
|
return 1;
|
|
}
|
|
printf("======READING FRAMES for 5 seconds========\n");
|
|
while (snd_pcm_readi(pcm_handle, (void *)buf, period_size) > 0 &&
|
|
periods_read < periods_to_read) {
|
|
// Because each frame is 2 bytes (format/bit rate), cast to int16
|
|
// int16_t *samples = (int16_t *)buf;
|
|
// for (int i = 0; i < period_size; i++) {
|
|
// printf("%d ", samples[i]);
|
|
// }
|
|
|
|
fwrite(buf, sizeof(int16_t), period_size, file);
|
|
periods_read++;
|
|
}
|
|
printf("done capturing audio\n");
|
|
|
|
fclose(file);
|
|
|
|
// Reopen file for reading and transcribe
|
|
file = fopen("output.pcm", "rb");
|
|
if (file != NULL) {
|
|
char *transcript = dg_transcribe(file);
|
|
if (transcript) {
|
|
printf("Transcript: %s\n", transcript);
|
|
|
|
char *ai_answer = ask_ai(transcript);
|
|
if (ai_answer) {
|
|
printf("AI answer: %s\n", ai_answer);
|
|
|
|
struct curl_response_data *tts_response = dg_text_to_speech(ai_answer);
|
|
int result = play_pcm_data(tts_response->data, tts_response->size);
|
|
if (result != 0) {
|
|
fprintf(stderr, "failure in playback");
|
|
}
|
|
|
|
free(tts_response->data);
|
|
free(tts_response);
|
|
|
|
// we have this response data, and buffer within it. we can loop through
|
|
// it and write the pcm data within it to alsa
|
|
free(ai_answer);
|
|
}
|
|
free(transcript);
|
|
} else {
|
|
fprintf(stderr, "unable to get transcript");
|
|
}
|
|
fclose(file);
|
|
} else {
|
|
fprintf(stderr, "unable to open file for reading\n");
|
|
}
|
|
|
|
free(buf);
|
|
snd_pcm_close(pcm_handle);
|
|
snd_pcm_hw_params_free(params);
|
|
|
|
return 0;
|
|
}
|