Files
research_and_development/ai-conversation/audio_capture.c
T
talksik c93267c9ee create full flow with stt, anthropic, tts, and playback
This organizes some components like deepgram into own module, and also
allows gets the full flow to work every time we run the program.
2025-07-27 14:18:12 -07:00

348 lines
10 KiB
C

#include "deepgram_client.h"
#include "utils.h"
#include <alsa/asoundlib.h>
#include <alsa/control.h>
#include <alsa/pcm.h>
#include <curl/curl.h>
#include <stdio.h>
#include <stdlib.h>
// NOTE: find this via `arecord -l`
#define ALSA_MIC_HARDWARE_DEVICE "hw:3"
#define ALSA_SPEAKER_HARDWARE_DEVICE "default"
void print_pcm_state(snd_pcm_t *pcm_handle) {
snd_pcm_state_t state = snd_pcm_state(pcm_handle);
const char *state_name = snd_pcm_state_name(state);
printf("State of pcm handle: %s\n", state_name);
}
char *find_ai_response(char *json_response) {
printf("Parsing response: %s \n", json_response);
char *start = strstr(json_response, "\"text\":\"");
if (!start)
return NULL;
start += 8; // Skip past "text":"
char *end = strchr(start, '"');
if (!end)
return NULL;
size_t len = end - start;
char *result = malloc(len + 1);
strncpy(result, start, len);
result[len] = '\0';
return result;
}
char *ask_ai(char *question) {
CURL *curl;
CURLcode res;
struct curl_slist *headers = NULL;
// Initialize response buffer
struct curl_response_data response = {0};
response.data = malloc(1);
response.size = 0;
// Create JSON payload
char json_payload[2048];
snprintf(json_payload, sizeof(json_payload),
"{"
"\"model\":\"claude-3-haiku-20240307\","
"\"max_tokens\":1024,"
"\"messages\":[{\"role\":\"user\",\"content\":\"%s\"}]"
"}",
question);
curl = curl_easy_init();
if (curl) {
char *anthropic_api_key = getenv("ANTHROPIC_API_KEY");
if (!anthropic_api_key) {
printf("ANTHROPIC_API_KEY not set\n");
free(response.data);
curl_easy_cleanup(curl);
return NULL;
}
char auth_header[256];
snprintf(auth_header, sizeof(auth_header), "x-api-key: %s",
anthropic_api_key);
headers = curl_slist_append(headers, "content-type: application/json");
headers = curl_slist_append(headers, auth_header);
headers = curl_slist_append(headers, "anthropic-version: 2023-06-01");
curl_easy_setopt(curl, CURLOPT_URL,
"https://api.anthropic.com/v1/messages");
curl_easy_setopt(curl, CURLOPT_HTTPHEADER, headers);
curl_easy_setopt(curl, CURLOPT_POST, 1L);
curl_easy_setopt(curl, CURLOPT_POSTFIELDS, json_payload);
curl_easy_setopt(curl, CURLOPT_WRITEFUNCTION, curl_write_callback_response);
curl_easy_setopt(curl, CURLOPT_WRITEDATA, &response);
res = curl_easy_perform(curl);
if (res != CURLE_OK) {
fprintf(stderr, "curl failed: %s\n", curl_easy_strerror(res));
free(response.data);
curl_slist_free_all(headers);
curl_easy_cleanup(curl);
return NULL;
}
curl_slist_free_all(headers);
curl_easy_cleanup(curl);
}
// Parse AI response from complete response
char *answer = find_ai_response(response.data);
free(response.data);
return answer;
}
/* count: how many bytes in the buffer
* returns int: 0 if success, -1 if failure
* NOTE: This function assumes that the audio data is using:
* - linear16 formatting (each sample is 16 bits)
* - one channel (1 sample)
* - with 48000 HZ sampling rate
* Hence, it's 2 bytes per frame.
*/
int play_pcm_data(char *buf, size_t byte_count) {
printf("going to play audio of %d bytes of data\n", (int)byte_count);
snd_pcm_t *pcm_handle;
// The total frames in the buffer is how many bytes divided by 2 because
// of linear16 encoding, one channel
size_t total_frames = byte_count / 2;
int err;
err = snd_pcm_open(&pcm_handle, ALSA_SPEAKER_HARDWARE_DEVICE,
SND_PCM_STREAM_PLAYBACK, 0);
if (err < 0) {
fprintf(stderr, "unable to open playback device");
return err;
}
print_pcm_state(pcm_handle);
const char *name = snd_pcm_name(pcm_handle);
printf("found this device: %s\n", name);
snd_pcm_info_t *pcm_info;
snd_pcm_info_malloc(&pcm_info);
err = snd_pcm_info(pcm_handle, pcm_info);
if (err < 0) {
snd_pcm_close(pcm_handle);
printf("unable to get info of pcm device");
return -1;
}
name = snd_pcm_info_get_name(pcm_info);
if (err < 0) {
printf("unable to get card info");
snd_pcm_close(pcm_handle);
snd_pcm_info_free(pcm_info);
return -1;
}
printf("got pcm name: %s \n", name);
print_pcm_state(pcm_handle);
snd_pcm_info_free(pcm_info);
if ((err = snd_pcm_set_params(pcm_handle, SND_PCM_FORMAT_S16_LE,
SND_PCM_ACCESS_RW_INTERLEAVED, 1, 48000, 1,
500000)) < 0) { /* 0.5sec */
printf("Playback open error: %s\n", snd_strerror(err));
exit(EXIT_FAILURE);
}
// we want to read in a chunk of data and writei to the pcm
size_t frames_written = 0;
size_t frames_in_each_chunk = 1024; // how many frames to read per chunk (2048
// bytes since each frame is 2 bytes)
while (frames_written < total_frames) {
size_t frames_to_write =
frames_in_each_chunk < (total_frames - frames_written)
? frames_in_each_chunk
: (total_frames - frames_written);
// We move the pointer x2 because each frame is 2 bytes so the next audio
// sample is 16 bits forward, not 8
char *start_ptr = buf + (frames_written * 2);
snd_pcm_sframes_t result =
snd_pcm_writei(pcm_handle, start_ptr, frames_to_write);
if (result < 0) {
result = snd_pcm_recover(pcm_handle, result, 0);
if (result < 0) {
printf("snd_pcm_writei failed: %s\n", snd_strerror(result));
break;
}
}
if (result > 0 && result < frames_to_write) {
printf("Short write \n");
}
frames_written += result;
}
/* pass the remaining samples, otherwise they're dropped in close */
err = snd_pcm_drain(pcm_handle);
if (err < 0) {
printf("snd_pcm_drain failed: %s\n", snd_strerror(err));
}
snd_pcm_close(pcm_handle);
return 0;
}
int main() {
snd_pcm_t *pcm_handle = NULL;
int err;
err = snd_pcm_open(&pcm_handle, ALSA_MIC_HARDWARE_DEVICE,
SND_PCM_STREAM_CAPTURE, 0);
if (err < 0) {
printf("unable to open pcm device");
return -1;
}
print_pcm_state(pcm_handle);
const char *name = snd_pcm_name(pcm_handle);
printf("found this device: %s\n", name);
snd_pcm_info_t *pcm_info;
snd_pcm_info_malloc(&pcm_info);
err = snd_pcm_info(pcm_handle, pcm_info);
if (err < 0) {
snd_pcm_close(pcm_handle);
printf("unable to get info of pcm device");
return -1;
}
name = snd_pcm_info_get_name(pcm_info);
if (err < 0) {
printf("unable to get card info");
snd_pcm_close(pcm_handle);
snd_pcm_info_free(pcm_info);
return -1;
}
printf("got pcm name: %s \n", name);
print_pcm_state(pcm_handle);
snd_pcm_info_free(pcm_info);
snd_pcm_hw_params_t *params;
err = snd_pcm_hw_params_malloc(&params);
if (err < 0) {
fprintf(stderr, "unable to mallow hw params");
return -1;
}
err = snd_pcm_hw_params_any(pcm_handle, params);
if (err < 0) {
fprintf(stderr, "unable to find ranges for hw device");
snd_pcm_close(pcm_handle);
snd_pcm_hw_params_free(params);
return -1;
}
uint period_size = 1024;
uint buffer_size = 4096;
uint channel_count = 1;
uint bytes_per_sample = 2; // FORMAT_S16_LE
uint samples_per_second = 48000;
snd_pcm_hw_params_set_access(pcm_handle, params,
SND_PCM_ACCESS_RW_INTERLEAVED);
snd_pcm_hw_params_set_channels(pcm_handle, params, channel_count);
snd_pcm_hw_params_set_buffer_size(pcm_handle, params, buffer_size);
snd_pcm_hw_params_set_format(pcm_handle, params, SND_PCM_FORMAT_S16_LE);
snd_pcm_hw_params_set_rate(pcm_handle, params, samples_per_second, 0);
snd_pcm_hw_params_set_periods(pcm_handle, params, period_size, 0);
err = snd_pcm_hw_params(pcm_handle, params);
if (err < 0) {
fprintf(stderr, "unable to prepare pcm\n");
snd_pcm_close(pcm_handle);
snd_pcm_hw_params_free(params);
return -1;
}
print_pcm_state(pcm_handle);
err = snd_pcm_start(pcm_handle);
if (err < 0) {
fprintf(stderr, "unable to start pcm");
snd_pcm_close(pcm_handle);
snd_pcm_hw_params_free(params);
return -1;
}
print_pcm_state(pcm_handle);
const uint seconds_to_capture = 5;
const uint periods_to_read =
(samples_per_second * seconds_to_capture) / period_size;
char *buf = malloc(period_size * channel_count * bytes_per_sample);
uint periods_read = 0;
/* Read one period at a time from the hardware buffer
* We set the hardware buffer to read 4096 frames.
* The period size is 1024 frames.
* Each period is 2048 BYTES because each frame is 2 bytes.
* Each frame is 2 bytes because of 1 channel, and 16 bit format
*/
FILE *file = fopen("output.pcm", "wb");
if (file == NULL) {
perror("Error opening file");
return 1;
}
printf("======READING FRAMES for 5 seconds========\n");
while (snd_pcm_readi(pcm_handle, (void *)buf, period_size) > 0 &&
periods_read < periods_to_read) {
// Because each frame is 2 bytes (format/bit rate), cast to int16
// int16_t *samples = (int16_t *)buf;
// for (int i = 0; i < period_size; i++) {
// printf("%d ", samples[i]);
// }
fwrite(buf, sizeof(int16_t), period_size, file);
periods_read++;
}
printf("done capturing audio\n");
fclose(file);
// Reopen file for reading and transcribe
file = fopen("output.pcm", "rb");
if (file != NULL) {
char *transcript = dg_transcribe(file);
if (transcript) {
printf("Transcript: %s\n", transcript);
char *ai_answer = ask_ai(transcript);
if (ai_answer) {
printf("AI answer: %s\n", ai_answer);
struct curl_response_data *tts_response = dg_text_to_speech(ai_answer);
int result = play_pcm_data(tts_response->data, tts_response->size);
if (result != 0) {
fprintf(stderr, "failure in playback");
}
free(tts_response->data);
free(tts_response);
// we have this response data, and buffer within it. we can loop through
// it and write the pcm data within it to alsa
free(ai_answer);
}
free(transcript);
} else {
fprintf(stderr, "unable to get transcript");
}
fclose(file);
} else {
fprintf(stderr, "unable to open file for reading\n");
}
free(buf);
snd_pcm_close(pcm_handle);
snd_pcm_hw_params_free(params);
return 0;
}