From 47cfccb2ecf12ed207bfaf1364f3c174e21a2556 Mon Sep 17 00:00:00 2001
From: pramodthe
Date: Mon, 13 Jul 2026 02:23:18 -0700
Subject: [PATCH 1/6] feat(speech): add Deepgram as an alternate STT/TTS
provider
Introduce a thin provider facade (speech_cfg / speech_stt / speech_tts) over the
existing Soniox clients and add Deepgram Listen v1 (STT) + Speak v1 (TTS)
WebSocket backends (linear16 @ 16 kHz, same ES8311 mic/speaker path). The
provider is chosen at runtime from the AMOLED-setup captive portal; Soniox
remains the default and is unchanged for existing devices (legacy soniox_key and
fresh flash both keep Soniox).
- speech_cfg: resolve provider + key from NVS with a cached provider value
(invalidated on portal save) plus legacy soniox_key migration.
- speech_stt / speech_tts: facades that tear down the previous backend's mic /
drain task before opening the other provider, so a no-reboot portal switch
never double-owns the ES8311.
- portal: provider dropdown + per-provider voice lists; requires a key when
switching provider and resets tts_voice to the new provider's default.
- soniox_client / soniox_tts_client: add *_deinit() for clean provider switching
and read the key via speech_cfg; otherwise unchanged.
Docs: docs/speech-providers.md. Builds clean on ESP-IDF 5.5.x / esp32s3.
Co-Authored-By: Claude Opus 4.8 (1M context)
---
.gitignore | 2 +
README.md | 9 +-
docs/speech-providers.md | 47 +++
.../components/app_cfg/include/app_cfg.h | 6 +-
.../components/deepgram_stt/CMakeLists.txt | 5 +
.../components/deepgram_stt/deepgram_stt.c | 395 ++++++++++++++++++
.../deepgram_stt/include/deepgram_stt.h | 40 ++
.../components/deepgram_tts/CMakeLists.txt | 4 +
.../components/deepgram_tts/deepgram_tts.c | 333 +++++++++++++++
.../deepgram_tts/include/deepgram_tts.h | 23 +
esp32-agui/components/net_prov/CMakeLists.txt | 2 +-
esp32-agui/components/net_prov/portal.c | 167 ++++++--
.../soniox_client/include/soniox_client.h | 4 +
.../components/soniox_client/soniox_client.c | 14 +
.../soniox_tts_client/CMakeLists.txt | 2 +-
.../include/soniox_tts_client.h | 6 +
.../soniox_tts_client/soniox_tts_client.c | 51 ++-
.../components/speech_cfg/CMakeLists.txt | 4 +
.../speech_cfg/include/speech_cfg.h | 36 ++
esp32-agui/components/speech_cfg/speech_cfg.c | 53 +++
.../components/speech_stt/CMakeLists.txt | 4 +
.../speech_stt/include/speech_stt.h | 38 ++
esp32-agui/components/speech_stt/speech_stt.c | 118 ++++++
.../components/speech_tts/CMakeLists.txt | 4 +
.../speech_tts/include/speech_tts.h | 27 ++
esp32-agui/components/speech_tts/speech_tts.c | 82 ++++
esp32-agui/main/CMakeLists.txt | 5 +-
esp32-agui/main/esp32_agui_main.c | 43 +-
28 files changed, 1440 insertions(+), 84 deletions(-)
create mode 100644 docs/speech-providers.md
create mode 100644 esp32-agui/components/deepgram_stt/CMakeLists.txt
create mode 100644 esp32-agui/components/deepgram_stt/deepgram_stt.c
create mode 100644 esp32-agui/components/deepgram_stt/include/deepgram_stt.h
create mode 100644 esp32-agui/components/deepgram_tts/CMakeLists.txt
create mode 100644 esp32-agui/components/deepgram_tts/deepgram_tts.c
create mode 100644 esp32-agui/components/deepgram_tts/include/deepgram_tts.h
create mode 100644 esp32-agui/components/speech_cfg/CMakeLists.txt
create mode 100644 esp32-agui/components/speech_cfg/include/speech_cfg.h
create mode 100644 esp32-agui/components/speech_cfg/speech_cfg.c
create mode 100644 esp32-agui/components/speech_stt/CMakeLists.txt
create mode 100644 esp32-agui/components/speech_stt/include/speech_stt.h
create mode 100644 esp32-agui/components/speech_stt/speech_stt.c
create mode 100644 esp32-agui/components/speech_tts/CMakeLists.txt
create mode 100644 esp32-agui/components/speech_tts/include/speech_tts.h
create mode 100644 esp32-agui/components/speech_tts/speech_tts.c
diff --git a/.gitignore b/.gitignore
index 7f7dd65..5b8cf14 100644
--- a/.gitignore
+++ b/.gitignore
@@ -66,6 +66,8 @@ dependencies.lock
.devcontainer/
.vscode/
.clangd
+# clangd / editor indexes (e.g. esp32-agui/.cache/clangd/)
+.cache/
# --- Flash images — add later as a convenience ---
*.bin
diff --git a/README.md b/README.md
index 806376a..3c03961 100644
--- a/README.md
+++ b/README.md
@@ -7,6 +7,8 @@ The device captures mic audio and streams it to **[Soniox](https://soniox.com)**
speech-to-text, is itself the **AG-UI client** (it POSTs `RunAgentInput` and consumes the SSE
event stream directly on-device), renders the conversation plus live agent activity on the
1.8″ AMOLED via **LVGL 8.4**, and **speaks the reply back** with streaming Soniox text-to-speech.
+STT/TTS are also pluggable: **[Deepgram](https://deepgram.com)** (Listen / Speak v1) is available
+as an alternate provider from the captive portal (see [docs/speech-providers.md](docs/speech-providers.md)).
Because the AG-UI client lives on the device, agent events (`TOOL_CALL_*`, reasoning, run
lifecycle) drive the screen, and the board exposes its own sensors / screen / clock back to the
agent as tools and ambient context.
@@ -27,6 +29,7 @@ mic ─ES8311/I²S(16k s16le)─▶ Soniox STT (streaming WSS) ─▶ live trans
─▶ reply text ─▶ Soniox TTS (streaming WSS) ─▶ ES8311 speaker (barge-in to interrupt)
─▶ client tools run on-device (set_timer → ringing alarm)
```
+(Same pipeline with Deepgram when that provider is selected.)
---
@@ -107,6 +110,8 @@ esp32-agui/ ← repo root
|---|---|
| [`agui_client`](esp32-agui/components/agui_client/) | Thin `extern "C"` shim over the vendored AG-UI SDK: POST `RunAgentInput` → handler callbacks |
| [`agui_sdk`](esp32-agui/components/agui_sdk/) | Vendored + ESP-ported community C++ AG-UI SDK (streaming SSE parser, event router) |
+| [`speech_stt`](esp32-agui/components/speech_stt/) / [`speech_tts`](esp32-agui/components/speech_tts/) | Facades over Deepgram or Soniox (see [docs/speech-providers.md](docs/speech-providers.md)) |
+| [`deepgram_stt`](esp32-agui/components/deepgram_stt/) / [`deepgram_tts`](esp32-agui/components/deepgram_tts/) | Deepgram Listen / Speak v1 WSS backends |
| [`soniox_client`](esp32-agui/components/soniox_client/) | ES8311 mic capture → Soniox WSS streaming STT → partial/final transcript callbacks |
| [`soniox_tts_client`](esp32-agui/components/soniox_tts_client/) | Reply text → Soniox WSS streaming TTS → PCM → ES8311 speaker (cancelable for barge-in) |
| [`chat_ui`](esp32-agui/components/chat_ui/) | LVGL chat bubbles, status line, touch-to-talk, configurable screen power saver, ringing-alarm overlay (uploaded graphic), idle screensaver |
@@ -196,10 +201,10 @@ are not emulated. Useful for boot / app-logic / networking, not the UI.
1. **Provision.** With no saved credentials, the device starts a SoftAP captive portal named
**`AMOLED-setup`**. Join it from a phone, and the form lets you set:
- WiFi SSID + password
- - **Soniox API key**
+ - **Soniox API key** (or switch **Speech provider** to Deepgram and enter a Deepgram key)
- **AG-UI endpoint URL** (+ optional bearer token)
- **Time zone** (POSIX `TZ` string, for the agent's ambient time context)
- - **TTS voice** (dropdown of the 28 Soniox `tts-rt-v1` voices; default **Adrian**)
+ - **TTS voice** (Soniox `tts-rt-v1` voices, default **Adrian**; Deepgram Aura-2 voices when that provider is selected)
- **Screen blank timeout** (seconds; default 60, `0` = always on)
- **Idle animation** (checkbox; gently pulse the uploaded alarm image when idle)
- **Alarm image** (file picker; any image is cropped in-browser to 240×240, converted to
diff --git a/docs/speech-providers.md b/docs/speech-providers.md
new file mode 100644
index 0000000..f3fc924
--- /dev/null
+++ b/docs/speech-providers.md
@@ -0,0 +1,47 @@
+# Speech providers (STT / TTS)
+
+The firmware speaks to cloud STT and TTS through a thin facade so the rest of the
+app (`main`, push-to-talk, AG-UI) stays provider-agnostic.
+
+## Supported providers
+
+| Provider | STT | TTS | Default |
+|----------|-----|-----|---------|
+| **Soniox** | Real-time WSS (`stt-rt-v5`) | Real-time WSS (`tts-rt-v1`) | Yes (fresh flash + legacy) |
+| **Deepgram** | Listen v1 WSS (`nova-3`, linear16 @ 16 kHz) | Speak v1 WSS (Aura-2, linear16 @ 16 kHz) | Opt-in via portal |
+
+Select the provider and paste that vendor’s API key in the **AMOLED-setup** captive portal.
+Switching providers requires a new API key; the portal resets TTS voice to the new provider’s default.
+
+`speech_provider_get()` caches the NVS value in RAM (invalidated on portal save). STT/TTS facades
+tear down the previous backend’s mic / drain task before opening the other provider, so a
+portal switch without reboot does not double-own the ES8311.
+
+## NVS keys (`appcfg` namespace)
+
+| Key | Meaning |
+|-----|---------|
+| `speech_prov` | `deepgram` or `soniox` |
+| `speech_key` | API key for the active provider |
+| `soniox_key` | Legacy Soniox-only key (still read as fallback when provider is Soniox) |
+| `tts_voice` | Provider-specific voice / model id |
+
+## Components
+
+```
+speech_cfg resolve provider + key (incl. migration)
+speech_stt facade → deepgram_stt | soniox_client
+speech_tts facade → deepgram_tts | soniox_tts_client
+deepgram_stt Deepgram Listen v1
+deepgram_tts Deepgram Speak v1
+soniox_* unchanged original backends
+```
+
+## Adding a third provider
+
+1. Implement `*_stt` / `*_tts` components with the same session / open-feed-finish APIs.
+2. Extend `speech_provider_t` + portal dropdown + voice list.
+3. Dispatch in `speech_stt.c` / `speech_tts.c`.
+4. Document auth + endpoints here.
+
+Do **not** call provider backends from `main` — always go through the facades.
diff --git a/esp32-agui/components/app_cfg/include/app_cfg.h b/esp32-agui/components/app_cfg/include/app_cfg.h
index 255caff..819e968 100644
--- a/esp32-agui/components/app_cfg/include/app_cfg.h
+++ b/esp32-agui/components/app_cfg/include/app_cfg.h
@@ -10,11 +10,13 @@
extern "C" {
#endif
-#define APP_CFG_SONIOX_KEY "soniox_key" // Soniox API key (permanent; ephemeral mint is P8)
+#define APP_CFG_SONIOX_KEY "soniox_key" // legacy Soniox-only key (still read via speech_cfg)
+#define APP_CFG_SPEECH_PROVIDER "speech_prov" // "soniox" | "deepgram" (default soniox)
+#define APP_CFG_SPEECH_KEY "speech_key" // active speech provider API key
#define APP_CFG_AGUI_URL "agui_url" // AG-UI endpoint (P2)
#define APP_CFG_AGUI_TOKEN "agui_token" // AG-UI bearer (P2)
#define APP_CFG_TZ "tz" // POSIX TZ string for local_time context (P5; default UTC0)
-#define APP_CFG_TTS_VOICE "tts_voice" // Soniox TTS voice name (P8; default "Adrian")
+#define APP_CFG_TTS_VOICE "tts_voice" // provider-specific TTS voice / model id
#define APP_CFG_TTS_VOL "tts_vol" // spoken-reply volume 0-100 (volume buttons; default 90)
#define APP_CFG_SCREEN_TO "scr_to" // screen blank timeout, seconds (default 60; 0 = always on)
#define APP_CFG_IDLE_ANIM "idle_anim" // idle screensaver flag "0"/"1" (default 0; pulses alarm image)
diff --git a/esp32-agui/components/deepgram_stt/CMakeLists.txt b/esp32-agui/components/deepgram_stt/CMakeLists.txt
new file mode 100644
index 0000000..31354ef
--- /dev/null
+++ b/esp32-agui/components/deepgram_stt/CMakeLists.txt
@@ -0,0 +1,5 @@
+idf_component_register(
+ SRCS "deepgram_stt.c"
+ INCLUDE_DIRS "include"
+ REQUIRES esp_websocket_client esp-tls json esp_codec_dev
+ esp32_s3_touch_amoled_1_8 app_cfg)
diff --git a/esp32-agui/components/deepgram_stt/deepgram_stt.c b/esp32-agui/components/deepgram_stt/deepgram_stt.c
new file mode 100644
index 0000000..3869940
--- /dev/null
+++ b/esp32-agui/components/deepgram_stt/deepgram_stt.c
@@ -0,0 +1,395 @@
+// deepgram_stt — mic (16 kHz mono s16le) -> Deepgram Listen v1 WSS -> transcript callbacks.
+#include "deepgram_stt.h"
+
+#include
+#include
+#include "freertos/FreeRTOS.h"
+#include "freertos/task.h"
+#include "freertos/semphr.h"
+#include "freertos/stream_buffer.h"
+#include "esp_log.h"
+#include "esp_heap_caps.h"
+#include "cJSON.h"
+#include "esp_websocket_client.h"
+#include "esp_crt_bundle.h"
+#include "esp_codec_dev.h"
+#include "bsp/esp32_s3_touch_amoled_1_8.h"
+
+static const char *TAG = "dg_stt";
+
+#define DEFAULT_MODEL "nova-3"
+#define DEFAULT_SR 16000
+#define MIC_GAIN_DB 30.0f
+#define READ_CHUNK_BYTES 640
+#define DRAIN_CHUNKS 8
+#define WS_BUFFER_BYTES 8192
+#define SEND_MAX_BYTES WS_BUFFER_BYTES
+#define SEND_TRIGGER 4096
+#define AUDIO_SB_BYTES (32 * 1024)
+#define COMMITTED_MAX 512
+#define RUNNING_MAX 640
+#define MSG_MAX 32768
+#define URI_MAX 320
+#define AUTH_HDR_MAX (APP_CFG_VAL_MAX + 32)
+
+// APP_CFG_VAL_MAX lives in app_cfg.h — pull for auth buffer sizing
+#include "app_cfg.h"
+
+static esp_codec_dev_handle_t s_mic;
+static bool s_mic_open;
+static esp_websocket_client_handle_t s_ws;
+static SemaphoreHandle_t s_lock;
+static SemaphoreHandle_t s_cap_done;
+static SemaphoreHandle_t s_send_done;
+static TaskHandle_t s_cap_task;
+static TaskHandle_t s_send_task;
+static StreamBufferHandle_t s_audio_sb;
+static StaticStreamBuffer_t s_sb_ctrl;
+static uint8_t *s_sb_storage;
+static volatile bool s_stop;
+static volatile bool s_fatal;
+static volatile bool s_active;
+static volatile bool s_close_sent;
+
+static deepgram_stt_partial_cb s_partial_cb;
+static deepgram_stt_turn_cb s_turn_cb;
+static void *s_ctx;
+
+static char s_api_key[APP_CFG_VAL_MAX];
+static char s_model[48];
+static int s_sr;
+static char s_uri[URI_MAX];
+static char s_auth_hdr[AUTH_HDR_MAX];
+static char s_committed[COMMITTED_MAX]; // touched only from the ws task (same invariant as Soniox)
+static char s_last_error[128];
+
+static uint8_t *s_rx;
+static size_t s_rx_total;
+
+static void emit_partial(const char *interim)
+{
+ if (!s_partial_cb) return;
+ char running[RUNNING_MAX];
+ snprintf(running, sizeof running, "%s%s", s_committed, interim ? interim : "");
+ if (running[0]) s_partial_cb(running, s_ctx);
+}
+
+static void commit_turn(void)
+{
+ if (s_committed[0] && s_turn_cb) s_turn_cb(s_committed, s_ctx);
+ s_committed[0] = '\0';
+}
+
+static void parse_message(const char *json)
+{
+ cJSON *root = cJSON_Parse(json);
+ if (!root) return;
+
+ cJSON *err = cJSON_GetObjectItem(root, "error");
+ if (cJSON_IsString(err) && err->valuestring) {
+ strlcpy(s_last_error, err->valuestring, sizeof s_last_error);
+ s_fatal = true;
+ ESP_LOGE(TAG, "deepgram error: %s", err->valuestring);
+ cJSON_Delete(root);
+ return;
+ }
+
+ cJSON *type = cJSON_GetObjectItem(root, "type");
+ const char *t = cJSON_IsString(type) ? type->valuestring : "";
+
+ if (strcmp(t, "Results") == 0) {
+ cJSON *ch = cJSON_GetObjectItem(root, "channel");
+ cJSON *alts = ch ? cJSON_GetObjectItem(ch, "alternatives") : NULL;
+ cJSON *alt0 = (cJSON_IsArray(alts) && cJSON_GetArraySize(alts) > 0)
+ ? cJSON_GetArrayItem(alts, 0) : NULL;
+ cJSON *tr = alt0 ? cJSON_GetObjectItem(alt0, "transcript") : NULL;
+ const char *transcript = (cJSON_IsString(tr) && tr->valuestring) ? tr->valuestring : "";
+ bool is_final = cJSON_IsTrue(cJSON_GetObjectItem(root, "is_final"));
+ bool speech_final = cJSON_IsTrue(cJSON_GetObjectItem(root, "speech_final"));
+
+ if (is_final) {
+ if (transcript[0]) {
+ if (s_committed[0] && s_committed[strlen(s_committed) - 1] != ' ')
+ strlcat(s_committed, " ", sizeof s_committed);
+ strlcat(s_committed, transcript, sizeof s_committed);
+ }
+ emit_partial("");
+ if (speech_final) commit_turn();
+ } else {
+ emit_partial(transcript);
+ }
+ } else if (strcmp(t, "UtteranceEnd") == 0) {
+ commit_turn();
+ emit_partial("");
+ }
+
+ cJSON_Delete(root);
+}
+
+static void ws_event(void *arg, esp_event_base_t base, int32_t id, void *data)
+{
+ (void)arg; (void)base;
+ esp_websocket_event_data_t *e = data;
+ switch (id) {
+ case WEBSOCKET_EVENT_CONNECTED:
+ ESP_LOGI(TAG, "ws connected");
+ s_committed[0] = '\0';
+ s_close_sent = false;
+ break;
+ case WEBSOCKET_EVENT_DISCONNECTED:
+ case WEBSOCKET_EVENT_CLOSED:
+ if (s_rx) { heap_caps_free(s_rx); s_rx = NULL; }
+ break;
+ case WEBSOCKET_EVENT_DATA: {
+ if (!e || e->payload_len == 0) break;
+ if (e->op_code == 0x08 || e->op_code == 0x09 || e->op_code == 0x0A) break;
+ if (e->op_code == 0x02) break; // ignore unexpected binary from server
+ if (e->payload_len > MSG_MAX) { ESP_LOGW(TAG, "oversized msg %d", (int)e->payload_len); break; }
+ if (e->payload_offset == 0) {
+ if (s_rx) heap_caps_free(s_rx);
+ s_rx = heap_caps_malloc(e->payload_len + 1, MALLOC_CAP_SPIRAM);
+ s_rx_total = e->payload_len;
+ }
+ if (!s_rx) break;
+ if (e->payload_offset + e->data_len <= s_rx_total)
+ memcpy(s_rx + e->payload_offset, e->data_ptr, e->data_len);
+ if (e->payload_offset + e->data_len >= s_rx_total) {
+ s_rx[s_rx_total] = '\0';
+ parse_message((const char *)s_rx);
+ heap_caps_free(s_rx);
+ s_rx = NULL;
+ }
+ break;
+ }
+ case WEBSOCKET_EVENT_ERROR:
+ ESP_LOGW(TAG, "ws transport error");
+ break;
+ default: break;
+ }
+}
+
+static void capture_task(void *arg)
+{
+ (void)arg;
+ uint8_t *buf = heap_caps_malloc(READ_CHUNK_BYTES, MALLOC_CAP_DEFAULT);
+ if (buf) {
+ esp_codec_dev_set_in_gain(s_mic, MIC_GAIN_DB);
+ for (int i = 0; i < DRAIN_CHUNKS && !s_stop && !s_fatal; i++)
+ esp_codec_dev_read(s_mic, buf, READ_CHUNK_BYTES);
+ while (!s_stop && !s_fatal) {
+ if (esp_codec_dev_read(s_mic, buf, READ_CHUNK_BYTES) != ESP_OK) continue;
+ if (xStreamBufferSpacesAvailable(s_audio_sb) >= READ_CHUNK_BYTES)
+ xStreamBufferSend(s_audio_sb, buf, READ_CHUNK_BYTES, 0);
+ }
+ heap_caps_free(buf);
+ } else {
+ ESP_LOGE(TAG, "no mem for capture buf");
+ }
+ s_cap_task = NULL;
+ xSemaphoreGive(s_cap_done);
+ vTaskDelete(NULL);
+}
+
+static void sender_task(void *arg)
+{
+ (void)arg;
+ uint8_t *buf = heap_caps_malloc(SEND_MAX_BYTES, MALLOC_CAP_DEFAULT);
+ if (buf) {
+ while (true) {
+ if (!esp_websocket_client_is_connected(s_ws)) {
+ if (s_stop || s_fatal) break;
+ vTaskDelay(pdMS_TO_TICKS(50));
+ continue;
+ }
+ size_t n = xStreamBufferReceive(s_audio_sb, buf, SEND_MAX_BYTES, pdMS_TO_TICKS(100));
+ if (n == 0) {
+ if (s_stop || s_fatal) break;
+ continue;
+ }
+ if (s_fatal) continue;
+ esp_websocket_client_send_bin(s_ws, (const char *)buf, n, pdMS_TO_TICKS(500));
+ }
+ // End-of-audio: ask Deepgram to flush finals before we tear the socket down.
+ if (s_ws && esp_websocket_client_is_connected(s_ws) && !s_close_sent) {
+ const char *msg = "{\"type\":\"CloseStream\"}";
+ esp_websocket_client_send_text(s_ws, msg, strlen(msg), pdMS_TO_TICKS(1000));
+ s_close_sent = true;
+ // Brief wait for final Results / UtteranceEnd on the ws task.
+ // Do NOT commit_turn() here — s_committed is ws-task-only (same as Soniox).
+ vTaskDelay(pdMS_TO_TICKS(800));
+ }
+ heap_caps_free(buf);
+ } else {
+ ESP_LOGE(TAG, "no mem for sender buf");
+ }
+ s_send_task = NULL;
+ xSemaphoreGive(s_send_done);
+ vTaskDelete(NULL);
+}
+
+esp_err_t deepgram_stt_init(void)
+{
+ if (!s_lock) s_lock = xSemaphoreCreateMutex();
+ if (!s_cap_done) s_cap_done = xSemaphoreCreateBinary();
+ if (!s_send_done) s_send_done = xSemaphoreCreateBinary();
+ if (!s_audio_sb) {
+ s_sb_storage = heap_caps_malloc(AUDIO_SB_BYTES + 1, MALLOC_CAP_SPIRAM);
+ if (!s_sb_storage) { ESP_LOGE(TAG, "no PSRAM for audio ring"); return ESP_ERR_NO_MEM; }
+ s_audio_sb = xStreamBufferCreateStatic(AUDIO_SB_BYTES, SEND_TRIGGER, s_sb_storage, &s_sb_ctrl);
+ }
+ if (s_mic) return ESP_OK;
+ s_mic = bsp_audio_codec_microphone_init();
+ if (!s_mic) { ESP_LOGE(TAG, "mic init failed"); return ESP_FAIL; }
+ esp_codec_dev_set_in_gain(s_mic, MIC_GAIN_DB);
+ esp_codec_dev_sample_info_t fs = {
+ .bits_per_sample = 16, .channel = 1, .sample_rate = DEFAULT_SR,
+ };
+ esp_err_t err = esp_codec_dev_open(s_mic, &fs);
+ if (err != ESP_OK) { ESP_LOGE(TAG, "mic open failed: %s", esp_err_to_name(err)); return err; }
+ s_mic_open = true;
+ ESP_LOGI(TAG, "mic ready (16k mono)");
+ return ESP_OK;
+}
+
+void deepgram_stt_deinit(void)
+{
+ if (s_active) {
+ ESP_LOGW(TAG, "deinit while session active — stop first");
+ deepgram_stt_session_stop();
+ }
+ if (s_mic) {
+ if (s_mic_open) { esp_codec_dev_close(s_mic); s_mic_open = false; }
+ esp_codec_dev_delete(s_mic);
+ s_mic = NULL;
+ ESP_LOGI(TAG, "mic released");
+ }
+}
+
+esp_err_t deepgram_stt_mic_stop(void)
+{
+ if (s_mic && s_mic_open) { esp_codec_dev_close(s_mic); s_mic_open = false; }
+ return ESP_OK;
+}
+
+esp_err_t deepgram_stt_mic_start(void)
+{
+ if (!s_mic) return deepgram_stt_init();
+ if (s_mic_open) return ESP_OK;
+ esp_codec_dev_sample_info_t fs = { .bits_per_sample = 16, .channel = 1, .sample_rate = DEFAULT_SR };
+ esp_err_t err = esp_codec_dev_open(s_mic, &fs);
+ if (err != ESP_OK) { ESP_LOGE(TAG, "mic reopen failed: %s", esp_err_to_name(err)); return err; }
+ esp_codec_dev_set_in_gain(s_mic, MIC_GAIN_DB);
+ s_mic_open = true;
+ return ESP_OK;
+}
+
+esp_err_t deepgram_stt_session_start(const deepgram_stt_cfg_t *cfg,
+ deepgram_stt_partial_cb on_partial,
+ deepgram_stt_turn_cb on_turn, void *ctx)
+{
+ if (!s_mic) { esp_err_t e = deepgram_stt_init(); if (e != ESP_OK) return e; }
+ xSemaphoreTake(s_lock, portMAX_DELAY);
+ if (s_active) { xSemaphoreGive(s_lock); return ESP_ERR_INVALID_STATE; }
+
+ if (!cfg || !cfg->api_key || !cfg->api_key[0]) {
+ ESP_LOGE(TAG, "missing Deepgram API key");
+ xSemaphoreGive(s_lock);
+ return ESP_ERR_NOT_FOUND;
+ }
+ strlcpy(s_api_key, cfg->api_key, sizeof s_api_key);
+ strlcpy(s_model, (cfg->model && cfg->model[0]) ? cfg->model : DEFAULT_MODEL, sizeof s_model);
+ s_sr = cfg->sample_rate ? cfg->sample_rate : DEFAULT_SR;
+
+ if (cfg->endpoint && cfg->endpoint[0]) {
+ strlcpy(s_uri, cfg->endpoint, sizeof s_uri);
+ } else {
+ snprintf(s_uri, sizeof s_uri,
+ "wss://api.deepgram.com/v1/listen?model=%s&encoding=linear16"
+ "&sample_rate=%d&channels=1&interim_results=true&punctuate=true"
+ "&endpointing=300&utterance_end_ms=1000",
+ s_model, s_sr);
+ }
+ snprintf(s_auth_hdr, sizeof s_auth_hdr, "Authorization: Token %s\r\n", s_api_key);
+
+ s_partial_cb = on_partial;
+ s_turn_cb = on_turn;
+ s_ctx = ctx;
+ s_committed[0] = '\0';
+ s_last_error[0] = '\0';
+ s_stop = false;
+ s_fatal = false;
+ s_close_sent = false;
+ xSemaphoreTake(s_cap_done, 0);
+ xSemaphoreTake(s_send_done, 0);
+ xStreamBufferReset(s_audio_sb);
+
+ esp_websocket_client_config_t wcfg = {
+ .uri = s_uri,
+ .headers = s_auth_hdr,
+ .buffer_size = WS_BUFFER_BYTES,
+ .task_stack = 8192,
+ .crt_bundle_attach = esp_crt_bundle_attach,
+ .reconnect_timeout_ms = 5000,
+ .network_timeout_ms = 10000,
+ .ping_interval_sec = 20,
+ };
+ s_ws = esp_websocket_client_init(&wcfg);
+ if (!s_ws) { xSemaphoreGive(s_lock); return ESP_FAIL; }
+ esp_websocket_register_events(s_ws, WEBSOCKET_EVENT_ANY, ws_event, NULL);
+ esp_err_t err = esp_websocket_client_start(s_ws);
+ if (err != ESP_OK) {
+ esp_websocket_client_destroy(s_ws); s_ws = NULL;
+ xSemaphoreGive(s_lock);
+ return err;
+ }
+
+ s_active = true;
+ if (xTaskCreate(capture_task, "dg_cap", 4096, NULL, 6, &s_cap_task) != pdPASS ||
+ xTaskCreate(sender_task, "dg_snd", 4096, NULL, 5, &s_send_task) != pdPASS) {
+ s_stop = true;
+ if (s_cap_task) xSemaphoreTake(s_cap_done, portMAX_DELAY);
+ if (s_send_task) xSemaphoreTake(s_send_done, portMAX_DELAY);
+ esp_websocket_client_stop(s_ws);
+ esp_websocket_client_destroy(s_ws);
+ s_ws = NULL;
+ s_active = false;
+ xSemaphoreGive(s_lock);
+ return ESP_FAIL;
+ }
+ ESP_LOGI(TAG, "session started (%s @ %d Hz)", s_model, s_sr);
+ xSemaphoreGive(s_lock);
+ return ESP_OK;
+}
+
+esp_err_t deepgram_stt_session_finalize(void)
+{
+ xSemaphoreTake(s_lock, portMAX_DELAY);
+ esp_err_t ret = ESP_ERR_INVALID_STATE;
+ if (s_active && s_ws && !s_close_sent) {
+ const char *msg = "{\"type\":\"Finalize\"}";
+ int n = esp_websocket_client_send_text(s_ws, msg, strlen(msg), pdMS_TO_TICKS(1000));
+ ret = n < 0 ? ESP_FAIL : ESP_OK;
+ }
+ xSemaphoreGive(s_lock);
+ return ret;
+}
+
+void deepgram_stt_session_stop(void)
+{
+ xSemaphoreTake(s_lock, portMAX_DELAY);
+ if (!s_active) { xSemaphoreGive(s_lock); return; }
+ s_stop = true;
+ if (s_cap_task) xSemaphoreTake(s_cap_done, portMAX_DELAY);
+ if (s_send_task) xSemaphoreTake(s_send_done, portMAX_DELAY);
+ esp_websocket_client_handle_t ws = s_ws;
+ s_ws = NULL;
+ if (ws) esp_websocket_client_destroy(ws);
+ if (s_rx) { heap_caps_free(s_rx); s_rx = NULL; }
+ s_active = false;
+ xSemaphoreGive(s_lock);
+ ESP_LOGI(TAG, "session stopped");
+}
+
+bool deepgram_stt_session_active(void) { return s_active && !s_fatal; }
+
+const char *deepgram_stt_last_error(void) { return s_last_error[0] ? s_last_error : NULL; }
diff --git a/esp32-agui/components/deepgram_stt/include/deepgram_stt.h b/esp32-agui/components/deepgram_stt/include/deepgram_stt.h
new file mode 100644
index 0000000..40175ef
--- /dev/null
+++ b/esp32-agui/components/deepgram_stt/include/deepgram_stt.h
@@ -0,0 +1,40 @@
+// deepgram_stt — live STT via Deepgram Listen v1 WebSocket.
+#pragma once
+
+#include
+#include
+#include
+#include "esp_err.h"
+
+#ifdef __cplusplus
+extern "C" {
+#endif
+
+typedef struct {
+ const char *endpoint; // NULL -> wss://api.deepgram.com/v1/listen?...
+ const char *api_key; // required (facade supplies from NVS)
+ const char *model; // NULL -> "nova-3"
+ int sample_rate; // 0 -> 16000
+} deepgram_stt_cfg_t;
+
+typedef void (*deepgram_stt_partial_cb)(const char *running_text, void *ctx);
+typedef void (*deepgram_stt_turn_cb)(const char *final_text, void *ctx);
+
+esp_err_t deepgram_stt_init(void);
+// Release the mic codec handle so another STT backend can own I2S-RX (provider switch).
+void deepgram_stt_deinit(void);
+esp_err_t deepgram_stt_mic_stop(void);
+esp_err_t deepgram_stt_mic_start(void);
+
+esp_err_t deepgram_stt_session_start(const deepgram_stt_cfg_t *cfg,
+ deepgram_stt_partial_cb on_partial,
+ deepgram_stt_turn_cb on_turn, void *ctx);
+
+esp_err_t deepgram_stt_session_finalize(void);
+void deepgram_stt_session_stop(void);
+bool deepgram_stt_session_active(void);
+const char *deepgram_stt_last_error(void);
+
+#ifdef __cplusplus
+}
+#endif
diff --git a/esp32-agui/components/deepgram_tts/CMakeLists.txt b/esp32-agui/components/deepgram_tts/CMakeLists.txt
new file mode 100644
index 0000000..ef5a428
--- /dev/null
+++ b/esp32-agui/components/deepgram_tts/CMakeLists.txt
@@ -0,0 +1,4 @@
+idf_component_register(
+ SRCS "deepgram_tts.c"
+ INCLUDE_DIRS "include"
+ REQUIRES esp_websocket_client esp-tls json app_cfg speech_cfg)
diff --git a/esp32-agui/components/deepgram_tts/deepgram_tts.c b/esp32-agui/components/deepgram_tts/deepgram_tts.c
new file mode 100644
index 0000000..c82254b
--- /dev/null
+++ b/esp32-agui/components/deepgram_tts/deepgram_tts.c
@@ -0,0 +1,333 @@
+// deepgram_tts — text -> Deepgram Speak v1 WSS -> pcm_s16le @16k/mono -> sink.
+#include "deepgram_tts.h"
+
+#include
+#include
+#include "freertos/FreeRTOS.h"
+#include "freertos/task.h"
+#include "freertos/event_groups.h"
+#include "freertos/stream_buffer.h"
+#include "freertos/semphr.h"
+#include "esp_log.h"
+#include "esp_heap_caps.h"
+#include "esp_websocket_client.h"
+#include "esp_crt_bundle.h"
+#include "cJSON.h"
+#include "app_cfg.h"
+#include "speech_cfg.h"
+
+static const char *TAG = "dg_tts";
+
+#define TTS_VOICE_DEFAULT "aura-2-asteria-en"
+#define TTS_SR 16000
+#define WS_BUFFER_BYTES 8192
+#define MSG_MAX 65536
+#define RING_BYTES (128 * 1024)
+#define DRAIN_CHUNK 640
+#define URI_MAX 320
+#define AUTH_HDR_MAX (APP_CFG_VAL_MAX + 32)
+
+#define BIT_CONNECTED (1u << 0)
+#define BIT_FLUSHED (1u << 1)
+#define BIT_WSERR (1u << 2)
+#define BIT_CANCEL (1u << 3)
+#define BIT_CLOSED (1u << 4)
+#define BIT_DRAIN_EXIT (1u << 5)
+
+typedef enum { TTS_IDLE, TTS_OPEN, TTS_FINISHING } tts_state_t;
+static tts_state_t s_state;
+static volatile bool s_cancel;
+static volatile uint32_t s_audio_rx;
+static SemaphoreHandle_t s_send_mutex;
+static void (*s_sink)(const void *pcm, size_t bytes);
+static esp_websocket_client_handle_t s_ws;
+static StreamBufferHandle_t s_ring;
+static StaticStreamBuffer_t s_ring_ctrl;
+static uint8_t *s_ring_buf;
+static EventGroupHandle_t s_eg;
+static SemaphoreHandle_t s_lock;
+static TaskHandle_t s_drain_task;
+static uint8_t *s_rx;
+static size_t s_rx_total;
+static char s_api_key[APP_CFG_VAL_MAX];
+static char s_voice[APP_CFG_VAL_MAX];
+static char s_uri[URI_MAX];
+static char s_auth_hdr[AUTH_HDR_MAX];
+
+static void drain_task(void *arg)
+{
+ (void)arg;
+ uint8_t *buf = heap_caps_malloc(DRAIN_CHUNK, MALLOC_CAP_DEFAULT);
+ if (!buf) { ESP_LOGE(TAG, "drain buf alloc failed"); s_drain_task = NULL; vTaskDelete(NULL); return; }
+ for (;;) {
+ if (s_eg && (xEventGroupGetBits(s_eg) & BIT_DRAIN_EXIT)) break;
+ size_t n = xStreamBufferReceive(s_ring, buf, DRAIN_CHUNK, pdMS_TO_TICKS(100));
+ if (s_cancel) continue;
+ if (n && s_sink) s_sink(buf, n);
+ }
+ heap_caps_free(buf);
+ s_drain_task = NULL;
+ vTaskDelete(NULL);
+}
+
+static void parse_json(const char *json)
+{
+ cJSON *root = cJSON_Parse(json);
+ if (!root) return;
+ cJSON *type = cJSON_GetObjectItem(root, "type");
+ const char *t = cJSON_IsString(type) ? type->valuestring : "";
+ if (strcmp(t, "Flushed") == 0) {
+ xEventGroupSetBits(s_eg, BIT_FLUSHED);
+ } else if (strcmp(t, "Warning") == 0) {
+ cJSON *desc = cJSON_GetObjectItem(root, "description");
+ ESP_LOGW(TAG, "tts warning: %s", cJSON_IsString(desc) ? desc->valuestring : json);
+ } else if (cJSON_GetObjectItem(root, "error") || strcmp(t, "Error") == 0) {
+ ESP_LOGW(TAG, "tts server error: %s", json);
+ xEventGroupSetBits(s_eg, BIT_WSERR);
+ }
+ cJSON_Delete(root);
+}
+
+static void ws_event(void *arg, esp_event_base_t base, int32_t id, void *data)
+{
+ (void)arg; (void)base;
+ esp_websocket_event_data_t *e = data;
+ switch (id) {
+ case WEBSOCKET_EVENT_CONNECTED:
+ ESP_LOGI(TAG, "ws connected");
+ xEventGroupSetBits(s_eg, BIT_CONNECTED);
+ break;
+ case WEBSOCKET_EVENT_DISCONNECTED:
+ case WEBSOCKET_EVENT_CLOSED:
+ xEventGroupSetBits(s_eg, BIT_CLOSED);
+ if (s_rx) { heap_caps_free(s_rx); s_rx = NULL; }
+ break;
+ case WEBSOCKET_EVENT_DATA: {
+ if (!e || e->payload_len == 0) break;
+ if (e->op_code == 0x08 || e->op_code == 0x09 || e->op_code == 0x0A) break;
+ // Binary = PCM audio
+ if (e->op_code == 0x02) {
+ if (e->data_len && !s_cancel) {
+ xStreamBufferSend(s_ring, e->data_ptr, e->data_len, pdMS_TO_TICKS(2000));
+ s_audio_rx += e->data_len;
+ }
+ break;
+ }
+ // Text = Metadata / Flushed / Warning
+ if (e->payload_len > MSG_MAX) break;
+ if (e->payload_offset == 0) {
+ if (s_rx) heap_caps_free(s_rx);
+ s_rx = heap_caps_malloc(e->payload_len + 1, MALLOC_CAP_SPIRAM);
+ s_rx_total = e->payload_len;
+ }
+ if (!s_rx) break;
+ if (e->payload_offset + e->data_len <= s_rx_total)
+ memcpy(s_rx + e->payload_offset, e->data_ptr, e->data_len);
+ if (e->payload_offset + e->data_len >= s_rx_total) {
+ s_rx[s_rx_total] = '\0';
+ parse_json((const char *)s_rx);
+ heap_caps_free(s_rx);
+ s_rx = NULL;
+ }
+ break;
+ }
+ case WEBSOCKET_EVENT_ERROR:
+ ESP_LOGW(TAG, "ws transport error");
+ xEventGroupSetBits(s_eg, BIT_WSERR);
+ break;
+ default: break;
+ }
+}
+
+esp_err_t deepgram_tts_init(void (*sink)(const void *pcm, size_t bytes))
+{
+ s_sink = sink;
+ if (!s_eg) s_eg = xEventGroupCreate();
+ if (!s_lock) s_lock = xSemaphoreCreateMutex();
+ if (!s_send_mutex) s_send_mutex = xSemaphoreCreateMutex();
+ if (!s_eg || !s_lock || !s_send_mutex) {
+ ESP_LOGE(TAG, "init alloc failed");
+ return ESP_ERR_NO_MEM;
+ }
+ if (!s_ring_buf) {
+ s_ring_buf = heap_caps_malloc(RING_BYTES, MALLOC_CAP_SPIRAM);
+ if (!s_ring_buf) { ESP_LOGE(TAG, "ring alloc failed"); return ESP_ERR_NO_MEM; }
+ s_ring = xStreamBufferCreateStatic(RING_BYTES, 1, s_ring_buf, &s_ring_ctrl);
+ if (!s_ring) return ESP_ERR_NO_MEM;
+ }
+ if (s_drain_task) return ESP_OK;
+ xEventGroupClearBits(s_eg, BIT_DRAIN_EXIT);
+ if (xTaskCreate(drain_task, "dg_tts_drain", 3072, NULL, 5, &s_drain_task) != pdPASS) {
+ ESP_LOGE(TAG, "drain task create failed");
+ return ESP_FAIL;
+ }
+ ESP_LOGI(TAG, "tts inited (ring %d KB PSRAM)", RING_BYTES / 1024);
+ return ESP_OK;
+}
+
+void deepgram_tts_deinit(void)
+{
+ if (!s_eg && !s_drain_task) return;
+ deepgram_tts_cancel();
+ if (s_eg) xEventGroupSetBits(s_eg, BIT_DRAIN_EXIT);
+ for (int i = 0; i < 50 && s_drain_task; i++) vTaskDelay(pdMS_TO_TICKS(20));
+ if (s_ws) { esp_websocket_client_stop(s_ws); esp_websocket_client_destroy(s_ws); s_ws = NULL; }
+ if (s_rx) { heap_caps_free(s_rx); s_rx = NULL; }
+ if (s_ring_buf) { heap_caps_free(s_ring_buf); s_ring_buf = NULL; }
+ s_ring = NULL;
+ s_state = TTS_IDLE;
+ s_sink = NULL;
+ ESP_LOGI(TAG, "tts deinit");
+}
+
+static esp_err_t send_json_locked(cJSON *root)
+{
+ char *s = cJSON_PrintUnformatted(root);
+ cJSON_Delete(root);
+ if (!s) return ESP_ERR_NO_MEM;
+ xSemaphoreTake(s_send_mutex, portMAX_DELAY);
+ int n = esp_websocket_client_send_text(s_ws, s, strlen(s), pdMS_TO_TICKS(2000));
+ xSemaphoreGive(s_send_mutex);
+ cJSON_free(s);
+ return n < 0 ? ESP_FAIL : ESP_OK;
+}
+
+static void ws_teardown(void)
+{
+ if (s_ws) {
+ esp_websocket_client_stop(s_ws);
+ esp_websocket_client_destroy(s_ws);
+ s_ws = NULL;
+ }
+ if (s_rx) { heap_caps_free(s_rx); s_rx = NULL; }
+}
+
+esp_err_t deepgram_tts_open(void)
+{
+ if (!s_lock || !s_sink) return ESP_ERR_INVALID_STATE;
+ xSemaphoreTake(s_lock, portMAX_DELAY);
+
+ if (!speech_cfg_get_key(s_api_key, sizeof s_api_key)) {
+ ESP_LOGW(TAG, "no speech API key for TTS");
+ xSemaphoreGive(s_lock);
+ return ESP_ERR_NOT_FOUND;
+ }
+ if (!app_cfg_get(APP_CFG_TTS_VOICE, s_voice, sizeof s_voice) || !s_voice[0])
+ strlcpy(s_voice, TTS_VOICE_DEFAULT, sizeof s_voice);
+
+ s_cancel = false;
+ s_audio_rx = 0;
+ s_state = TTS_IDLE;
+ xStreamBufferReset(s_ring);
+ xEventGroupClearBits(s_eg, BIT_CONNECTED | BIT_FLUSHED | BIT_WSERR | BIT_CANCEL | BIT_CLOSED);
+
+ snprintf(s_uri, sizeof s_uri,
+ "wss://api.deepgram.com/v1/speak?model=%s&encoding=linear16&sample_rate=%d",
+ s_voice, TTS_SR);
+ snprintf(s_auth_hdr, sizeof s_auth_hdr, "Authorization: Token %s\r\n", s_api_key);
+
+ ESP_LOGI(TAG, "open: internal free %u largest %u",
+ (unsigned)heap_caps_get_free_size(MALLOC_CAP_INTERNAL),
+ (unsigned)heap_caps_get_largest_free_block(MALLOC_CAP_INTERNAL));
+
+ esp_websocket_client_config_t wcfg = {
+ .uri = s_uri,
+ .headers = s_auth_hdr,
+ .buffer_size = WS_BUFFER_BYTES,
+ .task_stack = 8192,
+ .crt_bundle_attach = esp_crt_bundle_attach,
+ .reconnect_timeout_ms = 5000,
+ .network_timeout_ms = 10000,
+ .ping_interval_sec = 20,
+ };
+ s_ws = esp_websocket_client_init(&wcfg);
+ if (!s_ws) { xSemaphoreGive(s_lock); return ESP_FAIL; }
+ esp_websocket_register_events(s_ws, WEBSOCKET_EVENT_ANY, ws_event, NULL);
+ if (esp_websocket_client_start(s_ws) != ESP_OK) { ws_teardown(); xSemaphoreGive(s_lock); return ESP_FAIL; }
+
+ EventBits_t b = xEventGroupWaitBits(s_eg, BIT_CONNECTED | BIT_WSERR | BIT_CANCEL,
+ pdFALSE, pdFALSE, pdMS_TO_TICKS(5000));
+ if (!(b & BIT_CONNECTED)) {
+ ESP_LOGW(TAG, "tts connect failed/timeout → batch");
+ ws_teardown();
+ xSemaphoreGive(s_lock);
+ return ESP_FAIL;
+ }
+
+ s_state = TTS_OPEN;
+ ESP_LOGI(TAG, "tts stream open (%s)", s_voice);
+ return ESP_OK;
+}
+
+esp_err_t deepgram_tts_feed(const char *text)
+{
+ if (s_state != TTS_OPEN || !text || !text[0]) return ESP_ERR_INVALID_STATE;
+ cJSON *t = cJSON_CreateObject();
+ cJSON_AddStringToObject(t, "type", "Speak");
+ cJSON_AddStringToObject(t, "text", text);
+ return send_json_locked(t);
+}
+
+esp_err_t deepgram_tts_finish(void)
+{
+ if (s_state != TTS_OPEN) return ESP_OK;
+ cJSON *flush = cJSON_CreateObject();
+ cJSON_AddStringToObject(flush, "type", "Flush");
+ esp_err_t r = send_json_locked(flush);
+ cJSON *close = cJSON_CreateObject();
+ cJSON_AddStringToObject(close, "type", "Close");
+ send_json_locked(close);
+ s_state = TTS_FINISHING;
+ return r;
+}
+
+esp_err_t deepgram_tts_wait_drained(uint32_t stall_ms)
+{
+ if (s_state == TTS_IDLE) return ESP_OK;
+
+ for (;;) {
+ uint32_t rx_before = s_audio_rx;
+ EventBits_t b = xEventGroupWaitBits(s_eg, BIT_FLUSHED | BIT_CLOSED | BIT_WSERR | BIT_CANCEL,
+ pdFALSE, pdFALSE, pdMS_TO_TICKS(stall_ms));
+ if (b & BIT_CANCEL) { ESP_LOGI(TAG, "tts cancelled (barge-in)"); break; }
+ if (b & BIT_WSERR) { ESP_LOGW(TAG, "tts error"); break; }
+ if ((b & BIT_FLUSHED) || (b & BIT_CLOSED)) {
+ int spins = 0;
+ while (xStreamBufferBytesAvailable(s_ring) > 0 && spins++ < 600)
+ vTaskDelay(pdMS_TO_TICKS(50));
+ vTaskDelay(pdMS_TO_TICKS(120));
+ break;
+ }
+ if (s_audio_rx != rx_before) continue;
+ ESP_LOGW(TAG, "tts drain stall (no audio for %u ms)", (unsigned)stall_ms);
+ break;
+ }
+
+ ws_teardown();
+ if (s_cancel) {
+ int spins = 0;
+ while (xStreamBufferBytesAvailable(s_ring) > 0 && spins++ < 200)
+ vTaskDelay(pdMS_TO_TICKS(2));
+ s_cancel = false;
+ }
+ s_state = TTS_IDLE;
+ xSemaphoreGive(s_lock);
+ return ESP_OK;
+}
+
+esp_err_t deepgram_tts_speak(const char *text)
+{
+ if (!text || !text[0]) return ESP_ERR_INVALID_ARG;
+ esp_err_t e = deepgram_tts_open();
+ if (e != ESP_OK) return e;
+ deepgram_tts_feed(text);
+ deepgram_tts_finish();
+ return deepgram_tts_wait_drained(30000);
+}
+
+void deepgram_tts_cancel(void)
+{
+ if (!s_eg) return;
+ s_cancel = true;
+ xEventGroupSetBits(s_eg, BIT_CANCEL);
+}
diff --git a/esp32-agui/components/deepgram_tts/include/deepgram_tts.h b/esp32-agui/components/deepgram_tts/include/deepgram_tts.h
new file mode 100644
index 0000000..e520107
--- /dev/null
+++ b/esp32-agui/components/deepgram_tts/include/deepgram_tts.h
@@ -0,0 +1,23 @@
+// deepgram_tts — spoken replies via Deepgram Speak v1 WebSocket (linear16 @ 16 kHz).
+#pragma once
+
+#include
+#include
+#include "esp_err.h"
+
+#ifdef __cplusplus
+extern "C" {
+#endif
+
+esp_err_t deepgram_tts_init(void (*sink)(const void *pcm, size_t bytes));
+void deepgram_tts_deinit(void);
+esp_err_t deepgram_tts_speak(const char *text);
+esp_err_t deepgram_tts_open(void);
+esp_err_t deepgram_tts_feed(const char *text);
+esp_err_t deepgram_tts_finish(void);
+esp_err_t deepgram_tts_wait_drained(uint32_t timeout_ms);
+void deepgram_tts_cancel(void);
+
+#ifdef __cplusplus
+}
+#endif
diff --git a/esp32-agui/components/net_prov/CMakeLists.txt b/esp32-agui/components/net_prov/CMakeLists.txt
index 8653d8d..2a45b66 100644
--- a/esp32-agui/components/net_prov/CMakeLists.txt
+++ b/esp32-agui/components/net_prov/CMakeLists.txt
@@ -1,4 +1,4 @@
idf_component_register(
SRCS "net_prov.c" "portal.c"
INCLUDE_DIRS "include"
- REQUIRES esp_wifi esp_netif esp_event esp_timer nvs_flash esp_http_server esp_http_client esp-tls lwip app_cfg alarm_img)
+ REQUIRES esp_wifi esp_netif esp_event esp_timer nvs_flash esp_http_server esp_http_client esp-tls lwip app_cfg speech_cfg alarm_img)
diff --git a/esp32-agui/components/net_prov/portal.c b/esp32-agui/components/net_prov/portal.c
index 5f3a269..79c0175 100644
--- a/esp32-agui/components/net_prov/portal.c
+++ b/esp32-agui/components/net_prov/portal.c
@@ -3,8 +3,11 @@
#include "net_prov.h"
#include "app_cfg.h"
+#include "speech_cfg.h"
+#include
#include
+#include
#include
#include
#include "freertos/FreeRTOS.h"
@@ -41,10 +44,15 @@ static const char FORM_HEAD[] =
"input,select{display:block;width:100%;padding:.6em;margin:.4em 0;font-size:1em}"
"button{padding:.7em 1.2em;font-size:1em}"
"AG-UI device setup "
- ""
- ""
- // Alarm graphic: any image, cropped in-browser to 240x240 and converted to RGB565 (high byte
- // first, matching the device's LV_COLOR_16_SWAP) so the device stores the raw bytes with no decode.
+ "needed. Leave WiFi blank if already connected. Soniox is the default; choose Deepgram to use "
+ "that provider instead. Paste that provider's API key (required when switching) and set the "
+ "AG-UI URL.
"
+ "Save & connect ";
+
+static const char FORM_TAIL_ALARM[] =
"Alarm image (optional) "
"Shown when a timer goes off. Any image - it's cropped to "
"a 240x240 square (preview below) and sent to the device.
"
@@ -157,22 +167,38 @@ static bool form_field(const char *body, const char *name, char *out, size_t out
static esp_err_t root_get(httpd_req_t *req)
{
httpd_resp_set_type(req, "text/html");
- httpd_resp_set_hdr(req, "Cache-Control", "no-store"); // always serve the current form (no stale cache)
+ httpd_resp_set_hdr(req, "Cache-Control", "no-store");
httpd_resp_send_chunk(req, FORM_HEAD, HTTPD_RESP_USE_STRLEN);
- httpd_resp_send_chunk(req, TZ_OPTIONS, HTTPD_RESP_USE_STRLEN); // ~24 KB timezone list
- httpd_resp_send_chunk(req, FORM_MID, HTTPD_RESP_USE_STRLEN); // close tz , open voice
-
- // TTS-voice s, pre-selecting the saved voice (so re-saving keeps it); default Adrian.
- char saved[APP_CFG_VAL_MAX];
- if (!app_cfg_get(APP_CFG_TTS_VOICE, saved, sizeof saved)) strlcpy(saved, TTS_VOICE_DEFAULT, sizeof saved);
- char opts[1500];
+ httpd_resp_send_chunk(req, TZ_OPTIONS, HTTPD_RESP_USE_STRLEN);
+ httpd_resp_send_chunk(req, FORM_MID, HTTPD_RESP_USE_STRLEN);
+
+ speech_provider_t prov = speech_provider_get();
+ char saved_voice[APP_CFG_VAL_MAX];
+ if (!app_cfg_get(APP_CFG_TTS_VOICE, saved_voice, sizeof saved_voice))
+ strlcpy(saved_voice,
+ (prov == SPEECH_PROVIDER_SONIOX) ? SONIOX_VOICE_DEFAULT : DEEPGRAM_VOICE_DEFAULT,
+ sizeof saved_voice);
+
+ // Pre-select speech provider in the form (patch via tiny script after options).
+ char opts[3500];
size_t o = 0;
- for (size_t i = 0; i < sizeof(TTS_VOICES) / sizeof(TTS_VOICES[0]); i++)
+ // Emit both voice lists as JS arrays + initial options for current provider.
+ o += snprintf(opts + o, sizeof(opts) - o, "");
+
+ const char *const *list = (prov == SPEECH_PROVIDER_SONIOX) ? SONIOX_VOICES : DEEPGRAM_VOICES;
+ size_t nlist = (prov == SPEECH_PROVIDER_SONIOX)
+ ? sizeof(SONIOX_VOICES) / sizeof(SONIOX_VOICES[0])
+ : sizeof(DEEPGRAM_VOICES) / sizeof(DEEPGRAM_VOICES[0]);
+ for (size_t i = 0; i < nlist; i++)
o += snprintf(opts + o, sizeof(opts) - o, " %s ",
- strcmp(TTS_VOICES[i], saved) == 0 ? " selected" : "", TTS_VOICES[i]);
+ strcmp(list[i], saved_voice) == 0 ? " selected" : "", list[i]);
- // Close the voice , then the screen-timeout field, pre-filled with the saved value (so
- // re-saving keeps it). Default 60 s; 0 = always on.
char scr_to[8];
if (!app_cfg_get(APP_CFG_SCREEN_TO, scr_to, sizeof scr_to)) strlcpy(scr_to, "60", sizeof scr_to);
o += snprintf(opts + o, sizeof(opts) - o,
@@ -180,8 +206,6 @@ static esp_err_t root_get(httpd_req_t *req)
"Screen blank timeout (seconds, 0 = always on) "
" ", scr_to);
- // Idle-animation checkbox, pre-checked from the saved state (so re-saving keeps it). Inline style
- // overrides the form's block/full-width input rule.
char ia[4];
bool ia_on = app_cfg_get(APP_CFG_IDLE_ANIM, ia, sizeof ia) && ia[0] == '1';
o += snprintf(opts + o, sizeof(opts) - o,
@@ -191,8 +215,29 @@ static esp_err_t root_get(httpd_req_t *req)
ia_on ? " checked" : "");
httpd_resp_send_chunk(req, opts, o);
- httpd_resp_send_chunk(req, FORM_TAIL, HTTPD_RESP_USE_STRLEN);
- return httpd_resp_send_chunk(req, NULL, 0); // terminate the chunked response
+ httpd_resp_send_chunk(req, FORM_TAIL_INTRO, HTTPD_RESP_USE_STRLEN);
+
+ // Provider select sync + timezone auto-detect.
+ char script[900];
+ snprintf(script, sizeof script,
+ "",
+ speech_provider_name(prov), SONIOX_VOICE_DEFAULT, DEEPGRAM_VOICE_DEFAULT);
+ httpd_resp_send_chunk(req, script, HTTPD_RESP_USE_STRLEN);
+ httpd_resp_send_chunk(req, FORM_TAIL_ALARM, HTTPD_RESP_USE_STRLEN);
+ return httpd_resp_send_chunk(req, NULL, 0);
}
static esp_err_t save_post(httpd_req_t *req)
@@ -207,19 +252,22 @@ static esp_err_t save_post(httpd_req_t *req)
}
body[got] = '\0';
- char ssid[33] = {0}, pass[65] = {0}, soniox[APP_CFG_VAL_MAX] = {0};
+ char ssid[33] = {0}, pass[65] = {0}, speech_key[APP_CFG_VAL_MAX] = {0};
char agui_url[APP_CFG_VAL_MAX] = {0}, agui_token[APP_CFG_VAL_MAX] = {0}, tz[64] = {0};
- char voice[APP_CFG_VAL_MAX] = {0}, scr_to[8] = {0};
+ char voice[APP_CFG_VAL_MAX] = {0}, scr_to[8] = {0}, speech_prov[24] = {0};
bool have_ssid = form_field(body, "ssid", ssid, sizeof(ssid)) && ssid[0];
- bool have_soniox = form_field(body, "soniox", soniox, sizeof(soniox)) && soniox[0];
+ bool have_key = form_field(body, "speech_key", speech_key, sizeof(speech_key)) && speech_key[0];
+ // Legacy field name from older portal builds.
+ if (!have_key) have_key = form_field(body, "soniox", speech_key, sizeof(speech_key)) && speech_key[0];
+ bool have_prov = form_field(body, "speech_prov", speech_prov, sizeof(speech_prov)) && speech_prov[0];
bool have_url = form_field(body, "agui_url", agui_url, sizeof(agui_url)) && agui_url[0];
bool have_token = form_field(body, "agui_token", agui_token, sizeof(agui_token)) && agui_token[0];
bool have_tz = form_field(body, "tz", tz, sizeof(tz)) && tz[0];
bool have_voice = form_field(body, "voice", voice, sizeof(voice)) && voice[0];
bool have_scr = form_field(body, "scr_to", scr_to, sizeof(scr_to)) && scr_to[0];
- ESP_LOGI(TAG, "save: body=%dB ssid=%d soniox=%d url=%d token=%d tz=%d voice=%d scr=%d",
- got, have_ssid, have_soniox, have_url, have_token, have_tz, have_voice, have_scr);
- if (!have_ssid && !have_soniox && !have_url && !have_token && !have_tz && !have_voice && !have_scr) {
+ ESP_LOGI(TAG, "save: body=%dB ssid=%d key=%d prov=%d url=%d token=%d tz=%d voice=%d scr=%d",
+ got, have_ssid, have_key, have_prov, have_url, have_token, have_tz, have_voice, have_scr);
+ if (!have_ssid && !have_key && !have_prov && !have_url && !have_token && !have_tz && !have_voice && !have_scr) {
httpd_resp_send_err(req, HTTPD_400_BAD_REQUEST, "nothing to save");
return ESP_FAIL;
}
@@ -227,7 +275,38 @@ static esp_err_t save_post(httpd_req_t *req)
form_field(body, "pass", pass, sizeof(pass));
net_creds_add(ssid, pass);
}
- if (have_soniox) app_cfg_set(APP_CFG_SONIOX_KEY, soniox);
+
+ speech_provider_t old_prov = speech_provider_get();
+ speech_provider_t new_prov = old_prov;
+ if (have_prov) {
+ new_prov = (strcmp(speech_prov, SPEECH_PROVIDER_SONIOX_STR) == 0)
+ ? SPEECH_PROVIDER_SONIOX : SPEECH_PROVIDER_DEEPGRAM;
+ }
+
+ // Switching provider requires a key for the new vendor (legacy soniox_key is Soniox-only).
+ if (have_prov && new_prov != old_prov && !have_key) {
+ httpd_resp_send_err(req, HTTPD_400_BAD_REQUEST,
+ "API key required when switching speech provider");
+ return ESP_FAIL;
+ }
+
+ if (have_prov) {
+ app_cfg_set(APP_CFG_SPEECH_PROVIDER, speech_provider_name(new_prov));
+ // Reset voice to the new provider's default unless this POST also sets a voice.
+ if (!have_voice) {
+ app_cfg_set(APP_CFG_TTS_VOICE,
+ (new_prov == SPEECH_PROVIDER_SONIOX) ? SONIOX_VOICE_DEFAULT
+ : DEEPGRAM_VOICE_DEFAULT);
+ }
+ }
+ if (have_key) {
+ app_cfg_set(APP_CFG_SPEECH_KEY, speech_key);
+ // Keep legacy key in sync when Soniox is selected (older tools / migration).
+ if (new_prov == SPEECH_PROVIDER_SONIOX)
+ app_cfg_set(APP_CFG_SONIOX_KEY, speech_key);
+ }
+ if (have_prov || have_key) speech_cfg_invalidate();
+
if (have_url) app_cfg_set(APP_CFG_AGUI_URL, agui_url);
if (have_token) app_cfg_set(APP_CFG_AGUI_TOKEN, agui_token);
if (have_tz) app_cfg_set(APP_CFG_TZ, tz);
@@ -262,7 +341,8 @@ static esp_err_t save_post(httpd_req_t *req)
"Saved "
"This is exactly what the device received:
"
"WiFi: %s "
- "Soniox key: %s "
+ "Speech provider: %s "
+ "Speech API key: %s "
"AG-UI URL: %s "
"AG-UI token: %s "
"Timezone: %s "
@@ -272,7 +352,8 @@ static esp_err_t save_post(httpd_req_t *req)
" If AG-UI URL says \"unchanged\" but you typed one, your phone submitted a cached "
"form — reload 192.168.4.1 and try again.
",
have_ssid ? "updated" : "unchanged",
- have_soniox ? "updated" : "unchanged",
+ have_prov ? speech_prov : "unchanged",
+ have_key ? "updated" : "unchanged",
have_url ? agui_url : "unchanged",
have_token ? "updated" : "unchanged",
have_tz ? tz : "unchanged",
diff --git a/esp32-agui/components/soniox_client/include/soniox_client.h b/esp32-agui/components/soniox_client/include/soniox_client.h
index 7761576..50d6104 100644
--- a/esp32-agui/components/soniox_client/include/soniox_client.h
+++ b/esp32-agui/components/soniox_client/include/soniox_client.h
@@ -32,6 +32,10 @@ typedef void (*soniox_turn_cb)(const char *final_text, void *ctx);
// Bring up the microphone codec (16 kHz mono). Call once.
esp_err_t soniox_client_init(void);
+// Release the mic codec handle so another STT backend can own I2S-RX (provider switch).
+// No-op if a session is still active — stop the session first.
+void soniox_client_deinit(void);
+
// Idle low-power: stop/restart the mic codec (disables/enables the I2S so it releases its
// NO_LIGHT_SLEEP lock, letting the CPU light-sleep when idle). mic_start() must precede a session.
esp_err_t soniox_client_mic_stop(void);
diff --git a/esp32-agui/components/soniox_client/soniox_client.c b/esp32-agui/components/soniox_client/soniox_client.c
index 8c820e2..ff0f396 100644
--- a/esp32-agui/components/soniox_client/soniox_client.c
+++ b/esp32-agui/components/soniox_client/soniox_client.c
@@ -283,6 +283,20 @@ esp_err_t soniox_client_init(void)
return ESP_OK;
}
+void soniox_client_deinit(void)
+{
+ if (s_active) {
+ ESP_LOGW(TAG, "deinit while session active — stop first");
+ soniox_session_stop();
+ }
+ if (s_mic) {
+ if (s_mic_open) { esp_codec_dev_close(s_mic); s_mic_open = false; }
+ esp_codec_dev_delete(s_mic);
+ s_mic = NULL;
+ ESP_LOGI(TAG, "mic released");
+ }
+}
+
// Idle low-power: stop the mic codec when idle so the I2S RX channel is disabled and releases its
// NO_LIGHT_SLEEP lock (a running I2S blocks light sleep). esp_codec_dev_close() disables the channel;
// mic_start() reopens it (re-asserting the 30 dB gain) before a turn. Both idempotent.
diff --git a/esp32-agui/components/soniox_tts_client/CMakeLists.txt b/esp32-agui/components/soniox_tts_client/CMakeLists.txt
index c5a6bdf..4211583 100644
--- a/esp32-agui/components/soniox_tts_client/CMakeLists.txt
+++ b/esp32-agui/components/soniox_tts_client/CMakeLists.txt
@@ -1,4 +1,4 @@
idf_component_register(
SRCS "soniox_tts_client.c"
INCLUDE_DIRS "include"
- REQUIRES esp_websocket_client esp-tls json mbedtls app_cfg)
+ REQUIRES esp_websocket_client esp-tls json mbedtls app_cfg speech_cfg)
diff --git a/esp32-agui/components/soniox_tts_client/include/soniox_tts_client.h b/esp32-agui/components/soniox_tts_client/include/soniox_tts_client.h
index bfba167..ede1d54 100644
--- a/esp32-agui/components/soniox_tts_client/include/soniox_tts_client.h
+++ b/esp32-agui/components/soniox_tts_client/include/soniox_tts_client.h
@@ -16,11 +16,17 @@ extern "C" {
// PCM sink: raw 16-bit/16k/mono samples to play. Provided by main (which owns the ES8311 OUT / s_spk),
// so there is a single speaker owner (the beep + low-power codec-close never race a second handle).
+#ifndef TTS_PCM_SINK_T_DEFINED
+#define TTS_PCM_SINK_T_DEFINED
typedef void (*tts_pcm_sink_t)(const void *pcm, size_t bytes);
+#endif
// One-time: register the playback sink + spin up the drain task. Call once at boot.
esp_err_t soniox_tts_init(tts_pcm_sink_t sink);
+// Tear down drain task + ring so another TTS backend can own the speaker sink.
+void soniox_tts_deinit(void);
+
// Speak `text` and BLOCK until playback finishes (or error/timeout). Opens the TTS WSS, sends the
// text, plays the returned PCM, closes. Soniox key comes from NVS (app_cfg). Serialized internally.
// Equivalent to open()+feed(text)+finish()+wait_drained(); used as the P-a batch fallback.
diff --git a/esp32-agui/components/soniox_tts_client/soniox_tts_client.c b/esp32-agui/components/soniox_tts_client/soniox_tts_client.c
index 6ec9c1d..6399634 100644
--- a/esp32-agui/components/soniox_tts_client/soniox_tts_client.c
+++ b/esp32-agui/components/soniox_tts_client/soniox_tts_client.c
@@ -7,6 +7,7 @@
// Opened only AFTER the AG-UI run has finished, so the TTS TLS never overlaps the SSE TLS (P-a
// "sequential TLS"); streaming-while-the-run-speaks is P-b.
#include "soniox_tts_client.h"
+#include "speech_cfg.h"
#include
#include "freertos/FreeRTOS.h"
@@ -39,6 +40,7 @@ static const char *TAG = "tts";
#define BIT_TERMINATED (1u << 1)
#define BIT_WSERR (1u << 2)
#define BIT_CANCEL (1u << 3) // PTT barge-in: stop speaking NOW (set by soniox_tts_cancel)
+#define BIT_DRAIN_EXIT (1u << 4) // deinit: ask drain task to exit
typedef enum { TTS_IDLE, TTS_OPEN, TTS_FINISHING } tts_state_t;
static tts_state_t s_state; // lifecycle (ptt_task only); IDLE between turns
@@ -52,6 +54,7 @@ static StaticStreamBuffer_t s_ring_ctrl;
static uint8_t *s_ring_buf;
static EventGroupHandle_t s_eg;
static SemaphoreHandle_t s_lock;
+static TaskHandle_t s_drain_task;
static char *s_rx; // WS message reassembly (PSRAM)
static size_t s_rx_total;
static char s_api_key[APP_CFG_VAL_MAX];
@@ -61,9 +64,11 @@ static char s_voice[APP_CFG_VAL_MAX]; // portal-sele
static void drain_task(void *arg)
{
+ (void)arg;
uint8_t *buf = heap_caps_malloc(DRAIN_CHUNK, MALLOC_CAP_DEFAULT);
- if (!buf) { ESP_LOGE(TAG, "drain buf alloc failed"); vTaskDelete(NULL); return; }
+ if (!buf) { ESP_LOGE(TAG, "drain buf alloc failed"); s_drain_task = NULL; vTaskDelete(NULL); return; }
for (;;) {
+ if (s_eg && (xEventGroupGetBits(s_eg) & BIT_DRAIN_EXIT)) break;
size_t n = xStreamBufferReceive(s_ring, buf, DRAIN_CHUNK, pdMS_TO_TICKS(100));
// Barge-in: keep RECEIVING (so the ring drains and speak() can finish) but DISCARD — don't
// feed the codec. This empties the ring deterministically without an xStreamBufferReset
@@ -71,6 +76,9 @@ static void drain_task(void *arg)
if (s_cancel) continue;
if (n && s_sink) s_sink(buf, n);
}
+ heap_caps_free(buf);
+ s_drain_task = NULL;
+ vTaskDelete(NULL);
}
// ---- websocket RX ----------------------------------------------------------
@@ -155,16 +163,20 @@ static void ws_event(void *arg, esp_event_base_t base, int32_t id, void *data)
esp_err_t soniox_tts_init(tts_pcm_sink_t sink)
{
- if (s_lock) return ESP_OK; // already inited
s_sink = sink;
- s_eg = xEventGroupCreate();
- s_lock = xSemaphoreCreateMutex();
- s_send_mutex = xSemaphoreCreateMutex();
- s_ring_buf = heap_caps_malloc(RING_BYTES, MALLOC_CAP_SPIRAM);
- if (!s_eg || !s_lock || !s_send_mutex || !s_ring_buf) { ESP_LOGE(TAG, "init alloc failed"); return ESP_ERR_NO_MEM; }
- s_ring = xStreamBufferCreateStatic(RING_BYTES, 1, s_ring_buf, &s_ring_ctrl);
- if (!s_ring) return ESP_ERR_NO_MEM;
- if (xTaskCreate(drain_task, "tts_drain", 3072, NULL, 5, NULL) != pdPASS) {
+ if (!s_eg) s_eg = xEventGroupCreate();
+ if (!s_lock) s_lock = xSemaphoreCreateMutex();
+ if (!s_send_mutex) s_send_mutex = xSemaphoreCreateMutex();
+ if (!s_eg || !s_lock || !s_send_mutex) { ESP_LOGE(TAG, "init alloc failed"); return ESP_ERR_NO_MEM; }
+ if (!s_ring_buf) {
+ s_ring_buf = heap_caps_malloc(RING_BYTES, MALLOC_CAP_SPIRAM);
+ if (!s_ring_buf) { ESP_LOGE(TAG, "ring alloc failed"); return ESP_ERR_NO_MEM; }
+ s_ring = xStreamBufferCreateStatic(RING_BYTES, 1, s_ring_buf, &s_ring_ctrl);
+ if (!s_ring) return ESP_ERR_NO_MEM;
+ }
+ if (s_drain_task) return ESP_OK;
+ xEventGroupClearBits(s_eg, BIT_DRAIN_EXIT);
+ if (xTaskCreate(drain_task, "tts_drain", 3072, NULL, 5, &s_drain_task) != pdPASS) {
ESP_LOGE(TAG, "drain task create failed");
return ESP_FAIL;
}
@@ -172,6 +184,21 @@ esp_err_t soniox_tts_init(tts_pcm_sink_t sink)
return ESP_OK;
}
+void soniox_tts_deinit(void)
+{
+ if (!s_eg && !s_drain_task) return;
+ soniox_tts_cancel();
+ if (s_eg) xEventGroupSetBits(s_eg, BIT_DRAIN_EXIT);
+ for (int i = 0; i < 50 && s_drain_task; i++) vTaskDelay(pdMS_TO_TICKS(20));
+ if (s_ws) { esp_websocket_client_stop(s_ws); esp_websocket_client_destroy(s_ws); s_ws = NULL; }
+ if (s_rx) { heap_caps_free(s_rx); s_rx = NULL; }
+ if (s_ring_buf) { heap_caps_free(s_ring_buf); s_ring_buf = NULL; }
+ s_ring = NULL;
+ s_state = TTS_IDLE;
+ s_sink = NULL;
+ ESP_LOGI(TAG, "tts deinit");
+}
+
static esp_err_t send_json_locked(cJSON *root)
{
char *s = cJSON_PrintUnformatted(root);
@@ -206,8 +233,8 @@ esp_err_t soniox_tts_open(void)
if (!s_lock || !s_sink) return ESP_ERR_INVALID_STATE;
xSemaphoreTake(s_lock, portMAX_DELAY);
- if (!app_cfg_get(APP_CFG_SONIOX_KEY, s_api_key, sizeof s_api_key)) {
- ESP_LOGW(TAG, "no Soniox key for TTS");
+ if (!speech_cfg_get_key(s_api_key, sizeof s_api_key)) {
+ ESP_LOGW(TAG, "no speech API key for TTS");
xSemaphoreGive(s_lock);
return ESP_ERR_NOT_FOUND;
}
diff --git a/esp32-agui/components/speech_cfg/CMakeLists.txt b/esp32-agui/components/speech_cfg/CMakeLists.txt
new file mode 100644
index 0000000..939f03d
--- /dev/null
+++ b/esp32-agui/components/speech_cfg/CMakeLists.txt
@@ -0,0 +1,4 @@
+idf_component_register(
+ SRCS "speech_cfg.c"
+ INCLUDE_DIRS "include"
+ REQUIRES app_cfg)
diff --git a/esp32-agui/components/speech_cfg/include/speech_cfg.h b/esp32-agui/components/speech_cfg/include/speech_cfg.h
new file mode 100644
index 0000000..ef2ca2b
--- /dev/null
+++ b/esp32-agui/components/speech_cfg/include/speech_cfg.h
@@ -0,0 +1,36 @@
+// speech_cfg — resolve speech provider + API key from NVS (portal / migration).
+#pragma once
+
+#include
+#include
+
+#ifdef __cplusplus
+extern "C" {
+#endif
+
+typedef enum {
+ SPEECH_PROVIDER_SONIOX = 0,
+ SPEECH_PROVIDER_DEEPGRAM = 1,
+} speech_provider_t;
+
+#define SPEECH_PROVIDER_DEEPGRAM_STR "deepgram"
+#define SPEECH_PROVIDER_SONIOX_STR "soniox"
+
+// Cached after first NVS read. Call speech_cfg_invalidate() after portal saves.
+speech_provider_t speech_provider_get(void);
+
+const char *speech_provider_name(speech_provider_t p);
+
+// Drop the in-memory provider cache so the next get() re-reads NVS.
+void speech_cfg_invalidate(void);
+
+// Load the active speech API key. Prefer APP_CFG_SPEECH_KEY; for Soniox also fall back to
+// legacy APP_CFG_SONIOX_KEY.
+bool speech_cfg_get_key(char *buf, size_t len);
+
+// True if a usable speech API key is configured for the active provider.
+bool speech_cfg_has_key(void);
+
+#ifdef __cplusplus
+}
+#endif
diff --git a/esp32-agui/components/speech_cfg/speech_cfg.c b/esp32-agui/components/speech_cfg/speech_cfg.c
new file mode 100644
index 0000000..21dcf15
--- /dev/null
+++ b/esp32-agui/components/speech_cfg/speech_cfg.c
@@ -0,0 +1,53 @@
+#include "speech_cfg.h"
+#include "app_cfg.h"
+
+#include
+
+static bool s_prov_cached;
+static speech_provider_t s_prov = SPEECH_PROVIDER_SONIOX;
+
+static speech_provider_t load_provider(void)
+{
+ char p[24];
+ if (app_cfg_get(APP_CFG_SPEECH_PROVIDER, p, sizeof p)) {
+ if (strcmp(p, SPEECH_PROVIDER_SONIOX_STR) == 0) return SPEECH_PROVIDER_SONIOX;
+ if (strcmp(p, SPEECH_PROVIDER_DEEPGRAM_STR) == 0) return SPEECH_PROVIDER_DEEPGRAM;
+ }
+ // No explicit provider saved → keep Soniox (legacy soniox_key and fresh flash alike).
+ return SPEECH_PROVIDER_SONIOX;
+}
+
+speech_provider_t speech_provider_get(void)
+{
+ if (!s_prov_cached) {
+ s_prov = load_provider();
+ s_prov_cached = true;
+ }
+ return s_prov;
+}
+
+void speech_cfg_invalidate(void)
+{
+ s_prov_cached = false;
+}
+
+const char *speech_provider_name(speech_provider_t p)
+{
+ return (p == SPEECH_PROVIDER_SONIOX) ? SPEECH_PROVIDER_SONIOX_STR
+ : SPEECH_PROVIDER_DEEPGRAM_STR;
+}
+
+bool speech_cfg_get_key(char *buf, size_t len)
+{
+ if (!buf || len == 0) return false;
+ if (app_cfg_get(APP_CFG_SPEECH_KEY, buf, len)) return true;
+ if (speech_provider_get() == SPEECH_PROVIDER_SONIOX)
+ return app_cfg_get(APP_CFG_SONIOX_KEY, buf, len);
+ return false;
+}
+
+bool speech_cfg_has_key(void)
+{
+ char tmp[APP_CFG_VAL_MAX];
+ return speech_cfg_get_key(tmp, sizeof tmp);
+}
diff --git a/esp32-agui/components/speech_stt/CMakeLists.txt b/esp32-agui/components/speech_stt/CMakeLists.txt
new file mode 100644
index 0000000..2bb28f9
--- /dev/null
+++ b/esp32-agui/components/speech_stt/CMakeLists.txt
@@ -0,0 +1,4 @@
+idf_component_register(
+ SRCS "speech_stt.c"
+ INCLUDE_DIRS "include"
+ REQUIRES speech_cfg soniox_client deepgram_stt app_cfg)
diff --git a/esp32-agui/components/speech_stt/include/speech_stt.h b/esp32-agui/components/speech_stt/include/speech_stt.h
new file mode 100644
index 0000000..efa911d
--- /dev/null
+++ b/esp32-agui/components/speech_stt/include/speech_stt.h
@@ -0,0 +1,38 @@
+// speech_stt — facade over Deepgram / Soniox streaming STT (provider from NVS).
+#pragma once
+
+#include
+#include
+#include
+#include "esp_err.h"
+
+#ifdef __cplusplus
+extern "C" {
+#endif
+
+typedef struct {
+ const char *endpoint; // optional provider override
+ const char *api_key; // NULL -> speech_cfg_get_key()
+ const char *model; // optional
+ int sample_rate;// 0 -> 16000
+} speech_stt_cfg_t;
+
+typedef void (*speech_stt_partial_cb)(const char *running_text, void *ctx);
+typedef void (*speech_stt_turn_cb)(const char *final_text, void *ctx);
+
+esp_err_t speech_stt_init(void);
+esp_err_t speech_stt_mic_stop(void);
+esp_err_t speech_stt_mic_start(void);
+
+esp_err_t speech_stt_session_start(const speech_stt_cfg_t *cfg,
+ speech_stt_partial_cb on_partial,
+ speech_stt_turn_cb on_turn, void *ctx);
+
+esp_err_t speech_stt_session_finalize(void);
+void speech_stt_session_stop(void);
+bool speech_stt_session_active(void);
+const char *speech_stt_last_error(void);
+
+#ifdef __cplusplus
+}
+#endif
diff --git a/esp32-agui/components/speech_stt/speech_stt.c b/esp32-agui/components/speech_stt/speech_stt.c
new file mode 100644
index 0000000..c5031b9
--- /dev/null
+++ b/esp32-agui/components/speech_stt/speech_stt.c
@@ -0,0 +1,118 @@
+#include "speech_stt.h"
+#include "speech_cfg.h"
+#include "app_cfg.h"
+#include "soniox_client.h"
+#include "deepgram_stt.h"
+
+#include
+#include "esp_log.h"
+
+static const char *TAG = "speech_stt";
+static speech_provider_t s_inited_for = (speech_provider_t)-1;
+
+static void teardown_backend(speech_provider_t p)
+{
+ if (p == SPEECH_PROVIDER_SONIOX) {
+ soniox_client_deinit();
+ } else if (p == SPEECH_PROVIDER_DEEPGRAM) {
+ deepgram_stt_deinit();
+ }
+}
+
+static esp_err_t ensure_backend(void)
+{
+ speech_provider_t p = speech_provider_get();
+ if (s_inited_for == p) return ESP_OK;
+
+ // Tear down the previous backend's mic before opening another ES8311 handle
+ // on the shared I2S-RX (bsp_audio_codec_microphone_init mints a new instance).
+ if (s_inited_for == SPEECH_PROVIDER_SONIOX || s_inited_for == SPEECH_PROVIDER_DEEPGRAM) {
+ ESP_LOGI(TAG, "switching STT %s → %s",
+ speech_provider_name(s_inited_for), speech_provider_name(p));
+ teardown_backend(s_inited_for);
+ s_inited_for = (speech_provider_t)-1;
+ }
+
+ esp_err_t e = (p == SPEECH_PROVIDER_SONIOX) ? soniox_client_init() : deepgram_stt_init();
+ if (e == ESP_OK) {
+ s_inited_for = p;
+ ESP_LOGI(TAG, "backend ready: %s", speech_provider_name(p));
+ }
+ return e;
+}
+
+esp_err_t speech_stt_init(void) { return ensure_backend(); }
+
+esp_err_t speech_stt_mic_stop(void)
+{
+ if (ensure_backend() != ESP_OK) return ESP_FAIL;
+ return (speech_provider_get() == SPEECH_PROVIDER_SONIOX)
+ ? soniox_client_mic_stop() : deepgram_stt_mic_stop();
+}
+
+esp_err_t speech_stt_mic_start(void)
+{
+ if (ensure_backend() != ESP_OK) return ESP_FAIL;
+ return (speech_provider_get() == SPEECH_PROVIDER_SONIOX)
+ ? soniox_client_mic_start() : deepgram_stt_mic_start();
+}
+
+esp_err_t speech_stt_session_start(const speech_stt_cfg_t *cfg,
+ speech_stt_partial_cb on_partial,
+ speech_stt_turn_cb on_turn, void *ctx)
+{
+ if (ensure_backend() != ESP_OK) return ESP_FAIL;
+
+ char key[APP_CFG_VAL_MAX];
+ const char *api_key = (cfg && cfg->api_key) ? cfg->api_key : NULL;
+ if (!api_key) {
+ if (!speech_cfg_get_key(key, sizeof key)) {
+ ESP_LOGE(TAG, "no speech API key — set via portal");
+ return ESP_ERR_NOT_FOUND;
+ }
+ api_key = key;
+ }
+
+ speech_provider_t p = speech_provider_get();
+ ESP_LOGI(TAG, "session_start provider=%s", speech_provider_name(p));
+ if (p == SPEECH_PROVIDER_SONIOX) {
+ soniox_cfg_t sc = {
+ .endpoint = cfg ? cfg->endpoint : NULL,
+ .api_key = api_key,
+ .model = cfg ? cfg->model : NULL,
+ .sample_rate = cfg ? cfg->sample_rate : 0,
+ };
+ return soniox_session_start(&sc, on_partial, on_turn, ctx);
+ }
+ deepgram_stt_cfg_t dc = {
+ .endpoint = cfg ? cfg->endpoint : NULL,
+ .api_key = api_key,
+ .model = cfg ? cfg->model : NULL,
+ .sample_rate = cfg ? cfg->sample_rate : 0,
+ };
+ return deepgram_stt_session_start(&dc, on_partial, on_turn, ctx);
+}
+
+esp_err_t speech_stt_session_finalize(void)
+{
+ return (speech_provider_get() == SPEECH_PROVIDER_SONIOX)
+ ? soniox_session_finalize() : deepgram_stt_session_finalize();
+}
+
+void speech_stt_session_stop(void)
+{
+ if (speech_provider_get() == SPEECH_PROVIDER_SONIOX) soniox_session_stop();
+ else deepgram_stt_session_stop();
+}
+
+bool speech_stt_session_active(void)
+{
+ return (speech_provider_get() == SPEECH_PROVIDER_SONIOX)
+ ? soniox_session_active() : deepgram_stt_session_active();
+}
+
+const char *speech_stt_last_error(void)
+{
+ return (speech_provider_get() == SPEECH_PROVIDER_SONIOX)
+ ? soniox_last_error() : deepgram_stt_last_error();
+}
diff --git a/esp32-agui/components/speech_tts/CMakeLists.txt b/esp32-agui/components/speech_tts/CMakeLists.txt
new file mode 100644
index 0000000..e3d4308
--- /dev/null
+++ b/esp32-agui/components/speech_tts/CMakeLists.txt
@@ -0,0 +1,4 @@
+idf_component_register(
+ SRCS "speech_tts.c"
+ INCLUDE_DIRS "include"
+ REQUIRES speech_cfg soniox_tts_client deepgram_tts)
diff --git a/esp32-agui/components/speech_tts/include/speech_tts.h b/esp32-agui/components/speech_tts/include/speech_tts.h
new file mode 100644
index 0000000..33175bb
--- /dev/null
+++ b/esp32-agui/components/speech_tts/include/speech_tts.h
@@ -0,0 +1,27 @@
+// speech_tts — facade over Deepgram / Soniox streaming TTS (provider from NVS).
+#pragma once
+
+#include
+#include
+#include "esp_err.h"
+
+#ifdef __cplusplus
+extern "C" {
+#endif
+
+#ifndef TTS_PCM_SINK_T_DEFINED
+#define TTS_PCM_SINK_T_DEFINED
+typedef void (*tts_pcm_sink_t)(const void *pcm, size_t bytes);
+#endif
+
+esp_err_t speech_tts_init(tts_pcm_sink_t sink);
+esp_err_t speech_tts_speak(const char *text);
+esp_err_t speech_tts_open(void);
+esp_err_t speech_tts_feed(const char *text);
+esp_err_t speech_tts_finish(void);
+esp_err_t speech_tts_wait_drained(uint32_t timeout_ms);
+void speech_tts_cancel(void);
+
+#ifdef __cplusplus
+}
+#endif
diff --git a/esp32-agui/components/speech_tts/speech_tts.c b/esp32-agui/components/speech_tts/speech_tts.c
new file mode 100644
index 0000000..b3f21b9
--- /dev/null
+++ b/esp32-agui/components/speech_tts/speech_tts.c
@@ -0,0 +1,82 @@
+#include "speech_tts.h"
+#include "speech_cfg.h"
+#include "soniox_tts_client.h"
+#include "deepgram_tts.h"
+
+#include "esp_log.h"
+
+static const char *TAG = "speech_tts";
+static tts_pcm_sink_t s_sink;
+static speech_provider_t s_inited_for = (speech_provider_t)-1;
+
+static void teardown_backend(speech_provider_t p)
+{
+ if (p == SPEECH_PROVIDER_SONIOX) soniox_tts_deinit();
+ else if (p == SPEECH_PROVIDER_DEEPGRAM) deepgram_tts_deinit();
+}
+
+static esp_err_t ensure_backend(void)
+{
+ speech_provider_t p = speech_provider_get();
+ if (s_inited_for == p) return ESP_OK;
+
+ if (s_inited_for == SPEECH_PROVIDER_SONIOX || s_inited_for == SPEECH_PROVIDER_DEEPGRAM) {
+ ESP_LOGI(TAG, "switching TTS %s → %s",
+ speech_provider_name(s_inited_for), speech_provider_name(p));
+ teardown_backend(s_inited_for);
+ s_inited_for = (speech_provider_t)-1;
+ }
+
+ esp_err_t e = (p == SPEECH_PROVIDER_SONIOX)
+ ? soniox_tts_init(s_sink)
+ : deepgram_tts_init(s_sink);
+ if (e == ESP_OK) {
+ s_inited_for = p;
+ ESP_LOGI(TAG, "backend ready: %s", speech_provider_name(p));
+ }
+ return e;
+}
+
+esp_err_t speech_tts_init(tts_pcm_sink_t sink)
+{
+ s_sink = sink;
+ return ensure_backend();
+}
+
+esp_err_t speech_tts_speak(const char *text)
+{
+ if (ensure_backend() != ESP_OK) return ESP_FAIL;
+ if (speech_provider_get() == SPEECH_PROVIDER_SONIOX) return soniox_tts_speak(text);
+ return deepgram_tts_speak(text);
+}
+
+esp_err_t speech_tts_open(void)
+{
+ if (ensure_backend() != ESP_OK) return ESP_FAIL;
+ if (speech_provider_get() == SPEECH_PROVIDER_SONIOX) return soniox_tts_open();
+ return deepgram_tts_open();
+}
+
+esp_err_t speech_tts_feed(const char *text)
+{
+ if (speech_provider_get() == SPEECH_PROVIDER_SONIOX) return soniox_tts_feed(text);
+ return deepgram_tts_feed(text);
+}
+
+esp_err_t speech_tts_finish(void)
+{
+ if (speech_provider_get() == SPEECH_PROVIDER_SONIOX) return soniox_tts_finish();
+ return deepgram_tts_finish();
+}
+
+esp_err_t speech_tts_wait_drained(uint32_t timeout_ms)
+{
+ if (speech_provider_get() == SPEECH_PROVIDER_SONIOX) return soniox_tts_wait_drained(timeout_ms);
+ return deepgram_tts_wait_drained(timeout_ms);
+}
+
+void speech_tts_cancel(void)
+{
+ if (s_inited_for == SPEECH_PROVIDER_SONIOX) soniox_tts_cancel();
+ else if (s_inited_for == SPEECH_PROVIDER_DEEPGRAM) deepgram_tts_cancel();
+}
diff --git a/esp32-agui/main/CMakeLists.txt b/esp32-agui/main/CMakeLists.txt
index 042e8a2..92e9a39 100644
--- a/esp32-agui/main/CMakeLists.txt
+++ b/esp32-agui/main/CMakeLists.txt
@@ -5,8 +5,9 @@ idf_component_register(
nvs_flash
net_prov
app_cfg
- soniox_client
- soniox_tts_client
+ speech_cfg
+ speech_stt
+ speech_tts
agui_client
device_tools
chat_ui
diff --git a/esp32-agui/main/esp32_agui_main.c b/esp32-agui/main/esp32_agui_main.c
index 5879052..6e843f8 100644
--- a/esp32-agui/main/esp32_agui_main.c
+++ b/esp32-agui/main/esp32_agui_main.c
@@ -27,8 +27,9 @@
#include "net_prov.h"
#include "app_cfg.h"
-#include "soniox_client.h"
-#include "soniox_tts_client.h"
+#include "speech_stt.h"
+#include "speech_tts.h"
+#include "speech_cfg.h"
#include "agui_client.h"
#include "device_tools.h"
#include "chat_ui.h"
@@ -141,9 +142,9 @@ static void h_text_delta(const char *d, void *c)
// on ptt_task, so open() briefly stalls the SSE — lossless). If it can't open (e.g. concurrent-
// TLS OOM), s_tts_streaming stays false and we fall back to batch after the run.
if (!s_tts_streaming) {
- if (soniox_tts_open() == ESP_OK) s_tts_streaming = true;
+ if (speech_tts_open() == ESP_OK) s_tts_streaming = true;
}
- if (s_tts_streaming) soniox_tts_feed(d);
+ if (s_tts_streaming) speech_tts_feed(d);
// Always also buffer the whole reply: the batch fallback for a delta-less / open-failed turn.
size_t dl = strlen(d);
if (s_tts_len + dl < sizeof s_tts_text) {
@@ -272,14 +273,14 @@ static void run_agent_turn(const char *text)
cJSON_Delete(tools); // cJSON_Delete(NULL) is safe
if (s_aborting || s_run_error) { // barge-in or real error → tear down any live stream
- if (s_tts_streaming) { soniox_tts_cancel(); soniox_tts_wait_drained(3000); } // BIT_CANCEL → quick close + unlock
+ if (s_tts_streaming) { speech_tts_cancel(); speech_tts_wait_drained(3000); } // BIT_CANCEL → quick close + unlock
s_responding = false;
return; // abort: new turn owns the UI; error: keep "Error" up
}
if (s_tts_streaming) { // P-b: the reply streamed live → finalize it
chat_ui_status("Speaking...");
- soniox_tts_finish(); // text_end:true ONCE for the whole turn
- soniox_tts_wait_drained(30000); // play out the tail (30 s no-audio stall cap, not total)
+ speech_tts_finish(); // text_end:true ONCE for the whole turn
+ speech_tts_wait_drained(30000); // play out the tail (30 s no-audio stall cap, not total)
if (s_aborting) { s_responding = false; return; } // barged in during the spoken tail
} else if (s_tts_len > 0) { // delta-less / streaming-open-failed → P-a batch fallback
chat_ui_status("Speaking...");
@@ -292,7 +293,7 @@ static void run_agent_turn(const char *text)
// ---- PTT "go ahead and talk" beep -------------------------------------------------------------
// A short, subtle sine cue played on the ES8311 speaker the instant a hold starts. It plays from
-// ptt_task BEFORE soniox_session_start() creates capture_task, so it never glitches an in-flight
+// ptt_task BEFORE speech_stt_session_start() creates capture_task, so it never glitches an in-flight
// mic read. The speaker is a second esp_codec_dev OUT handle that coexists with the always-open mic
// IN handle on the single ES8311; both MUST use the same 16k/16/1 format (the shared codec/I2S
// clock's last set_fs wins — and it is NOT auto-enforced, so BEEP_SR is hard-pinned). Opening the
@@ -392,7 +393,7 @@ static void tts_pcm_write(const void *pcm, size_t bytes)
// Batch fallback used when streaming didn't open (delta-less / open-failed run). Volume is handled by
// the sink (tts_pcm_write), so this is just the blocking speak.
-static void tts_speak_reply(const char *text) { soniox_tts_speak(text); }
+static void tts_speak_reply(const char *text) { speech_tts_speak(text); }
// --- Volume control -----------------------------------------------------------------------------
// BOOT single-click = volume up, PWR short-press = volume down. The button/PWR callbacks just bump
@@ -452,7 +453,7 @@ static void lp_wake(void) // bring WiFi + codec back if shed; exactly-once
if (s_lp_suspended) {
s_lp_suspended = false;
if (s_lp_lock) esp_pm_lock_acquire(s_lp_lock); // no light sleep while active
- soniox_client_mic_start(); // re-enable the mic I2S (+ gain) for the turn
+ speech_stt_mic_start(); // re-enable the mic I2S (+ gain) for the turn
net_wifi_resume();
}
xSemaphoreGive(s_lp_mutex);
@@ -465,7 +466,7 @@ static void lp_idle(void) // shed everything when idle — only on battery (pl
xSemaphoreTake(s_lp_mutex, portMAX_DELAY);
if (!s_lp_suspended) {
s_lp_suspended = true;
- soniox_client_mic_stop(); // disable the mic I2S (RX)
+ speech_stt_mic_stop(); // disable the mic I2S (RX)
if (s_spk) { esp_codec_dev_close(s_spk); s_spk = NULL; s_spk_vol = -1; } // disable the speaker I2S (TX)
net_wifi_suspend();
if (s_lp_lock) esp_pm_lock_release(s_lp_lock); // WiFi+I2S now off → allow light sleep
@@ -529,8 +530,8 @@ static void ptt_task(void *arg)
turn_perf(true); // low-latency WiFi + 240 MHz CPU for the whole turn
chat_ui_status("Listening...");
play_ptt_beep(); // "go ahead" cue; plays & returns before capture starts
- soniox_cfg_t scfg = { 0 }; // api_key from NVS
- if (soniox_session_start(&scfg, on_partial, on_turn, NULL) != ESP_OK) {
+ speech_stt_cfg_t scfg = { 0 }; // api_key from NVS via speech_cfg
+ if (speech_stt_session_start(&scfg, on_partial, on_turn, NULL) != ESP_OK) {
s_listening = false;
turn_perf(false); // no turn will run; restore power-save now
chat_ui_status("STT error");
@@ -538,9 +539,9 @@ static void ptt_task(void *arg)
}
} else if (ev == 0 && s_listening) { // RELEASE
s_listening = false;
- soniox_session_stop(); // ws task is gone after this; buffers are stable
- if (soniox_last_error()) { // STT upload/transport died (e.g. hotspot congestion)
- ESP_LOGW(TAG, "STT failed: %s", soniox_last_error());
+ speech_stt_session_stop(); // ws task is gone after this; buffers are stable
+ if (speech_stt_last_error()) { // STT upload/transport died (e.g. hotspot congestion)
+ ESP_LOGW(TAG, "STT failed: %s", speech_stt_last_error());
chat_ui_status("Network — hold to retry"); // don't silently drop the turn
turn_perf(false);
continue;
@@ -589,7 +590,7 @@ static void ptt_down_cb(void *btn, void *ctx)
ESP_LOGI(TAG, "barge-in: aborting active reply");
s_aborting = true; // set BEFORE agui_abort so h_error (fires inline) suppresses "Error"
agui_abort(); // cancel the AG-UI run (no-op if already past it)
- soniox_tts_cancel(); // stop TTS playback (no-op if not speaking)
+ speech_tts_cancel(); // stop TTS playback (no-op if not speaking)
}
int e = 1; xQueueSend(s_ptt_q, &e, 0); // enqueue the press; ptt_task starts a fresh turn once unblocked
}
@@ -607,7 +608,7 @@ static void talk_cb(int ev, void *ctx)
ESP_LOGI(TAG, "barge-in (touch): aborting active reply");
s_aborting = true;
agui_abort();
- soniox_tts_cancel();
+ speech_tts_cancel();
}
int e = ev; xQueueSend(s_ptt_q, &e, 0);
}
@@ -717,7 +718,7 @@ void app_main(void)
ESP_ERROR_CHECK(net_prov_init());
for (;;) {
bool wifi_ok = net_is_connected() || (net_connect_saved(15000) == ESP_OK);
- bool key_ok = app_cfg_has(APP_CFG_SONIOX_KEY);
+ bool key_ok = speech_cfg_has_key();
bool url_ok = app_cfg_has(APP_CFG_AGUI_URL);
if (wifi_ok && key_ok && url_ok) break;
chat_ui_status("Setup: join 'AMOLED-setup'");
@@ -732,8 +733,8 @@ void app_main(void)
ESP_LOGI(TAG, "network + keys ready");
// Bring the mic up now; the Soniox WSS opens only on a PTT press.
- if (soniox_client_init() != ESP_OK) ESP_LOGE(TAG, "mic init failed");
- if (soniox_tts_init(tts_pcm_write) != ESP_OK) ESP_LOGE(TAG, "tts init failed"); // P-a: spoken replies
+ if (speech_stt_init() != ESP_OK) ESP_LOGE(TAG, "mic init failed");
+ if (speech_tts_init(tts_pcm_write) != ESP_OK) ESP_LOGE(TAG, "tts init failed"); // P-a: spoken replies
// Spoken-reply volume from NVS (set by the volume buttons); default VOL_DEFAULT if unset.
{ char v[8]; if (app_cfg_get(APP_CFG_TTS_VOL, v, sizeof v)) { int n = atoi(v); if (n >= 0 && n <= 100) s_tts_vol = n; } }
From 9bf7f72cfb95b6586de0068a6215c0d4ab3c0bcf Mon Sep 17 00:00:00 2001
From: pramodthe
Date: Mon, 13 Jul 2026 13:49:27 -0700
Subject: [PATCH 2/6] feat(ui): on-device NIMO companion face + speech/tooling
refinements
On-device companion UI, ported from the NIMO reference sketch and scaled to
the 368x448 AMOLED:
- face_engine: NIMO-style animated face rendered into a PSRAM canvas via a
small software GFX (rounded-rect eyes, rect pupils + glint, overlay-masked
expressions, spring physics, blink). Moods: normal/listen/think/speak/happy/
sleepy/sad/angry/surprised/dizzy/love/suspicious. Shake -> dizzy (orbiting
pupils + stars) -> angry (red eyes, brows, "!" marks, teeth). Talking mouth
opens/closes during speech.
- companion_pages: Eyes / Clock / Chat with tap-to-cycle and voice-state
auto-switch; face visibility/activity gated to the Eyes page.
- imu_qmi8658: QMI8658 6-axis IMU driver (smoothed tilt for gaze, shake
intensity, orientation label for ambient context).
- chat_ui: drive face moods from voice/turn status; wire companion pages.
Speech + tooling refinements across the provider facade and device layer:
deepgram STT/TTS, soniox client, speech_stt/speech_tts, device_tools,
net_prov/portal, app_cfg, and main (PTT/turn wiring). Docs + project guide.
Co-Authored-By: Claude Opus 4.8 (1M context)
---
docs/PR_SPEECH_PROVIDERS.md | 36 ++
esp32-agui/CLAUDE.md | 134 +++++
.../components/app_cfg/include/app_cfg.h | 1 -
esp32-agui/components/chat_ui/CMakeLists.txt | 9 +-
esp32-agui/components/chat_ui/chat_ui.c | 276 +++++++++-
.../components/chat_ui/companion_pages.c | 150 ++++++
.../components/chat_ui/companion_pages.h | 28 +
esp32-agui/components/chat_ui/face_engine.c | 488 ++++++++++++++++++
esp32-agui/components/chat_ui/face_engine.h | 22 +
.../components/chat_ui/idf_component.yml | 2 +
.../components/chat_ui/include/chat_ui.h | 22 +
.../components/deepgram_stt/deepgram_stt.c | 105 +++-
.../components/deepgram_tts/deepgram_tts.c | 28 +-
.../components/device_tools/CMakeLists.txt | 5 +-
.../components/device_tools/device_tools.c | 107 +++-
.../device_tools/include/device_tools.h | 9 +
.../components/imu_qmi8658/CMakeLists.txt | 4 +
.../components/imu_qmi8658/imu_qmi8658.c | 181 +++++++
.../imu_qmi8658/include/imu_qmi8658.h | 25 +
esp32-agui/components/net_prov/net_prov.c | 105 +++-
esp32-agui/components/net_prov/portal.c | 8 +-
.../components/soniox_client/soniox_client.c | 1 +
esp32-agui/components/speech_stt/speech_stt.c | 74 ++-
esp32-agui/components/speech_tts/speech_tts.c | 19 +-
esp32-agui/main/esp32_agui_main.c | 95 +++-
esp32-agui/sdkconfig.defaults | 1 +
26 files changed, 1815 insertions(+), 120 deletions(-)
create mode 100644 docs/PR_SPEECH_PROVIDERS.md
create mode 100644 esp32-agui/CLAUDE.md
create mode 100644 esp32-agui/components/chat_ui/companion_pages.c
create mode 100644 esp32-agui/components/chat_ui/companion_pages.h
create mode 100644 esp32-agui/components/chat_ui/face_engine.c
create mode 100644 esp32-agui/components/chat_ui/face_engine.h
create mode 100644 esp32-agui/components/chat_ui/idf_component.yml
create mode 100644 esp32-agui/components/imu_qmi8658/CMakeLists.txt
create mode 100644 esp32-agui/components/imu_qmi8658/imu_qmi8658.c
create mode 100644 esp32-agui/components/imu_qmi8658/include/imu_qmi8658.h
diff --git a/docs/PR_SPEECH_PROVIDERS.md b/docs/PR_SPEECH_PROVIDERS.md
new file mode 100644
index 0000000..04c368b
--- /dev/null
+++ b/docs/PR_SPEECH_PROVIDERS.md
@@ -0,0 +1,36 @@
+# PR draft: Add Deepgram as an alternate speech provider
+
+**Target repo:** https://github.com/contextablemark/esp32-agui
+**Branch (local):** `feat/speech-providers-deepgram`
+**Working tree:** `~/Desktop/esp32-agui`
+
+## Summary
+
+- **Add** Deepgram as an optional STT/TTS provider — Soniox stays the default and is unchanged as a feature.
+- Add a thin provider facade (`speech_stt` / `speech_tts` + `speech_cfg`) over the existing Soniox clients.
+- Implement **Deepgram** Listen v1 + Speak v1 WebSocket backends (16 kHz linear16, same mic/speaker path).
+- Select provider at runtime from the **AMOLED-setup** captive portal.
+- Existing devices (legacy `soniox_key` or no `speech_prov`) keep using Soniox.
+
+## Test plan
+
+- [ ] `idf.py build` on ESP-IDF 5.5.x / esp32s3
+- [ ] Flash Waveshare ESP32-S3-Touch-AMOLED-1.8
+- [ ] Portal: provider=Deepgram + Deepgram API key + Wi‑Fi + AG-UI URL → PTT → transcript → spoken reply
+- [ ] Portal: switch to Soniox + Soniox key → same flow
+- [ ] Barge-in (new PTT during TTS) still cancels speech
+
+## Notes for maintainers
+
+See [docs/speech-providers.md](../docs/speech-providers.md) for NVS keys and how to add a third provider.
+
+### Suggested `gh` flow (after fork)
+
+```bash
+cd ~/Desktop/esp32-agui
+git remote add fork git@github.com:/esp32-agui.git # if needed
+git push -u fork feat/speech-providers-deepgram
+gh pr create --repo contextablemark/esp32-agui \
+ --title "Add Deepgram as an alternate speech provider (Soniox remains default)" \
+ --body-file docs/PR_SPEECH_PROVIDERS.md
+```
diff --git a/esp32-agui/CLAUDE.md b/esp32-agui/CLAUDE.md
new file mode 100644
index 0000000..870ce17
--- /dev/null
+++ b/esp32-agui/CLAUDE.md
@@ -0,0 +1,134 @@
+# CLAUDE.md
+
+This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository.
+
+This is the **ESP-IDF firmware project** for the on-device AG-UI voice client. It is the buildable
+root (`idf.py` runs here). The **parent guide** [`../CLAUDE.md`](../CLAUDE.md) covers the hardware,
+the **ESP-IDF 5.5.x-only** toolchain constraint, flashing (incl. the devcontainer RFC2217 bridge),
+and DeepWiki driver lookup — read it for those; it is not repeated here. The parent doc frames the
+client as a "roadmap"; it is now implemented in this directory. Design docs live in
+[`../docs/`](../docs/) (`esp32-agui-plan.md`, `speech-providers.md`, `flashing.md`).
+
+## Build / flash / test
+
+Source the IDF env first (`. ~/esp/esp-idf/export.sh`), then from **this directory**:
+
+- **Build:** `idf.py build` — target `esp32s3` comes from `sdkconfig.defaults`. Do **not** run
+ `idf.py set-target` (it wipes `sdkconfig`). Slash-commands `/idf-build` `/idf-flash` `/idf-monitor`
+ `/idf-qemu` `/idf-size` wrap the common invocations.
+- **Flash + monitor:** `idf.py -p flash monitor` (`/dev/cu.usbmodem*` on macOS).
+- **QEMU:** boots app logic + networking only — no AMOLED/touch/codec/I²C. Not for UI work.
+- **AG-UI SDK host tests:** `components/agui_sdk/test/run_host_tests.sh` — plain `g++`, no ESP-IDF or
+ hardware. Round-trips the device SDK extensions (REASONING_*/interrupt/resume). **Run this after
+ re-syncing the vendored SDK** (see below); it fails loudly if a delta was dropped.
+
+Partitions (`partitions.csv`): `factory` app = 3 MB; `alarmimg` (custom type `0x40`, 0x40000) holds
+the portal-uploaded alarm graphic. A normal `idf.py flash` preserves `nvs` (saved WiFi/keys).
+
+## Component architecture
+
+First-party components under `components/` (managed deps live in `managed_components/`, vendored BSP
++ drivers are also under `components/`). The design is layered — `main` never talks to a cloud
+provider or the raw SDK directly, only to facades:
+
+```
+main/esp32_agui_main.c app entry, PTT state machine, audio/beep/alarm, low-power (see below)
+ ├─ net_prov WiFi (multi-SSID) + SoftAP captive portal ("AMOLED-setup") + SNTP/HTTPS clock
+ ├─ app_cfg NVS string store (namespace "appcfg"): keys, secrets, prefs (APP_CFG_* macros)
+ ├─ speech_cfg resolves active provider + API key from NVS; caches, invalidate on portal save
+ ├─ speech_stt ──────► soniox_client | deepgram_stt streaming STT facade (session start/stop/finalize)
+ ├─ speech_tts ──────► soniox_tts_client | deepgram_tts streaming TTS facade (open/feed/finish + speak)
+ ├─ agui_client ─────► agui_sdk AG-UI client (see "AG-UI SDK" below)
+ ├─ device_tools tool registry + impls (set_timer/set_alarm/show_qr/show_image) + ambient context
+ ├─ chat_ui LVGL UI: chat bubbles, status pill, face overlay, screensaver, alarm overlay
+ └─ alarm_img read/write the alarmimg flash partition
+```
+
+**Speech facade (`speech_*`).** The rest of the app is provider-agnostic. `speech_stt`/`speech_tts`
+dispatch to Soniox (default; real-time WSS) or Deepgram (opt-in; Listen/Speak v1 WSS) based on
+`speech_provider_get()`. **Never call a provider backend (`soniox_*`, `deepgram_*`) from `main` —
+always go through the facade.** Both providers use **16 kHz / s16le / mono** end-to-end (see the
+ES8311 constraint below). Adding a provider: implement `*_stt`/`*_tts` with the same
+session/open-feed-finish API, extend `speech_provider_t` + the portal dropdown, dispatch in
+`speech_stt.c`/`speech_tts.c`. Details: [`../docs/speech-providers.md`](../docs/speech-providers.md).
+
+**AG-UI SDK (`agui_sdk` + `agui_client`).** `agui_sdk` **vendors the upstream AG-UI community C++
+SDK** (`ag-ui-protocol/ag-ui` `sdks/community/c++`, MIT) with device patches; `agui_client` is a thin
+`extern "C"` shim over it (`main` is C). Ports: libcurl→`esp_http_client` (`EspHttpService`
+implements the SDK's `IHttpService`), and the SDK keeps `nlohmann/json` (managed dep
+`johboh__nlohmann-json`) while the shim exposes cJSON. Device extensions: per-run ambient `context`,
+first-class `REASONING_*` events, Interrupt→resume, client-tool dispatch. **Re-syncing the vendor
+snapshot:** all deltas are marked `// [device]` in-tree (`grep -rn "\[device\]" components/agui_sdk/src`);
+follow `components/agui_sdk/PATCHES.md` and re-run the host tests. C++ components pin `-std=gnu++17`.
+
+## Runtime model — the part you must understand before editing `main`
+
+**One task drives everything: `ptt_task`.** There is no continuous STT session and no barge-in
+audio mixer. A FreeRTOS queue (`s_ptt_q`) serializes all input events (button/touch/PWR callbacks
+only flip flags + enqueue — they never block or dispatch). Event codes: `1`=press, `0`=release,
+`2`=setup-portal, `3`/`4`=volume up/down.
+
+**Turn model (push-to-talk).** Hold BOOT **or** long-press the screen → open the STT session, stream
+mic, show the live transcript in the status line. Release → stop the session, assemble the utterance
+(accumulated finals + last interim), and run one AG-UI turn. The button *defines the turn boundary*,
+so there is no idle STT timeout and no streamed-silence garbage. BOOT **tap** = volume up, BOOT
+**double-tap** = reopen the setup portal, PWR short-press = volume down.
+
+**Agent turn + client-tool loop (`run_agent_turn`).** `agui_run()` **blocks** on `ptt_task` and
+fires handlers inline. Critical rule: the `on_tool_call` handler runs *inside* `agui_run` under the
+SDK lock, so it may only **record** client-tool calls into `s_pending[]` (copy strings) — never
+dispatch or re-run (deadlock). After `agui_run` returns, `ptt_task` drains `s_pending[]`, executes
+each tool via `device_tools_dispatch`, appends a result with `agui_tool_result`, and re-runs with
+`user_text=NULL`. Bounded by `AGUI_TOOL_MAX_ITERS`. Only tools where `device_tools_is_client(name)`
+is true are captured — server/agent tools are the agent's to run.
+
+**TTS: streaming with batch fallback.** On the first speakable text delta, `run_agent_turn` opens a
+live TTS stream (`speech_tts_open`/`_feed`) so audio starts before the reply finishes; it also buffers
+the whole reply. If the stream never opened (delta-less reply / OOM), it falls back to
+`speech_tts_speak` on the buffered text after the run.
+
+**Barge-in (`s_responding`/`s_aborting`).** A press while a reply is in flight aborts it:
+`agui_abort()` flips the SDK's atomic cancel flag (lock-free, callable from any task) and
+`speech_tts_cancel()` stops playback. `on_error` checks `s_aborting` to suppress the deliberate
+cancel as a non-error; the partial assistant message is dropped (`agui_drop_partial_assistant`) so it
+can't poison the next run.
+
+**Two hard concurrency invariants:**
+- **Single ES8311, full-duplex.** The mic IN handle stays open; the speaker is a *second*
+ `esp_codec_dev` OUT handle on the same chip. Both **must** use the same 16 kHz/16-bit/mono format
+ (the shared I2S clock's last `set_fs` wins and is not auto-enforced). Opening the speaker
+ soft-resets the chip and clobbers the mic ADC gain, and beep crosstalk lands in the RX DMA ring —
+ the capture task re-asserts gain and drains the ring before forwarding audio. See the long comment
+ above `BEEP_SR` in `main`.
+- **Sequential TLS, PSRAM buffers.** STT, AG-UI, and TTS TLS sessions run **one at a time** (listen →
+ respond); the device can't afford concurrent handshakes. The heartbeat logs
+ `internal_max` (largest contiguous *internal* block) because TLS/lwIP send buffers need contiguous
+ internal RAM — that number, not total free heap, predicts session failures.
+
+**Low-power idle.** PM is configured with light-sleep enabled and a `NO_LIGHT_SLEEP` lock held while
+active. On **battery** idle (display blanked, via the `chat_ui` power-cb) `lp_idle` sheds WiFi + the
+codec and releases the lock so the CPU light-sleeps; `lp_wake` (a PTT press) reverses it. Plugged-in
+stays fully connected. A separate `CPU_FREQ_MAX` lock is held only for the span of a turn (`turn_perf`)
+so the three TLS handshakes run at 240 MHz. Removing this block breaks the STT upload even on good
+WiFi — do not "simplify" it away.
+
+**UI liveness.** `chat_ui` bumps a 1 Hz counter from inside the LVGL task; the heartbeat flags a
+**UI STALL** if it doesn't move (historically an AMOLED brightness `tx_param` racing an LVGL flush —
+see the fix in commit `f8bb8b9`).
+
+## Config & secrets
+
+All runtime config/secrets live in **NVS** (`app_cfg`, namespace `appcfg`; `APP_CFG_*` keys in
+`components/app_cfg/include/app_cfg.h`), written by the captive portal (`net_prov`). Includes WiFi,
+speech provider + API key, AG-UI URL + bearer, TZ, TTS voice/volume, screen timeout, idle-anim flag.
+`main` boots straight into provisioning (opens `AMOLED-setup` SoftAP) until WiFi + a speech key +
+`agui_url` are all present. Legacy `soniox_key` is still read as a fallback for the Soniox provider.
+
+## Conventions
+
+- Source comments tag work by build phase (**P0**–**P7**, plus **P-a/-b/-c** for the TTS/barge-in
+ sub-phases). The phase map is in [`../docs/esp32-agui-plan.md`](../docs/esp32-agui-plan.md); keep
+ using the same tags when extending a feature.
+- `device_tools` and `chat_ui` avoid a circular dependency by having `main` wire runtime callbacks
+ after init (`device_tools_set_show_image_handler`, `chat_ui_set_*_cb`) rather than one `#include`
+ the other. Preserve this when adding cross-component hooks.
diff --git a/esp32-agui/components/app_cfg/include/app_cfg.h b/esp32-agui/components/app_cfg/include/app_cfg.h
index 819e968..288550a 100644
--- a/esp32-agui/components/app_cfg/include/app_cfg.h
+++ b/esp32-agui/components/app_cfg/include/app_cfg.h
@@ -20,7 +20,6 @@ extern "C" {
#define APP_CFG_TTS_VOL "tts_vol" // spoken-reply volume 0-100 (volume buttons; default 90)
#define APP_CFG_SCREEN_TO "scr_to" // screen blank timeout, seconds (default 60; 0 = always on)
#define APP_CFG_IDLE_ANIM "idle_anim" // idle screensaver flag "0"/"1" (default 0; pulses alarm image)
-
// Max stored value length (incl. NUL). One number shared by the portal form, the NVS
// writer, and every reader so a long value can't pass provisioning then fail to load.
#define APP_CFG_VAL_MAX 192
diff --git a/esp32-agui/components/chat_ui/CMakeLists.txt b/esp32-agui/components/chat_ui/CMakeLists.txt
index dd306c2..0e55014 100644
--- a/esp32-agui/components/chat_ui/CMakeLists.txt
+++ b/esp32-agui/components/chat_ui/CMakeLists.txt
@@ -1,6 +1,5 @@
idf_component_register(
- SRCS "chat_ui.c"
- INCLUDE_DIRS "include"
- REQUIRES json lvgl esp32_s3_touch_amoled_1_8 device_tools alarm_img esp_timer)
-# P3 adds the LVGL chat list/bubbles; P4 status; P6 interrupt prompt + lv_qrcode.
-# P7 screen-power saver: device_tools (PWR-key poll) + esp_timer (monotonic ms).
+ SRCS "chat_ui.c" "face_engine.c" "companion_pages.c"
+ INCLUDE_DIRS "include" "."
+ REQUIRES json lvgl esp32_s3_touch_amoled_1_8 device_tools alarm_img esp_timer
+ esp_http_client esp_jpeg app_cfg imu_qmi8658)
diff --git a/esp32-agui/components/chat_ui/chat_ui.c b/esp32-agui/components/chat_ui/chat_ui.c
index c815122..6f7a551 100644
--- a/esp32-agui/components/chat_ui/chat_ui.c
+++ b/esp32-agui/components/chat_ui/chat_ui.c
@@ -5,15 +5,21 @@
// each wraps its LVGL work in bsp_display_lock()/unlock(). Interrupt prompt + QR are P6.
#include "chat_ui.h"
+#include
#include
#include "freertos/FreeRTOS.h"
#include "freertos/task.h"
#include "esp_log.h"
#include "esp_timer.h"
+#include "esp_heap_caps.h"
+#include "esp_http_client.h"
+#include "jpeg_decoder.h"
#include "bsp/esp32_s3_touch_amoled_1_8.h"
#include "device_tools.h"
#include "alarm_img.h"
#include "lvgl.h"
+#include "face_engine.h"
+#include "companion_pages.h"
static const char *TAG = "chat_ui";
@@ -40,10 +46,63 @@ static const char *TAG = "chat_ui";
static lv_obj_t *s_chat; // scrollable flex column of message rows
static lv_obj_t *s_status; // top status label
+static lv_obj_t *s_status_box; // clipping box for status (kept above the face)
static lv_obj_t *s_assist_lbl; // label of the in-progress assistant bubble (streaming)
static char s_assist_buf[2048]; // accumulated assistant text (to re-measure on each delta)
static size_t s_assist_len;
+// --- show_image overlay ----------------------------------------------------------------------
+static lv_obj_t *s_img_overlay;
+static lv_obj_t *s_img_view;
+static lv_img_dsc_t s_img_dsc;
+static uint8_t *s_img_pixels; // RGB565 in PSRAM
+static uint16_t s_img_w, s_img_h;
+
+// Forward decls used by page-tap / face helpers (defined with screen-power / talk code below).
+static volatile bool s_alarm_active;
+static bool s_talk_armed;
+
+// --- NIMO-style face + companion pages (Eyes/Clock/Chat) ------------------------------------
+static void chat_ui_page_tap_cb(lv_event_t *e)
+{
+ if (s_alarm_active) return;
+ if (s_talk_armed) return; // long-press PTT owns this gesture
+ if (lv_event_get_code(e) != LV_EVENT_SHORT_CLICKED) return;
+ companion_pages_cycle();
+}
+
+void chat_ui_set_face(chat_ui_face_mood_t mood)
+{
+ if (!bsp_display_lock(1000)) {
+ face_engine_set_mood(mood == CHAT_UI_FACE_HIDDEN ? CHAT_UI_FACE_IDLE : mood);
+ return;
+ }
+ if (mood == CHAT_UI_FACE_HIDDEN) {
+ face_engine_set_mood(CHAT_UI_FACE_IDLE);
+ companion_pages_show(COMPANION_PAGE_CHAT);
+ } else {
+ face_engine_set_mood(mood);
+ companion_page_t cur = companion_pages_current();
+ if (cur == COMPANION_PAGE_CHAT || cur == COMPANION_PAGE_EYES)
+ companion_pages_show(COMPANION_PAGE_EYES);
+ if (s_status_box) lv_obj_move_foreground(s_status_box);
+ }
+ bsp_display_unlock();
+}
+
+static void face_from_status(const char *text)
+{
+ if (!text || !text[0]) return;
+ companion_pages_on_voice_status(text);
+ if (!strncmp(text, "Listening", 9)) chat_ui_set_face(CHAT_UI_FACE_LISTEN);
+ else if (!strncmp(text, "Thinking", 8) ||
+ !strncmp(text, "Reasoning", 9) ||
+ !strncmp(text, "Using ", 6)) chat_ui_set_face(CHAT_UI_FACE_THINK);
+ else if (!strncmp(text, "Speaking", 8)) chat_ui_set_face(CHAT_UI_FACE_SPEAK);
+ else if (!strncmp(text, "Hold ", 5) ||
+ !strcmp(text, "Ready")) chat_ui_set_face(CHAT_UI_FACE_HAPPY);
+}
+
// Make text renderable by the (Latin-only) Montserrat font: transliterate common punctuation
// (em/en dash, curly quotes, ellipsis, nbsp) to ASCII, and DROP any other multi-byte codepoint
// (emoji, CJK, accents) the font has no glyph for — otherwise they render as tofu. v1 defers
@@ -160,13 +219,13 @@ esp_err_t chat_ui_init(void)
// Status line: a fixed-width clipping box holding a single-line label. Long live transcripts
// scroll left so the tail (latest words) stays on screen; short messages center. (clips to the
// box edges, respecting the round-corner safe zone — not the screen edge.)
- lv_obj_t *box = lv_obj_create(scr);
- lv_obj_remove_style_all(box);
- lv_obj_set_size(box, CHAT_W, STATUS_H);
- lv_obj_align(box, LV_ALIGN_TOP_MID, 0, SAFE_INSET);
- lv_obj_clear_flag(box, LV_OBJ_FLAG_SCROLLABLE);
+ s_status_box = lv_obj_create(scr);
+ lv_obj_remove_style_all(s_status_box);
+ lv_obj_set_size(s_status_box, CHAT_W, STATUS_H);
+ lv_obj_align(s_status_box, LV_ALIGN_TOP_MID, 0, SAFE_INSET);
+ lv_obj_clear_flag(s_status_box, LV_OBJ_FLAG_SCROLLABLE);
- s_status = lv_label_create(box);
+ s_status = lv_label_create(s_status_box);
lv_label_set_long_mode(s_status, LV_LABEL_LONG_CLIP); // one line, full content width, no dots
lv_obj_set_style_text_color(s_status, lv_palette_main(LV_PALETTE_GREY), 0);
lv_obj_set_style_text_font(s_status, CHAT_FONT, 0);
@@ -184,6 +243,10 @@ esp_err_t chat_ui_init(void)
lv_obj_set_scroll_dir(s_chat, LV_DIR_VER);
lv_obj_set_scrollbar_mode(s_chat, LV_SCROLLBAR_MODE_OFF);
+ face_engine_create(scr); // Eyes page (spring-physics face)
+ companion_pages_create(scr); // Clock overlay
+ lv_obj_move_foreground(s_status_box);
+
// Touch/scroll = activity for the screen-power saver. This event cb runs INSIDE the LVGL task
// (already holding lvgl_mutex), so it bumps s_last_activity_ms without ever taking a lock — i.e.
// it works even when screen_power_task can't grab the lock to read lv_disp_get_inactive_time().
@@ -198,10 +261,20 @@ esp_err_t chat_ui_init(void)
// (presses over the blank/status area) AND the chat list (which covers most of the screen — a press
// there goes to s_chat and does NOT bubble to scr). A drag on the chat still scrolls, because LVGL
// suppresses LONG_PRESSED once a scroll begins; a still hold fires it.
+ // Handle PRESS_LOST as well as RELEASED: a hold that ends any way other than a clean lift — the
+ // gesture turning into a scroll, focus change, or a stuck/phantom capacitive touch — emits
+ // LV_EVENT_PRESS_LOST, NOT RELEASED. Without it, s_talk_armed never clears and the app latches in
+ // "Listening..." forever (the release cb never fires). Register both exit events on both objects.
lv_obj_add_event_cb(scr, chat_ui_talk_evt_cb, LV_EVENT_LONG_PRESSED, NULL);
lv_obj_add_event_cb(scr, chat_ui_talk_evt_cb, LV_EVENT_RELEASED, NULL);
+ lv_obj_add_event_cb(scr, chat_ui_talk_evt_cb, LV_EVENT_PRESS_LOST, NULL);
lv_obj_add_event_cb(s_chat, chat_ui_talk_evt_cb, LV_EVENT_LONG_PRESSED, NULL);
lv_obj_add_event_cb(s_chat, chat_ui_talk_evt_cb, LV_EVENT_RELEASED, NULL);
+ lv_obj_add_event_cb(s_chat, chat_ui_talk_evt_cb, LV_EVENT_PRESS_LOST, NULL);
+
+ // Short tap cycles Eyes → Clock → Chat (long-press still owns PTT via s_talk_armed).
+ lv_obj_add_event_cb(scr, chat_ui_page_tap_cb, LV_EVENT_SHORT_CLICKED, NULL);
+ lv_obj_add_event_cb(s_chat, chat_ui_page_tap_cb, LV_EVENT_SHORT_CLICKED, NULL);
// Raise the touch scroll threshold so a STILL hold (with the few px of capacitive jitter) over the
// scrollable chat isn't read as a scroll — which would cancel the long-press. A deliberate drag
@@ -220,6 +293,7 @@ esp_err_t chat_ui_init(void)
void chat_ui_add_user(const char *text)
{
chat_ui_note_activity();
+ chat_ui_set_face(CHAT_UI_FACE_HIDDEN); // flip to chat for reading
if (!s_chat || !bsp_display_lock(1000)) return;
add_bubble(true, COL_USER, text);
scroll_bottom();
@@ -229,6 +303,7 @@ void chat_ui_add_user(const char *text)
void *chat_ui_begin_assistant(void)
{
chat_ui_note_activity();
+ chat_ui_set_face(CHAT_UI_FACE_HIDDEN); // flip to chat while the reply streams
if (!s_chat || !bsp_display_lock(1000)) return NULL;
s_assist_buf[0] = '\0'; s_assist_len = 0;
s_assist_lbl = add_bubble(false, COL_ASSIST, "");
@@ -283,6 +358,7 @@ void chat_ui_append_assistant(const char *delta)
void chat_ui_status(const char *text)
{
chat_ui_note_activity();
+ face_from_status(text);
if (!s_status || !bsp_display_lock(1000)) return;
char clean[1024]; // live transcript can be long
sanitize(text ? text : "", clean, sizeof clean);
@@ -291,6 +367,7 @@ void chat_ui_status(const char *text)
lv_coord_t lw = lv_obj_get_width(s_status), bw = CHAT_W;
lv_obj_set_x(s_status, lw > bw ? (bw - lw) // overflow → show the tail (scroll left)
: (bw - lw) / 2); // fits → center
+ if (s_status_box) lv_obj_move_foreground(s_status_box);
bsp_display_unlock();
}
@@ -308,7 +385,6 @@ static bool s_idle_disabled; // "always on": never bl
static bool s_screen_on = true;
static bool s_force_off_armed; // PWR-tapped off; stays off until newer activity
static uint32_t s_force_off_ms; // when the force-off press happened
-static volatile bool s_alarm_active; // a timer is ringing → it owns the screen
static lv_obj_t *s_alarm_overlay; // full-screen black overlay shown while ringing
static lv_obj_t *s_alarm_ring; // default graphic: red ring (toggled to flash)
static lv_obj_t *s_alarm_img; // user graphic (if uploaded): pulsed via img_opa
@@ -514,7 +590,6 @@ void chat_ui_set_power_cb(chat_ui_power_cb cb) { s_power_cb = cb; }
// with the display lock already held, so it only flips flags + calls the (non-blocking) app callback.
static chat_ui_talk_cb s_talk_cb;
static void *s_talk_ctx;
-static bool s_talk_armed;
void chat_ui_set_talk_cb(chat_ui_talk_cb cb, void *ctx) { s_talk_cb = cb; s_talk_ctx = ctx; }
static void chat_ui_talk_evt_cb(lv_event_t *e)
@@ -526,9 +601,9 @@ static void chat_ui_talk_evt_cb(lv_event_t *e)
ESP_LOGI(TAG, "touch-to-talk: hold");
s_talk_armed = true;
if (s_talk_cb) s_talk_cb(1, s_talk_ctx); // hold start → like a BOOT down
- } else if (code == LV_EVENT_RELEASED) {
- if (s_talk_armed && s_talk_cb) s_talk_cb(0, s_talk_ctx); // release → stop + run
- s_talk_armed = false;
+ } else if (code == LV_EVENT_RELEASED || code == LV_EVENT_PRESS_LOST) {
+ if (s_talk_armed && s_talk_cb) s_talk_cb(0, s_talk_ctx); // release/press-lost → stop + run
+ s_talk_armed = false; // always disarm, even if never armed
}
}
@@ -640,6 +715,185 @@ void chat_ui_screen_power_start(int idle_timeout_s)
(unsigned)(s_idle_timeout_ms / 1000));
}
+// --- show_image: HTTPS JPEG → RGB565 overlay ------------------------------------------------
+#define IMG_DL_MAX (300 * 1024)
+#define IMG_DISP_MAX_W 320
+#define IMG_DISP_MAX_H 320
+
+static void img_overlay_dismiss(void)
+{
+ if (bsp_display_lock(1000)) {
+ if (s_img_overlay) {
+ lv_obj_del(s_img_overlay);
+ s_img_overlay = NULL;
+ s_img_view = NULL;
+ }
+ if (s_img_pixels) {
+ heap_caps_free(s_img_pixels);
+ s_img_pixels = NULL;
+ }
+ s_img_w = s_img_h = 0;
+ if (s_status_box) lv_obj_move_foreground(s_status_box);
+ bsp_display_unlock();
+ }
+ chat_ui_note_activity();
+}
+
+static void img_overlay_evt(lv_event_t *e)
+{
+ if (lv_event_get_code(e) == LV_EVENT_CLICKED) img_overlay_dismiss();
+}
+
+static esp_err_t http_get_psram(const char *url, uint8_t **out, int *out_len)
+{
+ *out = NULL;
+ *out_len = 0;
+ uint8_t *buf = heap_caps_malloc(IMG_DL_MAX, MALLOC_CAP_SPIRAM | MALLOC_CAP_8BIT);
+ if (!buf) return ESP_ERR_NO_MEM;
+
+ esp_http_client_config_t cfg = {
+ .url = url,
+ .timeout_ms = 20000,
+ .buffer_size = 4096,
+ .buffer_size_tx = 1024,
+ };
+ esp_http_client_handle_t client = esp_http_client_init(&cfg);
+ if (!client) { heap_caps_free(buf); return ESP_FAIL; }
+
+ esp_err_t err = esp_http_client_open(client, 0);
+ if (err != ESP_OK) {
+ ESP_LOGE(TAG, "http open: %s", esp_err_to_name(err));
+ esp_http_client_cleanup(client);
+ heap_caps_free(buf);
+ return err;
+ }
+ (void)esp_http_client_fetch_headers(client);
+ int status = esp_http_client_get_status_code(client);
+ int total = 0;
+ while (total < IMG_DL_MAX) {
+ int n = esp_http_client_read(client, (char *)buf + total, IMG_DL_MAX - total);
+ if (n < 0) { err = ESP_FAIL; break; }
+ if (n == 0) break;
+ total += n;
+ }
+ esp_http_client_close(client);
+ esp_http_client_cleanup(client);
+
+ if (err != ESP_OK || status < 200 || status >= 300 || total < 16) {
+ ESP_LOGE(TAG, "http get failed status=%d len=%d", status, total);
+ heap_caps_free(buf);
+ return ESP_FAIL;
+ }
+ *out = buf;
+ *out_len = total;
+ return ESP_OK;
+}
+
+static esp_err_t jpeg_to_rgb565(const uint8_t *jpg, int jpg_len,
+ uint8_t **pixels, uint16_t *w, uint16_t *h)
+{
+ // Largest-first: pick the biggest scale whose DECODED dims fit the display cap. esp_jpeg_get_image_info
+ // reports full-resolution width/height regardless of out_scale (only output_len is scaled), so the
+ // fit test must divide by the scale divisor. The old code compared full-res against the cap, so any
+ // image wider than the cap was rejected at every scale (never shown) and small ones were shrunk to 1/4.
+ static const struct { esp_jpeg_image_scale_t scale; int div; } scales[] = {
+ { JPEG_IMAGE_SCALE_0, 1 },
+ { JPEG_IMAGE_SCALE_1_2, 2 },
+ { JPEG_IMAGE_SCALE_1_4, 4 },
+ { JPEG_IMAGE_SCALE_1_8, 8 },
+ };
+ for (size_t i = 0; i < sizeof scales / sizeof scales[0]; i++) {
+ esp_jpeg_image_cfg_t probe = {
+ .indata = (uint8_t *)jpg,
+ .indata_size = (uint32_t)jpg_len,
+ .outbuf = NULL,
+ .outbuf_size = 0,
+ .out_format = JPEG_IMAGE_FORMAT_RGB565,
+ .out_scale = scales[i].scale,
+ .flags = { .swap_color_bytes = 1 },
+ };
+ esp_jpeg_image_output_t info = {0};
+ if (esp_jpeg_get_image_info(&probe, &info) != ESP_OK || info.width == 0 || info.height == 0)
+ continue;
+ if (info.width / scales[i].div > IMG_DISP_MAX_W || info.height / scales[i].div > IMG_DISP_MAX_H)
+ continue; // decoded size still too big at this scale → try the next smaller scale
+
+ size_t need = info.output_len ? info.output_len : (size_t)info.width * info.height * 2;
+ uint8_t *out = heap_caps_malloc(need, MALLOC_CAP_SPIRAM | MALLOC_CAP_8BIT);
+ if (!out) continue;
+
+ esp_jpeg_image_cfg_t cfg = probe;
+ cfg.outbuf = out;
+ cfg.outbuf_size = need;
+ esp_jpeg_image_output_t decoded = {0};
+ if (esp_jpeg_decode(&cfg, &decoded) == ESP_OK && decoded.width > 0) {
+ *pixels = out;
+ *w = decoded.width;
+ *h = decoded.height;
+ ESP_LOGI(TAG, "jpeg decoded %ux%u scale=1/%d", (unsigned)*w, (unsigned)*h, scales[i].div);
+ return ESP_OK;
+ }
+ heap_caps_free(out);
+ }
+ return ESP_FAIL;
+}
+
+esp_err_t chat_ui_show_image(const char *url)
+{
+ if (!url || !url[0]) return ESP_ERR_INVALID_ARG;
+ chat_ui_note_activity();
+ img_overlay_dismiss(); // replace any previous image
+
+ uint8_t *jpg = NULL;
+ int jpg_len = 0;
+ esp_err_t err = http_get_psram(url, &jpg, &jpg_len);
+ if (err != ESP_OK) return err;
+
+ uint8_t *pix = NULL;
+ uint16_t w = 0, h = 0;
+ err = jpeg_to_rgb565(jpg, jpg_len, &pix, &w, &h);
+ heap_caps_free(jpg);
+ if (err != ESP_OK) {
+ ESP_LOGE(TAG, "jpeg decode failed");
+ return err;
+ }
+
+ if (!bsp_display_lock(2000)) {
+ heap_caps_free(pix);
+ return ESP_ERR_TIMEOUT;
+ }
+ s_img_pixels = pix;
+ s_img_w = w;
+ s_img_h = h;
+ memset(&s_img_dsc, 0, sizeof s_img_dsc);
+ s_img_dsc.header.always_zero = 0;
+ s_img_dsc.header.cf = LV_IMG_CF_TRUE_COLOR;
+ s_img_dsc.header.w = w;
+ s_img_dsc.header.h = h;
+ s_img_dsc.data_size = (uint32_t)w * h * 2;
+ s_img_dsc.data = s_img_pixels;
+
+ s_img_overlay = lv_obj_create(lv_scr_act());
+ lv_obj_remove_style_all(s_img_overlay);
+ lv_obj_set_size(s_img_overlay, lv_pct(100), lv_pct(100));
+ lv_obj_set_style_bg_color(s_img_overlay, lv_color_black(), 0);
+ lv_obj_set_style_bg_opa(s_img_overlay, LV_OPA_COVER, 0);
+ lv_obj_clear_flag(s_img_overlay, LV_OBJ_FLAG_SCROLLABLE);
+ lv_obj_add_flag(s_img_overlay, LV_OBJ_FLAG_CLICKABLE);
+ lv_obj_add_event_cb(s_img_overlay, img_overlay_evt, LV_EVENT_CLICKED, NULL);
+
+ s_img_view = lv_img_create(s_img_overlay);
+ lv_img_set_src(s_img_view, &s_img_dsc);
+ lv_obj_center(s_img_view);
+ lv_obj_clear_flag(s_img_view, LV_OBJ_FLAG_SCROLLABLE);
+
+ if (s_status_box) lv_obj_move_foreground(s_status_box);
+ bsp_display_unlock();
+ chat_ui_set_face(CHAT_UI_FACE_HIDDEN);
+ ESP_LOGI(TAG, "show_image ok %ux%u", (unsigned)w, (unsigned)h);
+ return ESP_OK;
+}
+
// --- later phases ---
void chat_ui_show_qr(const char *data) { (void)data; } // P6
void chat_ui_idle_timer(int s, const char *label) { (void)s; (void)label; } // P7: set_timer countdown
diff --git a/esp32-agui/components/chat_ui/companion_pages.c b/esp32-agui/components/chat_ui/companion_pages.c
new file mode 100644
index 0000000..1a3d87c
--- /dev/null
+++ b/esp32-agui/components/chat_ui/companion_pages.c
@@ -0,0 +1,150 @@
+// Companion page overlays: Clock (+ Eyes/Chat orchestration).
+// Weather lives on the agent via web_search — no on-device OWM page.
+#include "companion_pages.h"
+
+#include
+#include
+#include
+
+#include "esp_log.h"
+#include "device_tools.h"
+#include "face_engine.h"
+
+static const char *TAG = "companion";
+
+#define SAFE_INSET 22
+
+static companion_page_t s_page = COMPANION_PAGE_EYES;
+static lv_obj_t *s_clock;
+static lv_obj_t *s_clock_time;
+static lv_obj_t *s_clock_date;
+static lv_obj_t *s_clock_batt;
+static lv_timer_t *s_page_timer;
+
+static void set_hidden(lv_obj_t *o, bool hide)
+{
+ if (!o) return;
+ if (hide) lv_obj_add_flag(o, LV_OBJ_FLAG_HIDDEN);
+ else lv_obj_clear_flag(o, LV_OBJ_FLAG_HIDDEN);
+}
+
+static lv_obj_t *make_page(lv_obj_t *scr)
+{
+ lv_obj_t *p = lv_obj_create(scr);
+ lv_obj_remove_style_all(p);
+ lv_obj_set_size(p, lv_pct(100), lv_pct(100));
+ lv_obj_set_style_bg_color(p, lv_color_black(), 0);
+ lv_obj_set_style_bg_opa(p, LV_OPA_COVER, 0);
+ lv_obj_clear_flag(p, LV_OBJ_FLAG_SCROLLABLE | LV_OBJ_FLAG_CLICKABLE);
+ lv_obj_add_flag(p, LV_OBJ_FLAG_HIDDEN);
+ return p;
+}
+
+static void clock_refresh(void)
+{
+ if (!s_clock_time) return;
+ time_t now = time(NULL);
+ struct tm lt;
+ localtime_r(&now, <);
+ char tbuf[32], dbuf[48];
+ strftime(tbuf, sizeof tbuf, "%I:%M:%S %p", <);
+ // Strip leading zero from hour for a cleaner look
+ if (tbuf[0] == '0') memmove(tbuf, tbuf + 1, strlen(tbuf));
+ strftime(dbuf, sizeof dbuf, "%a %b %d", <);
+ lv_label_set_text(s_clock_time, (now < 1700000000) ? "--:--:--" : tbuf);
+ lv_label_set_text(s_clock_date, (now < 1700000000) ? "Syncing time..." : dbuf);
+
+ int pct = -1;
+ bool plugged = false;
+ char status[40] = {0};
+ if (device_tools_battery_read(&pct, &plugged, status, sizeof status) && pct >= 0) {
+ char b[64];
+ snprintf(b, sizeof b, "%d%% %s", pct, plugged ? "USB" : "Batt");
+ lv_label_set_text(s_clock_batt, b);
+ } else {
+ lv_label_set_text(s_clock_batt, "");
+ }
+}
+
+static void page_timer_cb(lv_timer_t *t)
+{
+ (void)t;
+ if (s_page == COMPANION_PAGE_CLOCK) clock_refresh();
+}
+
+static void apply_visibility(void)
+{
+ bool eyes = (s_page == COMPANION_PAGE_EYES);
+ bool clock = (s_page == COMPANION_PAGE_CLOCK);
+ // Chat = none of the overlays
+
+ face_engine_set_visible(eyes);
+ face_engine_set_active(eyes);
+ set_hidden(s_clock, !clock);
+
+ if (clock) clock_refresh();
+}
+
+void companion_pages_create(lv_obj_t *scr)
+{
+ s_clock = make_page(scr);
+ s_clock_time = lv_label_create(s_clock);
+ lv_obj_set_style_text_color(s_clock_time, lv_color_white(), 0);
+ lv_obj_set_style_text_font(s_clock_time, &lv_font_montserrat_20, 0);
+ lv_label_set_text(s_clock_time, "--:--:--");
+ lv_obj_align(s_clock_time, LV_ALIGN_CENTER, 0, -30);
+
+ s_clock_date = lv_label_create(s_clock);
+ lv_obj_set_style_text_color(s_clock_date, lv_palette_main(LV_PALETTE_GREY), 0);
+ lv_obj_set_style_text_font(s_clock_date, &lv_font_montserrat_20, 0);
+ lv_label_set_text(s_clock_date, "");
+ lv_obj_align(s_clock_date, LV_ALIGN_CENTER, 0, 30);
+
+ s_clock_batt = lv_label_create(s_clock);
+ lv_obj_set_style_text_color(s_clock_batt, lv_palette_main(LV_PALETTE_GREY), 0);
+ lv_obj_set_style_text_font(s_clock_batt, &lv_font_montserrat_20, 0);
+ lv_label_set_text(s_clock_batt, "");
+ lv_obj_align(s_clock_batt, LV_ALIGN_BOTTOM_MID, 0, -SAFE_INSET - 8);
+
+ s_page_timer = lv_timer_create(page_timer_cb, 500, NULL);
+ s_page = COMPANION_PAGE_EYES;
+ apply_visibility();
+}
+
+void companion_pages_show(companion_page_t page)
+{
+ s_page = page;
+ apply_visibility();
+}
+
+void companion_pages_cycle(void)
+{
+ s_page = (companion_page_t)((s_page + 1) % 3);
+ ESP_LOGI(TAG, "page → %d", (int)s_page);
+ apply_visibility();
+}
+
+companion_page_t companion_pages_current(void) { return s_page; }
+
+void companion_pages_on_voice_status(const char *text)
+{
+ if (!text || !text[0]) return;
+ if (!strncmp(text, "Listening", 9) ||
+ !strncmp(text, "Thinking", 8) ||
+ !strncmp(text, "Reasoning", 9) ||
+ !strncmp(text, "Using ", 6) ||
+ !strncmp(text, "Speaking", 8) ||
+ !strncmp(text, "Hold ", 5)) {
+ companion_pages_show(COMPANION_PAGE_EYES);
+ }
+}
+
+void companion_pages_on_chat_stream(void)
+{
+ companion_pages_show(COMPANION_PAGE_CHAT);
+}
+
+void companion_pages_tick(void)
+{
+ // no-op; lv_timer handles refresh
+}
diff --git a/esp32-agui/components/chat_ui/companion_pages.h b/esp32-agui/components/chat_ui/companion_pages.h
new file mode 100644
index 0000000..b5329bc
--- /dev/null
+++ b/esp32-agui/components/chat_ui/companion_pages.h
@@ -0,0 +1,28 @@
+// Eyes / Clock / Chat page shell for the companion UI.
+#pragma once
+
+#include
+#include "lvgl.h"
+#include "chat_ui.h"
+
+#ifdef __cplusplus
+extern "C" {
+#endif
+
+typedef enum {
+ COMPANION_PAGE_EYES = 0,
+ COMPANION_PAGE_CLOCK,
+ COMPANION_PAGE_CHAT,
+} companion_page_t;
+
+void companion_pages_create(lv_obj_t *scr);
+void companion_pages_cycle(void);
+void companion_pages_show(companion_page_t page);
+companion_page_t companion_pages_current(void);
+void companion_pages_on_voice_status(const char *text); // Listening/Speaking/… → Eyes
+void companion_pages_on_chat_stream(void); // user/assistant text → Chat
+void companion_pages_tick(void); // clock refresh (from face/status)
+
+#ifdef __cplusplus
+}
+#endif
diff --git a/esp32-agui/components/chat_ui/face_engine.c b/esp32-agui/components/chat_ui/face_engine.c
new file mode 100644
index 0000000..4a94fed
--- /dev/null
+++ b/esp32-agui/components/chat_ui/face_engine.c
@@ -0,0 +1,488 @@
+// NIMO-faithful eyes/mouth for the AMOLED companion.
+//
+// Ported from the reference NIMO Arduino sketch (SH1106 128x64 mono OLED): the same
+// rounded-rect eyes, rounded-rect pupils + white glint, overlay-masked expressions
+// (happy/sleepy/sad/think), angry eyes with diagonal brows + "!" marks + teeth, dizzy
+// circle-eyes with two orbiting pupils + star bitmaps, and NIMO's pixel-parabola mouths
+// with heart/zzz particles.
+//
+// NIMO draws pixels straight into a framebuffer, so we reproduce that exactly: a small
+// software GFX (rounded rect / rect / circle / line / bitmap) writes white/black into a
+// PSRAM canvas, and every NIMO coordinate is scaled ~2.9x (128 -> 368) at draw time. The
+// canvas is one lv_canvas invalidated each tick. Physics (spring eyes) and mood triggers
+// (tilt -> suspicious, shake -> dizzy -> angry) are unchanged.
+#include "face_engine.h"
+
+#include
+#include
+#include
+
+#include "esp_random.h"
+#include "esp_timer.h"
+#include "esp_heap_caps.h"
+#include "esp_log.h"
+#include "imu_qmi8658.h"
+#include "bsp/esp32_s3_touch_amoled_1_8.h"
+
+static const char *TAG = "face_engine";
+
+#define FACE_TIMER_MS 70
+#define NIMO_W 128 // reference screen the coordinates are authored in
+#define NIMO_H 64
+#define SHAKE_THRESH 0.55f
+
+#define WHITE 0xFFFFu
+#define BLACK 0x0000u
+
+// Internal NIMO-style moods (drawing vocabulary). Voice states reuse the NORMAL path
+// with small tweaks (bigger eyes / look-up lid / talking mouth).
+enum {
+ NM_NORMAL = 0, NM_LISTEN, NM_THINK, NM_SPEAK,
+ NM_HAPPY, NM_SLEEPY, NM_SAD, NM_ANGRY, NM_SURPRISED, NM_DIZZY, NM_LOVE, NM_SUSPICIOUS,
+};
+
+// ---- 1bpp bitmaps copied verbatim from the NIMO sketch (16x16, MSB-first) ----
+static const uint8_t bmp_heart[] = {
+ 0x00,0x00,0x0c,0x60,0x1e,0xf0,0x3f,0xf8,0x7f,0xfc,0x7f,0xfc,0x7f,0xfc,0x3f,0xf8,
+ 0x1f,0xf0,0x0f,0xe0,0x07,0xc0,0x03,0x80,0x01,0x00,0x00,0x00,0x00,0x00,0x00,0x00
+};
+static const uint8_t bmp_zzz[] = {
+ 0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x3c,0x00,0x0c,0x00,0x18,0x00,0x30,0x00,0x7e,
+ 0x00,0x00,0x3c,0x00,0x0c,0x00,0x18,0x00,0x30,0x00,0x7c,0x00,0x00,0x00,0x00,0x00
+};
+static const uint8_t bmp_dizzy_stars[] = {
+ 0x08,0x20,0x14,0x50,0x22,0x88,0x41,0x04,0x82,0x02,0x41,0x04,0x22,0x88,0x14,0x50,
+ 0x08,0x20,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00
+};
+static const uint8_t bmp_angry_mark[] = {
+ 0x00,0x00,0x00,0x00,0x08,0x00,0x1c,0x00,0x3e,0x00,0x7f,0x00,0x3e,0x00,0x1c,0x00,
+ 0x08,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00
+};
+
+// ---- eye physics (NIMO coordinate space) ----
+typedef struct {
+ float x, y, w, h; // current
+ float tx, ty, tw, th; // target
+ float vx, vy, vw, vh; // velocity
+ float px, py; // pupil offset (current)
+ float tpx, tpy; // pupil offset (target)
+ float pvx, pvy;
+} eye_t;
+
+static lv_obj_t *s_root;
+static lv_obj_t *s_canvas;
+static uint16_t *s_fb; // W*H canvas (white/black + red for angry)
+static int s_W, s_H;
+static float s_scale;
+static uint16_t s_red; // RGB565 red as stored in the canvas (swap-aware)
+static lv_timer_t *s_timer;
+
+static chat_ui_face_mood_t s_mood = CHAT_UI_FACE_IDLE;
+static chat_ui_face_mood_t s_override = CHAT_UI_FACE_IDLE; // dizzy/angry/suspicious from motion
+static int64_t s_override_until_ms;
+static int64_t s_shake_start_ms;
+static bool s_active = true;
+static bool s_blinking;
+static uint32_t s_blink_last_ms, s_next_blink_ms;
+static int s_speak_phase;
+static eye_t s_L, s_R;
+
+static uint32_t face_now_ms(void) { return (uint32_t)(esp_timer_get_time() / 1000ULL); }
+static float clampf(float v, float lo, float hi) { return v < lo ? lo : (v > hi ? hi : v); }
+
+// ================= software GFX over the PSRAM canvas (real/buffer pixels) =============
+#define SC(v) ((int)lroundf((v) * s_scale))
+
+static inline void rpx(int x, int y, uint16_t c)
+{
+ if ((unsigned)x < (unsigned)s_W && (unsigned)y < (unsigned)s_H) s_fb[y * s_W + x] = c;
+}
+
+static void rfill(int x, int y, int w, int h, uint16_t c)
+{
+ if (x < 0) { w += x; x = 0; }
+ if (y < 0) { h += y; y = 0; }
+ if (x + w > s_W) w = s_W - x;
+ if (y + h > s_H) h = s_H - y;
+ for (int j = 0; j < h; j++) {
+ uint16_t *p = &s_fb[(y + j) * s_W + x];
+ for (int i = 0; i < w; i++) p[i] = c;
+ }
+}
+
+static void rfill_circle(int cx, int cy, int r, uint16_t c)
+{
+ if (r < 0) return;
+ for (int dy = -r; dy <= r; dy++) {
+ int dx = (int)lround(sqrt((double)r * r - (double)dy * dy));
+ rfill(cx - dx, cy + dy, 2 * dx + 1, 1, c);
+ }
+}
+
+static void rdraw_line(int x0, int y0, int x1, int y1, uint16_t c)
+{
+ int dx = abs(x1 - x0), sx = x0 < x1 ? 1 : -1;
+ int dy = -abs(y1 - y0), sy = y0 < y1 ? 1 : -1;
+ int err = dx + dy;
+ for (;;) {
+ rpx(x0, y0, c);
+ if (x0 == x1 && y0 == y1) break;
+ int e2 = 2 * err;
+ if (e2 >= dy) { err += dy; x0 += sx; }
+ if (e2 <= dx) { err += dx; y0 += sy; }
+ }
+}
+
+static void rfill_round_rect(int x, int y, int w, int h, int r, uint16_t c)
+{
+ if (w <= 0 || h <= 0) return;
+ if (r > w / 2) r = w / 2;
+ if (r > h / 2) r = h / 2;
+ if (r <= 0) { rfill(x, y, w, h, c); return; }
+ for (int j = 0; j < h; j++) {
+ int inset = 0;
+ if (j < r) {
+ int k = r - 1 - j;
+ inset = r - (int)floor(sqrt((double)r * r - (double)k * k));
+ } else if (j >= h - r) {
+ int k = r - 1 - (h - 1 - j);
+ inset = r - (int)floor(sqrt((double)r * r - (double)k * k));
+ }
+ rfill(x + inset, y + j, w - 2 * inset, 1, c);
+ }
+}
+
+// NIMO-coordinate wrappers (scaled to the panel)
+static void gfx_rrect(float x, float y, float w, float h, float r, uint16_t c)
+{ rfill_round_rect(SC(x), SC(y), SC(w), SC(h), SC(r), c); }
+static void gfx_rect(float x, float y, float w, float h, uint16_t c)
+{ rfill(SC(x), SC(y), SC(w), SC(h), c); }
+static void gfx_circle(float cx, float cy, float r, uint16_t c)
+{ rfill_circle(SC(cx), SC(cy), SC(r), c); }
+static void gfx_line(float x0, float y0, float x1, float y1, uint16_t c)
+{ rdraw_line(SC(x0), SC(y0), SC(x1), SC(y1), c); }
+static void gfx_px(float x, float y, uint16_t c) // one NIMO pixel = one scaled block
+{ int b = (int)ceilf(s_scale); rfill(SC(x), SC(y), b, b, c); }
+
+static void gfx_bitmap(float x, float y, const uint8_t *bmp, int bw, int bh, uint16_t c)
+{
+ int bytes_per_row = (bw + 7) / 8;
+ int b = (int)ceilf(s_scale);
+ int ox = SC(x), oy = SC(y);
+ for (int j = 0; j < bh; j++)
+ for (int i = 0; i < bw; i++)
+ if (bmp[j * bytes_per_row + i / 8] & (0x80 >> (i & 7)))
+ rfill(ox + (int)lroundf(i * s_scale), oy + (int)lroundf(j * s_scale), b, b, c);
+}
+
+// Solid slanted brow/lid band across the top of an eye (used for sad + angry, which NIMO
+// draws as a stack of diagonal lines — filled here so it stays solid when scaled up).
+static void gfx_slant_top(float ix, float iy, float iw, float drop, float thick, int down)
+{
+ int x0 = SC(ix), x1 = SC(ix + iw);
+ if (x1 <= x0) return;
+ for (int x = x0; x <= x1; x++) {
+ float f = (float)(x - x0) / (float)(x1 - x0);
+ float yy = iy + (down ? f * drop : (1.0f - f) * drop);
+ rfill(x, SC(yy), 1, SC(thick), BLACK);
+ }
+}
+
+// ================================= NIMO draw functions ================================
+static void draw_normal_eye(const eye_t *e, bool is_left, int nm)
+{
+ float ix = e->x, iy = e->y, iw = e->w, ih = e->h;
+ float r = (iw < 20) ? 3 : 8;
+ gfx_rrect(ix, iy, iw, ih, r, WHITE);
+
+ float cx = ix + iw / 2, cy = iy + ih / 2;
+ float pw = iw / 2.2f, ph = ih / 2.2f;
+ float pxp = cx + e->px - pw / 2, pyp = cy + e->py - ph / 2;
+ if (pxp < ix) pxp = ix;
+ if (pxp + pw > ix + iw) pxp = ix + iw - pw;
+ if (pyp < iy) pyp = iy;
+ if (pyp + ph > iy + ih) pyp = iy + ih - ph;
+ gfx_rrect(pxp, pyp, pw, ph, r / 2.0f, BLACK);
+ if (iw > 15 && ih > 15) gfx_circle(pxp + pw - 4, pyp + 4, 2, WHITE); // glint
+
+ if (nm == NM_HAPPY || nm == NM_LOVE) gfx_rect(ix, iy + ih - 10, iw, 12, BLACK);
+ else if (nm == NM_SLEEPY) gfx_rect(ix, iy, iw, ih / 2, BLACK);
+ else if (nm == NM_THINK) gfx_rect(ix, iy, iw, ih / 3, BLACK);
+ else if (nm == NM_SUSPICIOUS && !is_left) gfx_rect(ix, iy, iw, ih / 3, BLACK); // skeptic brow
+ else if (nm == NM_SAD) gfx_slant_top(ix, iy, iw, 4, 8, is_left ? 1 : 0);
+}
+
+static void draw_angry_eye(const eye_t *e, bool is_left)
+{
+ float ix = e->x, iy = e->y, iw = e->w, ih = e->h;
+ gfx_rrect(ix, iy, iw, ih, 8, s_red); // angry -> red eyes
+
+ float cx = ix + iw / 2, cy = iy + ih / 2;
+ float ps = iw / 2.5f;
+ gfx_rrect(cx - 2 - ps / 2, cy - ps / 2, ps, ps, ps / 2, BLACK);
+
+ gfx_slant_top(ix - 2, iy - 2, iw + 4, 8, 8, is_left ? 1 : 0); // frown brow
+ gfx_bitmap(is_left ? ix - 12 : ix + iw - 4, iy - 6, bmp_angry_mark, 16, 16, WHITE);
+ for (int i = 0; i < 3; i++) // gritted teeth
+ gfx_line(ix + 2 + i * 4, iy + ih - 2, ix + 6 + i * 4, iy + ih - 6, WHITE);
+}
+
+static void draw_dizzy_eye(const eye_t *e)
+{
+ float ix = e->x, iy = e->y, iw = e->w;
+ float cx = ix + iw / 2, cy = iy + e->h / 2;
+ gfx_circle(cx, cy, iw / 2, WHITE);
+
+ float ang = (face_now_ms() - (uint32_t)s_shake_start_ms) * 0.025f;
+ float rad = iw / 3;
+ gfx_circle(cx + cosf(ang) * rad, cy + sinf(ang) * rad, 4, BLACK);
+ gfx_circle(cx + cosf(ang) * rad + 2, cy + sinf(ang) * rad - 2, 1, WHITE);
+ gfx_circle(cx - cosf(ang) * rad, cy - sinf(ang) * rad, 3, BLACK);
+}
+
+static void draw_mouth(int nm)
+{
+ const float mx = 64, my = 55;
+
+ if (nm == NM_DIZZY) { // woozy wave
+ for (int i = -9; i <= 9; i++)
+ gfx_px(mx + i, my + sinf(i * 0.7f + face_now_ms() * 0.03f) * 4, WHITE);
+ return;
+ }
+ if (nm == NM_ANGRY) { // gritted teeth
+ gfx_rect(mx - 9, my - 3, 18, 6, BLACK);
+ for (int i = -6; i <= 6; i += 3) gfx_line(mx + i, my - 3, mx + i, my + 3, WHITE);
+ return;
+ }
+ switch (nm) {
+ case NM_HAPPY: // bold smile
+ for (int i = -10; i <= 10; i++) {
+ float y = my - (i * i / 22.0f);
+ gfx_px(mx + i, y, WHITE);
+ gfx_px(mx + i, y - 1, WHITE);
+ }
+ break;
+ case NM_SAD: // frown
+ for (int i = -9; i <= 9; i++) {
+ float y = my + (i * i / 26.0f);
+ gfx_px(mx + i, y, WHITE);
+ gfx_px(mx + i, y + 1, WHITE);
+ }
+ break;
+ case NM_SURPRISED: // open "O"
+ gfx_circle(mx, my + 1, 6, WHITE);
+ gfx_circle(mx, my + 1, 4, BLACK);
+ break;
+ case NM_LISTEN: // small attentive "o"
+ gfx_circle(mx, my, 4, WHITE);
+ gfx_circle(mx, my, 2, BLACK);
+ break;
+ case NM_LOVE: // heart mouth
+ gfx_bitmap(mx - 8, my - 4, bmp_heart, 16, 16, WHITE);
+ break;
+ case NM_SPEAK: { // talking: open/close cavity
+ float h = 4 + (s_speak_phase % 3) * 6.0f; // 4 -> 10 -> 16
+ gfx_rrect(mx - 9, my - h / 2, 18, h, 4, WHITE);
+ float ih = h - 5;
+ if (ih >= 1) gfx_rrect(mx - 6, my - ih / 2, 12, ih, 3, BLACK);
+ break;
+ }
+ default: // NORMAL / THINK / SLEEPY / SUSPICIOUS
+ for (int i = -6; i <= 6; i++) gfx_px(mx + i, my - (i * i / 45.0f), WHITE);
+ break;
+ }
+}
+
+static int nimo_mood(chat_ui_face_mood_t m)
+{
+ switch (m) {
+ case CHAT_UI_FACE_LISTEN: return NM_LISTEN;
+ case CHAT_UI_FACE_THINK: return NM_THINK;
+ case CHAT_UI_FACE_SPEAK: return NM_SPEAK;
+ case CHAT_UI_FACE_HAPPY: return NM_HAPPY;
+ case CHAT_UI_FACE_SLEEPY: return NM_SLEEPY;
+ case CHAT_UI_FACE_SAD: return NM_SAD;
+ case CHAT_UI_FACE_ANGRY: return NM_ANGRY;
+ case CHAT_UI_FACE_SURPRISED: return NM_SURPRISED;
+ case CHAT_UI_FACE_DIZZY: return NM_DIZZY;
+ case CHAT_UI_FACE_LOVE: return NM_LOVE;
+ case CHAT_UI_FACE_SUSPICIOUS: return NM_SUSPICIOUS;
+ default: return NM_NORMAL;
+ }
+}
+
+static chat_ui_face_mood_t effective_mood(void)
+{
+ if (s_override != CHAT_UI_FACE_IDLE && (int64_t)face_now_ms() < s_override_until_ms)
+ return s_override;
+ return s_mood;
+}
+
+// ---- physics targets, blink, spring ----
+static void apply_targets(int nm, float roll, float pitch)
+{
+ float lx = 28, ly = 18, rx = 80, ry = 18, ew = 32, eh = 32;
+ if (nm == NM_LISTEN) { ew = 36; eh = 36; lx = 26; ly = 16; rx = 78; ry = 16; } // attentive
+
+ float dx = roll / 15.0f, dy = pitch / 15.0f;
+ s_L.tx = lx + dx; s_L.ty = ly + dy; s_L.tw = ew; s_L.th = eh;
+ s_R.tx = rx + dx; s_R.ty = ry + dy; s_R.tw = ew; s_R.th = eh;
+
+ float tpx = clampf(roll / 5.0f, -12, 12);
+ float tpy = clampf(pitch / 5.0f, -10, 10);
+ if (nm == NM_THINK) tpy -= 6; // look up
+ if (nm == NM_SUSPICIOUS) tpx += 8; // side-eye
+ s_L.tpx = s_R.tpx = tpx;
+ s_L.tpy = s_R.tpy = tpy;
+}
+
+static void update_blink(int nm)
+{
+ uint32_t now = face_now_ms();
+ bool fast = (nm == NM_DIZZY || nm == NM_ANGRY);
+ uint32_t bd = fast ? 60 : 120;
+ uint32_t bmin = fast ? 300 : (nm == NM_SUSPICIOUS ? 3500 : 2000);
+ uint32_t bmax = fast ? 800 : (nm == NM_SUSPICIOUS ? 7000 : 6000);
+ if (now > s_next_blink_ms) {
+ s_blinking = true;
+ s_blink_last_ms = now;
+ s_next_blink_ms = now + bmin + (esp_random() % (bmax - bmin + 1));
+ }
+ if (s_blinking) {
+ s_L.th = s_R.th = 2; // eyelids down
+ if (now - s_blink_last_ms > bd) s_blinking = false;
+ }
+}
+
+static void eye_spring(eye_t *e)
+{
+ const float K = 0.15f, D = 0.65f, PK = 0.20f, PD = 0.60f;
+ e->vx = (e->vx + (e->tx - e->x) * K) * D; e->x += e->vx;
+ e->vy = (e->vy + (e->ty - e->y) * K) * D; e->y += e->vy;
+ e->vw = (e->vw + (e->tw - e->w) * K) * D; e->w += e->vw;
+ e->vh = (e->vh + (e->th - e->h) * K) * D; e->h += e->vh;
+ e->pvx = (e->pvx + (e->tpx - e->px) * PK) * PD; e->px += e->pvx;
+ e->pvy = (e->pvy + (e->tpy - e->py) * PK) * PD; e->py += e->pvy;
+}
+
+static void draw_scene(int nm)
+{
+ if (nm == NM_DIZZY) {
+ draw_dizzy_eye(&s_L);
+ draw_dizzy_eye(&s_R);
+ int off = (face_now_ms() / 80) % 4;
+ gfx_bitmap(6 - off, 0, bmp_dizzy_stars, 16, 16, WHITE);
+ gfx_bitmap(106 + off, 0, bmp_dizzy_stars, 16, 16, WHITE);
+ } else if (nm == NM_ANGRY) {
+ draw_angry_eye(&s_L, true);
+ draw_angry_eye(&s_R, false);
+ } else {
+ draw_normal_eye(&s_L, true, nm);
+ draw_normal_eye(&s_R, false, nm);
+ if (nm == NM_LOVE) gfx_bitmap(56, 0, bmp_heart, 16, 16, WHITE);
+ else if (nm == NM_SLEEPY) gfx_bitmap(110, 0, bmp_zzz, 16, 16, WHITE);
+ }
+ draw_mouth(nm);
+}
+
+static void face_tick(lv_timer_t *t)
+{
+ (void)t;
+ if (!s_root || !s_active || !s_fb) return;
+ if (lv_obj_has_flag(s_root, LV_OBJ_FLAG_HIDDEN)) return;
+
+ uint32_t now = face_now_ms();
+
+ // Shake -> dizzy (~1.5s) -> angry (~2s)
+ float shake = imu_qmi8658_shake_intensity();
+ if (shake > SHAKE_THRESH && s_override != CHAT_UI_FACE_DIZZY) {
+ s_override = CHAT_UI_FACE_DIZZY;
+ s_override_until_ms = (int64_t)now + 1500;
+ s_shake_start_ms = now;
+ } else if (s_override == CHAT_UI_FACE_DIZZY && (int64_t)now >= s_override_until_ms) {
+ s_override = CHAT_UI_FACE_ANGRY;
+ s_override_until_ms = (int64_t)now + 2000;
+ } else if (s_override == CHAT_UI_FACE_ANGRY && (int64_t)now >= s_override_until_ms) {
+ s_override = CHAT_UI_FACE_IDLE;
+ }
+
+ float roll = 0, pitch = 0;
+ imu_qmi8658_get_tilt(&roll, &pitch); // drives gaze only (see apply_targets)
+
+ int nm = nimo_mood(effective_mood());
+ if (nm == NM_SPEAK) s_speak_phase++;
+
+ apply_targets(nm, roll, pitch);
+ update_blink(nm);
+ eye_spring(&s_L);
+ eye_spring(&s_R);
+
+ memset(s_fb, 0, (size_t)s_W * s_H * sizeof(uint16_t)); // clear to black
+ draw_scene(nm);
+ lv_obj_invalidate(s_canvas);
+}
+
+void face_engine_create(lv_obj_t *scr)
+{
+ s_root = lv_obj_create(scr);
+ lv_obj_remove_style_all(s_root);
+ lv_obj_set_size(s_root, lv_pct(100), lv_pct(100));
+ lv_obj_set_style_bg_color(s_root, lv_color_black(), 0);
+ lv_obj_set_style_bg_opa(s_root, LV_OPA_COVER, 0);
+ lv_obj_clear_flag(s_root, LV_OBJ_FLAG_SCROLLABLE | LV_OBJ_FLAG_CLICKABLE);
+
+ s_scale = (float)BSP_LCD_H_RES / NIMO_W; // ~2.875
+ s_W = BSP_LCD_H_RES;
+ s_H = (int)lroundf(NIMO_H * s_scale);
+ s_fb = heap_caps_malloc((size_t)s_W * s_H * sizeof(uint16_t), MALLOC_CAP_SPIRAM);
+ if (!s_fb) {
+ ESP_LOGE(TAG, "canvas alloc failed (%d bytes)", s_W * s_H * 2);
+ return;
+ }
+ memset(s_fb, 0, (size_t)s_W * s_H * sizeof(uint16_t));
+ s_red = lv_color_hex(0xFF2222).full; // angry-eye colour (matches the alarm ring)
+
+ s_canvas = lv_canvas_create(s_root);
+ lv_canvas_set_buffer(s_canvas, s_fb, s_W, s_H, LV_IMG_CF_TRUE_COLOR);
+ lv_obj_align(s_canvas, LV_ALIGN_CENTER, 0, 0);
+ lv_obj_clear_flag(s_canvas, LV_OBJ_FLAG_SCROLLABLE | LV_OBJ_FLAG_CLICKABLE);
+
+ // Seed eyes at their NIMO rest pose.
+ memset(&s_L, 0, sizeof s_L);
+ memset(&s_R, 0, sizeof s_R);
+ s_L.x = s_L.tx = 28; s_L.y = s_L.ty = 18; s_L.w = s_L.tw = 32; s_L.h = s_L.th = 32;
+ s_R.x = s_R.tx = 80; s_R.y = s_R.ty = 18; s_R.w = s_R.tw = 32; s_R.h = s_R.th = 32;
+ s_next_blink_ms = face_now_ms() + 1500;
+
+ draw_scene(NM_NORMAL);
+ lv_obj_invalidate(s_canvas);
+
+ s_timer = lv_timer_create(face_tick, FACE_TIMER_MS, NULL);
+ ESP_LOGI(TAG, "NIMO face up (%dx%d canvas, scale %.2f)", s_W, s_H, s_scale);
+}
+
+void face_engine_set_mood(chat_ui_face_mood_t mood) { s_mood = mood; }
+chat_ui_face_mood_t face_engine_get_mood(void) { return s_mood; }
+
+void face_engine_set_visible(bool visible)
+{
+ if (!s_root) return;
+ if (visible) lv_obj_clear_flag(s_root, LV_OBJ_FLAG_HIDDEN);
+ else lv_obj_add_flag(s_root, LV_OBJ_FLAG_HIDDEN);
+}
+
+bool face_engine_is_visible(void)
+{
+ return s_root && !lv_obj_has_flag(s_root, LV_OBJ_FLAG_HIDDEN);
+}
+
+void face_engine_set_active(bool active)
+{
+ s_active = active;
+ if (s_timer) {
+ if (active) lv_timer_resume(s_timer);
+ else lv_timer_pause(s_timer);
+ }
+}
+
+lv_obj_t *face_engine_root(void) { return s_root; }
diff --git a/esp32-agui/components/chat_ui/face_engine.h b/esp32-agui/components/chat_ui/face_engine.h
new file mode 100644
index 0000000..ae76d8c
--- /dev/null
+++ b/esp32-agui/components/chat_ui/face_engine.h
@@ -0,0 +1,22 @@
+// NIMO-style spring-physics face (eyes + mouth). Owned by chat_ui.
+#pragma once
+
+#include
+#include "lvgl.h"
+#include "chat_ui.h"
+
+#ifdef __cplusplus
+extern "C" {
+#endif
+
+void face_engine_create(lv_obj_t *scr);
+void face_engine_set_mood(chat_ui_face_mood_t mood);
+chat_ui_face_mood_t face_engine_get_mood(void);
+void face_engine_set_visible(bool visible);
+bool face_engine_is_visible(void);
+void face_engine_set_active(bool active); // pause timer when Eyes page is not shown
+lv_obj_t *face_engine_root(void);
+
+#ifdef __cplusplus
+}
+#endif
diff --git a/esp32-agui/components/chat_ui/idf_component.yml b/esp32-agui/components/chat_ui/idf_component.yml
new file mode 100644
index 0000000..fc7bed4
--- /dev/null
+++ b/esp32-agui/components/chat_ui/idf_component.yml
@@ -0,0 +1,2 @@
+dependencies:
+ espressif/esp_jpeg: "^1.3.0"
diff --git a/esp32-agui/components/chat_ui/include/chat_ui.h b/esp32-agui/components/chat_ui/include/chat_ui.h
index af08427..0c54533 100644
--- a/esp32-agui/components/chat_ui/include/chat_ui.h
+++ b/esp32-agui/components/chat_ui/include/chat_ui.h
@@ -28,6 +28,28 @@ void chat_ui_clear_status(void);
void chat_ui_show_qr(const char *data); // lv_qrcode
void chat_ui_idle_timer(int seconds_left, const char *label);
+// NIMO-style companion face moods. HIDDEN shows chat (or other non-eyes pages).
+// Non-clickable face layer so long-press PTT still works; short tap cycles pages.
+typedef enum {
+ CHAT_UI_FACE_IDLE = 0,
+ CHAT_UI_FACE_LISTEN,
+ CHAT_UI_FACE_THINK,
+ CHAT_UI_FACE_SPEAK,
+ CHAT_UI_FACE_HAPPY,
+ CHAT_UI_FACE_SLEEPY,
+ CHAT_UI_FACE_SAD,
+ CHAT_UI_FACE_ANGRY,
+ CHAT_UI_FACE_SURPRISED,
+ CHAT_UI_FACE_DIZZY,
+ CHAT_UI_FACE_LOVE,
+ CHAT_UI_FACE_SUSPICIOUS,
+ CHAT_UI_FACE_HIDDEN,
+} chat_ui_face_mood_t;
+void chat_ui_set_face(chat_ui_face_mood_t mood);
+
+// Download a JPEG from `url`, decode to RGB565, show as a full-screen overlay (tap to dismiss).
+esp_err_t chat_ui_show_image(const char *url);
+
// Screen-power saver: blank the AMOLED (brightness 0) after `idle_timeout_s` of no touch /
// PTT / UI activity, and wake it on any of those. Pass 0 for "always on" (never blank); >0 sets
// the timeout in seconds. (The PWR button no longer toggles the screen — it's volume-down now, so
diff --git a/esp32-agui/components/deepgram_stt/deepgram_stt.c b/esp32-agui/components/deepgram_stt/deepgram_stt.c
index 3869940..544167e 100644
--- a/esp32-agui/components/deepgram_stt/deepgram_stt.c
+++ b/esp32-agui/components/deepgram_stt/deepgram_stt.c
@@ -7,6 +7,7 @@
#include "freertos/task.h"
#include "freertos/semphr.h"
#include "freertos/stream_buffer.h"
+#include "freertos/idf_additions.h" // xTaskCreateWithCaps — STT stacks in PSRAM
#include "esp_log.h"
#include "esp_heap_caps.h"
#include "cJSON.h"
@@ -22,9 +23,12 @@ static const char *TAG = "dg_stt";
#define MIC_GAIN_DB 30.0f
#define READ_CHUNK_BYTES 640
#define DRAIN_CHUNKS 8
-#define WS_BUFFER_BYTES 8192
+// Keep WS buffers/stack modest: rx+tx are calloc'd in *internal* RAM, and the ws
+// task stack is too. After AG-UI TLS, an 8 KB stack often fails → "websocket start: ESP_FAIL".
+#define WS_BUFFER_BYTES 4096
+#define WS_TASK_STACK 5120
#define SEND_MAX_BYTES WS_BUFFER_BYTES
-#define SEND_TRIGGER 4096
+#define SEND_TRIGGER 2048
#define AUDIO_SB_BYTES (32 * 1024)
#define COMMITTED_MAX 512
#define RUNNING_MAX 640
@@ -65,6 +69,7 @@ static char s_last_error[128];
static uint8_t *s_rx;
static size_t s_rx_total;
+static size_t s_rx_written; // bytes copied into s_rx so far; parse only when it reaches s_rx_total
static void emit_partial(const char *interim)
{
@@ -149,11 +154,17 @@ static void ws_event(void *arg, esp_event_base_t base, int32_t id, void *data)
if (s_rx) heap_caps_free(s_rx);
s_rx = heap_caps_malloc(e->payload_len + 1, MALLOC_CAP_SPIRAM);
s_rx_total = e->payload_len;
+ s_rx_written = 0;
}
if (!s_rx) break;
- if (e->payload_offset + e->data_len <= s_rx_total)
+ if (e->payload_offset + e->data_len <= s_rx_total) {
memcpy(s_rx + e->payload_offset, e->data_ptr, e->data_len);
- if (e->payload_offset + e->data_len >= s_rx_total) {
+ s_rx_written += e->data_len;
+ }
+ // Parse only when the whole message is assembled. Gate on bytes actually written (not on the
+ // offset reaching the end): an overshooting fragment is skipped above, so keying off the offset
+ // would parse a buffer with an unfilled hole → garbled/dropped transcript.
+ if (s_rx_written >= s_rx_total) {
s_rx[s_rx_total] = '\0';
parse_message((const char *)s_rx);
heap_caps_free(s_rx);
@@ -163,6 +174,9 @@ static void ws_event(void *arg, esp_event_base_t base, int32_t id, void *data)
}
case WEBSOCKET_EVENT_ERROR:
ESP_LOGW(TAG, "ws transport error");
+ if (!s_last_error[0])
+ strlcpy(s_last_error, "Deepgram websocket failed", sizeof s_last_error);
+ s_fatal = true;
break;
default: break;
}
@@ -187,7 +201,7 @@ static void capture_task(void *arg)
}
s_cap_task = NULL;
xSemaphoreGive(s_cap_done);
- vTaskDelete(NULL);
+ vTaskDeleteWithCaps(NULL);
}
static void sender_task(void *arg)
@@ -224,7 +238,7 @@ static void sender_task(void *arg)
}
s_send_task = NULL;
xSemaphoreGive(s_send_done);
- vTaskDelete(NULL);
+ vTaskDeleteWithCaps(NULL);
}
esp_err_t deepgram_stt_init(void)
@@ -289,10 +303,15 @@ esp_err_t deepgram_stt_session_start(const deepgram_stt_cfg_t *cfg,
{
if (!s_mic) { esp_err_t e = deepgram_stt_init(); if (e != ESP_OK) return e; }
xSemaphoreTake(s_lock, portMAX_DELAY);
- if (s_active) { xSemaphoreGive(s_lock); return ESP_ERR_INVALID_STATE; }
+ if (s_active) {
+ strlcpy(s_last_error, "STT session already active", sizeof s_last_error);
+ xSemaphoreGive(s_lock);
+ return ESP_ERR_INVALID_STATE;
+ }
if (!cfg || !cfg->api_key || !cfg->api_key[0]) {
ESP_LOGE(TAG, "missing Deepgram API key");
+ strlcpy(s_last_error, "no Deepgram API key", sizeof s_last_error);
xSemaphoreGive(s_lock);
return ESP_ERR_NOT_FOUND;
}
@@ -327,25 +346,53 @@ esp_err_t deepgram_stt_session_start(const deepgram_stt_cfg_t *cfg,
.uri = s_uri,
.headers = s_auth_hdr,
.buffer_size = WS_BUFFER_BYTES,
- .task_stack = 8192,
+ .task_stack = WS_TASK_STACK,
.crt_bundle_attach = esp_crt_bundle_attach,
- .reconnect_timeout_ms = 5000,
+ .disable_auto_reconnect = true, // we own session lifecycle; reconnect fights stop/destroy
.network_timeout_ms = 10000,
.ping_interval_sec = 20,
};
s_ws = esp_websocket_client_init(&wcfg);
- if (!s_ws) { xSemaphoreGive(s_lock); return ESP_FAIL; }
+ if (!s_ws) {
+ strlcpy(s_last_error, "websocket init failed", sizeof s_last_error);
+ xSemaphoreGive(s_lock);
+ return ESP_FAIL;
+ }
esp_websocket_register_events(s_ws, WEBSOCKET_EVENT_ANY, ws_event, NULL);
esp_err_t err = esp_websocket_client_start(s_ws);
if (err != ESP_OK) {
- esp_websocket_client_destroy(s_ws); s_ws = NULL;
+ // One settle+retry: prior AG-UI/TTS TLS often leaves internal heap fragmented for a beat.
+ ESP_LOGW(TAG, "ws start %s (int free=%u largest=%u) — retry",
+ esp_err_to_name(err),
+ (unsigned)heap_caps_get_free_size(MALLOC_CAP_INTERNAL),
+ (unsigned)heap_caps_get_largest_free_block(MALLOC_CAP_INTERNAL));
+ esp_websocket_client_destroy(s_ws);
+ s_ws = NULL;
xSemaphoreGive(s_lock);
- return err;
+ vTaskDelay(pdMS_TO_TICKS(150));
+ xSemaphoreTake(s_lock, portMAX_DELAY);
+ if (s_active) { xSemaphoreGive(s_lock); return ESP_ERR_INVALID_STATE; }
+ s_ws = esp_websocket_client_init(&wcfg);
+ if (s_ws) {
+ esp_websocket_register_events(s_ws, WEBSOCKET_EVENT_ANY, ws_event, NULL);
+ err = esp_websocket_client_start(s_ws);
+ } else {
+ err = ESP_FAIL;
+ }
+ if (err != ESP_OK) {
+ snprintf(s_last_error, sizeof s_last_error, "websocket start: %s", esp_err_to_name(err));
+ if (s_ws) { esp_websocket_client_destroy(s_ws); s_ws = NULL; }
+ xSemaphoreGive(s_lock);
+ return err;
+ }
}
s_active = true;
- if (xTaskCreate(capture_task, "dg_cap", 4096, NULL, 6, &s_cap_task) != pdPASS ||
- xTaskCreate(sender_task, "dg_snd", 4096, NULL, 5, &s_send_task) != pdPASS) {
+ // Stacks in PSRAM so websocket (internal) + capture/sender can coexist after AG-UI/TTS TLS.
+ if (xTaskCreateWithCaps(capture_task, "dg_cap", 4096, NULL, 6, &s_cap_task,
+ MALLOC_CAP_SPIRAM | MALLOC_CAP_8BIT) != pdPASS ||
+ xTaskCreateWithCaps(sender_task, "dg_snd", 4096, NULL, 5, &s_send_task,
+ MALLOC_CAP_SPIRAM | MALLOC_CAP_8BIT) != pdPASS) {
s_stop = true;
if (s_cap_task) xSemaphoreTake(s_cap_done, portMAX_DELAY);
if (s_send_task) xSemaphoreTake(s_send_done, portMAX_DELAY);
@@ -353,12 +400,26 @@ esp_err_t deepgram_stt_session_start(const deepgram_stt_cfg_t *cfg,
esp_websocket_client_destroy(s_ws);
s_ws = NULL;
s_active = false;
+ strlcpy(s_last_error, "STT tasks failed", sizeof s_last_error);
xSemaphoreGive(s_lock);
return ESP_FAIL;
}
- ESP_LOGI(TAG, "session started (%s @ %d Hz)", s_model, s_sr);
xSemaphoreGive(s_lock);
- return ESP_OK;
+
+ // Wait briefly for TLS/WSS so a broken hotspot/key surfaces as start failure, not silent listen.
+ for (int i = 0; i < 100; i++) { // ~5 s
+ if (s_fatal) break;
+ if (s_ws && esp_websocket_client_is_connected(s_ws)) {
+ ESP_LOGI(TAG, "session started (%s @ %d Hz)", s_model, s_sr);
+ return ESP_OK;
+ }
+ vTaskDelay(pdMS_TO_TICKS(50));
+ }
+ if (!s_last_error[0])
+ strlcpy(s_last_error, "Deepgram connect timeout", sizeof s_last_error);
+ ESP_LOGE(TAG, "ws not connected: %s", s_last_error);
+ deepgram_stt_session_stop();
+ return ESP_FAIL;
}
esp_err_t deepgram_stt_session_finalize(void)
@@ -383,13 +444,21 @@ void deepgram_stt_session_stop(void)
if (s_send_task) xSemaphoreTake(s_send_done, portMAX_DELAY);
esp_websocket_client_handle_t ws = s_ws;
s_ws = NULL;
- if (ws) esp_websocket_client_destroy(ws);
+ if (ws) {
+ esp_websocket_client_close(ws, pdMS_TO_TICKS(1000));
+ esp_websocket_client_stop(ws);
+ esp_websocket_client_destroy(ws);
+ }
if (s_rx) { heap_caps_free(s_rx); s_rx = NULL; }
s_active = false;
xSemaphoreGive(s_lock);
ESP_LOGI(TAG, "session stopped");
}
-bool deepgram_stt_session_active(void) { return s_active && !s_fatal; }
+// Report the raw session state: a fatal server error sets s_fatal but leaves the ws/tasks/mic up, so
+// the session still needs an explicit stop — masking it with !s_fatal would let callers skip teardown
+// and leak the ES8311 mic (next session_start then returns ESP_ERR_INVALID_STATE). Error text is
+// surfaced separately via deepgram_stt_last_error().
+bool deepgram_stt_session_active(void) { return s_active; }
const char *deepgram_stt_last_error(void) { return s_last_error[0] ? s_last_error : NULL; }
diff --git a/esp32-agui/components/deepgram_tts/deepgram_tts.c b/esp32-agui/components/deepgram_tts/deepgram_tts.c
index c82254b..c6a66a0 100644
--- a/esp32-agui/components/deepgram_tts/deepgram_tts.c
+++ b/esp32-agui/components/deepgram_tts/deepgram_tts.c
@@ -20,7 +20,8 @@ static const char *TAG = "dg_tts";
#define TTS_VOICE_DEFAULT "aura-2-asteria-en"
#define TTS_SR 16000
-#define WS_BUFFER_BYTES 8192
+#define WS_BUFFER_BYTES 4096
+#define WS_TASK_STACK 5120
#define MSG_MAX 65536
#define RING_BYTES (128 * 1024)
#define DRAIN_CHUNK 640
@@ -234,16 +235,35 @@ esp_err_t deepgram_tts_open(void)
.uri = s_uri,
.headers = s_auth_hdr,
.buffer_size = WS_BUFFER_BYTES,
- .task_stack = 8192,
+ .task_stack = WS_TASK_STACK,
.crt_bundle_attach = esp_crt_bundle_attach,
- .reconnect_timeout_ms = 5000,
+ .disable_auto_reconnect = true,
.network_timeout_ms = 10000,
.ping_interval_sec = 20,
};
s_ws = esp_websocket_client_init(&wcfg);
if (!s_ws) { xSemaphoreGive(s_lock); return ESP_FAIL; }
esp_websocket_register_events(s_ws, WEBSOCKET_EVENT_ANY, ws_event, NULL);
- if (esp_websocket_client_start(s_ws) != ESP_OK) { ws_teardown(); xSemaphoreGive(s_lock); return ESP_FAIL; }
+ if (esp_websocket_client_start(s_ws) != ESP_OK) {
+ ESP_LOGW(TAG, "ws start fail (int free=%u largest=%u) — retry",
+ (unsigned)heap_caps_get_free_size(MALLOC_CAP_INTERNAL),
+ (unsigned)heap_caps_get_largest_free_block(MALLOC_CAP_INTERNAL));
+ ws_teardown();
+ xSemaphoreGive(s_lock);
+ vTaskDelay(pdMS_TO_TICKS(150));
+ xSemaphoreTake(s_lock, portMAX_DELAY);
+ s_ws = esp_websocket_client_init(&wcfg);
+ if (!s_ws) {
+ xSemaphoreGive(s_lock);
+ return ESP_FAIL;
+ }
+ esp_websocket_register_events(s_ws, WEBSOCKET_EVENT_ANY, ws_event, NULL);
+ if (esp_websocket_client_start(s_ws) != ESP_OK) {
+ ws_teardown();
+ xSemaphoreGive(s_lock);
+ return ESP_FAIL;
+ }
+ }
EventBits_t b = xEventGroupWaitBits(s_eg, BIT_CONNECTED | BIT_WSERR | BIT_CANCEL,
pdFALSE, pdFALSE, pdMS_TO_TICKS(5000));
diff --git a/esp32-agui/components/device_tools/CMakeLists.txt b/esp32-agui/components/device_tools/CMakeLists.txt
index 1d113e0..9014c11 100644
--- a/esp32-agui/components/device_tools/CMakeLists.txt
+++ b/esp32-agui/components/device_tools/CMakeLists.txt
@@ -1,6 +1,5 @@
idf_component_register(
SRCS "device_tools.c"
INCLUDE_DIRS "include"
- REQUIRES json esp_driver_i2c esp_timer esp32_s3_touch_amoled_1_8 app_cfg)
-# esp_driver_i2c + the BSP (bsp_i2c_get_handle) for AXP2101 battery reads on the shared I2C bus.
-# P5/P7 add later: QMI8658 motion (device_motion) and PCF85063 RTC alarm (set_alarm tool).
+ REQUIRES json esp_driver_i2c esp_timer esp32_s3_touch_amoled_1_8 app_cfg imu_qmi8658)
+# AXP2101 battery + QMI8658 motion on the shared BSP I2C bus.
diff --git a/esp32-agui/components/device_tools/device_tools.c b/esp32-agui/components/device_tools/device_tools.c
index 0191e4c..9bd0b63 100644
--- a/esp32-agui/components/device_tools/device_tools.c
+++ b/esp32-agui/components/device_tools/device_tools.c
@@ -12,6 +12,7 @@
#include "driver/i2c_master.h"
#include "bsp/esp32_s3_touch_amoled_1_8.h"
#include "app_cfg.h"
+#include "imu_qmi8658.h"
static const char *TAG = "device_tools";
@@ -29,6 +30,9 @@ esp_err_t device_tools_init(void)
ESP_LOGI(TAG, "init");
s_timer_q = xQueueCreate(1, sizeof(uint8_t)); // set_timer fire signal (block on it, don't poll)
register_builtins(); // P7: register builtin client tools (set_timer, ...)
+ esp_err_t imu = imu_qmi8658_init();
+ if (imu != ESP_OK && imu != ESP_ERR_NOT_FOUND)
+ ESP_LOGW(TAG, "IMU init failed: %s", esp_err_to_name(imu));
return ESP_OK;
}
@@ -124,6 +128,26 @@ static void add_battery(cJSON *arr)
ctx_add(arr, "battery", val);
}
+bool device_tools_battery_read(int *percent_out, bool *plugged_out, char *status, size_t status_len)
+{
+ if (percent_out) *percent_out = -1;
+ if (plugged_out) *plugged_out = false;
+ if (status && status_len) status[0] = '\0';
+ if (!axp_ensure()) return false;
+ uint8_t s1, s2, pct;
+ if (axp_rd(AXP_REG_STATUS1, &s1) != ESP_OK || !(s1 & AXP_ST1_BAT_PRESENT)) return false;
+ if (axp_rd(AXP_REG_BAT_PERCENT, &pct) != ESP_OK || pct > 100) return false;
+ bool plugged = (s1 & AXP_ST1_VBUS_GOOD) != 0;
+ bool charging = (axp_rd(AXP_REG_STATUS2, &s2) == ESP_OK) && ((s2 >> 5) == 0x01);
+ const char *st = !plugged ? "on battery"
+ : charging ? "charging"
+ : "plugged in, not charging";
+ if (percent_out) *percent_out = (int)pct;
+ if (plugged_out) *plugged_out = plugged;
+ if (status && status_len) strlcpy(status, st, status_len);
+ return true;
+}
+
// PWR button (AXP2101 PWRKEY) short-press since the last call. The PMIC latches short/long-press
// events in INTSTS2 (we poll it — the AXP IRQ pin isn't wired to a GPIO on this board). A long press
// is a hardware power-off the PMIC does on its own, so only short presses are reported (used by the
@@ -157,6 +181,18 @@ static void add_voice(cJSON *arr)
ctx_add(arr, "tts_voice", voice);
}
+static void add_motion(cJSON *arr)
+{
+ if (!imu_qmi8658_ok()) return;
+ float roll = 0, pitch = 0;
+ imu_qmi8658_get_tilt(&roll, &pitch);
+ char val[128];
+ snprintf(val, sizeof val,
+ "{\"orientation\":\"%s\",\"roll_deg\":%.1f,\"pitch_deg\":%.1f,\"shake\":%.2f}",
+ imu_qmi8658_orientation(), roll, pitch, imu_qmi8658_shake_intensity());
+ ctx_add(arr, "device_motion", val);
+}
+
cJSON *device_context_build(void)
{
cJSON *arr = cJSON_CreateArray();
@@ -165,7 +201,7 @@ cJSON *device_context_build(void)
add_local_time(arr);
add_battery(arr);
add_voice(arr);
- // Next P5 increment appends here: device_motion (QMI8658).
+ add_motion(arr);
if (cJSON_GetArraySize(arr) == 0) { // nothing to report yet → send no context
cJSON_Delete(arr);
@@ -244,18 +280,53 @@ static esp_timer_handle_t s_timer_h;
static volatile int64_t s_timer_deadline_us; // 0 = no active timer
static volatile bool s_timer_fired;
static char s_timer_label[40];
+// Serializes the fired/deadline/label trio, written by tool_set_timer (agent task) and
+// timer_fire_cb (esp_timer task) and read by device_tools_timer_take_fired (alert task). Without it a
+// replace-during-fire tears the label or lets a stale fire clobber the new deadline.
+static portMUX_TYPE s_timer_mux = portMUX_INITIALIZER_UNLOCKED;
static void timer_fire_cb(void *arg)
{
(void)arg;
+ char lbl[sizeof s_timer_label];
+ taskENTER_CRITICAL(&s_timer_mux);
s_timer_deadline_us = 0;
s_timer_fired = true;
+ strlcpy(lbl, s_timer_label, sizeof lbl); // snapshot for the log outside the critical section
+ taskEXIT_CRITICAL(&s_timer_mux);
if (s_timer_q) { uint8_t sig = 1; xQueueSend(s_timer_q, &sig, 0); } // wake the alert task (cb runs in task ctx)
- ESP_LOGI(TAG, "timer fired: %s", s_timer_label[0] ? s_timer_label : "(timer)");
+ ESP_LOGI(TAG, "timer fired: %s", lbl[0] ? lbl : "(timer)");
}
QueueHandle_t device_tools_timer_queue(void) { return s_timer_q; }
+// --- builtin: show_image --------------------------------------------------------------------
+static device_show_image_fn s_show_image_fn;
+
+void device_tools_set_show_image_handler(device_show_image_fn fn) { s_show_image_fn = fn; }
+
+static esp_err_t tool_show_image(const cJSON *args, cJSON **result)
+{
+ const cJSON *url = cJSON_GetObjectItemCaseSensitive(args, "url");
+ if (!cJSON_IsString(url) || !url->valuestring || !url->valuestring[0]) {
+ if (result) *result = cJSON_CreateString("error: 'url' is required");
+ return ESP_OK;
+ }
+ if (!s_show_image_fn) {
+ if (result) *result = cJSON_CreateString("error: show_image handler not wired");
+ return ESP_OK;
+ }
+ esp_err_t err = s_show_image_fn(url->valuestring);
+ if (err != ESP_OK) {
+ char msg[80];
+ snprintf(msg, sizeof msg, "error: display failed (%s)", esp_err_to_name(err));
+ if (result) *result = cJSON_CreateString(msg);
+ return ESP_OK;
+ }
+ if (result) *result = cJSON_CreateString("Image displayed on screen");
+ return ESP_OK;
+}
+
static esp_err_t tool_set_timer(const cJSON *args, cJSON **result)
{
const cJSON *secs = cJSON_GetObjectItemCaseSensitive(args, "seconds");
@@ -265,8 +336,7 @@ static esp_err_t tool_set_timer(const cJSON *args, cJSON **result)
if (result) *result = cJSON_CreateString("error: 'seconds' must be an integer 1..86400");
return ESP_OK; // a (negative) tool RESULT, not a dispatch failure
}
- strlcpy(s_timer_label, (cJSON_IsString(label) && label->valuestring) ? label->valuestring : "",
- sizeof s_timer_label);
+ const char *newlabel = (cJSON_IsString(label) && label->valuestring) ? label->valuestring : "";
if (!s_timer_h) {
const esp_timer_create_args_t ta = { .callback = timer_fire_cb, .name = "devtimer" };
if (esp_timer_create(&ta, &s_timer_h) != ESP_OK) {
@@ -275,8 +345,15 @@ static esp_err_t tool_set_timer(const cJSON *args, cJSON **result)
}
}
esp_timer_stop(s_timer_h); // replace any timer already running
+ int64_t deadline = esp_timer_get_time() + (int64_t)seconds * 1000000;
+ // Drain a fire the OLD timer may have already queued (queue depth 1) so a replaced timer can't
+ // surface as a phantom alarm, then set label + deadline + fired atomically vs timer_fire_cb.
+ if (s_timer_q) { uint8_t x; while (xQueueReceive(s_timer_q, &x, 0) == pdTRUE) {} }
+ taskENTER_CRITICAL(&s_timer_mux);
+ strlcpy(s_timer_label, newlabel, sizeof s_timer_label);
s_timer_fired = false;
- s_timer_deadline_us = esp_timer_get_time() + (int64_t)seconds * 1000000;
+ s_timer_deadline_us = deadline;
+ taskEXIT_CRITICAL(&s_timer_mux);
esp_timer_start_once(s_timer_h, (uint64_t)seconds * 1000000);
char msg[80];
snprintf(msg, sizeof msg, "Timer set for %d second%s%s%s", seconds, seconds == 1 ? "" : "s",
@@ -295,10 +372,15 @@ int device_tools_timer_remaining(void)
bool device_tools_timer_take_fired(char *label, size_t n)
{
- if (!s_timer_fired) return false;
- s_timer_fired = false;
- if (label && n) strlcpy(label, s_timer_label, n);
- return true;
+ bool fired;
+ taskENTER_CRITICAL(&s_timer_mux);
+ fired = s_timer_fired;
+ if (fired) {
+ s_timer_fired = false;
+ if (label && n) strlcpy(label, s_timer_label, n); // consistent label for THIS fire
+ }
+ taskEXIT_CRITICAL(&s_timer_mux);
+ return fired;
}
// Build a tool def {description, parameters} from a description + a JSON-Schema string.
@@ -323,5 +405,12 @@ static void register_builtins(void)
"\"label\":{\"type\":\"string\",\"description\":\"Optional short name for the timer\"}},"
"\"required\":[\"seconds\"]}"),
tool_set_timer);
+ device_tools_register(
+ "show_image",
+ tool_def("Display an image on the device screen from an HTTPS JPEG URL.",
+ "{\"type\":\"object\",\"properties\":{"
+ "\"url\":{\"type\":\"string\",\"description\":\"HTTPS URL of a JPEG image\"}},"
+ "\"required\":[\"url\"]}"),
+ tool_show_image);
// set_alarm (PCF85063 RTC) and show_qr (lv_qrcode) register here next.
}
diff --git a/esp32-agui/components/device_tools/include/device_tools.h b/esp32-agui/components/device_tools/include/device_tools.h
index 232fe75..11cc868 100644
--- a/esp32-agui/components/device_tools/include/device_tools.h
+++ b/esp32-agui/components/device_tools/include/device_tools.h
@@ -30,6 +30,10 @@ bool device_power_key_short_press(void);
// True if running on battery (no USB/VBUS power). Gates idle WiFi-off (plugged-in stays connected).
bool device_tools_on_battery(void);
+// Battery gauge for UI (clock page). Returns false if PMIC/gauge unavailable.
+// percent_out: 0..100; plugged_out: USB present; status: short human string (optional).
+bool device_tools_battery_read(int *percent_out, bool *plugged_out, char *status, size_t status_len);
+
// A tool implementation: parse args, produce a result JSON (caller owns *result).
typedef esp_err_t (*device_tool_fn)(const cJSON *args, cJSON **result);
@@ -49,6 +53,11 @@ bool device_tools_timer_take_fired(char *label, size_t n); // true ONCE after a
// light-sleep until the deadline. Drain with device_tools_timer_take_fired().
QueueHandle_t device_tools_timer_queue(void);
+// show_image: device_tools owns the AG-UI tool registration; chat_ui owns download/display.
+// Main wires the handler after chat_ui_init() so we avoid a circular component dependency.
+typedef esp_err_t (*device_show_image_fn)(const char *url);
+void device_tools_set_show_image_handler(device_show_image_fn fn);
+
#ifdef __cplusplus
}
#endif
diff --git a/esp32-agui/components/imu_qmi8658/CMakeLists.txt b/esp32-agui/components/imu_qmi8658/CMakeLists.txt
new file mode 100644
index 0000000..cef9304
--- /dev/null
+++ b/esp32-agui/components/imu_qmi8658/CMakeLists.txt
@@ -0,0 +1,4 @@
+idf_component_register(
+ SRCS "imu_qmi8658.c"
+ INCLUDE_DIRS "include"
+ REQUIRES esp_driver_i2c esp_timer esp32_s3_touch_amoled_1_8)
diff --git a/esp32-agui/components/imu_qmi8658/imu_qmi8658.c b/esp32-agui/components/imu_qmi8658/imu_qmi8658.c
new file mode 100644
index 0000000..d4a9706
--- /dev/null
+++ b/esp32-agui/components/imu_qmi8658/imu_qmi8658.c
@@ -0,0 +1,181 @@
+// Minimal QMI8658 driver for Waveshare ESP32-S3-Touch-AMOLED-1.8 (shared BSP I2C).
+#include "imu_qmi8658.h"
+
+#include
+#include
+
+#include "freertos/FreeRTOS.h"
+#include "freertos/task.h"
+#include "esp_log.h"
+#include "driver/i2c_master.h"
+#include "bsp/esp32_s3_touch_amoled_1_8.h"
+
+static const char *TAG = "imu_qmi8658";
+
+#define QMI_WHO_AM_I 0x00
+#define QMI_CTRL1 0x02
+#define QMI_CTRL2 0x03
+#define QMI_CTRL3 0x04
+#define QMI_CTRL5 0x06
+#define QMI_CTRL7 0x08
+#define QMI_AX_L 0x35
+#define QMI_RESET 0x60
+#define QMI_WHO_VAL 0x05
+#define QMI_ADDR_A 0x6A
+#define QMI_ADDR_B 0x6B
+
+// ±4 g full-scale → ~8192 LSB/g (approx for QMI8658 ±4g)
+#define ACC_LSB_PER_G 8192.0f
+
+static i2c_master_dev_handle_t s_dev;
+static bool s_ok;
+static float s_roll, s_pitch; // smoothed degrees
+static float s_shake; // decayed intensity
+static float s_ax0, s_ay0, s_az0; // still calibration baseline
+static bool s_calibrated;
+static int s_still_n;
+static char s_orient[16] = "unknown";
+
+static esp_err_t qmi_rd(uint8_t reg, uint8_t *buf, size_t n)
+{
+ return i2c_master_transmit_receive(s_dev, ®, 1, buf, n, 50);
+}
+
+static esp_err_t qmi_wr(uint8_t reg, uint8_t val)
+{
+ uint8_t b[2] = { reg, val };
+ return i2c_master_transmit(s_dev, b, 2, 50);
+}
+
+static bool probe_addr(i2c_master_bus_handle_t bus, uint8_t addr)
+{
+ i2c_device_config_t cfg = {
+ .dev_addr_length = I2C_ADDR_BIT_LEN_7,
+ .device_address = addr,
+ .scl_speed_hz = 400000,
+ };
+ i2c_master_dev_handle_t dev = NULL;
+ if (i2c_master_bus_add_device(bus, &cfg, &dev) != ESP_OK) return false;
+ uint8_t who = 0;
+ uint8_t reg = QMI_WHO_AM_I;
+ esp_err_t e = i2c_master_transmit_receive(dev, ®, 1, &who, 1, 50);
+ if (e != ESP_OK || who != QMI_WHO_VAL) {
+ i2c_master_bus_rm_device(dev);
+ return false;
+ }
+ s_dev = dev;
+ ESP_LOGI(TAG, "QMI8658 at 0x%02X (WHO_AM_I=0x%02X)", addr, who);
+ return true;
+}
+
+static bool configure(void)
+{
+ // Soft reset then bring up accel+gyro at ~125 Hz.
+ qmi_wr(QMI_RESET, 0xB0);
+ vTaskDelay(pdMS_TO_TICKS(20));
+ if (qmi_wr(QMI_CTRL1, 0x40) != ESP_OK) return false; // address auto-increment
+ if (qmi_wr(QMI_CTRL2, 0x16) != ESP_OK) return false; // ±4g, aODR ~125 Hz
+ if (qmi_wr(QMI_CTRL3, 0x56) != ESP_OK) return false; // ±512 dps, gODR ~125 Hz
+ if (qmi_wr(QMI_CTRL5, 0x11) != ESP_OK) return false; // light LPF
+ if (qmi_wr(QMI_CTRL7, 0x03) != ESP_OK) return false; // aEN | gEN
+ vTaskDelay(pdMS_TO_TICKS(10));
+ return true;
+}
+
+static bool read_accel_g(float *ax, float *ay, float *az)
+{
+ uint8_t raw[6];
+ if (qmi_rd(QMI_AX_L, raw, 6) != ESP_OK) return false;
+ int16_t x = (int16_t)((raw[1] << 8) | raw[0]);
+ int16_t y = (int16_t)((raw[3] << 8) | raw[2]);
+ int16_t z = (int16_t)((raw[5] << 8) | raw[4]);
+ *ax = (float)x / ACC_LSB_PER_G;
+ *ay = (float)y / ACC_LSB_PER_G;
+ *az = (float)z / ACC_LSB_PER_G;
+ return true;
+}
+
+static void update_orient(float ax, float ay, float az)
+{
+ float mag = sqrtf(ax * ax + ay * ay + az * az);
+ if (mag < 0.3f) {
+ strlcpy(s_orient, "unknown", sizeof s_orient);
+ return;
+ }
+ float nx = ax / mag, ny = ay / mag, nz = az / mag;
+ if (fabsf(nz) > 0.85f) strlcpy(s_orient, "flat", sizeof s_orient);
+ else if (fabsf(ny) > 0.7f || fabsf(nx) > 0.7f) strlcpy(s_orient, "upright", sizeof s_orient);
+ else strlcpy(s_orient, "tilted", sizeof s_orient);
+}
+
+static void imu_task(void *arg)
+{
+ (void)arg;
+ float prev_ax = 0, prev_ay = 0, prev_az = 1;
+ bool have_prev = false;
+ for (;;) {
+ float ax, ay, az;
+ if (s_ok && read_accel_g(&ax, &ay, &az)) {
+ // Soft EMA for tilt (gravity-dominant).
+ float roll = atan2f(ay, az) * (180.0f / (float)M_PI);
+ float pitch = atan2f(-ax, sqrtf(ay * ay + az * az)) * (180.0f / (float)M_PI);
+ s_roll = s_roll * 0.85f + roll * 0.15f;
+ s_pitch = s_pitch * 0.85f + pitch * 0.15f;
+ update_orient(ax, ay, az);
+
+ if (!s_calibrated) {
+ float j = fabsf(ax - prev_ax) + fabsf(ay - prev_ay) + fabsf(az - prev_az);
+ if (have_prev && j < 0.05f) {
+ if (++s_still_n >= 20) { // ~800 ms still
+ s_ax0 = ax; s_ay0 = ay; s_az0 = az;
+ s_calibrated = true;
+ ESP_LOGI(TAG, "calibrated (still)");
+ }
+ } else {
+ s_still_n = 0;
+ }
+ }
+
+ if (have_prev) {
+ float d = fabsf(ax - prev_ax) + fabsf(ay - prev_ay) + fabsf(az - prev_az);
+ if (d > s_shake) s_shake = d;
+ else s_shake *= 0.92f;
+ }
+ prev_ax = ax; prev_ay = ay; prev_az = az;
+ have_prev = true;
+ }
+ vTaskDelay(pdMS_TO_TICKS(40)); // ~25 Hz
+ }
+}
+
+esp_err_t imu_qmi8658_init(void)
+{
+ if (s_ok) return ESP_OK;
+ if (bsp_i2c_init() != ESP_OK) return ESP_FAIL;
+ i2c_master_bus_handle_t bus = bsp_i2c_get_handle();
+ if (!bus) return ESP_FAIL;
+
+ if (!probe_addr(bus, QMI_ADDR_A) && !probe_addr(bus, QMI_ADDR_B)) {
+ ESP_LOGW(TAG, "QMI8658 not found on 0x6A/0x6B");
+ return ESP_ERR_NOT_FOUND;
+ }
+ if (!configure()) {
+ ESP_LOGE(TAG, "configure failed");
+ return ESP_FAIL;
+ }
+ s_ok = true;
+ xTaskCreate(imu_task, "imu_qmi", 3072, NULL, 4, NULL);
+ return ESP_OK;
+}
+
+bool imu_qmi8658_ok(void) { return s_ok; }
+
+void imu_qmi8658_get_tilt(float *roll_deg, float *pitch_deg)
+{
+ if (roll_deg) *roll_deg = s_ok ? s_roll : 0.f;
+ if (pitch_deg) *pitch_deg = s_ok ? s_pitch : 0.f;
+}
+
+float imu_qmi8658_shake_intensity(void) { return s_ok ? s_shake : 0.f; }
+
+const char *imu_qmi8658_orientation(void) { return s_ok ? s_orient : "unknown"; }
diff --git a/esp32-agui/components/imu_qmi8658/include/imu_qmi8658.h b/esp32-agui/components/imu_qmi8658/include/imu_qmi8658.h
new file mode 100644
index 0000000..10a6cb9
--- /dev/null
+++ b/esp32-agui/components/imu_qmi8658/include/imu_qmi8658.h
@@ -0,0 +1,25 @@
+// QMI8658 6-axis IMU on the Waveshare ESP32-S3-Touch-AMOLED-1.8 shared BSP I2C bus.
+#pragma once
+
+#include
+#include "esp_err.h"
+
+#ifdef __cplusplus
+extern "C" {
+#endif
+
+esp_err_t imu_qmi8658_init(void);
+bool imu_qmi8658_ok(void);
+
+// Smoothed tilt in degrees (roll around X, pitch around Y). 0 when unavailable.
+void imu_qmi8658_get_tilt(float *roll_deg, float *pitch_deg);
+
+// Recent peak |Δaccel| magnitude (g). Spike when shaken.
+float imu_qmi8658_shake_intensity(void);
+
+// Compact orientation label for ambient context: "upright" | "flat" | "tilted" | "unknown".
+const char *imu_qmi8658_orientation(void);
+
+#ifdef __cplusplus
+}
+#endif
diff --git a/esp32-agui/components/net_prov/net_prov.c b/esp32-agui/components/net_prov/net_prov.c
index 9af4281..58d2780 100644
--- a/esp32-agui/components/net_prov/net_prov.c
+++ b/esp32-agui/components/net_prov/net_prov.c
@@ -6,6 +6,7 @@
#include "net_prov_internal.h"
#include
+#include
#include
#include
#include "freertos/FreeRTOS.h"
@@ -266,29 +267,89 @@ static void on_sntp_sync(struct timeval *tv)
void net_sntp_start(void)
{
- // One-shot SNTP for wall-clock time (P5 ambient context "local_time"). Call once after WiFi is
- // up; it syncs in the background. Guarded so a later reconnect doesn't re-init the service.
+ // Prefer Google NTP — phone hotspots often block pool.ntp.org. Keep to 1 server
+ // (CONFIG_SNTP_MAX_SERVERS defaults to 1).
static bool started;
if (started) return;
- esp_sntp_config_t cfg = ESP_NETIF_SNTP_DEFAULT_CONFIG("pool.ntp.org");
- cfg.sync_cb = on_sntp_sync; // log + flag when the clock is actually set
+ esp_sntp_config_t cfg = ESP_NETIF_SNTP_DEFAULT_CONFIG("time.google.com");
+ cfg.sync_cb = on_sntp_sync;
esp_err_t e = esp_netif_sntp_init(&cfg);
- if (e == ESP_OK) { started = true; ESP_LOGI(TAG, "SNTP started (pool.ntp.org)"); }
+ if (e == ESP_OK) { started = true; ESP_LOGI(TAG, "SNTP started (time.google.com)"); }
else ESP_LOGW(TAG, "SNTP init failed: %s", esp_err_to_name(e));
}
bool net_time_synced(void) { return s_time_synced; }
-esp_err_t net_time_http_fallback(void)
+static bool apply_unix_time(time_t t, const char *via)
+{
+ if (t < 1700000000) return false;
+ struct timeval tv = { .tv_sec = t, .tv_usec = 0 };
+ settimeofday(&tv, NULL);
+ s_time_synced = true;
+ ESP_LOGI(TAG, "time set via %s: %lld", via, (long long)t);
+ return true;
+}
+
+static bool apply_build_time(void)
+{
+ // Last resort when NTP/HTTP are blocked (common on phone hotspots). Firmware build stamp is
+ // close enough for TLS cert validity windows (years), which unblocks Deepgram/AG-UI.
+ // __DATE__ = "Mmm dd yyyy", __TIME__ = "hh:mm:ss"
+ const char *d = __DATE__;
+ const char *t = __TIME__;
+ const char *months = "JanFebMarAprMayJunJulAugSepOctNovDec";
+ char mon[4] = { d[0], d[1], d[2], 0 };
+ const char *mp = strstr(months, mon);
+ if (!mp) return false;
+ struct tm tm = {0};
+ tm.tm_year = atoi(d + 7) - 1900;
+ tm.tm_mon = (int)(mp - months) / 3;
+ tm.tm_mday = atoi(d + 4);
+ tm.tm_hour = atoi(t);
+ tm.tm_min = atoi(t + 3);
+ tm.tm_sec = atoi(t + 6);
+ if (!apply_unix_time(tm_to_utc(&tm), "firmware-build")) return false;
+ ESP_LOGW(TAG, "NTP/HTTP blocked — using build time so TLS can proceed");
+ return true;
+}
+
+static esp_err_t time_from_http_body(const char *url)
{
- // Time-via-HTTPS for networks that block NTP (UDP/123) — common on hotspots/guest WiFi. HEAD a
- // tiny TLS endpoint, parse the server's Date: header (RFC 1123, UTC) -> settimeofday. Cheap; the
- // heartbeat retries it until the clock is set, then stops.
- if (s_time_synced) return ESP_OK;
esp_http_client_config_t hcfg = {
- .url = "https://www.google.com/generate_204",
+ .url = url,
+ .method = HTTP_METHOD_GET,
+ .timeout_ms = 5000,
+ .disable_auto_redirect = true,
+ };
+ esp_http_client_handle_t c = esp_http_client_init(&hcfg);
+ if (!c) return ESP_FAIL;
+ char body[512];
+ int got = 0;
+ esp_err_t ret = ESP_FAIL;
+ if (esp_http_client_open(c, 0) == ESP_OK) {
+ (void)esp_http_client_fetch_headers(c);
+ int n;
+ while (got < (int)sizeof(body) - 1 &&
+ (n = esp_http_client_read(c, body + got, (int)sizeof(body) - 1 - got)) > 0)
+ got += n;
+ body[got] = '\0';
+ esp_http_client_close(c);
+ const char *p = strstr(body, "\"unixtime\"");
+ if (p) {
+ p = strchr(p, ':');
+ if (p && apply_unix_time((time_t)strtoll(p + 1, NULL, 10), url)) ret = ESP_OK;
+ }
+ }
+ esp_http_client_cleanup(c);
+ return ret;
+}
+
+static esp_err_t time_from_https_date(const char *url)
+{
+ esp_http_client_config_t hcfg = {
+ .url = url,
.method = HTTP_METHOD_HEAD,
- .timeout_ms = 8000,
+ .timeout_ms = 5000,
.crt_bundle_attach = esp_crt_bundle_attach,
};
esp_http_client_handle_t c = esp_http_client_init(&hcfg);
@@ -297,19 +358,21 @@ esp_err_t net_time_http_fallback(void)
if (esp_http_client_perform(c) == ESP_OK) {
char *date = NULL;
if (esp_http_client_get_header(c, "Date", &date) == ESP_OK && date) {
- struct tm tm = {0}; // e.g. "Wed, 23 Jun 2026 18:00:00 GMT"
+ struct tm tm = {0};
if (strptime(date, "%a, %d %b %Y %H:%M:%S", &tm)) {
- time_t t = tm_to_utc(&tm);
- if (t > 1700000000) {
- struct timeval tv = { .tv_sec = t, .tv_usec = 0 };
- settimeofday(&tv, NULL);
- s_time_synced = true;
- ESP_LOGI(TAG, "time set via HTTPS Date: %s", date);
- ret = ESP_OK;
- }
+ if (apply_unix_time(tm_to_utc(&tm), date)) ret = ESP_OK;
}
}
}
esp_http_client_cleanup(c);
return ret;
}
+
+esp_err_t net_time_http_fallback(void)
+{
+ if (s_time_synced) return ESP_OK;
+ // Instant provisional clock FIRST so PTT/TLS never wait on a blocked hotspot.
+ // SNTP (time.google.com) may refine later via on_sntp_sync.
+ if (!apply_build_time()) return ESP_FAIL;
+ return ESP_OK;
+}
diff --git a/esp32-agui/components/net_prov/portal.c b/esp32-agui/components/net_prov/portal.c
index 79c0175..f1b97dd 100644
--- a/esp32-agui/components/net_prov/portal.c
+++ b/esp32-agui/components/net_prov/portal.c
@@ -180,7 +180,7 @@ static esp_err_t root_get(httpd_req_t *req)
sizeof saved_voice);
// Pre-select speech provider in the form (patch via tiny script after options).
- char opts[3500];
+ char opts[4200];
size_t o = 0;
// Emit both voice lists as JS arrays + initial options for current provider.
o += snprintf(opts + o, sizeof(opts) - o, "