From 8ae3451c902b32a6bf8b205c7bacfd583e89af78 Mon Sep 17 00:00:00 2001
From: Takao Akaki
Date: Mon, 21 Sep 2026 18:01:32 +0900
Subject: [PATCH 01/10] =?UTF-8?q?fix(hts=5Fengine):=20float**=20=E9=85=8D?=
=?UTF-8?q?=E5=88=97=E3=82=92=20sizeof(float=20*)=20=E3=81=A7=E7=A2=BA?=
=?UTF-8?q?=E4=BF=9D=E3=81=99=E3=82=8B?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
HTS_pstream.c / HTS_sstream.c の win_coefficient (float**) を
HTS_calloc(win_size, sizeof(float)) で確保していた (double → float 化の際の
取り違え)。ESP32 (32bit) ではポインタも 4 バイトなので顕在化しないが、64bit
ホストでは win_size 個のポインタが確保域を超えてヒープを壊し、合成後の
HTS_Engine_refresh で abort する。
ホストで HMM 合成を回すテスト (jtts) に必要。実機の挙動は変わらない。
Co-Authored-By: Claude Sonnet 5
---
components/hts_engine/lib/HTS_pstream.c | 2 +-
components/hts_engine/lib/HTS_sstream.c | 2 +-
2 files changed, 2 insertions(+), 2 deletions(-)
diff --git a/components/hts_engine/lib/HTS_pstream.c b/components/hts_engine/lib/HTS_pstream.c
index 3d52557..87bbf84 100644
--- a/components/hts_engine/lib/HTS_pstream.c
+++ b/components/hts_engine/lib/HTS_pstream.c
@@ -331,7 +331,7 @@ HTS_Boolean HTS_PStreamSet_create(HTS_PStreamSet * pss, HTS_SStreamSet * sss, fl
/* copy dynamic window */
pst->win_l_width = (int *) HTS_calloc(pst->win_size, sizeof(int));
pst->win_r_width = (int *) HTS_calloc(pst->win_size, sizeof(int));
- pst->win_coefficient = (float **) HTS_calloc(pst->win_size, sizeof(float));
+ pst->win_coefficient = (float **) HTS_calloc(pst->win_size, sizeof(float *));
for (j = 0; j < pst->win_size; j++) {
pst->win_l_width[j] = HTS_SStreamSet_get_window_left_width(sss, i, j);
pst->win_r_width[j] = HTS_SStreamSet_get_window_right_width(sss, i, j);
diff --git a/components/hts_engine/lib/HTS_sstream.c b/components/hts_engine/lib/HTS_sstream.c
index d1a5f09..9464ff1 100644
--- a/components/hts_engine/lib/HTS_sstream.c
+++ b/components/hts_engine/lib/HTS_sstream.c
@@ -305,7 +305,7 @@ HTS_Boolean HTS_SStreamSet_create(HTS_SStreamSet * sss, HTS_ModelSet * ms, HTS_L
sst->win_max_width = HTS_ModelSet_get_window_max_width(ms, i);
sst->win_l_width = (int *) HTS_calloc(sst->win_size, sizeof(int));
sst->win_r_width = (int *) HTS_calloc(sst->win_size, sizeof(int));
- sst->win_coefficient = (float **) HTS_calloc(sst->win_size, sizeof(float));
+ sst->win_coefficient = (float **) HTS_calloc(sst->win_size, sizeof(float *));
for (j = 0; j < sst->win_size; j++) {
sst->win_l_width[j] = HTS_ModelSet_get_window_left_width(ms, i, j);
sst->win_r_width[j] = HTS_ModelSet_get_window_right_width(ms, i, j);
From 9115e515f9d0d0b9d5abd1cf6af742ca5824b3ed Mon Sep 17 00:00:00 2001
From: Takao Akaki
Date: Mon, 21 Sep 2026 18:01:32 +0900
Subject: [PATCH 02/10] =?UTF-8?q?feat(avatar):=20=E5=90=B9=E3=81=8D?=
=?UTF-8?q?=E5=87=BA=E3=81=97=E3=82=92=E5=B7=A6=E8=A9=B0=E3=82=81=E3=81=AB?=
=?UTF-8?q?=E3=81=97=E3=80=81=E6=BA=A2=E3=82=8C=E3=82=8B=E6=99=82=E3=81=A0?=
=?UTF-8?q?=E3=81=91=E6=96=87=E9=A0=AD=E3=81=8B=E3=82=89=E4=B8=80=E5=BA=A6?=
=?UTF-8?q?=E3=82=B9=E3=82=AF=E3=83=AD=E3=83=BC=E3=83=AB=E3=81=99=E3=82=8B?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
従来は、収まる文字列は中央寄せ、溢れる文字列は右の外から入ってきて左へ
流れ切るループ表示だった (最初は文字が見えない)。
- 左詰めで表示する。収まる文字列はスクロールしない。
- 溢れる文字列は、文頭が見える状態で少し止まり、溢れた分だけ左へ動かして、
文末が右端に揃ったところで止める。ループしない。
- hold_ms が指定されているときは、その時間内に文末まで読めるよう速度を
合わせる (30〜120 px/s、前後の静止は各 1/5・最大 1 秒)。
位置の計算は副作用のない関数 (balloon_layout.hpp) に切り出し、ホストで
テストする (components/avatar/test/host)。
Co-Authored-By: Claude Sonnet 5
---
.github/workflows/host-tests.yml | 9 ++
components/avatar/balloon.cpp | 50 ++--------
components/avatar/balloon_layout.hpp | 65 +++++++++++++
components/avatar/include/avatar/avatar.hpp | 10 +-
components/avatar/test/host/CMakeLists.txt | 18 ++++
.../avatar/test/host/test_balloon_layout.cpp | 94 +++++++++++++++++++
.../avatar_vm/include/avatar/draw_context.hpp | 11 ++-
main/shared_state.hpp | 7 +-
8 files changed, 208 insertions(+), 56 deletions(-)
create mode 100644 components/avatar/balloon_layout.hpp
create mode 100644 components/avatar/test/host/CMakeLists.txt
create mode 100644 components/avatar/test/host/test_balloon_layout.cpp
diff --git a/.github/workflows/host-tests.yml b/.github/workflows/host-tests.yml
index 65e53f1..742fc78 100644
--- a/.github/workflows/host-tests.yml
+++ b/.github/workflows/host-tests.yml
@@ -60,6 +60,15 @@ jobs:
cmake --build build-san
ctest --test-dir build-san --output-on-failure
+ # --- avatar: balloon scroll layout (pure functions, header only) ------
+ - name: avatar host tests
+ working-directory: components/avatar/test/host
+ run: |
+ set -euo pipefail
+ cmake -S . -B build -G Ninja -DCMAKE_BUILD_TYPE=Debug
+ cmake --build build
+ ctest --test-dir build --output-on-failure
+
# --- flash_layout: ADR-001 exttab / bootctl format --------------------
# format.c is the exact source the custom bootloader compiles, so the
# validation / A-B selection / boot-decision logic is proven here.
diff --git a/components/avatar/balloon.cpp b/components/avatar/balloon.cpp
index d03e159..cd2537b 100644
--- a/components/avatar/balloon.cpp
+++ b/components/avatar/balloon.cpp
@@ -3,7 +3,7 @@
#include "balloon.hpp"
-#include
+#include "balloon_layout.hpp"
namespace stackchan::avatar::internal {
@@ -21,16 +21,6 @@ constexpr std::int16_t kSmallPanelHeightThreshold = 160;
constexpr std::int16_t kBigPanelH = 40; // for 24-px font
constexpr std::int16_t kSmallPanelH = 22; // for 12-px font
-// Marquee tuning.
-constexpr std::int32_t kScrollSpeedPxPerSec = 60;
-// Gap (px) of "empty space" between the trailing edge of one pass and the
-// leading edge of the next so the user perceives the message restarting.
-constexpr std::int32_t kRepeatGapPx = 60;
-
-// Default minimum display time for short (non-scrolling) text. The application
-// can override with `Avatar::set_balloon_text(text, hold_ms)`.
-constexpr std::uint32_t kDefaultStaticHoldMs = 3000;
-
} // namespace
void draw_balloon(RichCanvas& canvas, DrawContext& ctx)
@@ -71,42 +61,14 @@ void draw_balloon(RichCanvas& canvas, DrawContext& ctx)
const std::int32_t mid_y = panel_y + panel_h / 2;
const std::uint32_t elapsed_ms = ctx.now_ms - ctx.balloon_set_ms;
- if (text_w <= inner_w) {
- // Text fits — static centered. Mark done after the configured hold.
- canvas.setTextDatum(lgfx::textdatum_t::middle_center);
- canvas.drawString(text.c_str(), panel_x + panel_w / 2, mid_y);
-
- const std::uint32_t hold_ms =
- std::max(ctx.balloon_hold_ms, kDefaultStaticHoldMs);
- if (elapsed_ms >= hold_ms) {
- ctx.balloon_done = true;
- }
- canvas.end_group();
- return;
- }
-
- // Marquee: text starts just past the right inner edge and scrolls left.
- // A single "pass" travels `text_w + inner_w` pixels (entry + traverse +
- // exit). One full cycle adds `kRepeatGapPx` so the message restarts with
- // a perceivable gap.
- const std::int32_t one_pass_px = text_w + inner_w;
- const std::int32_t cycle_px = one_pass_px + kRepeatGapPx;
- const std::int32_t offset_in_cycle =
- static_cast(elapsed_ms) * kScrollSpeedPxPerSec / 1000 % cycle_px;
- const std::int32_t x = inner_x + inner_w - offset_in_cycle;
-
+ // Left-aligned; scrolls only when the text overflows the balloon (starts at
+ // the beginning of the text, then reveals the rest — see balloon_layout.hpp).
+ const BalloonScroll scroll = compute_balloon_scroll(text_w, inner_w, elapsed_ms, ctx.balloon_hold_ms);
canvas.setClipRect(inner_x, panel_y, inner_w, panel_h);
canvas.setTextDatum(lgfx::textdatum_t::middle_left);
- canvas.drawString(text.c_str(), x, mid_y);
+ canvas.drawString(text.c_str(), inner_x - scroll.offset_px, mid_y);
canvas.clearClipRect();
-
- // Mark done once the message has scrolled across at least once
- // (or the caller-requested hold time has elapsed, whichever is longer).
- const std::uint32_t one_pass_ms =
- static_cast(one_pass_px) * 1000u /
- static_cast(kScrollSpeedPxPerSec);
- const std::uint32_t complete_at = std::max(ctx.balloon_hold_ms, one_pass_ms);
- if (elapsed_ms >= complete_at) {
+ if (scroll.done) {
ctx.balloon_done = true;
}
canvas.end_group();
diff --git a/components/avatar/balloon_layout.hpp b/components/avatar/balloon_layout.hpp
new file mode 100644
index 0000000..8fb50a3
--- /dev/null
+++ b/components/avatar/balloon_layout.hpp
@@ -0,0 +1,65 @@
+// SPDX-FileCopyrightText: 2026 Kenta IDA
+// SPDX-License-Identifier: BSL-1.0
+//
+// 吹き出しの文字列の表示位置 (スクロール) の計算。描画から切り離した純粋な関数で、
+// ホストでテストできる。
+//
+// - 文字列が吹き出しに収まるなら、左詰めで静止表示する (スクロールしない)。
+// - 溢れるときだけスクロールする: 最初は左詰め (文頭が見える) で少し止まり、
+// 溢れた分だけ左へ動かして、文末が見えたところで止める。ループはしない。
+// - 表示時間 (hold_ms) が分かっているときは、その間に文末まで読めるよう速度を
+// 合わせる。チャンクごとに切り替わる吹き出しなら、次のチャンクが始まるまでに
+// 文末が見える。
+#pragma once
+
+#include
+#include
+
+namespace stackchan::avatar::internal {
+
+// 収まる文字列を最低限見せておく時間 (hold_ms 未指定のとき)。
+constexpr std::uint32_t kBalloonStaticHoldMs = 3000;
+// 溢れる文字列: スクロール前後の静止時間の上限と、既定 / 上限 / 下限のスクロール速度。
+constexpr std::uint32_t kBalloonEdgeHoldMs = 1000;
+constexpr std::int32_t kBalloonScrollPxPerSec = 60;
+constexpr std::int32_t kBalloonScrollMaxPxPerSec = 120;
+constexpr std::int32_t kBalloonScrollMinPxPerSec = 30;
+
+struct BalloonScroll {
+ std::int32_t offset_px = 0; // 左へ動かした量 (0 = 左詰め、travel = 文末が右端に揃う)
+ bool done = false; // 表示を終えてよい (呼び出し側が吹き出しを閉じる)
+};
+
+// text_w: 文字列の幅、inner_w: 吹き出しの内側の幅 [px]。
+// elapsed_ms: 表示してからの時間、hold_ms: 呼び出し側の指定表示時間 (0 = 未指定)。
+constexpr BalloonScroll compute_balloon_scroll(std::int32_t text_w, std::int32_t inner_w, std::uint32_t elapsed_ms,
+ std::uint32_t hold_ms)
+{
+ if (text_w <= inner_w) {
+ return {0, elapsed_ms >= std::max(hold_ms, kBalloonStaticHoldMs)};
+ }
+ const auto travel = static_cast(text_w - inner_w);
+ const std::uint32_t default_ms = travel * 1000u / kBalloonScrollPxPerSec;
+ const std::uint32_t min_ms = travel * 1000u / kBalloonScrollMaxPxPerSec;
+ const std::uint32_t max_ms = travel * 1000u / kBalloonScrollMinPxPerSec;
+
+ std::uint32_t edge_ms = kBalloonEdgeHoldMs; // 動き出し前 / 動いた後の静止
+ std::uint32_t move_ms = default_ms;
+ if (hold_ms > 0) {
+ // 指定時間の中で、前後の静止 (各 1/5、最大 1 秒) を除いた分を動かす。
+ edge_ms = std::min(kBalloonEdgeHoldMs, hold_ms / 5);
+ const std::uint32_t avail = hold_ms > 2 * edge_ms ? hold_ms - 2 * edge_ms : 0;
+ move_ms = std::clamp(avail, min_ms, max_ms);
+ }
+
+ BalloonScroll s;
+ if (elapsed_ms > edge_ms) {
+ const std::uint32_t t = elapsed_ms - edge_ms;
+ s.offset_px = t >= move_ms ? static_cast(travel)
+ : static_cast(static_cast(travel) * t / move_ms);
+ }
+ s.done = elapsed_ms >= std::max(hold_ms, edge_ms + move_ms + edge_ms);
+ return s;
+}
+
+} // namespace stackchan::avatar::internal
diff --git a/components/avatar/include/avatar/avatar.hpp b/components/avatar/include/avatar/avatar.hpp
index d4722f3..911086c 100644
--- a/components/avatar/include/avatar/avatar.hpp
+++ b/components/avatar/include/avatar/avatar.hpp
@@ -36,13 +36,15 @@ class Avatar {
// apply its face/background colours. Takes effect on the next tick(); safe
// to call live (e.g. from the render task on a config change).
void set_face_tuning(const FaceTuning& tuning) noexcept;
- // Show `text` in the balloon. `hold_ms` overrides the default display
- // time (0 = use balloon defaults: short text holds for a few seconds,
- // long text plays one full marquee pass).
+ // Show `text` left-aligned in the balloon. Text that does not fit scrolls
+ // once (start of the text first, then the rest is revealed); text that fits
+ // never scrolls. `hold_ms` is the on-screen time: 0 = defaults (fitting text
+ // holds a few seconds, overflowing text scrolls at a fixed speed); when set,
+ // overflowing text is scrolled so its end is reached within that time.
void set_balloon_text(std::string_view text, std::uint32_t hold_ms = 0);
void clear_balloon() noexcept;
// True once the current balloon has been fully displayed (hold elapsed
- // or one marquee pass completed). Stays true until the next
+ // or the scroll finished). Stays true until the next
// set_balloon_text / clear_balloon.
bool is_balloon_done() const noexcept;
diff --git a/components/avatar/test/host/CMakeLists.txt b/components/avatar/test/host/CMakeLists.txt
new file mode 100644
index 0000000..796a3a8
--- /dev/null
+++ b/components/avatar/test/host/CMakeLists.txt
@@ -0,0 +1,18 @@
+# SPDX-FileCopyrightText: 2026 Kenta IDA
+# SPDX-License-Identifier: BSL-1.0
+#
+# Host-side (ESP-IDF independent) tests for the balloon scroll layout
+# (components/avatar/balloon_layout.hpp — pure functions, header only).
+
+cmake_minimum_required(VERSION 3.16)
+project(avatar_host_test CXX)
+
+set(CMAKE_CXX_STANDARD 20)
+set(CMAKE_CXX_STANDARD_REQUIRED ON)
+set(CMAKE_CXX_EXTENSIONS OFF)
+
+enable_testing()
+
+add_executable(avatar_test_balloon_layout test_balloon_layout.cpp)
+target_compile_options(avatar_test_balloon_layout PRIVATE -Wall -Wextra)
+add_test(NAME balloon_layout COMMAND avatar_test_balloon_layout)
diff --git a/components/avatar/test/host/test_balloon_layout.cpp b/components/avatar/test/host/test_balloon_layout.cpp
new file mode 100644
index 0000000..06971c8
--- /dev/null
+++ b/components/avatar/test/host/test_balloon_layout.cpp
@@ -0,0 +1,94 @@
+// SPDX-FileCopyrightText: 2026 Kenta IDA
+// SPDX-License-Identifier: BSL-1.0
+//
+// 吹き出しのスクロール計算 (balloon_layout.hpp) の検証。
+#include
+
+#include "../../balloon_layout.hpp"
+
+using namespace stackchan::avatar::internal;
+
+namespace {
+
+int g_failures = 0;
+
+#define CHECK(cond) \
+ do { \
+ if (!(cond)) { \
+ std::fprintf(stderr, "FAIL %s:%d: %s\n", __FILE__, __LINE__, #cond); \
+ ++g_failures; \
+ } \
+ } while (0)
+
+} // namespace
+
+int main()
+{
+ constexpr std::int32_t kInner = 288; // CoreS3 の吹き出し内側 (320 - 2*4 - 2*8)
+
+ // 収まる文字列はスクロールしない (常に 0 = 左詰め)。既定では 3 秒見せて終了、
+ // hold_ms があればそれまで。
+ for (std::uint32_t t : {0u, 500u, 2999u}) {
+ const auto s = compute_balloon_scroll(200, kInner, t, 0);
+ CHECK(s.offset_px == 0 && !s.done);
+ }
+ CHECK(compute_balloon_scroll(200, kInner, 3000, 0).done);
+ CHECK(compute_balloon_scroll(kInner, kInner, 0, 0).offset_px == 0); // ちょうど収まる
+ CHECK(!compute_balloon_scroll(200, kInner, 3500, 5000).done);
+ CHECK(compute_balloon_scroll(200, kInner, 5000, 5000).done);
+
+ // 溢れる文字列 (hold なし): 最初は左詰めで 1 秒止まり、60 px/s で溢れた分だけ動き、
+ // 文末が右端に揃ったところで 1 秒止まって終了。ループしない。
+ {
+ const std::int32_t text_w = kInner + 300; // 溢れは 300 px → 5 秒で動く
+ CHECK(compute_balloon_scroll(text_w, kInner, 0, 0).offset_px == 0); // 文頭が見えている
+ CHECK(compute_balloon_scroll(text_w, kInner, 1000, 0).offset_px == 0); // まだ動かない
+ CHECK(compute_balloon_scroll(text_w, kInner, 2000, 0).offset_px == 60); // 1 秒動いた = 60 px
+ CHECK(compute_balloon_scroll(text_w, kInner, 6000, 0).offset_px == 300); // 文末が右端
+ CHECK(compute_balloon_scroll(text_w, kInner, 60000, 0).offset_px == 300); // それ以上は動かない
+ CHECK(!compute_balloon_scroll(text_w, kInner, 6999, 0).done);
+ CHECK(compute_balloon_scroll(text_w, kInner, 7000, 0).done); // 1 + 5 + 1 秒
+ // 単調増加 (途中で戻らない)。
+ std::int32_t prev = 0;
+ for (std::uint32_t t = 0; t <= 8000; t += 50) {
+ const auto s = compute_balloon_scroll(text_w, kInner, t, 0);
+ CHECK(s.offset_px >= prev && s.offset_px >= 0 && s.offset_px <= 300);
+ prev = s.offset_px;
+ }
+ }
+
+ // 溢れる文字列 (hold あり): その時間内に文末まで動く。前後の静止は 1/5 ずつ。
+ {
+ const std::int32_t text_w = kInner + 100; // 溢れ 100 px
+ const std::uint32_t hold = 3000; // 前後 600 ms、動くのは 1800 ms (≈ 55 px/s)
+ CHECK(compute_balloon_scroll(text_w, kInner, 0, hold).offset_px == 0);
+ CHECK(compute_balloon_scroll(text_w, kInner, 600, hold).offset_px == 0);
+ CHECK(compute_balloon_scroll(text_w, kInner, 2400, hold).offset_px == 100); // 動き終わり
+ CHECK(compute_balloon_scroll(text_w, kInner, hold - 1, hold).offset_px == 100);
+ CHECK(!compute_balloon_scroll(text_w, kInner, hold - 1, hold).done);
+ CHECK(compute_balloon_scroll(text_w, kInner, hold, hold).done);
+ }
+ // hold が短くて間に合わないときは最大速度 (120 px/s) で動く (それでも hold 内には終わらない)。
+ {
+ const std::int32_t text_w = kInner + 600; // 溢れ 600 px → 最速 120 px/s でも 5 秒
+ // hold=2000 → 前後の静止は各 400 ms。動き出して 800 ms 後 (elapsed=1200) は 120*0.8=96 px。
+ CHECK(compute_balloon_scroll(text_w, kInner, 1200, 2000).offset_px == 96);
+ CHECK(compute_balloon_scroll(text_w, kInner, 2000, 2000).offset_px == 192); // 動き出して 1600 ms = 120*1.6。hold を過ぎても動き続ける
+ CHECK(!compute_balloon_scroll(text_w, kInner, 2000, 2000).done);
+ CHECK(compute_balloon_scroll(text_w, kInner, 5400, 2000).offset_px == 600);
+ CHECK(!compute_balloon_scroll(text_w, kInner, 5799, 2000).done);
+ CHECK(compute_balloon_scroll(text_w, kInner, 5800, 2000).done); // 0.4 + 5 + 0.4 秒
+ }
+ // hold が長いときは最低速度 (30 px/s) まで落とす (それ以上は遅くしない)。
+ {
+ const std::int32_t text_w = kInner + 60; // 溢れ 60 px → 最長 2 秒
+ // hold = 30 秒: 前後は各 1 秒 (上限)、動くのは 2 秒 (30 px/s)、あとは静止のまま hold まで。
+ CHECK(compute_balloon_scroll(text_w, kInner, 1000, 30000).offset_px == 0);
+ CHECK(compute_balloon_scroll(text_w, kInner, 3000, 30000).offset_px == 60);
+ CHECK(!compute_balloon_scroll(text_w, kInner, 29999, 30000).done);
+ CHECK(compute_balloon_scroll(text_w, kInner, 30000, 30000).done);
+ }
+
+ if (g_failures == 0) std::puts("test_balloon_layout: all passed");
+ return g_failures == 0 ? 0 : 1;
+}
diff --git a/components/avatar_vm/include/avatar/draw_context.hpp b/components/avatar_vm/include/avatar/draw_context.hpp
index cd2b645..661cae6 100644
--- a/components/avatar_vm/include/avatar/draw_context.hpp
+++ b/components/avatar_vm/include/avatar/draw_context.hpp
@@ -30,15 +30,16 @@ struct DrawContext {
Palette palette{kDefaultPalette};
std::uint32_t rng_state{0xC0FFEEu};
std::optional balloon_text{};
- // Wall-clock used for time-based animation (e.g. balloon marquee).
+ // Wall-clock used for time-based animation (e.g. balloon scroll).
std::uint32_t now_ms{0};
- // Set to `now_ms` whenever balloon_text changes — drives marquee phase.
+ // Set to `now_ms` whenever balloon_text changes — drives the scroll phase.
std::uint32_t balloon_set_ms{0};
- // Minimum display time. 0 means "use balloon defaults" (short = a fixed
- // hold, long = one marquee pass).
+ // Display time. 0 means "use balloon defaults" (fitting text = a fixed
+ // hold, overflowing text = one scroll at a fixed speed); otherwise the
+ // scroll of overflowing text is timed to finish within it.
std::uint32_t balloon_hold_ms{0};
// Set by balloon rendering once the message has been displayed in full
- // (i.e. hold time elapsed for short text, or one marquee cycle for long
+ // (i.e. hold time elapsed for short text, or the scroll finished for long
// text). The render task polls this and notifies the application.
bool balloon_done{false};
};
diff --git a/main/shared_state.hpp b/main/shared_state.hpp
index 161ae06..b70ba43 100644
--- a/main/shared_state.hpp
+++ b/main/shared_state.hpp
@@ -322,10 +322,11 @@ class SharedState {
// --- Balloon (mutex + completion callback; render_task consumes) -------
// Show `text` in the balloon.
- // - hold_ms: minimum on-screen time (0 = use avatar defaults — short
- // text holds a few seconds, long text plays one marquee pass).
+ // - hold_ms: on-screen time (0 = use avatar defaults — fitting text holds
+ // a few seconds, overflowing text scrolls once; with a value the scroll
+ // is timed to reach the end of the text within it).
// - on_complete: invoked once when the balloon finishes (after hold or
- // after a marquee pass). Fired from the render task; the
+ // after the scroll). Fired from the render task; the
// implementation must be cheap and thread-safe.
void set_balloon_text(std::string_view text,
std::uint32_t hold_ms = 0,
From 89ae6e3d97044a1b90ab93bdb443bee90b76298a Mon Sep 17 00:00:00 2001
From: Takao Akaki
Date: Mon, 21 Sep 2026 18:01:54 +0900
Subject: [PATCH 03/10] =?UTF-8?q?fix(balloon):=20=E5=89=8D=E3=81=AE?=
=?UTF-8?q?=E5=90=B9=E3=81=8D=E5=87=BA=E3=81=97=E3=81=AE=E5=AE=8C=E4=BA=86?=
=?UTF-8?q?=E9=80=9A=E7=9F=A5=E3=81=8C=E6=AC=A1=E3=81=AE=E5=90=B9=E3=81=8D?=
=?UTF-8?q?=E5=87=BA=E3=81=97=E3=82=92=E6=B6=88=E3=81=99=E7=AB=B6=E5=90=88?=
=?UTF-8?q?=E3=82=92=E8=A7=A3=E6=B6=88?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
描画タスクは毎フレーム「新しい吹き出しを反映 → 描画 → 表示が終わっていれば
notify_balloon_complete()」の順に処理する。notify_balloon_complete() は
どの吹き出しの完了かを見ずに共有状態の *現在の* 吹き出しを消していたため、
反映と通知の間に次の吹き出しが set されると、それが一度も描画されないまま
消えていた。吹き出しを短い間隔で差し替える使い方 (チャンクごとの字幕) で
表示時間と切り替えが重なり、「たまに吹き出しが出ない」症状になる。
- notify_balloon_complete(version): 終わった吹き出しの版を受け取り、現在の
版と違えば古い通知として無視する。
- snapshot_balloon(): テキスト・hold・版を同じロックの下で取得する。
- 描画タスクは反映した吹き出しの版を保持して通知に渡す。
Co-Authored-By: Claude Sonnet 5
---
main/render_task.cpp | 11 ++++++++---
main/shared_state.hpp | 25 +++++++++++++++++--------
2 files changed, 25 insertions(+), 11 deletions(-)
diff --git a/main/render_task.cpp b/main/render_task.cpp
index 10a9e8d..8c0a78c 100644
--- a/main/render_task.cpp
+++ b/main/render_task.cpp
@@ -138,6 +138,7 @@ void render_task_entry(void* arg)
int last_expression = -1;
std::uint32_t last_balloon_version = 0;
+ std::uint32_t balloon_applied_version = 0; // version of the balloon the avatar is showing
std::uint32_t last_face_config_version = 0;
std::uint32_t last_face_bytecode_version = 0;
std::string balloon_scratch;
@@ -225,14 +226,15 @@ void render_task_entry(void* arg)
if (balloon_version != last_balloon_version) {
if (args.state->balloon_visible()) {
std::uint32_t hold_ms = 0;
- args.state->snapshot_balloon(balloon_scratch, hold_ms);
+ args.state->snapshot_balloon(balloon_scratch, hold_ms, balloon_applied_version);
avatar.set_balloon_text(balloon_scratch, hold_ms);
balloon_pending = true;
+ last_balloon_version = balloon_applied_version;
} else {
avatar.clear_balloon();
balloon_pending = false;
+ last_balloon_version = balloon_version;
}
- last_balloon_version = balloon_version;
}
// avatar.tick() opens the frame (begin_frame) and draws the face;
@@ -263,7 +265,10 @@ void render_task_entry(void* arg)
if (balloon_pending && avatar.is_balloon_done()) {
balloon_pending = false;
- args.state->notify_balloon_complete();
+ // Completion of *this* balloon only: if the next one was set in the
+ // meantime (chunk subtitles), the notify is ignored and it gets applied
+ // on the next loop instead of being wiped.
+ args.state->notify_balloon_complete(balloon_applied_version);
}
// Use vTaskDelay (not vTaskDelayUntil) so the IDLE task on this core
diff --git a/main/shared_state.hpp b/main/shared_state.hpp
index b70ba43..d04084e 100644
--- a/main/shared_state.hpp
+++ b/main/shared_state.hpp
@@ -351,16 +351,22 @@ class SharedState {
balloon_visible_.store(false, std::memory_order_release);
}
- // Called by the render task when the avatar finishes displaying the
- // current balloon. Hides the balloon and invokes the completion callback
- // (if any) outside the lock.
- void notify_balloon_complete()
+ // Called by the render task when the avatar finishes displaying a balloon.
+ // `version` is the balloon_version() of the balloon that finished (the one
+ // snapshot_balloon() returned when the render task applied it). If the
+ // balloon has been replaced or cleared since — e.g. the next chunk's
+ // subtitle was set while the previous one was timing out — the completion is
+ // stale and is ignored, so it can never hide the newer balloon. Otherwise
+ // hides the balloon and invokes the completion callback (if any) outside the
+ // lock.
+ void notify_balloon_complete(std::uint32_t version)
{
BalloonCompletionCallback cb;
{
std::lock_guard lock{balloon_mutex_};
- if (!balloon_visible_.load(std::memory_order_relaxed)) {
- return; // already cleared
+ if (!balloon_visible_.load(std::memory_order_relaxed) ||
+ balloon_version_.load(std::memory_order_relaxed) != version) {
+ return; // already cleared, or replaced by a newer balloon
}
balloon_text_.clear();
balloon_hold_ms_ = 0;
@@ -385,12 +391,15 @@ class SharedState {
return balloon_visible_.load(std::memory_order_acquire);
}
- // Copies the current text + hold time into the supplied outputs.
- void snapshot_balloon(std::string& text_out, std::uint32_t& hold_ms_out) const
+ // Copies the current text + hold time + version into the supplied outputs
+ // (one consistent snapshot: all three are read under the lock). Pass the
+ // version back to notify_balloon_complete() when this balloon finishes.
+ void snapshot_balloon(std::string& text_out, std::uint32_t& hold_ms_out, std::uint32_t& version_out) const
{
std::lock_guard lock{balloon_mutex_};
text_out = balloon_text_;
hold_ms_out = balloon_hold_ms_;
+ version_out = balloon_version_.load(std::memory_order_relaxed);
}
// --- Versioned slots (VersionedValue facade — see the template above) --
From 523112c8d05d9ab78eae9696a8f0d458969b2d29 Mon Sep 17 00:00:00 2001
From: Takao Akaki
Date: Mon, 21 Sep 2026 18:01:54 +0900
Subject: [PATCH 04/10] =?UTF-8?q?feat(avatar):=20DSL=20=E5=A4=89=E6=95=B0?=
=?UTF-8?q?=20mouth=5Fform=20=E3=82=92=E8=BF=BD=E5=8A=A0=20=E2=80=94=20?=
=?UTF-8?q?=E5=8F=A3=E3=81=AE=E5=B9=85=E3=82=92=E9=96=8B=E3=81=8D=E3=81=8B?=
=?UTF-8?q?=E3=82=89=E7=8B=AC=E7=AB=8B=E3=81=AB=E5=88=B6=E5=BE=A1=E3=81=99?=
=?UTF-8?q?=E3=82=8B?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
従来の口は幅・高さとも mouth_open だけで決まり、「開くほど幅が狭くなる」
形しか作れなかった。母音ごとに口の形を変えるため、幅を決める変数
mouth_form (0 = 横広 .. 1 = すぼめ) を追加する。
- Var::MouthForm = 0x1F (opcodes.hpp / opcodes.js)。
- Avatar::set_mouth_form()。負の値 (既定 -1) は「未設定」で、mouth_open と
同値になるため、mouth_open だけを駆動する既存の使い方 (マイクのレベル
メーター等) は見た目が変わらない。
- default_face.avdsl の口幅を mouth_form で決める。
- wasm に avatar_set_mouth_form を export。
- docs/avatar_dsl.md の変数表 (未記載だった cheek 系も追記)、VM ホスト テスト。
Co-Authored-By: Claude Sonnet 5
---
assets/default_face.avdsl | 12 ++++---
components/avatar/avatar.cpp | 10 ++++++
components/avatar/include/avatar/avatar.hpp | 3 ++
.../avatar_vm/include/avatar/draw_context.hpp | 5 +++
.../avatar_vm/include/avatar_vm/opcodes.hpp | 1 +
components/avatar_vm/test/host/test_vm.cpp | 35 +++++++++++++++++--
components/avatar_vm/vm.cpp | 1 +
docs/avatar_dsl.md | 9 +++--
tools/avatar_dsl/README.md | 1 +
tools/avatar_dsl/opcodes.js | 1 +
wasm/avatar_wasm.cpp | 6 ++++
wasm/build.sh | 2 +-
12 files changed, 76 insertions(+), 10 deletions(-)
diff --git a/assets/default_face.avdsl b/assets/default_face.avdsl
index 17ceeee..14d727c 100644
--- a/assets/default_face.avdsl
+++ b/assets/default_face.avdsl
@@ -10,13 +10,17 @@
-- A `fn draw()` is mandatory (entry point, fn id 0). Helper functions follow.
----------------------------------------------------------------------
--- Mouth: single rectangle whose width shrinks and height grows with
--- mouth_open. The group box covers the maximum extent so the direct
--- canvas strategy clears stale frames correctly.
+-- Mouth: single rectangle whose height grows with mouth_open and whose width
+-- shrinks with mouth_form (0 = wide, 1 = narrow). Hosts that only drive
+-- mouth_open (level meters) get mouth_form == mouth_open, i.e. the classic
+-- "narrower as it opens" look; TTS vowel lip-sync sets the two independently
+-- (い = wide + slightly open, う = narrow, あ = wide-ish + fully open).
+-- The group box covers the maximum extent so the direct canvas strategy
+-- clears stale frames correctly.
----------------------------------------------------------------------
fn mouth(cx, cy, min_w, max_w, min_h, max_h, bo)
let h = min_h + (max_h - min_h) * mouth_open
- let w = min_w + (max_w - min_w) * (1 - mouth_open)
+ let w = min_w + (max_w - min_w) * (1 - mouth_form)
let x = cx - w / 2
let y = cy - h / 2 + breath * 2 + bo
let gx = cx - max_w / 2 - 1
diff --git a/components/avatar/avatar.cpp b/components/avatar/avatar.cpp
index 5e064ae..6dfc08d 100644
--- a/components/avatar/avatar.cpp
+++ b/components/avatar/avatar.cpp
@@ -134,6 +134,16 @@ void Avatar::set_mouth_open(float ratio) noexcept
impl_->context().mouth_open_ratio = ratio;
}
+void Avatar::set_mouth_form(float ratio) noexcept
+{
+ if (ratio > 1.0f) {
+ ratio = 1.0f;
+ } else if (ratio < 0.0f) {
+ ratio = -1.0f;
+ }
+ impl_->context().mouth_form_ratio = ratio;
+}
+
void Avatar::set_gaze(float horizontal, float vertical) noexcept
{
impl_->context().gaze_horizontal = horizontal;
diff --git a/components/avatar/include/avatar/avatar.hpp b/components/avatar/include/avatar/avatar.hpp
index 911086c..82bbc6c 100644
--- a/components/avatar/include/avatar/avatar.hpp
+++ b/components/avatar/include/avatar/avatar.hpp
@@ -30,6 +30,9 @@ class Avatar {
void set_expression(Expression expression) noexcept;
void set_mouth_open(float ratio) noexcept;
+ // Mouth shape, 0 = wide .. 1 = narrow (see DrawContext::mouth_form_ratio).
+ // A negative value clears it: the shape then follows set_mouth_open.
+ void set_mouth_form(float ratio) noexcept;
void set_gaze(float horizontal, float vertical) noexcept;
void set_palette(const Palette& palette) noexcept;
// Rebuild the face layout from user tuning (eye/eyebrow/mouth geometry) and
diff --git a/components/avatar_vm/include/avatar/draw_context.hpp b/components/avatar_vm/include/avatar/draw_context.hpp
index 661cae6..a50ef27 100644
--- a/components/avatar_vm/include/avatar/draw_context.hpp
+++ b/components/avatar_vm/include/avatar/draw_context.hpp
@@ -27,6 +27,11 @@ struct DrawContext {
float gaze_saccade_v{0.0f};
float eye_open_ratio{1.0f};
float mouth_open_ratio{0.0f};
+ // Mouth shape: 0 = wide, 1 = narrow (pursed). Set by Avatar::set_mouth_form
+ // (e.g. TTS vowel lip-sync: い is wide, う is narrow). Negative = "not set":
+ // Var::MouthForm then falls back to mouth_open_ratio so faces driven only
+ // by a level meter keep the classic "narrower as it opens" behaviour.
+ float mouth_form_ratio{-1.0f};
Palette palette{kDefaultPalette};
std::uint32_t rng_state{0xC0FFEEu};
std::optional balloon_text{};
diff --git a/components/avatar_vm/include/avatar_vm/opcodes.hpp b/components/avatar_vm/include/avatar_vm/opcodes.hpp
index 026f003..0698f11 100644
--- a/components/avatar_vm/include/avatar_vm/opcodes.hpp
+++ b/components/avatar_vm/include/avatar_vm/opcodes.hpp
@@ -103,6 +103,7 @@ enum class Var : std::uint8_t {
CheekRadius = 0x1C,
CheekOffX = 0x1D,
CheekOffY = 0x1E,
+ MouthForm = 0x1F,
VarCount,
};
diff --git a/components/avatar_vm/test/host/test_vm.cpp b/components/avatar_vm/test/host/test_vm.cpp
index ff507f7..78f92b0 100644
--- a/components/avatar_vm/test/host/test_vm.cpp
+++ b/components/avatar_vm/test/host/test_vm.cpp
@@ -48,7 +48,8 @@ struct RunResult {
RecordingCanvas canvas{320, 240};
};
-RunResult run_program(const std::vector& buf)
+RunResult run_program(const std::vector& buf,
+ const stackchan::avatar::DrawContext& ctx = stackchan::avatar::DrawContext{})
{
RunResult rr;
rr.decoded = decode(std::span(buf));
@@ -56,7 +57,6 @@ RunResult run_program(const std::vector& buf)
rr.ran = tl::unexpected(rr.decoded.error());
return rr;
}
- stackchan::avatar::DrawContext ctx;
stackchan::avatar::FaceTuning tuning;
Vm vm;
rr.ran = vm.run(*rr.decoded, rr.canvas, ctx, tuning);
@@ -187,6 +187,37 @@ int main()
CHECK(rr.canvas.ops.size() == 1 && rr.canvas.ops[0].a == 320);
}
+ // --- PushVar MouthForm: explicit value wins, unset follows MouthOpen ---
+ {
+ BytecodeBuilder b;
+ b.code(PUSH_VAR);
+ b.code(0x1F); // Var::MouthForm
+ b.code(PUSH_I8);
+ b.code(1);
+ b.code(PUSH_I8);
+ b.code(1);
+ b.code(PUSH_I8);
+ b.code(3);
+ b.code(FILL_CIRCLE);
+ b.code(RET);
+ b.add_fn(0, 0, 0);
+ const auto buf = b.build(0);
+
+ stackchan::avatar::DrawContext ctx;
+ ctx.mouth_open_ratio = 1.0f;
+ auto rr = run_program(buf, ctx); // form unset (-1) → follows open
+ CHECK(rr.ran.has_value() && rr.canvas.ops.size() == 1 && rr.canvas.ops[0].a == 1);
+
+ ctx.mouth_form_ratio = 0.0f; // explicitly wide, even though fully open
+ rr = run_program(buf, ctx);
+ CHECK(rr.ran.has_value() && rr.canvas.ops.size() == 1 && rr.canvas.ops[0].a == 0);
+
+ ctx.mouth_open_ratio = 0.0f;
+ ctx.mouth_form_ratio = 1.0f; // explicitly narrow, even though closed
+ rr = run_program(buf, ctx);
+ CHECK(rr.ran.has_value() && rr.canvas.ops.size() == 1 && rr.canvas.ops[0].a == 1);
+ }
+
// --- function call with a parameter ----------------------------------
{
BytecodeBuilder b;
diff --git a/components/avatar_vm/vm.cpp b/components/avatar_vm/vm.cpp
index 5c754c8..70b331f 100644
--- a/components/avatar_vm/vm.cpp
+++ b/components/avatar_vm/vm.cpp
@@ -74,6 +74,7 @@ inline float read_var(Var v, const avatar::Canvas& canvas, const avatar::DrawCon
case Var::GazeH: return ctx.gaze_horizontal + ctx.gaze_saccade_h;
case Var::GazeV: return ctx.gaze_vertical + ctx.gaze_saccade_v;
case Var::MouthOpen: return ctx.mouth_open_ratio;
+ case Var::MouthForm: return ctx.mouth_form_ratio >= 0.0f ? ctx.mouth_form_ratio : ctx.mouth_open_ratio;
case Var::Expr: return expression_to_f(ctx.expression);
case Var::Primary: return static_cast(ctx.palette.primary);
case Var::Background: return static_cast(ctx.palette.background);
diff --git a/docs/avatar_dsl.md b/docs/avatar_dsl.md
index 5de4fd4..8185cc4 100644
--- a/docs/avatar_dsl.md
+++ b/docs/avatar_dsl.md
@@ -213,6 +213,7 @@ end_group()
| `eye_open` | float | 0..1 | まばたき (0 = 閉) |
| `gaze_h`, `gaze_v` | float | -1..+1 | 視線サッカード |
| `mouth_open` | float | 0..1 | 口の開き |
+| `mouth_form` | float | 0..1 | 口の形 (0 = 横広、1 = すぼめ)。ホストが設定しない場合は `mouth_open` と同値 |
| `expr` | enum | 0..5 | 表情 (下記定数で名前指定可) |
| `primary` | u16 → float | RGB565 | 前景色 (デフォルト 白 `0xFFFF`) |
| `background` | u16 → float | RGB565 | 背景色 (デフォルト 黒 `0x0000`) |
@@ -436,9 +437,11 @@ end
| 0x04 | `breath` | 0x0E | `balloon_bg` | 0x18 | `mouth_min_h` |
| 0x05 | `eye_open` | 0x0F | `eye_radius` | 0x19 | `mouth_max_h` |
| 0x06 | `gaze_h` | 0x10 | `eye_off_x` | 0x1A | `eyebrows_visible` |
-| 0x07 | `gaze_v` | 0x11 | `eye_off_y` | | |
-| 0x08 | `mouth_open` | 0x12 | `brow_off_x` | | |
-| 0x09 | `expr` | 0x13 | `brow_off_y` | | |
+| 0x07 | `gaze_v` | 0x11 | `eye_off_y` | 0x1B | `cheeks_visible` |
+| 0x08 | `mouth_open` | 0x12 | `brow_off_x` | 0x1C | `cheek_radius` |
+| 0x09 | `expr` | 0x13 | `brow_off_y` | 0x1D | `cheek_off_x` |
+| | | | | 0x1E | `cheek_off_y` |
+| | | | | 0x1F | `mouth_form` |
> 真実源: [components/avatar_vm/include/avatar_vm/opcodes.hpp](https://github.com/ciniml/stackchan-idf/blob/main/components/avatar_vm/include/avatar_vm/opcodes.hpp)
> (C++ 側) / [tools/avatar_dsl/opcodes.js](https://github.com/ciniml/stackchan-idf/blob/main/tools/avatar_dsl/opcodes.js) (JS 側 ミラー)
diff --git a/tools/avatar_dsl/README.md b/tools/avatar_dsl/README.md
index 3651419..83a4c40 100644
--- a/tools/avatar_dsl/README.md
+++ b/tools/avatar_dsl/README.md
@@ -69,6 +69,7 @@ while cond do ... end -- 無限ループ防止策は無し、
-- read-only コンテキスト変数 (ホストから注入):
-- canvas_w canvas_h canvas_scale now_ms
-- breath eye_open gaze_h gaze_v mouth_open -- 0..1 / -1..1
+-- mouth_form -- 0 (wide) .. 1 (narrow)。未設定時は mouth_open と同値
-- expr -- enum 0..5
-- primary background secondary balloon_fg balloon_bg -- RGB565
-- eye_radius eye_off_x eye_off_y
diff --git a/tools/avatar_dsl/opcodes.js b/tools/avatar_dsl/opcodes.js
index acf52d9..6bdb25a 100644
--- a/tools/avatar_dsl/opcodes.js
+++ b/tools/avatar_dsl/opcodes.js
@@ -72,6 +72,7 @@ export const Var = Object.freeze({
cheek_radius: 0x1C,
cheek_off_x: 0x1D,
cheek_off_y: 0x1E,
+ mouth_form: 0x1F,
});
export const ConstTag = Object.freeze({
diff --git a/wasm/avatar_wasm.cpp b/wasm/avatar_wasm.cpp
index bbfec1c..17b36d2 100644
--- a/wasm/avatar_wasm.cpp
+++ b/wasm/avatar_wasm.cpp
@@ -248,6 +248,12 @@ EMSCRIPTEN_KEEPALIVE void avatar_set_mouth(float ratio)
g_ctx.mouth_open_ratio = ratio < 0.0f ? 0.0f : (ratio > 1.0f ? 1.0f : ratio);
}
+// Mouth shape, 0 = wide .. 1 = narrow; negative = follow mouth_open.
+EMSCRIPTEN_KEEPALIVE void avatar_set_mouth_form(float ratio)
+{
+ g_ctx.mouth_form_ratio = ratio < 0.0f ? -1.0f : (ratio > 1.0f ? 1.0f : ratio);
+}
+
EMSCRIPTEN_KEEPALIVE void avatar_set_manual_gaze(int on, float h, float v)
{
g_manual_gaze = on != 0;
diff --git a/wasm/build.sh b/wasm/build.sh
index 17d31fa..1c21338 100755
--- a/wasm/build.sh
+++ b/wasm/build.sh
@@ -76,7 +76,7 @@ node "$ROOT/tools/avatar_dsl/inject.mjs" \
"omega=$ROOT/assets/omega_mouth.avdsl" \
"aokko=$ROOT/assets/aokko_face.avdsl"
-EXPORTS='_avatar_init,_avatar_set_size,_avatar_width,_avatar_height,_avatar_framebuffer,_avatar_set_expression,_avatar_set_mouth,_avatar_set_manual_gaze,_avatar_set_saccade,_avatar_set_blink,_avatar_set_breath,_avatar_set_colors,_avatar_set_eyebrows_visible,_avatar_set_eye_params,_avatar_set_eyebrow_params,_avatar_set_mouth_params,_avatar_set_cheeks_visible,_avatar_set_cheek_params,_avatar_set_direct,_avatar_tick,_avatar_load_bytecode,_avatar_reset_bytecode,_malloc,_free'
+EXPORTS='_avatar_init,_avatar_set_size,_avatar_width,_avatar_height,_avatar_framebuffer,_avatar_set_expression,_avatar_set_mouth,_avatar_set_mouth_form,_avatar_set_manual_gaze,_avatar_set_saccade,_avatar_set_blink,_avatar_set_breath,_avatar_set_colors,_avatar_set_eyebrows_visible,_avatar_set_eye_params,_avatar_set_eyebrow_params,_avatar_set_mouth_params,_avatar_set_cheeks_visible,_avatar_set_cheek_params,_avatar_set_direct,_avatar_tick,_avatar_load_bytecode,_avatar_reset_bytecode,_malloc,_free'
# Shared compile inputs/flags for both outputs (same C++, same exports). Kept
# on single lines so the values interpolate cleanly into the `bash -c` script
From 5369a6b3587e7813c55f24c43ca10594019d9e89 Mon Sep 17 00:00:00 2001
From: Takao Akaki
Date: Mon, 21 Sep 2026 18:01:54 +0900
Subject: [PATCH 05/10] =?UTF-8?q?feat(speech):=20=E3=82=A2=E3=82=A4?=
=?UTF-8?q?=E3=83=89=E3=83=AB=E7=99=BA=E8=A9=B1=E3=83=95=E3=83=AC=E3=83=BC?=
=?UTF-8?q?=E3=82=BA=E3=81=AE=E9=A0=86=E5=BA=8F=20(=E3=83=A9=E3=83=B3?=
=?UTF-8?q?=E3=83=80=E3=83=A0=20/=20=E9=80=A3=E7=B6=9A)=20=E3=82=92?=
=?UTF-8?q?=E9=81=B8=E3=81=B9=E3=82=8B=E3=82=88=E3=81=86=E3=81=AB=E3=81=99?=
=?UTF-8?q?=E3=82=8B?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
jtts 設定 JSON に "phrase_order" ("random" | "sequential") を追加する。
random (既定・従来どおり) は毎回どれかを選び、sequential は一覧を上の行から
順に発話して最後まで行ったら先頭に戻る (設定を送り直すと先頭から)。
未指定・不明な値は random。
設定ページ (BLE 版 tools/settings.html、Wi-Fi 版 settings_wifi.html) の
アイドル発話フレーズ欄の下に「発話の順序」を追加する。random のときは
キーを出さないので、古いファームでもそのまま読める。
Co-Authored-By: Claude Sonnet 5
---
.../web/settings_wifi.html | 7 ++++++
main/speech.cpp | 24 ++++++++++++++++---
main/speech.hpp | 10 +++++++-
tools/settings.html | 11 +++++++++
4 files changed, 48 insertions(+), 4 deletions(-)
diff --git a/components/wifi_config_service/web/settings_wifi.html b/components/wifi_config_service/web/settings_wifi.html
index bf65ca7..78c6140 100644
--- a/components/wifi_config_service/web/settings_wifi.html
+++ b/components/wifi_config_service/web/settings_wifi.html
@@ -716,6 +716,11 @@ 単位連結 音声 DB (.jvox)
無視されます)。| を省くと表示と読みが同じになります。
例: こんにちは | こんにちわ
+
+
LT タイムキーパー
@@ -1142,6 +1147,7 @@
// `表示 | 読み` pair support shared with the BLE page (settings_common.js).
const phrases = StackchanSettings.jttsPhrasesFromText($('jtts-phrases').value);
if (phrases.length) obj.phrases = phrases;
+ if ($('jtts-phrase-order').value === 'sequential') obj.phrase_order = 'sequential';
return Object.keys(obj).length ? JSON.stringify(obj) : '';
}
@@ -1160,6 +1166,7 @@
if (Array.isArray(obj.phrases)) {
$('jtts-phrases').value = StackchanSettings.jttsPhrasesToText(obj.phrases);
}
+ $('jtts-phrase-order').value = obj.phrase_order === 'sequential' ? 'sequential' : 'random';
}
// --- Servo limits + range-setting calibration ---
diff --git a/main/speech.cpp b/main/speech.cpp
index f902e10..b90be89 100644
--- a/main/speech.cpp
+++ b/main/speech.cpp
@@ -177,6 +177,8 @@ void Speech::configure(const std::string& json)
for (const auto& p : kDefaultPhrases) {
phrases_.push_back({std::string(p.display), std::u32string(p.reading)});
}
+ phrase_order_ = PhraseOrder::Random;
+ next_phrase_ = 0;
initialised_ = true;
if (json.empty()) {
@@ -190,6 +192,14 @@ void Speech::configure(const std::string& json)
apply_options_json(opts_, root);
+ // phrase_order: "random" (default) or "sequential" (top to bottom, looping).
+ // Unknown / missing values keep the default.
+ const cJSON* order = cJSON_GetObjectItemCaseSensitive(root, "phrase_order");
+ if (cJSON_IsString(order) && order->valuestring != nullptr &&
+ std::strcmp(order->valuestring, "sequential") == 0) {
+ phrase_order_ = PhraseOrder::Sequential;
+ }
+
// phrases: array whose elements are either
// - a string "こんにちわ" (display == reading), or
// - an object {"text":"こんにちは","reading":"こんにちわ"}
@@ -220,9 +230,10 @@ void Speech::configure(const std::string& json)
if (!parsed.empty()) phrases_ = std::move(parsed);
}
cJSON_Delete(root);
- ESP_LOGI(kTag, "jtts config: voice=%s f0=%.0f mora=%.0fms phrases=%zu",
+ ESP_LOGI(kTag, "jtts config: voice=%s f0=%.0f mora=%.0fms phrases=%zu order=%s",
opts_.voice == jtts::Voice::Female ? "female" : "male",
- opts_.f0_hz, opts_.mora_ms, phrases_.size());
+ opts_.f0_hz, opts_.mora_ms, phrases_.size(),
+ phrase_order_ == PhraseOrder::Sequential ? "sequential" : "random");
}
std::string Speech::babble(std::uint32_t seed)
@@ -233,7 +244,14 @@ std::string Speech::babble(std::uint32_t seed)
if (phrases_.empty()) {
return {};
}
- const Phrase& phrase = phrases_[seed % phrases_.size()];
+ std::size_t index;
+ if (phrase_order_ == PhraseOrder::Sequential) {
+ index = next_phrase_ % phrases_.size();
+ next_phrase_ = index + 1; // stays < size + 1, so it never overflows
+ } else {
+ index = seed % phrases_.size();
+ }
+ const Phrase& phrase = phrases_[index];
// Couldn't pronounce → still return the display text so the caller shows
// the matching balloon (no audio / mouth movement in that case).
(void)say(phrase.reading);
diff --git a/main/speech.hpp b/main/speech.hpp
index a2d0df6..48e8a00 100644
--- a/main/speech.hpp
+++ b/main/speech.hpp
@@ -40,8 +40,14 @@ class Speech {
// ignored entirely. Call once at startup, before the first babble.
void configure(const std::string& json);
+ // Which phrase babble() picks next. Random (default) uses the caller's
+ // seed; Sequential walks the phrase list top to bottom and wraps around.
+ enum class PhraseOrder : std::uint8_t { Random, Sequential };
+
// Start a fresh utterance (non-blocking — M5.Speaker queues it).
- // `seed` selects which phrase to speak (seed % phrase count). Returns the
+ // Random order: `seed` selects the phrase (seed % phrase count).
+ // Sequential order: `seed` is ignored, the next line in the list is used
+ // (restarting from the first line after configure()). Returns the
// *display* text (発話内容) of the chosen phrase so the caller can show a
// matching balloon — synthesis uses that phrase's separate *reading*
// (発声内容, kana). Returns an empty string only when there are no phrases.
@@ -85,6 +91,8 @@ class Speech {
// preset, ~8 short Japanese phrases).
jtts::Options opts_;
std::vector phrases_;
+ PhraseOrder phrase_order_{PhraseOrder::Random};
+ std::size_t next_phrase_{0}; // Sequential: index of the phrase babble() speaks next
bool initialised_{false};
std::atomic start_ms_{0};
diff --git a/tools/settings.html b/tools/settings.html
index 9b04e8c..22fef55 100644
--- a/tools/settings.html
+++ b/tools/settings.html
@@ -809,6 +809,11 @@ jtts ボイス
(漢字は発声時に無視されます)。| を省くと表示と読みが同じになります。
例: こんにちは | こんにちわ
+
+
@@ -1779,6 +1784,8 @@
ファームウェア更新 (OTA)
const phrases = StackchanSettings.jttsPhrasesFromText(
document.getElementById('jtts-phrases').value);
if (phrases.length > 0) obj.phrases = phrases;
+ const order = document.getElementById('jtts-phrase-order').value;
+ if (order === 'sequential') obj.phrase_order = order;
return obj;
}
@@ -1791,6 +1798,7 @@
ファームウェア更新 (OTA)
document.getElementById(id).value = '';
}
document.getElementById('jtts-phrases').value = '';
+ document.getElementById('jtts-phrase-order').value = 'random';
if (!json) return;
let obj;
try { obj = JSON.parse(json); }
@@ -1813,6 +1821,9 @@
ファームウェア更新 (OTA)
document.getElementById('jtts-phrases').value =
StackchanSettings.jttsPhrasesToText(obj.phrases);
}
+ if (obj.phrase_order === 'sequential') {
+ document.getElementById('jtts-phrase-order').value = 'sequential';
+ }
}
// Serialise form state for change-detection. JSON.stringify on a fresh object
From ee4822571ec5f3b32085a67908a4adf09e4eae3e Mon Sep 17 00:00:00 2001
From: Takao Akaki
Date: Mon, 21 Sep 2026 18:02:16 +0900
Subject: [PATCH 06/10] =?UTF-8?q?feat(jtts):=20=E5=8F=A3=E5=BD=A2=E3=82=A4?=
=?UTF-8?q?=E3=83=99=E3=83=B3=E3=83=88=E3=83=BB=E9=95=B7=E6=96=87=E3=81=AE?=
=?UTF-8?q?=E5=88=86=E5=89=B2=E5=90=88=E6=88=90=E3=83=BB=E3=82=B9=E3=83=88?=
=?UTF-8?q?=E3=83=AA=E3=83=BC=E3=83=9F=E3=83=B3=E3=82=B0=E5=90=88=E6=88=90?=
=?UTF-8?q?=20API=20=E3=82=92=E8=BF=BD=E5=8A=A0=E3=81=99=E3=82=8B?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
■ 口形イベント (リップシンク用)
- synthesize(kana, pcm, visemes, opt): PCM と時間軸の揃った VisemeEvent 列
(母音 あ/い/う/え/お、閉口 = Vowel::None) を返す。
- フォルマント: Segment に口形を持たせる。両唇音 (m b p) は閉口、それ以外の
子音は後続母音の形を先取り、無声化母音・撥音・促音は閉口、長音は直前を保つ。
- HMM: 音素ラベル (p3) と hts_engine の状態継続長から作る (同じ規則)。
sanoTTS / 単位連結は口形を出せない (空) ので、呼び出し側は音量で代用する。
■ 長文の分割合成 (HMM)
- hts_engine は合成中フレーム数に比例して作業メモリ (≈ 2 KB / 5 ms フレーム)
を確保し続け、確保失敗はクラッシュになる (146 文字 ≈ 11 MB > PSRAM)。
空き PSRAM から 1 チャンクで扱えるモーラ数を見積もり、句読点 (、。)・
アクセント句境界 (/)、最後の手段でモーラ境界で分割して順に合成する。
境界の無音は元の pau と同じ長さに揃える。全体が収まる文は従来と同一出力。
- 予算に 1 モーラも入らない場合は HMM を諦めて次のエンジンにフォールバック。
- PCM が空きメモリに収まらない巨大な発話は合成前に Error::OutOfMemory で断る
(std::vector の確保失敗は例外無効ビルドで abort)。
■ ストリーミング合成
- synthesize_stream(kana, sink, opt): チャンクごとに sink へ渡す。最初のチャンクを
小さく (≈ 14 モーラ)、以降を直前の 1.3 倍まで徐々に大きくして、再生しながら
次を合成しても途切れにくくする。sink が false で中断 (Error::Cancelled)。
- SynthChunk::sample_rate: チャンクの出力レート。sanoTTS (Engine::Sano、22.05 kHz
固定) は synthesize_ex と同様に全体を 1 チャンクで渡す。Auto / Sano では Sano、
次に HMM、単位連結、フォルマントの順 (従来どおり)。
- SubtitleMapper (jtts/subtitle.hpp): 表示テキストを句読点で区切り、チャンクの
読みに対応する部分を返す (吹き出しの同期用)。合わないときは全文を最初に出す。
テスト: test_visemes / test_hmm_chunk (ホスト、.htsvoice を渡すと HMM も検証)。
Co-Authored-By: Claude Sonnet 5
---
.github/workflows/host-tests.yml | 8 +
components/jtts/CMakeLists.txt | 2 +
components/jtts/include/jtts/jtts.hpp | 53 ++
components/jtts/include/jtts/subtitle.hpp | 45 ++
components/jtts/src/hmm_chunk.cpp | 185 +++++++
components/jtts/src/hmm_synth.cpp | 296 +++++++++--
components/jtts/src/internal.hpp | 85 +++-
components/jtts/src/jtts.cpp | 203 ++++++--
components/jtts/src/phoneme.cpp | 29 +-
components/jtts/src/subtitle.cpp | 82 +++
components/jtts/test/host/CMakeLists.txt | 13 +
components/jtts/test/host/test_hmm_chunk.cpp | 496 +++++++++++++++++++
components/jtts/test/host/test_visemes.cpp | 195 ++++++++
13 files changed, 1623 insertions(+), 69 deletions(-)
create mode 100644 components/jtts/include/jtts/subtitle.hpp
create mode 100644 components/jtts/src/hmm_chunk.cpp
create mode 100644 components/jtts/src/subtitle.cpp
create mode 100644 components/jtts/test/host/test_hmm_chunk.cpp
create mode 100644 components/jtts/test/host/test_visemes.cpp
diff --git a/.github/workflows/host-tests.yml b/.github/workflows/host-tests.yml
index 742fc78..6e91365 100644
--- a/.github/workflows/host-tests.yml
+++ b/.github/workflows/host-tests.yml
@@ -128,3 +128,11 @@ jobs:
cmake --build build
./build/jtts_test_vowels
./build/jtts_test_sano_ir
+ # Vowel lip-sync events, long-text chunking, streaming synthesis and
+ # subtitle mapping. Without arguments only the formant / pure-logic
+ # cases run; with .htsvoice files the HMM cases run too (16 kHz voice
+ # = native rate, fast).
+ ./build/jtts_test_visemes
+ ./build/jtts_test_hmm_chunk
+ ./build/jtts_test_visemes ../../../../assets/voices/mei16.htsvoice
+ ./build/jtts_test_hmm_chunk ../../../../assets/voices/mei16.htsvoice
diff --git a/components/jtts/CMakeLists.txt b/components/jtts/CMakeLists.txt
index 55f8523..5510d6c 100644
--- a/components/jtts/CMakeLists.txt
+++ b/components/jtts/CMakeLists.txt
@@ -14,6 +14,8 @@ idf_component_register(
"src/hmm_synth.cpp"
"src/sano_ir.cpp"
"src/sano_synth.cpp"
+ "src/hmm_chunk.cpp"
+ "src/subtitle.cpp"
INCLUDE_DIRS "include"
PRIV_INCLUDE_DIRS "src"
REQUIRES tl_expected
diff --git a/components/jtts/include/jtts/jtts.hpp b/components/jtts/include/jtts/jtts.hpp
index 204bfe4..75b2ad5 100644
--- a/components/jtts/include/jtts/jtts.hpp
+++ b/components/jtts/include/jtts/jtts.hpp
@@ -3,12 +3,16 @@
#pragma once
#include
+#include
#include
+#include
#include
#include
#include
+#include "jtts/phoneme.hpp"
+
namespace stackchan::jtts {
enum class Voice : std::uint8_t {
@@ -90,6 +94,7 @@ struct Options {
enum class Error {
InvalidKana,
OutOfMemory,
+ Cancelled, // synthesize_stream のシンクが false を返して中断した
};
const char* to_string(Error e);
@@ -107,6 +112,54 @@ tl::expected synthesize_ex(std::u32string_view kana,
std::vector& out,
const Options& opt = {});
+// リップシンク用の口形イベント。`start_ms` (発話先頭からの経過時間) から次の
+// イベントまで、口は `vowel` の形を取る。Vowel::None は閉口。
+struct VisemeEvent {
+ std::uint32_t start_ms = 0;
+ Vowel vowel = Vowel::None;
+};
+
+// synthesize に加えて、PCM と時間軸が揃った口形イベント列を `visemes` に返す
+// (時刻昇順、隣り合うイベントの vowel は異なる。最後は閉口で終わる)。
+// 口形を出せるのはフォルマントと HMM エンジン。単位連結エンジンで合成
+// された場合と失敗時は空になるので、呼び出し側は音量エンベロープなどに
+// フォールバックすること。
+tl::expected synthesize(std::u32string_view kana,
+ std::vector& out,
+ std::vector& visemes,
+ const Options& opt = {});
+
+// ストリーミング合成の 1 チャンク分。
+struct SynthChunk {
+ // このチャンクが読む部分 (synthesize_stream に渡した読みの一部。HMM で強制分割した
+ // ときはアクセント記号が落ちる)。発話全体が 1 チャンクなら読み全体。
+ // 吹き出しをチャンクに同期させるとき (jtts/subtitle.hpp) に使う。
+ std::u32string text;
+ std::vector pcm;
+ // pcm のサンプルレート [Hz]。HMM / フォルマント / 単位連結は opt.sample_rate_hz、
+ // sanoTTS は 22.05 kHz 固定 (synthesize_ex と同じ)。再生側がこのレートで鳴らすこと。
+ std::uint32_t sample_rate = 0;
+ // このチャンク先頭からの口形イベント (synthesize と同じ規約)。空ならこの
+ // エンジンは口形を出せない (音量エンベロープなどにフォールバックすること)。
+ std::vector visemes;
+};
+
+// チャンクの受け取り側。false を返すと合成を中断する (Error::Cancelled)。
+using ChunkSink = std::function;
+
+// synthesize と同じ合成を、チャンクごとに sink へ渡しながら行う。長い発話は
+// HMM エンジンが句読点などで分割して順に合成するので、最初のチャンクが出来た
+// 時点で再生を始め、再生中に次を合成できる (全体の合成完了を待たなくてよい)。
+// - チャンクの PCM は連続再生すればそのまま 1 本の発話になる (境界の無音は調整済み)。
+// - 最初のチャンクを小さく、以降を徐々に大きくして、再生が途切れにくくする。
+// - HMM 以外のエンジン (sanoTTS を含む) や分割不要の短文は、全体が 1 チャンクで渡される。
+// - sanoTTS は synthesize_ex と同様に 22.05 kHz で出力する (SynthChunk::sample_rate)。
+// - HMM で 1 つ以上渡した後にメモリ不足になったら Error::OutOfMemory (途中まで
+// 渡した分は取り消せない)。何も渡す前なら他エンジンへフォールバックする。
+// sink はこの関数を呼んだスレッドで、合成の合間に呼ばれる。
+tl::expected synthesize_stream(std::u32string_view kana, const ChunkSink& sink,
+ const Options& opt = {});
+
// 単位連結エンジン用の音声 DB (.jvox、codec=0 の生形式) を登録する。
// blob の寿命は呼び出し側が保証する (PSRAM バッファ / flash mmap)。
// パースに失敗すると false を返し、DB 未ロード状態のまま。空 span で解除。
diff --git a/components/jtts/include/jtts/subtitle.hpp b/components/jtts/include/jtts/subtitle.hpp
new file mode 100644
index 0000000..890658b
--- /dev/null
+++ b/components/jtts/include/jtts/subtitle.hpp
@@ -0,0 +1,45 @@
+// SPDX-FileCopyrightText: 2026 Kenta IDA
+// SPDX-License-Identifier: BSL-1.0
+#pragma once
+
+#include
+#include
+#include
+#include
+
+namespace stackchan::jtts {
+
+// 吹き出し (表示テキスト) を、合成チャンクの再生に同期させるための対応付け。
+//
+// jtts は「読み」(かな) しか持たず、表示テキスト (漢字まじり) との対応は知らない。
+// しかしチャンクは句読点 (、。,,..) を境に分けられるので、表示テキストも同じ句読点で
+// 区切れば「n 個目の句読点までの部分」が対応する。読みと表示の句読点の数が合わない
+// ときは対応を諦め、最初のチャンクで全文を出す。
+//
+// SubtitleMapper m(display_utf8, reading);
+// synthesize_stream(reading, [&](SynthChunk&& c) {
+// std::string text = m.next(c.text); // このチャンクの間に出す表示テキスト
+// ...
+// });
+class SubtitleMapper {
+public:
+ SubtitleMapper(std::string_view display_utf8, std::u32string_view reading);
+
+ // 句読点で対応が取れるか。false のときは next() が最初に全文を返し、以降は空を返す。
+ bool mapped() const { return mapped_; }
+
+ // 次のチャンク (その読み) の間に出す表示テキスト。チャンクは先頭から順に渡すこと。
+ // 句読点で終わるチャンクはその句までを、途中で切れたチャンクは続きの句を受け持つ
+ // (強制分割で同じ句を 2 チャンクにまたがって出す場合は同じ文字列が返る)。
+ // 出すものが無ければ空。
+ std::string next(std::u32string_view chunk_reading);
+
+private:
+ std::string whole_;
+ std::vector segments_; // 表示テキストを句読点の直後で区切ったもの (句読点を含む)
+ bool mapped_ = false;
+ bool first_ = true;
+ std::size_t consumed_ = 0; // ここまでのチャンクが消費した句読点の数
+};
+
+} // namespace stackchan::jtts
diff --git a/components/jtts/src/hmm_chunk.cpp b/components/jtts/src/hmm_chunk.cpp
new file mode 100644
index 0000000..0a1e6c3
--- /dev/null
+++ b/components/jtts/src/hmm_chunk.cpp
@@ -0,0 +1,185 @@
+// SPDX-FileCopyrightText: 2026 Kenta IDA
+// SPDX-License-Identifier: BSL-1.0
+//
+// HMM エンジン用のテキスト分割。hts_engine は 1 発話の合成中、フレーム数に
+// 比例した量 (≈ 2 KB / 5 ms フレーム) の作業メモリを確保し続けるため、長い
+// フレーズをそのまま合成すると PSRAM を使い切って確保失敗 → クラッシュする。
+// そこで発話を、メモリ予算に収まるモーラ数のチャンクに分けて順に合成する。
+//
+// 分割位置の優先順位:
+// 1. 句読点 (、。,,..) … アクセント句 + 呼気段落の境界。元々ポーズが入る所
+// 2. アクセント句境界 (/) … ポーズなし
+// 3. 最後の手段: モーラ境界 (1 つの句がそれ単体で予算を超える場合)。
+// 拗音・長音・アクセント核の直前では切らない。
+#include
+#include
+#include
+
+#include "internal.hpp"
+
+#if defined(ESP_PLATFORM)
+#include "esp_heap_caps.h"
+#endif
+
+namespace stackchan::jtts::internal {
+
+namespace {
+
+bool is_pause_char(char32_t c) {
+ return c == U'、' || c == U'。' || c == U',' || c == U',' || c == U'.' || c == U'.';
+}
+
+bool is_accent_mark(char32_t c) {
+ return c == U'\'' || c == U'’';
+}
+
+// 直前のモーラと一体になる文字 (この直前では切らない)。
+bool is_attached_to_prev(char32_t c) {
+ switch (c) {
+ case U'ゃ': case U'ゅ': case U'ょ': case U'ぁ': case U'ぃ': case U'ぅ': case U'ぇ': case U'ぉ':
+ case U'ゎ': case U'ャ': case U'ュ': case U'ョ': case U'ァ': case U'ィ': case U'ゥ': case U'ェ':
+ case U'ォ': case U'ヮ': case U'ー': case U'/':
+ return true;
+ default:
+ return is_accent_mark(c);
+ }
+}
+
+std::size_t count_moras(std::u32string_view s) {
+ std::vector moras;
+ return parse_kana(s, moras) ? moras.size() : 0;
+}
+
+struct Unit {
+ std::u32string text;
+ std::size_t moras = 0;
+ bool pause = false; // 末尾が句読点 (= 直後にポーズが入る)
+};
+
+// 1 つの句が max_moras を超えるとき、モーラ境界で切って units に積む。
+// 切った断片ではアクセント核 (') の位置が意味を失うので取り除く (平板になる)。
+void push_split_unit(const Unit& u, std::size_t max_moras, std::vector& units) {
+ std::u32string rest;
+ for (char32_t c : u.text) {
+ if (!is_accent_mark(c)) rest.push_back(c);
+ }
+ while (count_moras(rest) > max_moras) {
+ // max_moras 以内に収まる最長の接頭辞のうち、切ってよい位置を探す。
+ std::size_t cut = 0;
+ std::size_t fallback = 0;
+ for (std::size_t i = 1; i < rest.size(); ++i) {
+ if (count_moras(std::u32string_view(rest).substr(0, i)) > max_moras) break;
+ fallback = i;
+ if (!is_attached_to_prev(rest[i])) cut = i;
+ }
+ if (cut == 0) cut = fallback;
+ if (cut == 0) break; // 1 モーラで既に予算超過: これ以上は割れない
+ Unit piece;
+ piece.text = rest.substr(0, cut);
+ piece.moras = count_moras(piece.text);
+ units.push_back(std::move(piece));
+ rest.erase(0, cut);
+ }
+ Unit last;
+ last.text = std::move(rest);
+ last.moras = count_moras(last.text);
+ last.pause = u.pause;
+ units.push_back(std::move(last));
+}
+
+// テスト用: 0 以外なら「PCM に使える PSRAM」をこの値 [byte] とみなす。
+std::size_t g_pcm_free_override = 0;
+
+} // namespace
+
+float estimate_utterance_ms(std::u32string_view text, float mora_ms) {
+ // 実測: 先頭 + 末尾の sil ≈ 0.65 s、句読点の pau ≈ 0.46 s、1 モーラ ≈ 1.25 × mora_ms
+ // (HMM: 等速時 0.138〜0.150 s、フォルマントは mora_ms + 子音分)。上限寄りに見積もる。
+ std::size_t pauses = 0;
+ for (char32_t c : text) {
+ if (is_pause_char(c)) ++pauses;
+ }
+ return 700.0f + 500.0f * static_cast(pauses) +
+ 1.3f * mora_ms * static_cast(count_moras(text));
+}
+
+bool pcm_fits_in_memory(std::size_t samples) {
+ const std::size_t bytes = samples * sizeof(std::int16_t);
+ if (g_pcm_free_override != 0) return bytes * 2 + 256 * 1024 <= g_pcm_free_override;
+#if defined(ESP_PLATFORM)
+ // PCM は 1 つの連続ブロック (std::vector) で PSRAM に置く。合成中の他の確保
+ // (hts_engine の作業メモリ、再生バッファ) と一時的な倍増に備えて 2 倍 + 256 KB を要求する。
+ const std::size_t free_psram = heap_caps_get_free_size(MALLOC_CAP_SPIRAM);
+ const std::size_t largest = heap_caps_get_largest_free_block(MALLOC_CAP_SPIRAM);
+ return bytes <= largest && bytes * 2 + 256 * 1024 <= free_psram;
+#else
+ return true;
+#endif
+}
+
+void set_pcm_memory_limit_for_test(std::size_t bytes) { g_pcm_free_override = bytes; }
+
+bool split_hmm_text(std::u32string_view text, std::size_t max_moras, std::vector& out,
+ std::size_t first_moras) {
+ out.clear();
+ if (max_moras == 0) return false;
+ const bool ramp = first_moras > 0;
+
+ // 全体が収まるなら手を加えず 1 チャンク (従来と完全に同じ入力で合成する)。
+ // 低遅延モードでは句読点で分けて先頭を早く出したいので、この近道は使わない
+ // (句の途中では切らないので、句読点の無い短文は結局 1 チャンクになる)。
+ const std::size_t total = count_moras(text);
+ if (total == 0) return false;
+ if (!ramp && total <= max_moras) {
+ out.push_back({std::u32string(text), false, total});
+ return true;
+ }
+
+ // 句読点 / アクセント句境界で「句」に分ける。区切り文字は前の句に含める。
+ std::vector units;
+ Unit cur;
+ auto flush_unit = [&](bool pause) {
+ if (cur.text.empty()) return;
+ cur.moras = count_moras(cur.text);
+ cur.pause = pause;
+ if (cur.moras > max_moras) {
+ push_split_unit(cur, max_moras, units);
+ } else {
+ units.push_back(std::move(cur));
+ }
+ cur = Unit{};
+ };
+ for (char32_t c : text) {
+ cur.text.push_back(c);
+ if (is_pause_char(c)) {
+ flush_unit(true);
+ } else if (c == U'/') {
+ flush_unit(false);
+ }
+ }
+ flush_unit(false);
+
+ // 予算に収まるまで句を貪欲に詰める。低遅延モードでは上限を first_moras から
+ // 始め、チャンクを閉じるたびに直前のチャンクの 1.3 倍へ引き上げる。
+ std::size_t cap = ramp ? std::min(first_moras, max_moras) : max_moras;
+ HmmChunk chunk;
+ auto flush_chunk = [&] {
+ const std::size_t done = chunk.moras;
+ if (chunk.moras > 0) out.push_back(std::move(chunk));
+ chunk = HmmChunk{};
+ if (ramp && done > 0) {
+ const std::size_t grown = (done * 13 + 9) / 10; // ceil(done * 1.3)
+ cap = std::min(max_moras, std::max(first_moras, grown));
+ }
+ };
+ for (auto& u : units) {
+ if (chunk.moras > 0 && chunk.moras + u.moras > cap) flush_chunk();
+ chunk.text += u.text;
+ chunk.moras += u.moras;
+ chunk.pause_after = u.pause;
+ }
+ flush_chunk();
+ return !out.empty();
+}
+
+} // namespace stackchan::jtts::internal
diff --git a/components/jtts/src/hmm_synth.cpp b/components/jtts/src/hmm_synth.cpp
index c25eb90..4cf4430 100644
--- a/components/jtts/src/hmm_synth.cpp
+++ b/components/jtts/src/hmm_synth.cpp
@@ -11,6 +11,7 @@
#include "sdkconfig.h"
#endif
+#include
#include
#include
#include
@@ -31,10 +32,12 @@
#include
#include
#include
+#include
#include "HTS_engine.h"
#if defined(ESP_PLATFORM)
+#include "esp_heap_caps.h"
#include "esp_log.h"
#endif
@@ -99,26 +102,124 @@ const std::array& decim_coeffs() {
return coeffs;
}
-} // namespace
-
-bool render_hmm(std::u32string_view text, std::vector& out, const Options& opt) {
- std::lock_guard lock(g_engine_mutex);
- if (!g_loaded) return false;
+// full-context ラベル "p1^p2-p3+p4=p5/A:..." から現在音素 p3 を取り出す。
+std::string_view phoneme_of_label(std::string_view label) {
+ const std::size_t dash = label.find('-');
+ if (dash == std::string_view::npos) return {};
+ const std::size_t plus = label.find('+', dash + 1);
+ if (plus == std::string_view::npos) return {};
+ return label.substr(dash + 1, plus - dash - 1);
+}
- // ボイスはネイティブ レート (48 kHz) のまま合成し、出力レート (16 kHz) へは
- // FIR 1/3 デシメーションで落とす。ボコーダを 16 kHz で直接回す (α 再設定)
- // 近似も試したが、メルケプの周波数軸はどの α でも 48 kHz 分析軸と一致せず
- // フォルマントが下方に歪む (声が暗く低く聞こえる) ため不採用。
- const std::size_t voice_rate = g_engine.ms.sampling_frequency;
- std::size_t decim;
- if (opt.sample_rate_hz == voice_rate) {
- decim = 1;
- } else if (voice_rate == 3 * opt.sample_rate_hz) {
- decim = 3;
- } else {
- return false; // 対応外レート → フォールバック
+// 音素名 → 口形。フォルマント エンジン (build_segments) と同じ規則:
+// 母音 a i u e o … その母音の形
+// 無声化母音 A I U E O・撥音 N・促音 cl・無音 sil / pau … 閉口
+// 両唇音 m b p (拗音含む) … 閉口 (唇を閉じる)
+// その他の子音 … 後続母音の形を先取り (後続が無声化母音なら閉口)
+Vowel viseme_of_phoneme(std::string_view ph, std::string_view next) {
+ if (ph.size() == 1) {
+ switch (ph[0]) {
+ case 'a': return Vowel::A;
+ case 'i': return Vowel::I;
+ case 'u': return Vowel::U;
+ case 'e': return Vowel::E;
+ case 'o': return Vowel::O;
+ case 'A': case 'I': case 'U': case 'E': case 'O': case 'N':
+ return Vowel::None;
+ default: break;
+ }
+ if (ph[0] == 'm' || ph[0] == 'b' || ph[0] == 'p') return Vowel::None;
+ } else if (ph == "sil" || ph == "pau" || ph == "cl" || ph == "my" || ph == "by" || ph == "py") {
+ return Vowel::None;
}
+ // 子音: 後続が有声母音ならその形。
+ if (next.size() == 1) {
+ switch (next[0]) {
+ case 'a': return Vowel::A;
+ case 'i': return Vowel::I;
+ case 'u': return Vowel::U;
+ case 'e': return Vowel::E;
+ case 'o': return Vowel::O;
+ default: break;
+ }
+ }
+ return Vowel::None;
+}
+// ---- メモリ予算 / 見積り -------------------------------------------------
+//
+// hts_engine は 1 回の合成中、フレーム数に比例した作業メモリを確保し続け
+// (mean / ivar / wuw / par 行列で ≈ 2 KB / フレーム + 波形 4 B / サンプル)、
+// HTS_Engine_refresh まで解放しない。確保に失敗すると hts_engine は復帰できず
+// クラッシュするため、合成前に「収まる長さ」を見積もり、超えるなら分割する。
+// 実測 (mei16、5 ms フレーム): ≈ 470 KB / 発話秒、≈ 75 KB / 文字。
+
+// テスト用: 0 以外ならこの値 [byte] を予算として使う。
+std::size_t g_budget_override = 0;
+
+constexpr float kBytesPerFrame = 2300.0f; // 実測 ≈ 2.05 KB + ヒープ ヘッダ / 余裕
+constexpr float kEdgeSilSec = 0.85f; // 先頭 + 末尾の sil (各 ≈ 0.32 s) + 1 合成ごとの固定分 (≈ 100〜140 KB)
+constexpr float kSecPerMora = 0.15f; // 等速時の 1 モーラあたり秒 (ポーズ込み、実測 0.138〜0.150)
+
+// 今 1 回の合成に使ってよいバイト数。他タスクの分 (128 KB) を残し、残りの 3/4 を
+// 上限とする (見積りの誤差と断片化の余裕)。
+std::size_t memory_budget() {
+ if (g_budget_override != 0) return g_budget_override;
+#if defined(ESP_PLATFORM)
+ const std::size_t free_psram = heap_caps_get_free_size(MALLOC_CAP_SPIRAM);
+ const std::size_t reserve = 128 * 1024;
+ return free_psram > reserve ? (free_psram - reserve) / 4 * 3 : 0;
+#else
+ return SIZE_MAX; // ホストでは無制限
+#endif
+}
+
+// 直前までの分割計画が既に安全率を見込んでいるので、2 チャンク目以降の再確認は
+// 余裕 (3/4) を掛けずに「実際に足りるか」だけを見る。
+std::size_t memory_free_hard() {
+ if (g_budget_override != 0) return g_budget_override;
+#if defined(ESP_PLATFORM)
+ const std::size_t free_psram = heap_caps_get_free_size(MALLOC_CAP_SPIRAM);
+ return free_psram > 128 * 1024 ? free_psram - 128 * 1024 : 0;
+#else
+ return SIZE_MAX;
+#endif
+}
+
+// 予算に収まる 1 チャンクの最大モーラ数 (0 = 1 モーラも収まらない)。
+std::size_t max_moras_for_budget(std::size_t budget, std::size_t voice_rate, std::size_t fperiod, float speed) {
+ if (budget == SIZE_MAX) return SIZE_MAX;
+ const float frames_per_sec = fperiod > 0 ? static_cast(voice_rate) / static_cast(fperiod) : 200.0f;
+ const float bytes_per_sec = frames_per_sec * kBytesPerFrame + static_cast(voice_rate) * sizeof(float);
+ const float sec = static_cast(budget) / bytes_per_sec - kEdgeSilSec;
+ if (sec <= 0.0f) return 0;
+ return static_cast(sec * speed / kSecPerMora);
+}
+
+// ---- チャンク境界の無音 ---------------------------------------------------
+//
+// チャンクごとの合成は前後に sil (≈ 0.32 s) を持つので、そのままつなぐと
+// 間延びする。境界では両側の sil を切り詰め、句読点なら元の pau (≈ 0.46 s) と
+// 同じ長さ、それ以外 (アクセント句境界 / 強制分割) ならごく短い無音だけ残す。
+constexpr float kPauseHalfMs = 210.0f; // 句読点境界で片側に残す無音 (等速時)
+constexpr std::size_t kSoftKeepFrames = 2; // その他の境界で片側に残す無音 (≈ 10 ms)
+
+std::size_t boundary_keep_frames(bool pause, float ms_per_frame, float speed) {
+ if (!pause) return kSoftKeepFrames;
+ const auto frames = static_cast(kPauseHalfMs / speed / ms_per_frame + 0.5f);
+ return frames > kSoftKeepFrames ? frames : kSoftKeepFrames;
+}
+
+// 低遅延モードの最初のチャンクの目安モーラ数 (≈ 2 秒の音声 = 合成 ≈ 1.5 秒)。
+constexpr std::size_t kStreamFirstMoras = 14;
+
+// 1 チャンクを合成して pcm / spans に出力する。
+// lead_cap / trail_cap: 残す先頭 / 末尾 sil の最大フレーム数 (SIZE_MAX = 全部残す)。
+// need_frames: 継続長が取れないときは失敗にする (複数チャンクでは切り詰めに必須)。
+// 失敗時は false (pcm / spans は不定)。
+bool synth_chunk(const std::u32string& text, const Options& opt, std::size_t decim, std::size_t voice_rate,
+ std::size_t fperiod, float speed, bool need_frames, std::size_t lead_cap, std::size_t trail_cap,
+ std::vector& pcm, std::vector& spans) {
std::vector labels;
if (!build_hts_labels(text, labels)) return false;
@@ -126,10 +227,6 @@ bool render_hmm(std::u32string_view text, std::vector& out, const
lines.reserve(labels.size());
for (auto& l : labels) lines.push_back(l.data());
- // mora_ms は「1 モーラの長さ」なので speed は逆比。既定 110 ms = 等速。
- float speed = 110.0f / opt.mora_ms;
- if (speed < 0.5f) speed = 0.5f;
- if (speed > 2.0f) speed = 2.0f;
HTS_Engine_set_speed(&g_engine, speed);
HTS_Engine_add_half_tone(&g_engine, opt.hmm_half_tone);
@@ -141,25 +238,61 @@ bool render_hmm(std::u32string_view text, std::vector& out, const
const auto t1 = std::chrono::steady_clock::now();
const std::size_t nsamples = HTS_Engine_get_nsamples(&g_engine);
+
+ // ラベル (音素) ごとの継続長 [フレーム]。ラベル 1 行 = 1 音素 = nstate 状態で、
+ // 総和 × フレーム周期 = 波形長。噛み合わなければ口形も切り詰めも諦める。
+ const std::size_t nstate = HTS_Engine_get_nstate(&g_engine);
+ std::vector frames;
+ if (nstate > 0 && fperiod > 0 && labels.size() >= 2 &&
+ HTS_Engine_get_total_state(&g_engine) == labels.size() * nstate &&
+ HTS_Engine_get_total_frame(&g_engine) * fperiod == nsamples) {
+ frames.assign(labels.size(), 0);
+ for (std::size_t i = 0; i < labels.size(); ++i) {
+ for (std::size_t s = 0; s < nstate; ++s) {
+ frames[i] += HTS_Engine_get_state_duration(&g_engine, i * nstate + s);
+ }
+ }
+ } else if (need_frames) {
+ HTS_Engine_refresh(&g_engine);
+ return false;
+ }
+
+ // 先頭 / 末尾の sil を切り詰める (波形は [begin, end) だけ出力する)。
+ std::size_t begin = 0;
+ std::size_t end = nsamples;
+ if (!frames.empty()) {
+ const std::size_t lead = frames.front();
+ const std::size_t trail = frames.back();
+ const std::size_t keep_lead = lead < lead_cap ? lead : lead_cap;
+ const std::size_t keep_trail = trail < trail_cap ? trail : trail_cap;
+ begin = (lead - keep_lead) * fperiod;
+ end = nsamples - (trail - keep_trail) * fperiod;
+ frames.front() = keep_lead;
+ frames.back() = keep_trail;
+ }
+
+ pcm.clear();
const float* speech = g_engine.gss.gspeech; // per-sample getter は高いので直接参照
const float gain = opt.gain;
if (decim == 1) {
- out.reserve(out.size() + nsamples);
- for (std::size_t i = 0; i < nsamples; ++i) {
+ pcm.reserve(end - begin);
+ for (std::size_t i = begin; i < end; ++i) {
float v = speech[i] * gain;
if (v > 32767.0f) v = 32767.0f;
if (v < -32768.0f) v = -32768.0f;
- out.push_back(static_cast(v));
+ pcm.push_back(static_cast(v));
}
} else {
// 1/3 ポリフェーズ デシメーション (45-tap Hamming sinc、fc=7.2 kHz)。
// 出力サンプルあたり実質 15 MAC なので合成コストに対して無視できる。
+ // フィルタは切り詰め前の全波形を参照するので、境界にもエッジ アーチファクトは出ない。
const auto& h = decim_coeffs();
constexpr int mid = static_cast(kDecimTaps) / 2;
- const std::size_t nout = nsamples / 3;
- out.reserve(out.size() + nout);
- for (std::size_t n = 0; n < nout; ++n) {
- const long center = static_cast(n) * 3;
+ const std::size_t n_begin = begin / decim;
+ const std::size_t n_end = end / decim;
+ pcm.reserve(n_end - n_begin);
+ for (std::size_t n = n_begin; n < n_end; ++n) {
+ const long center = static_cast(n) * static_cast(decim);
long lo = center - mid;
long hi = center + mid; // inclusive
int skip = 0;
@@ -175,11 +308,23 @@ bool render_hmm(std::u32string_view text, std::vector& out, const
acc *= gain;
if (acc > 32767.0f) acc = 32767.0f;
if (acc < -32768.0f) acc = -32768.0f;
- out.push_back(static_cast(acc));
+ pcm.push_back(static_cast(acc));
}
}
const auto t2 = std::chrono::steady_clock::now();
+ spans.clear();
+ if (!frames.empty()) {
+ const float ms_per_frame = 1000.0f * static_cast(fperiod) / static_cast(voice_rate);
+ spans.reserve(labels.size());
+ for (std::size_t i = 0; i < labels.size(); ++i) {
+ const std::string_view next =
+ i + 1 < labels.size() ? phoneme_of_label(labels[i + 1]) : std::string_view{};
+ spans.push_back({viseme_of_phoneme(phoneme_of_label(labels[i]), next),
+ static_cast(frames[i]) * ms_per_frame});
+ }
+ }
+
#if defined(ESP_PLATFORM)
ESP_LOGI("jtts-hmm", "synth %u ms + copy %u ms for %u samples @%u Hz",
static_cast(
@@ -187,12 +332,100 @@ bool render_hmm(std::u32string_view text, std::vector& out, const
static_cast(
std::chrono::duration_cast(t2 - t1).count()),
static_cast(nsamples), static_cast(voice_rate));
+#else
+ (void)t0;
+ (void)t1;
+ (void)t2;
#endif
HTS_Engine_refresh(&g_engine);
return true;
}
+} // namespace
+
+void set_hmm_memory_budget_for_test(std::size_t bytes) { g_budget_override = bytes; }
+
+HmmOutcome render_hmm_stream(std::u32string_view text, const Options& opt, const ChunkFn& emit, bool stream) {
+ std::lock_guard lock(g_engine_mutex);
+ if (!g_loaded) return HmmOutcome::NoOutput;
+
+ // ボイスはネイティブ レート (48 kHz) のまま合成し、出力レート (16 kHz) へは
+ // FIR 1/3 デシメーションで落とす。ボコーダを 16 kHz で直接回す (α 再設定)
+ // 近似も試したが、メルケプの周波数軸はどの α でも 48 kHz 分析軸と一致せず
+ // フォルマントが下方に歪む (声が暗く低く聞こえる) ため不採用。
+ const std::size_t voice_rate = g_engine.ms.sampling_frequency;
+ std::size_t decim;
+ if (opt.sample_rate_hz == voice_rate) {
+ decim = 1;
+ } else if (voice_rate == 3 * opt.sample_rate_hz) {
+ decim = 3;
+ } else {
+ return HmmOutcome::NoOutput; // 対応外レート → フォールバック
+ }
+ const std::size_t fperiod = HTS_Engine_get_fperiod(&g_engine);
+
+ // mora_ms は「1 モーラの長さ」なので speed は逆比。既定 110 ms = 等速。
+ float speed = 110.0f / opt.mora_ms;
+ if (speed < 0.5f) speed = 0.5f;
+ if (speed > 2.0f) speed = 2.0f;
+
+ // 長い発話はメモリ予算に収まるチャンクに分けて順に合成する。どうしても
+ // 収まらない (空きメモリが足りない) ときは諦めて呼び出し側にフォールバック
+ // (フォルマント合成) させる — 確保失敗は hts_engine 内でクラッシュになる。
+ const std::size_t max_moras = max_moras_for_budget(memory_budget(), voice_rate, fperiod, speed);
+ std::vector chunks;
+ if (!split_hmm_text(text, max_moras, chunks, stream ? kStreamFirstMoras : 0)) {
+#if defined(ESP_PLATFORM)
+ ESP_LOGW("jtts-hmm", "no memory for HMM synthesis (budget allows %u moras) → fallback",
+ static_cast(max_moras));
+#endif
+ return HmmOutcome::NoOutput;
+ }
+ const bool multi = chunks.size() > 1;
+ // 切り詰めは fperiod 単位 (デシメーション後も整数サンプル) で行う。
+ if (multi && (fperiod == 0 || fperiod % decim != 0)) return HmmOutcome::NoOutput;
+#if defined(ESP_PLATFORM)
+ if (multi) {
+ ESP_LOGI("jtts-hmm", "long text: %u chunks (max %u moras each, PSRAM free %u KB)",
+ static_cast(chunks.size()), static_cast(max_moras),
+ static_cast(heap_caps_get_free_size(MALLOC_CAP_SPIRAM) / 1024));
+ }
+#endif
+
+ const float ms_per_frame =
+ fperiod > 0 ? 1000.0f * static_cast(fperiod) / static_cast(voice_rate) : 5.0f;
+ std::size_t emitted = 0;
+ const auto failed = [&] { return emitted > 0 ? HmmOutcome::Aborted : HmmOutcome::NoOutput; };
+
+ std::vector pcm;
+ std::vector spans;
+ for (std::size_t k = 0; k < chunks.size(); ++k) {
+ // 2 チャンク目以降は、その間に空きメモリが減っている (再生待ちの PCM など) ので
+ // 収まるか確かめ直す。ストリーミングでは分割計画が安全率を見込み済みなので
+ // 「実際に足りるか」だけを見る。
+ if (k > 0) {
+ const std::size_t budget = stream ? memory_free_hard() : memory_budget();
+ if (max_moras_for_budget(budget, voice_rate, fperiod, speed) < chunks[k].moras) return failed();
+ }
+ const bool first = (k == 0);
+ const bool last = (k + 1 == chunks.size());
+ const std::size_t lead_cap =
+ first ? SIZE_MAX : boundary_keep_frames(chunks[k - 1].pause_after, ms_per_frame, speed);
+ const std::size_t trail_cap =
+ last ? SIZE_MAX : boundary_keep_frames(chunks[k].pause_after, ms_per_frame, speed);
+ if (!synth_chunk(chunks[k].text, opt, decim, voice_rate, fperiod, speed, multi, lead_cap, trail_cap, pcm,
+ spans)) {
+ return failed();
+ }
+ ++emitted;
+ if (!emit(std::move(pcm), std::move(spans), chunks[k].text)) return HmmOutcome::Cancelled;
+ pcm = {};
+ spans = {};
+ }
+ return HmmOutcome::Ok;
+}
+
} // namespace internal
} // namespace stackchan::jtts
@@ -204,7 +437,10 @@ bool set_hmm_voice(std::span) { return false; }
bool hmm_voice_loaded() { return false; }
namespace internal {
-bool render_hmm(std::u32string_view, std::vector&, const Options&) { return false; }
+HmmOutcome render_hmm_stream(std::u32string_view, const Options&, const ChunkFn&, bool) {
+ return HmmOutcome::NoOutput;
+}
+void set_hmm_memory_budget_for_test(std::size_t) {}
} // namespace internal
} // namespace stackchan::jtts
diff --git a/components/jtts/src/internal.hpp b/components/jtts/src/internal.hpp
index 09afb88..f803ef1 100644
--- a/components/jtts/src/internal.hpp
+++ b/components/jtts/src/internal.hpp
@@ -2,8 +2,11 @@
// SPDX-License-Identifier: BSL-1.0
#pragma once
+#include
#include
+#include
#include
+#include
#include
#include
@@ -52,6 +55,9 @@ struct Segment {
FormantFrame start;
FormantFrame end;
float duration_ms = 0.0f;
+ // リップシンク用: この区間で口が取る母音形。None = 閉口 (無音・「ん」・
+ // 両唇子音の閉鎖・無声化母音)。音の合成には使わない。
+ Vowel vowel = Vowel::None;
};
bool parse_kana(std::u32string_view kana, std::vector& out);
@@ -90,6 +96,22 @@ void render_segments(std::span segs, std::vector& o
void render_segments_classic(std::span segs, std::vector& out,
const Options& opt);
+// 口形イベント列の組み立て (jtts.cpp)。区間を時間順に add() していくと、
+// 口形が変わる所だけがイベントになる。finish() で最後が閉口でなければ
+// 終端に閉口イベントを付ける。
+class VisemeBuilder {
+public:
+ explicit VisemeBuilder(std::vector& out) : out_(out) {}
+ void add(Vowel v, float duration_ms);
+ void finish();
+
+private:
+ std::vector& out_;
+ float t_ms_ = 0.0f;
+ Vowel last_ = Vowel::None;
+ bool have_last_ = false;
+};
+
} // namespace stackchan::jtts::internal
namespace stackchan::jtts::jvox {
@@ -112,9 +134,66 @@ bool render_units(std::span moras, const jvox::Db& db,
// 検証リファレンス: tools/jvox/hts_label_kana.py
bool build_hts_labels(std::u32string_view text, std::vector& labels);
-// HMM エンジン本体 (hmm_synth.cpp)。ボイス未ロード・ラベル生成失敗・
-// レート非対応時は out を触らず false (呼び出し側がフォールバック)。
-bool render_hmm(std::u32string_view text, std::vector& out, const Options& opt);
+// 口形の 1 区間 (口形 + 継続時間)。HMM は音素ごとの継続長からこれを作る。
+struct VisemeSpan {
+ Vowel vowel = Vowel::None;
+ float duration_ms = 0.0f;
+};
+
+// spans → 口形イベント列 (先頭を 0 ms とし、同じ口形は連結、最後は閉口で終わる)。
+void spans_to_events(std::span spans, std::vector& out);
+
+// HMM の 1 チャンク分の受け取り側 (PCM と口形区間)。false で中断。
+using ChunkFn =
+ std::function&&, std::vector&&, const std::u32string& text)>;
+
+enum class HmmOutcome {
+ Ok, // 全チャンクを emit した
+ NoOutput, // 何も emit せずに諦めた (ボイス未ロード / メモリ不足など): 呼び出し側がフォールバック
+ Aborted, // 1 つ以上 emit した後にメモリ不足で諦めた
+ Cancelled, // emit が false を返した
+};
+
+// HMM エンジン本体 (hmm_synth.cpp)。長い発話はメモリ予算に収まるチャンクに分けて
+// 順に合成し、チャンクごとに emit する。
+// stream = false: 一括 (synthesize 用)。チャンクは予算いっぱいまで詰める。
+// stream = true : 低遅延 (synthesize_stream 用)。最初のチャンクを小さく、以降を
+// 徐々に大きくして、再生しながら次を合成しても途切れにくくする。
+HmmOutcome render_hmm_stream(std::u32string_view text, const Options& opt, const ChunkFn& emit, bool stream);
+
+// HMM 合成 1 回分のテキスト チャンク (hmm_chunk.cpp)。
+struct HmmChunk {
+ std::u32string text;
+ bool pause_after = false; // 末尾が句読点 (次のチャンクとの間に本来ポーズが入る)
+ std::size_t moras = 0;
+};
+
+// text を、各チャンクが max_moras 以下になるよう句読点 / アクセント句境界
+// (最後の手段でモーラ境界) で分割する。全体が収まるなら text をそのまま
+// 1 チャンクにする。max_moras == 0 や発声できる内容が無いときは false。
+//
+// first_moras > 0 のときは低遅延モード: 最初のチャンクを first_moras 程度に抑え、
+// 以降は直前のチャンクの 1.3 倍まで (max_moras を上限に) 徐々に大きくする。合成時間は
+// 音声長の約 0.72 倍なので、次のチャンクの合成が前のチャンクの再生中に終わる。分割は
+// 句読点 / アクセント句境界だけで行い、ここでは句の途中では切らない。
+bool split_hmm_text(std::u32string_view text, std::size_t max_moras, std::vector& out,
+ std::size_t first_moras = 0);
+
+// かな文字列 → 発話長 [ms] の粗い上限見積り (hmm_chunk.cpp)。PCM バッファの
+// 確保量の事前見積りに使う。
+float estimate_utterance_ms(std::u32string_view text, float mora_ms);
+
+// samples 個の int16 PCM (+ 合成中の作業余裕) が空きメモリに収まるか。ESP では
+// 空き PSRAM を見る。ホストでは常に true。収まらない発話は合成前に断り、
+// std::vector の確保失敗 (例外無効なので abort) を避ける。
+bool pcm_fits_in_memory(std::size_t samples);
+
+// テスト用: PCM に使える空きメモリ [byte] を固定する (0 で実機同様に自動判定)。
+void set_pcm_memory_limit_for_test(std::size_t bytes);
+
+// テスト用: HMM 合成のメモリ予算 [byte] を固定する (0 で実機同様に自動算出 /
+// ホストでは無制限)。分割合成をホストで検証するために使う。
+void set_hmm_memory_budget_for_test(std::size_t bytes);
// ---- sanoTTS-jp エンジン ----
diff --git a/components/jtts/src/jtts.cpp b/components/jtts/src/jtts.cpp
index cdfebd9..9eb5971 100644
--- a/components/jtts/src/jtts.cpp
+++ b/components/jtts/src/jtts.cpp
@@ -2,6 +2,7 @@
// SPDX-License-Identifier: BSL-1.0
#include "jtts/jtts.hpp"
+#include
#include
#include
#include
@@ -41,6 +42,7 @@ const char* to_string(Error e) {
switch (e) {
case Error::InvalidKana: return "InvalidKana";
case Error::OutOfMemory: return "OutOfMemory";
+ case Error::Cancelled: return "Cancelled";
}
return "Unknown";
}
@@ -76,35 +78,34 @@ void apply_formant_scale(std::vector& segs, float scale) {
}
}
-} // namespace
-
-namespace {
+// セグメント列 (時間軸は PCM と一致) を口形イベント列にまとめる。
+void collect_visemes(std::span segs, std::vector& out) {
+ internal::VisemeBuilder b(out);
+ for (const auto& s : segs) b.add(s.vowel, s.duration_ms);
+ b.finish();
+}
-tl::expected synthesize_impl(std::u32string_view kana,
- std::vector& out,
- const Options& opt_in, bool allow_native_rate) {
- out.clear();
- Options opt = resolve_defaults(opt_in);
+// sanoTTS: Auto では最優先、Sano 指定では必須。出力は 22.05 kHz 固定なので、呼び出し側が
+// レートを受け取れる (synthesize_ex / synthesize_stream) か、要求レートが一致するとき
+// だけ使う。
+bool wants_sano(const Options& opt) {
+ return opt.engine == Engine::Auto || opt.engine == Engine::Sano;
+}
- // sanoTTS エンジン: 重みがロード済みなら最優先。出力は 22.05 kHz 固定なので、
- // 呼び出し側がレートを受け取れる (synthesize_ex) か、要求レートが一致する
- // ときだけ使う。
- if (opt.engine == Engine::Auto || opt.engine == Engine::Sano) {
- if (allow_native_rate || opt.sample_rate_hz == 22050u) {
- std::uint32_t rate = 0;
- if (internal::render_sano(kana, out, opt, rate)) {
- return rate;
- }
- }
- }
+// HMM: Auto / Hmm、および Sano 指定 (重み未ロード時のフォールバック先)。
+bool wants_hmm(const Options& opt) {
+ return opt.engine == Engine::Auto || opt.engine == Engine::Hmm || opt.engine == Engine::Sano;
+}
- // HMM エンジン: ボイスがロード済みなら最優先 (品質最良)。
- // アクセント記号 (' と /) は HMM のみ解釈し、他エンジンでは
- // parse_kana が読み飛ばす。
- if (opt.engine == Engine::Auto || opt.engine == Engine::Hmm || opt.engine == Engine::Sano) {
- if (internal::render_hmm(kana, out, opt)) {
- return opt.sample_rate_hz;
- }
+// HMM 以外のエンジン (単位連結 → フォルマント) で発話全体を 1 本の PCM にする。
+// 全体を 1 つの std::vector に作るので、空きメモリに収まらない長さは合成前に断る
+// (std::vector の確保失敗は例外無効ビルドでは abort = 再起動になる)。
+tl::expected render_whole(std::u32string_view kana, const Options& opt,
+ std::vector& out, std::vector* visemes) {
+ const auto est_samples = static_cast(
+ internal::estimate_utterance_ms(kana, opt.mora_ms) * static_cast(opt.sample_rate_hz) / 1000.0f);
+ if (!internal::pcm_fits_in_memory(est_samples)) {
+ return tl::make_unexpected(Error::OutOfMemory);
}
std::vector moras;
@@ -120,7 +121,7 @@ tl::expected synthesize_impl(std::u32string_view kana,
auto db = g_voice_db.load();
if (db && db->sample_rate() == opt.sample_rate_hz &&
internal::render_units(moras, *db, out, opt)) {
- return opt.sample_rate_hz;
+ return {};
}
}
@@ -138,22 +139,162 @@ tl::expected synthesize_impl(std::u32string_view kana,
}
out.reserve(estimated_samples);
+ if (visemes) collect_visemes(segs, *visemes);
+
internal::render_segments(segs, out, opt);
+ return {};
+}
+
+// 一括合成。戻り値は出力 PCM のサンプルレート (sanoTTS は 22.05 kHz、他は opt.sample_rate_hz)。
+tl::expected synthesize_impl(std::u32string_view kana, std::vector& out,
+ std::vector* visemes, const Options& opt_in,
+ bool allow_native_rate) {
+ out.clear();
+ if (visemes) visemes->clear();
+ Options opt = resolve_defaults(opt_in);
+
+ // 発話が長すぎて PCM が空きメモリに収まらないなら、どのエンジンでも合成せず断る。
+ const std::uint32_t est_rate =
+ wants_sano(opt) && sano_weights_loaded() ? std::max(opt.sample_rate_hz, 22050u) : opt.sample_rate_hz;
+ const auto est_samples = static_cast(
+ internal::estimate_utterance_ms(kana, opt.mora_ms) * static_cast(est_rate) / 1000.0f);
+ if (!internal::pcm_fits_in_memory(est_samples)) {
+ return tl::make_unexpected(Error::OutOfMemory);
+ }
+
+ // sanoTTS エンジン: 重みがロード済みなら最優先 (口形は出せない)。
+ if (wants_sano(opt) && (allow_native_rate || opt.sample_rate_hz == 22050u)) {
+ std::uint32_t rate = 0;
+ if (internal::render_sano(kana, out, opt, rate)) {
+ return rate;
+ }
+ out.clear();
+ }
+
+ // HMM エンジン: ボイスがロード済みなら次に優先 (品質最良)。
+ // アクセント記号 (' と /) は HMM のみ解釈し、他エンジンでは
+ // parse_kana が読み飛ばす。全チャンクを 1 本の PCM に連結する。
+ if (wants_hmm(opt) && hmm_voice_loaded()) {
+ // PCM は最初に 1 回だけ確保し (途中の再確保 = 旧 + 新の一時倍増を避ける)、
+ // 合成後に余りを返す。
+ out.reserve(est_samples);
+ std::vector scratch;
+ internal::VisemeBuilder builder(visemes ? *visemes : scratch);
+ const auto outcome = internal::render_hmm_stream(
+ kana, opt,
+ [&](std::vector&& pcm, std::vector&& spans, const std::u32string&) {
+ out.insert(out.end(), pcm.begin(), pcm.end());
+ for (const auto& sp : spans) builder.add(sp.vowel, sp.duration_ms);
+ return true;
+ },
+ /*stream=*/false);
+ if (outcome == internal::HmmOutcome::Ok) {
+ builder.finish();
+ if (out.capacity() - out.size() > 32 * 1024) out.shrink_to_fit();
+ return opt.sample_rate_hz;
+ }
+ // 諦めた (メモリ不足など): 途中まで作った分は捨てて他エンジンへ。
+ out.clear();
+ if (visemes) visemes->clear();
+ }
+
+ if (auto r = render_whole(kana, opt, out, visemes); !r) return tl::make_unexpected(r.error());
return opt.sample_rate_hz;
}
} // namespace
-tl::expected synthesize(std::u32string_view kana,
- std::vector& out, const Options& opt) {
- auto r = synthesize_impl(kana, out, opt, /*allow_native_rate=*/false);
+namespace internal {
+
+void VisemeBuilder::add(Vowel v, float duration_ms) {
+ if (duration_ms <= 0.0f) return;
+ if (!have_last_ || v != last_) {
+ out_.push_back({static_cast(t_ms_ + 0.5f), v});
+ last_ = v;
+ have_last_ = true;
+ }
+ t_ms_ += duration_ms;
+}
+
+void spans_to_events(std::span spans, std::vector& out) {
+ out.clear();
+ VisemeBuilder b(out);
+ for (const auto& sp : spans) b.add(sp.vowel, sp.duration_ms);
+ b.finish();
+}
+
+void VisemeBuilder::finish() {
+ if (have_last_ && last_ != Vowel::None) {
+ out_.push_back({static_cast(t_ms_ + 0.5f), Vowel::None});
+ last_ = Vowel::None;
+ }
+}
+
+} // namespace internal
+
+tl::expected synthesize(std::u32string_view kana, std::vector& out,
+ const Options& opt) {
+ auto r = synthesize_impl(kana, out, nullptr, opt, /*allow_native_rate=*/false);
+ if (!r) return tl::make_unexpected(r.error());
+ return {};
+}
+
+tl::expected synthesize(std::u32string_view kana, std::vector& out,
+ std::vector& visemes, const Options& opt) {
+ auto r = synthesize_impl(kana, out, &visemes, opt, /*allow_native_rate=*/false);
if (!r) return tl::make_unexpected(r.error());
return {};
}
tl::expected synthesize_ex(std::u32string_view kana,
std::vector& out, const Options& opt) {
- return synthesize_impl(kana, out, opt, /*allow_native_rate=*/true);
+ return synthesize_impl(kana, out, nullptr, opt, /*allow_native_rate=*/true);
+}
+
+tl::expected synthesize_stream(std::u32string_view kana, const ChunkSink& sink, const Options& opt_in) {
+ const Options opt = resolve_defaults(opt_in);
+
+ // sanoTTS: 全体が 1 チャンク (22.05 kHz)。重み未ロードなら false で次へ。
+ if (wants_sano(opt)) {
+ SynthChunk chunk;
+ std::uint32_t rate = 0;
+ if (internal::render_sano(kana, chunk.pcm, opt, rate) && !chunk.pcm.empty()) {
+ chunk.text = std::u32string(kana);
+ chunk.sample_rate = rate;
+ if (!sink(std::move(chunk))) return tl::make_unexpected(Error::Cancelled);
+ return {};
+ }
+ }
+
+ if (wants_hmm(opt) && hmm_voice_loaded()) {
+ const auto outcome = internal::render_hmm_stream(
+ kana, opt,
+ [&](std::vector&& pcm, std::vector&& spans, const std::u32string& text) {
+ SynthChunk chunk;
+ chunk.text = text;
+ chunk.pcm = std::move(pcm);
+ chunk.sample_rate = opt.sample_rate_hz;
+ internal::spans_to_events(spans, chunk.visemes);
+ return sink(std::move(chunk));
+ },
+ /*stream=*/true);
+ switch (outcome) {
+ case internal::HmmOutcome::Ok: return {};
+ case internal::HmmOutcome::Cancelled: return tl::make_unexpected(Error::Cancelled);
+ // 途中まで渡してしまった分は取り消せないので、他エンジンでやり直さない。
+ case internal::HmmOutcome::Aborted: return tl::make_unexpected(Error::OutOfMemory);
+ case internal::HmmOutcome::NoOutput: break; // 何も渡していない → 他エンジンへ
+ }
+ }
+
+ // HMM 以外 / HMM を使えなかった: 発話全体が 1 チャンク。
+ SynthChunk chunk;
+ chunk.text = std::u32string(kana);
+ chunk.sample_rate = opt.sample_rate_hz;
+ if (auto r = render_whole(kana, opt, chunk.pcm, &chunk.visemes); !r) return r;
+ if (chunk.pcm.empty()) return tl::make_unexpected(Error::InvalidKana);
+ if (!sink(std::move(chunk))) return tl::make_unexpected(Error::Cancelled);
+ return {};
}
} // namespace stackchan::jtts
diff --git a/components/jtts/src/phoneme.cpp b/components/jtts/src/phoneme.cpp
index 9bb87b3..157cda2 100644
--- a/components/jtts/src/phoneme.cpp
+++ b/components/jtts/src/phoneme.cpp
@@ -20,7 +20,8 @@ FormantFrame silent_frame(float f0) {
// prenasal_zero_hz > 0 のとき、後続モーラが「ん」なので母音末尾 ~30 ms で
// nasal を 0→0.5 に立ち上げる (先行母音の鼻音化。ん への移行でスペクトルが
// 急変するのを防ぎ、自然な渡りになる)。値は後続の鼻音ゼロ周波数。
-void add_cv_segments(Consonant c, Vowel v, bool palatalized, bool devoiced, float mora_ms,
+// 戻り値は out 内で母音本体 (push_vowel_tail が積む区間) が始まる添字。
+std::size_t add_cv_segments(Consonant c, Vowel v, bool palatalized, bool devoiced, float mora_ms,
float f0, float prenasal_zero_hz, std::vector& out) {
FormantFrame vowel = vowel_frame(v, palatalized);
vowel.f0_hz = f0;
@@ -37,7 +38,13 @@ void add_cv_segments(Consonant c, Vowel v, bool palatalized, bool devoiced, floa
// 母音末尾セグメントを積む。prenasal (次モーラが「ん」) なら末尾 ~30 ms
// で nasal を 0→0.5 に上げて先行母音を鼻音化する。
+ std::size_t tail_begin = out.size();
+ bool tail_marked = false;
auto push_vowel_tail = [&](float consumed) {
+ if (!tail_marked) {
+ tail_begin = out.size();
+ tail_marked = true;
+ }
float v_ms = std::max(20.0f, mora_ms - consumed);
if (prenasal_zero_hz > 0.0f) {
FormantFrame nasalized = vowel;
@@ -56,7 +63,7 @@ void add_cv_segments(Consonant c, Vowel v, bool palatalized, bool devoiced, floa
if (c == Consonant::None) {
push_vowel_tail(0.0f);
- return;
+ return tail_begin;
}
FormantFrame burst = consonant_burst(c, v);
@@ -140,6 +147,7 @@ void add_cv_segments(Consonant c, Vowel v, bool palatalized, bool devoiced, floa
} else {
out.push_back({vowel, vowel, mora_ms});
}
+ return tail_begin;
}
} // namespace
@@ -179,8 +187,17 @@ void build_segments(std::span moras, std::vector& out, cons
if (i + 1 < moras.size() && moras[i + 1].kind == MoraKind::MoraicN) {
prenasal_zero_hz = moraic_n_frame(i + 1).nasal_zero_hz;
}
- add_cv_segments(m.c, m.v, m.palatalized, m.devoiced, mora_ms, f0,
- prenasal_zero_hz, out);
+ const std::size_t first = out.size();
+ const std::size_t tail = add_cv_segments(m.c, m.v, m.palatalized, m.devoiced, mora_ms, f0,
+ prenasal_zero_hz, out);
+ // 口形: 無声化母音は閉口のまま。子音区間は後続母音の形を先取りするが、
+ // 両唇音 (m b p) の閉鎖だけは唇を閉じる。
+ if (!m.devoiced) {
+ const bool bilabial = m.c == Consonant::M || m.c == Consonant::B || m.c == Consonant::P;
+ for (std::size_t k = first; k < out.size(); ++k) {
+ out[k].vowel = (bilabial && k < tail) ? Vowel::None : m.v;
+ }
+ }
break;
}
case MoraKind::MoraicN: {
@@ -196,7 +213,9 @@ void build_segments(std::span moras, std::vector& out, cons
case MoraKind::Chouon: {
if (!out.empty()) {
FormantFrame ref = out.back().end;
- out.push_back({ref, ref, mora_ms});
+ Segment held{ref, ref, mora_ms};
+ held.vowel = out.back().vowel; // 長音は直前の口形を保つ
+ out.push_back(held);
}
break;
}
diff --git a/components/jtts/src/subtitle.cpp b/components/jtts/src/subtitle.cpp
new file mode 100644
index 0000000..05b8e35
--- /dev/null
+++ b/components/jtts/src/subtitle.cpp
@@ -0,0 +1,82 @@
+// SPDX-FileCopyrightText: 2026 Kenta IDA
+// SPDX-License-Identifier: BSL-1.0
+#include "jtts/subtitle.hpp"
+
+#include
+#include
+
+namespace stackchan::jtts {
+
+namespace {
+
+bool is_pause_char(char32_t c) {
+ return c == U'、' || c == U'。' || c == U',' || c == U',' || c == U'.' || c == U'.';
+}
+
+std::size_t count_pauses(std::u32string_view s) {
+ return static_cast(std::count_if(s.begin(), s.end(), is_pause_char));
+}
+
+// チャンクの読みが句読点で終わるか (末尾の空白・アクセント記号・/ は無視)。
+bool ends_with_pause(std::u32string_view s) {
+ for (std::size_t i = s.size(); i > 0; --i) {
+ const char32_t c = s[i - 1];
+ if (c == U' ' || c == U' ' || c == U'\n' || c == U'\'' || c == U'’' || c == U'/') continue;
+ return is_pause_char(c);
+ }
+ return false;
+}
+
+// UTF-8 で s[i] から始まる句読点の長さ (バイト)。句読点でなければ 0。
+std::size_t pause_len_utf8(std::string_view s, std::size_t i) {
+ const auto b = [&](std::size_t k) { return i + k < s.size() ? static_cast(s[i + k]) : 0u; };
+ if (b(0) == ',' || b(0) == '.') return 1;
+ if (b(0) == 0xE3 && b(1) == 0x80 && (b(2) == 0x81 || b(2) == 0x82)) return 3; // 、 。
+ if (b(0) == 0xEF && b(1) == 0xBC && (b(2) == 0x8C || b(2) == 0x8E)) return 3; // , .
+ return 0;
+}
+
+} // namespace
+
+SubtitleMapper::SubtitleMapper(std::string_view display_utf8, std::u32string_view reading) : whole_(display_utf8) {
+ // 表示テキストを句読点の直後で区切る (句読点は前の句に含める)。
+ std::string cur;
+ std::size_t pauses = 0;
+ for (std::size_t i = 0; i < display_utf8.size();) {
+ const std::size_t n = pause_len_utf8(display_utf8, i);
+ if (n > 0) {
+ cur.append(display_utf8.substr(i, n));
+ segments_.push_back(std::move(cur));
+ cur.clear();
+ ++pauses;
+ i += n;
+ } else {
+ cur.push_back(display_utf8[i]);
+ ++i;
+ }
+ }
+ segments_.push_back(std::move(cur)); // 最後の句読点より後ろ (空のこともある)
+ mapped_ = !display_utf8.empty() && pauses == count_pauses(reading);
+}
+
+std::string SubtitleMapper::next(std::u32string_view chunk_reading) {
+ const bool first = first_;
+ first_ = false;
+ if (!mapped_) {
+ return first ? whole_ : std::string{};
+ }
+ const std::size_t pauses = count_pauses(chunk_reading);
+ const std::size_t begin = consumed_;
+ const std::size_t after = consumed_ + pauses;
+ consumed_ = after;
+ // 句読点で終わるチャンクは、その句読点を含む句までが担当 (次のチャンクは次の句から)。
+ // 途中で切れたチャンクは、まだ句読点に届いていない句 (after) も担当する。
+ const std::size_t last = segments_.size() - 1;
+ const std::size_t a = std::min(begin, last);
+ const std::size_t b = std::min(pauses > 0 && ends_with_pause(chunk_reading) ? after - 1 : after, last);
+ std::string out;
+ for (std::size_t k = a; k <= b; ++k) out += segments_[k];
+ return out;
+}
+
+} // namespace stackchan::jtts
diff --git a/components/jtts/test/host/CMakeLists.txt b/components/jtts/test/host/CMakeLists.txt
index 8914eb6..6cb0818 100644
--- a/components/jtts/test/host/CMakeLists.txt
+++ b/components/jtts/test/host/CMakeLists.txt
@@ -34,6 +34,8 @@ add_library(jtts STATIC
${JTTS_ROOT}/src/hmm_synth.cpp
${JTTS_ROOT}/src/sano_ir.cpp
${JTTS_ROOT}/src/sano_synth.cpp
+ ${JTTS_ROOT}/src/hmm_chunk.cpp
+ ${JTTS_ROOT}/src/subtitle.cpp
)
target_include_directories(jtts
PUBLIC ${JTTS_ROOT}/include ${EXPECTED_INC}
@@ -97,3 +99,14 @@ target_compile_options(jtts_test_sano_ir PRIVATE -Wall -Wextra)
add_executable(jtts_sano_demo sano_demo.cpp wav_writer.cpp)
target_link_libraries(jtts_sano_demo PRIVATE jtts)
target_compile_options(jtts_sano_demo PRIVATE -Wall -Wextra)
+
+# フォルマント エンジンの口形イベント (リップシンク用)
+add_executable(jtts_test_visemes test_visemes.cpp)
+target_link_libraries(jtts_test_visemes PRIVATE jtts)
+target_compile_options(jtts_test_visemes PRIVATE -Wall -Wextra)
+
+# HMM 長文分割 (メモリ予算に応じたチャンク分割合成)
+add_executable(jtts_test_hmm_chunk test_hmm_chunk.cpp)
+target_link_libraries(jtts_test_hmm_chunk PRIVATE jtts)
+target_include_directories(jtts_test_hmm_chunk PRIVATE ${JTTS_ROOT}/src)
+target_compile_options(jtts_test_hmm_chunk PRIVATE -Wall -Wextra)
diff --git a/components/jtts/test/host/test_hmm_chunk.cpp b/components/jtts/test/host/test_hmm_chunk.cpp
new file mode 100644
index 0000000..3954e1c
--- /dev/null
+++ b/components/jtts/test/host/test_hmm_chunk.cpp
@@ -0,0 +1,496 @@
+// SPDX-FileCopyrightText: 2026 Kenta IDA
+// SPDX-License-Identifier: BSL-1.0
+//
+// HMM 合成の長文分割 (split_hmm_text / 分割合成 / メモリ不足フォールバック) の検証。
+// jtts_test_hmm_chunk [voice.htsvoice ...]
+// 分割ロジックは常に検証する。.htsvoice を渡すと、それぞれをロードして、メモリ予算を
+// 絞った分割合成も検証する (例: assets/voices/mei16.htsvoice)。
+#include
+#include
+#include
+#include
+#include
+#include
+#include
+#include
+
+#include "internal.hpp"
+#include "jtts/jtts.hpp"
+#include "jtts/subtitle.hpp"
+
+using namespace stackchan::jtts;
+
+namespace {
+
+int g_failures = 0;
+
+#define CHECK(cond) \
+ do { \
+ if (!(cond)) { \
+ std::fprintf(stderr, "FAIL %s:%d: %s\n", __FILE__, __LINE__, #cond); \
+ ++g_failures; \
+ } \
+ } while (0)
+
+std::u32string join(const std::vector& chunks)
+{
+ std::u32string s;
+ for (const auto& c : chunks) s += c.text;
+ return s;
+}
+
+std::u32string strip_accent(std::u32string s)
+{
+ s.erase(std::remove(s.begin(), s.end(), U'\''), s.end());
+ return s;
+}
+
+void test_split()
+{
+ using internal::HmmChunk;
+ using internal::split_hmm_text;
+ std::vector ch;
+
+ // 収まるなら手を加えず 1 チャンク (アクセント記号もそのまま)。
+ CHECK(split_hmm_text(U"こ'んにちは", 100, ch));
+ CHECK(ch.size() == 1 && ch[0].text == U"こ'んにちは" && ch[0].moras == 5);
+
+ // 不正入力。
+ CHECK(!split_hmm_text(U"こんにちは", 0, ch));
+ CHECK(!split_hmm_text(U"、。", 10, ch));
+ CHECK(!split_hmm_text(U"", 10, ch));
+
+ // 句読点で分け、予算に収まる範囲で貪欲に詰める。区切りは前のチャンクに残る。
+ CHECK(split_hmm_text(U"あいう、えお。かき", 5, ch));
+ CHECK(ch.size() == 2);
+ CHECK(ch[0].text == U"あいう、えお。" && ch[0].moras == 5 && ch[0].pause_after);
+ CHECK(ch[1].text == U"かき" && ch[1].moras == 2 && !ch[1].pause_after);
+ CHECK(join(ch) == U"あいう、えお。かき");
+
+ CHECK(split_hmm_text(U"あいう、えお。かき", 3, ch));
+ CHECK(ch.size() == 3);
+ CHECK(ch[0].text == U"あいう、" && ch[0].pause_after);
+ CHECK(ch[1].text == U"えお。" && ch[1].pause_after);
+ CHECK(ch[2].text == U"かき");
+
+ // アクセント句境界 (/) ではポーズなし。
+ CHECK(split_hmm_text(U"あいう/えお", 3, ch));
+ CHECK(ch.size() == 2 && ch[0].text == U"あいう/" && !ch[0].pause_after && ch[1].text == U"えお");
+
+ // 句読点だけの句は前後に吸収され、モーラを持たないチャンクは作らない。
+ CHECK(split_hmm_text(U"あい、、、うえ", 2, ch));
+ for (const auto& c : ch) CHECK(c.moras > 0 && c.moras <= 2);
+ CHECK(join(ch) == U"あい、、、うえ");
+
+ // 句が単体で予算を超えるとモーラ境界で強制分割する。各チャンクは予算内。
+ const std::u32string flat = U"あいうえおかきくけこさしすせそたちつてとなにぬねの";
+ CHECK(split_hmm_text(flat, 8, ch));
+ CHECK(ch.size() == 4);
+ for (const auto& c : ch) CHECK(c.moras <= 8 && c.moras > 0);
+ CHECK(join(ch) == flat);
+
+ // 拗音・長音・アクセント核の直前では切らない。強制分割ではアクセント核を落とす。
+ const std::u32string youon = U"きゃきゅきょきゃきゅきょきゃきゅきょ";
+ CHECK(split_hmm_text(youon, 4, ch));
+ for (const auto& c : ch) {
+ CHECK(c.moras <= 4);
+ CHECK(c.text.front() != U'ゃ' && c.text.front() != U'ゅ' && c.text.front() != U'ょ');
+ }
+ CHECK(join(ch) == youon);
+ CHECK(split_hmm_text(U"あ'いう'えおかきくけこ", 4, ch));
+ for (const auto& c : ch) CHECK(c.text.find(U'\'') == std::u32string::npos);
+ CHECK(join(ch) == strip_accent(U"あ'いう'えおかきくけこ"));
+
+ // 実際の長文 (146 文字) が予算 20 モーラで全て収まる。
+ const std::u32string longtext =
+ U"すたっくちゃんは、ちいさくてかわいい、てのひらさいずのろぼっとです。"
+ U"かおのひょうじをかえたり、くびをうごかしたり、おしゃべりしたりできます。"
+ U"じぶんでぷろぐらむをつくって、いろいろなことをさせられるのも、たのしいところです。"
+ U"つくるひとによって、いろいろなこせいがうまれる、たのしいろぼっとです。";
+ CHECK(split_hmm_text(longtext, 20, ch));
+ CHECK(ch.size() > 4);
+ for (const auto& c : ch) CHECK(c.moras <= 20 && c.moras > 0);
+ CHECK(join(ch) == longtext);
+ // 句読点で終わるチャンクが大半 (強制分割は不要な長さの句ばかり)。
+ std::size_t pauses = 0;
+ for (const auto& c : ch) pauses += c.pause_after ? 1 : 0;
+ CHECK(pauses + 1 >= ch.size());
+
+ // 低遅延モード (first_moras > 0): 先頭を小さく出し、句の途中では切らない。
+ CHECK(split_hmm_text(longtext, 23, ch, 14));
+ CHECK(join(ch) == longtext);
+ CHECK(ch.front().moras <= 14);
+ CHECK(ch.size() >= 8);
+ for (std::size_t i = 0; i < ch.size(); ++i) {
+ CHECK(ch[i].moras > 0 && ch[i].moras <= 23);
+ CHECK(ch[i].pause_after); // 全て句読点で終わる = 句の途中では切っていない
+ // 直前のチャンクの 1.3 倍を超えて急に大きくならない (先頭 14 モーラ以内は除く)。
+ if (i > 0) CHECK(ch[i].moras <= std::max(14, (ch[i - 1].moras * 13 + 9) / 10) ||
+ ch[i].moras <= ch[i - 1].moras + 4);
+ }
+ // 句読点の無い短文は 1 チャンクのまま (原文どおり)。
+ CHECK(split_hmm_text(U"こんにちは、", 23, ch, 14));
+ CHECK(ch.size() == 1 && ch[0].text == U"こんにちは、");
+ CHECK(split_hmm_text(U"こ'んにちはありが'とうございま'す", 23, ch, 14));
+ CHECK(ch.size() == 1 && ch[0].text == U"こ'んにちはありが'とうございま'す");
+ // 全体が予算内でも、句読点があれば先頭を早く出すために分ける。
+ CHECK(split_hmm_text(U"あいうえお、かきくけこさし、たちつてと", 23, ch, 14));
+ CHECK(ch.size() == 2 && ch[0].moras == 12 && ch[1].moras == 5);
+ // 非低遅延 (first_moras = 0) では従来どおり 1 チャンク。
+ CHECK(split_hmm_text(U"あいうえお、かきくけこさし、たちつてと", 23, ch));
+ CHECK(ch.size() == 1);
+}
+
+// 吹き出しをチャンクに対応させる。
+void test_subtitle()
+{
+ // 句読点ごとに対応する。
+ {
+ SubtitleMapper m("スタックチャンは、小さくて、かわいい。ロボットです。",
+ U"すたっくちゃんは、ちいさくて、かわいい。ろぼっとです。");
+ CHECK(m.mapped());
+ CHECK(m.next(U"すたっくちゃんは、") == "スタックチャンは、");
+ CHECK(m.next(U"ちいさくて、かわいい。") == "小さくて、かわいい。");
+ CHECK(m.next(U"ろぼっとです。") == "ロボットです。");
+ }
+ // 1 チャンクが複数の句を受け持つ。
+ {
+ SubtitleMapper m("A、B、C。", U"あ、い、う。");
+ CHECK(m.mapped());
+ CHECK(m.next(U"あ、い、") == "A、B、");
+ CHECK(m.next(U"う。") == "C。");
+ }
+ // 句の途中で切れたチャンク (強制分割) は、同じ句を続けて返す。
+ {
+ SubtitleMapper m("あいうえおかきくけこ、さしすせそ。", U"あいうえおかきくけこ、さしすせそ。");
+ CHECK(m.next(U"あいうえお") == "あいうえおかきくけこ、");
+ CHECK(m.next(U"かきくけこ、") == "あいうえおかきくけこ、");
+ CHECK(m.next(U"さしすせそ。") == "さしすせそ。");
+ }
+ // アクセント記号・空白が付いた読み。
+ {
+ SubtitleMapper m("今日は。天気。", U"きょ'うは。 てんき。");
+ CHECK(m.next(U"きょ'うは。 ") == "今日は。");
+ CHECK(m.next(U"てんき。") == "天気。");
+ }
+ // 句読点が無ければ全体を 1 つの句として返す。
+ {
+ SubtitleMapper m("こんにちは", U"こんにちは");
+ CHECK(m.mapped() && m.next(U"こんにちは") == "こんにちは");
+ }
+ // 対応が取れない (句読点の数が違う / 表示が空): 最初に全文、以降は空。
+ for (const auto* d : {"こんにちは!げんき?", ""}) {
+ SubtitleMapper m(d, U"こんにちは、げんき?");
+ CHECK(!m.mapped());
+ CHECK(m.next(U"こんにちは、") == d);
+ CHECK(m.next(U"げんき?").empty());
+ }
+}
+
+std::string vowels_only(const std::vector& ev)
+{
+ std::string s;
+ for (const auto& e : ev) {
+ switch (e.vowel) {
+ case Vowel::A: s.push_back('a'); break;
+ case Vowel::I: s.push_back('i'); break;
+ case Vowel::U: s.push_back('u'); break;
+ case Vowel::E: s.push_back('e'); break;
+ case Vowel::O: s.push_back('o'); break;
+ case Vowel::None: break;
+ }
+ }
+ return s;
+}
+
+// 母音列の編集距離 (Levenshtein)。
+std::size_t edit_distance(const std::string& a, const std::string& b)
+{
+ std::vector prev(b.size() + 1), cur(b.size() + 1);
+ for (std::size_t j = 0; j <= b.size(); ++j) prev[j] = j;
+ for (std::size_t i = 1; i <= a.size(); ++i) {
+ cur[0] = i;
+ for (std::size_t j = 1; j <= b.size(); ++j) {
+ cur[j] = std::min({prev[j] + 1, cur[j - 1] + 1, prev[j - 1] + (a[i - 1] != b[j - 1] ? 1u : 0u)});
+ }
+ std::swap(prev, cur);
+ }
+ return prev[b.size()];
+}
+
+// synthesize_stream: チャンクごとに渡され、最初のチャンクが早く出て、連結すると
+// 一括合成と同じ発話になる。
+void test_stream(const std::u32string& longtext, const Options& opt, const std::string& ref_vowels, double ref_ms)
+{
+ internal::set_hmm_memory_budget_for_test(0);
+ std::vector got;
+ auto r = synthesize_stream(
+ longtext,
+ [&](SynthChunk&& c) {
+ got.push_back(std::move(c));
+ return true;
+ },
+ opt);
+ CHECK(r.has_value());
+ std::printf(" stream: %zu chunks", got.size());
+ CHECK(got.size() >= 6);
+
+ // 連結: PCM の長さと母音列 (チャンク先頭からの口形を積算して並べ直す)。
+ std::size_t total_samples = 0;
+ std::string vowels;
+ for (const auto& c : got) {
+ CHECK(!c.pcm.empty());
+ total_samples += c.pcm.size();
+ CHECK(!c.visemes.empty() && c.visemes.front().start_ms == 0);
+ CHECK(c.visemes.back().vowel == Vowel::None);
+ for (std::size_t i = 1; i < c.visemes.size(); ++i) CHECK(c.visemes[i].start_ms > c.visemes[i - 1].start_ms);
+ vowels += vowels_only(c.visemes);
+ }
+ const double ms = 1000.0 * static_cast(total_samples) / opt.sample_rate_hz;
+ const double first_ms = 1000.0 * static_cast(got.front().pcm.size()) / opt.sample_rate_hz;
+ std::printf(", total %.0f ms (%+.1f%%), first chunk %.0f ms\n", ms, (ms - ref_ms) * 100.0 / ref_ms, first_ms);
+ CHECK(vowels == ref_vowels);
+ CHECK(std::abs(ms - ref_ms) < 0.05 * ref_ms);
+ CHECK(first_ms < 0.2 * ms && first_ms < 3000.0); // 低遅延: 先頭は全体の 1/5 未満・3 秒未満
+ // 各チャンクの口形は自分の PCM の長さに収まる。
+ for (const auto& c : got) {
+ const double len = 1000.0 * static_cast(c.pcm.size()) / opt.sample_rate_hz;
+ CHECK(static_cast(c.visemes.back().start_ms) <= len + 1.0);
+ }
+
+ // 各チャンクの読み (text) を連結すると元の読みになり、それに合わせて表示テキストを
+ // 切り出すと、全チャンクで空でなく、連結すると表示テキスト全体になる。
+ {
+ const std::string display =
+ "スタックチャンは、小さくてかわいい、手のひらサイズのロボットです。"
+ "顔の表情を変えたり、首を動かしたり、おしゃべりしたりできます。"
+ "自分でプログラムを作って、いろいろなことをさせられるのも、楽しいところです。"
+ "作る人によって、いろいろな個性が生まれる、楽しいロボットです。";
+ SubtitleMapper m(display, longtext);
+ CHECK(m.mapped());
+ std::u32string joined;
+ std::string shown;
+ for (const auto& c : got) {
+ joined += c.text;
+ const std::string t = m.next(c.text);
+ CHECK(!t.empty());
+ shown += t;
+ }
+ CHECK(joined == longtext);
+ CHECK(shown == display);
+ }
+
+ // sink が false を返すと中断: それ以降のチャンクは合成されない。
+ int calls = 0;
+ r = synthesize_stream(
+ longtext,
+ [&](SynthChunk&&) { return ++calls < 2; },
+ opt);
+ CHECK(!r.has_value() && r.error() == Error::Cancelled);
+ CHECK(calls == 2);
+
+ // 全体の PCM が収まらない空き (400 KB) でも、HMM はチャンクごとなので発話できる。
+ internal::set_pcm_memory_limit_for_test(400 * 1024);
+ std::size_t n = 0;
+ r = synthesize_stream(
+ longtext,
+ [&](SynthChunk&& c) {
+ ++n;
+ CHECK(c.pcm.size() * 2 < 400 * 1024); // 1 チャンクなら十分小さい
+ return true;
+ },
+ opt);
+ CHECK(r.has_value() && n >= 6);
+ // 一括版は同じ条件だと断る (全体を 1 本の PCM に持つため)。
+ std::vector whole;
+ auto rw = synthesize(longtext, whole, opt);
+ CHECK(!rw.has_value() && rw.error() == Error::OutOfMemory);
+ internal::set_pcm_memory_limit_for_test(0);
+
+ // 予算を絞っても (チャンクがさらに小さくなるだけで) 最後まで発話できる。
+ internal::set_hmm_memory_budget_for_test(std::size_t{1500} * 1024);
+ n = 0;
+ r = synthesize_stream(
+ longtext,
+ [&](SynthChunk&&) {
+ ++n;
+ return true;
+ },
+ opt);
+ CHECK(r.has_value() && n >= 6);
+
+ // 1 モーラも収まらない予算: 何も渡す前なので他エンジン (フォルマント) の 1 チャンクにフォールバック。
+ internal::set_hmm_memory_budget_for_test(64 * 1024);
+ got.clear();
+ r = synthesize_stream(
+ longtext,
+ [&](SynthChunk&& c) {
+ got.push_back(std::move(c));
+ return true;
+ },
+ opt);
+ CHECK(r.has_value() && got.size() == 1 && !got[0].pcm.empty());
+ internal::set_hmm_memory_budget_for_test(0);
+}
+
+void test_hmm(const char* voice_path)
+{
+ std::ifstream f(voice_path, std::ios::binary);
+ static std::vector blob; // set_hmm_voice は blob の寿命を要求する
+ blob.assign(std::istreambuf_iterator(f), std::istreambuf_iterator());
+ CHECK(set_hmm_voice(blob));
+ std::printf("HMM voice: %s\n", voice_path);
+
+ Options opt;
+ opt.engine = Engine::Hmm;
+ opt.mora_ms = 120.0f; // 実機の既定
+ const std::u32string longtext =
+ U"すたっくちゃんは、ちいさくてかわいい、てのひらさいずのろぼっとです。"
+ U"かおのひょうじをかえたり、くびをうごかしたり、おしゃべりしたりできます。"
+ U"じぶんでぷろぐらむをつくって、いろいろなことをさせられるのも、たのしいところです。"
+ U"つくるひとによって、いろいろなこせいがうまれる、たのしいろぼっとです。";
+
+ // 無制限 (1 チャンク) の基準。
+ internal::set_hmm_memory_budget_for_test(0);
+ std::vector ref_pcm;
+ std::vector ref_ev;
+ CHECK(synthesize(longtext, ref_pcm, ref_ev, opt).has_value());
+ const std::string ref_vowels = vowels_only(ref_ev);
+ const double ref_ms = 1000.0 * static_cast(ref_pcm.size()) / opt.sample_rate_hz;
+ std::printf(" unchunked: %.0f ms, %zu vowels\n", ref_ms, ref_vowels.size());
+ CHECK(!ref_vowels.empty());
+
+ // 予算を絞って分割合成: 落ちずに合成でき、母音列も長さも基準に近い。
+ // 実機で想定する予算 (2 MB 前後) では句読点でだけ分かれ、母音列は完全に一致する。
+ // 極端に小さい予算 (句の途中で強制分割される) では、切れ目で無声化の文脈が
+ // 失われて母音が数個ずれるのを許容する。
+ struct Case {
+ std::size_t budget_kb;
+ std::size_t max_edit; // 母音列の許容編集距離
+ double tol; // 長さの許容誤差 (割合)
+ };
+ for (const Case c : {Case{2000, 0, 0.05}, Case{1500, 1, 0.05}, Case{1000, 4, 0.20}}) {
+ internal::set_hmm_memory_budget_for_test(c.budget_kb * 1024);
+ std::vector pcm;
+ std::vector ev;
+ CHECK(synthesize(longtext, pcm, ev, opt).has_value());
+ const double ms = 1000.0 * static_cast(pcm.size()) / opt.sample_rate_hz;
+ const std::size_t dist = edit_distance(vowels_only(ev), ref_vowels);
+ std::printf(" budget %zu KB: %.0f ms (%+.1f%%), vowel edit distance %zu\n", c.budget_kb, ms,
+ (ms - ref_ms) * 100.0 / ref_ms, dist);
+ CHECK(dist <= c.max_edit);
+ CHECK(std::abs(ms - ref_ms) < c.tol * ref_ms);
+
+ // 口形イベントは PCM と時間軸が揃う: 昇順、隣接は異なる、最後は PCM 長以内で閉口。
+ CHECK(!ev.empty() && ev.front().start_ms == 0);
+ for (std::size_t i = 1; i < ev.size(); ++i) {
+ CHECK(ev[i].start_ms > ev[i - 1].start_ms);
+ CHECK(ev[i].vowel != ev[i - 1].vowel);
+ }
+ CHECK(ev.back().vowel == Vowel::None);
+ CHECK(static_cast(ev.back().start_ms) <= ms + 1.0);
+ CHECK(ms - static_cast(ev.back().start_ms) < 1000.0);
+ }
+
+ // 母音の位置と実際の音: 分割合成でも各母音区間に音が出ている (無音のまま
+ // 母音イベントが立っていない)。全母音区間の RMS が背景 (先頭 sil) を上回る割合を見る。
+ {
+ internal::set_hmm_memory_budget_for_test(std::size_t{800} * 1024);
+ std::vector pcm;
+ std::vector ev;
+ CHECK(synthesize(longtext, pcm, ev, opt).has_value());
+ std::size_t voiced = 0, total = 0;
+ for (std::size_t i = 0; i + 1 < ev.size(); ++i) {
+ if (ev[i].vowel == Vowel::None) continue;
+ const double len = ev[i + 1].start_ms - ev[i].start_ms;
+ if (len < 40.0) continue;
+ const std::size_t a = static_cast((ev[i].start_ms + 0.25 * len) * 16.0);
+ const std::size_t b = std::min(pcm.size(), static_cast((ev[i].start_ms + 0.75 * len) * 16.0));
+ double acc = 0;
+ for (std::size_t k = a; k < b; ++k) acc += static_cast(pcm[k]) * pcm[k];
+ const double rms = std::sqrt(acc / static_cast(std::max(1, b - a)));
+ ++total;
+ if (rms > 300.0) ++voiced;
+ }
+ std::printf(" vowel spans with sound: %zu / %zu\n", voiced, total);
+ CHECK(total > 20 && voiced * 10 >= total * 9);
+ }
+
+ test_stream(longtext, opt, ref_vowels, ref_ms);
+
+ // 1 モーラも収まらない予算: HMM を諦めて (クラッシュせず) フォールバックする。
+ internal::set_hmm_memory_budget_for_test(64 * 1024);
+ {
+ std::vector pcm;
+ std::vector ev;
+ CHECK(synthesize(longtext, pcm, ev, opt).has_value());
+ CHECK(!pcm.empty()); // フォルマント合成で音が出る
+ }
+ internal::set_hmm_memory_budget_for_test(0);
+ set_hmm_voice({});
+}
+
+// 発話が長すぎて PCM が空きメモリに収まらないときは、どのエンジンでも合成せず
+// OutOfMemory を返す (確保失敗によるクラッシュを避ける)。
+void test_pcm_limit()
+{
+ Options opt;
+ opt.engine = Engine::Formant;
+ std::vector pcm;
+ std::vector ev;
+ const std::u32string longtext =
+ U"すたっくちゃんは、ちいさくてかわいい、てのひらさいずのろぼっとです。"
+ U"かおのひょうじをかえたり、くびをうごかしたり、おしゃべりしたりできます。"
+ U"じぶんでぷろぐらむをつくって、いろいろなことをさせられるのも、たのしいところです。"
+ U"つくるひとによって、いろいろなこせいがうまれる、たのしいろぼっとです。";
+
+ // 空きが 400 KB: 短い発話は通り、長文 (PCM ≈ 1 MB) は断る。
+ internal::set_pcm_memory_limit_for_test(400 * 1024);
+ CHECK(synthesize(U"こんにちは", pcm, ev, opt).has_value());
+ CHECK(!pcm.empty());
+ auto r = synthesize(longtext, pcm, ev, opt);
+ CHECK(!r.has_value() && r.error() == Error::OutOfMemory);
+ CHECK(pcm.empty() && ev.empty());
+ // 3 引数版も同じ。
+ r = synthesize(longtext, pcm, opt);
+ CHECK(!r.has_value() && r.error() == Error::OutOfMemory);
+
+ // ストリーミングでも、HMM 以外は全体 1 チャンクなので同じく断る (何も渡さない)。
+ std::size_t chunks = 0;
+ auto rs = synthesize_stream(
+ longtext,
+ [&](SynthChunk&&) {
+ ++chunks;
+ return true;
+ },
+ opt);
+ CHECK(!rs.has_value() && rs.error() == Error::OutOfMemory && chunks == 0);
+
+ // 上限なし (ホスト既定) なら長文も合成できる。
+ internal::set_pcm_memory_limit_for_test(0);
+ // フォルマントのストリーミングは全体が 1 チャンクで、口形付き。
+ chunks = 0;
+ rs = synthesize_stream(
+ longtext,
+ [&](SynthChunk&& c) {
+ ++chunks;
+ CHECK(!c.pcm.empty() && !c.visemes.empty() && c.visemes.front().start_ms == 0);
+ return true;
+ },
+ opt);
+ CHECK(rs.has_value() && chunks == 1);
+ CHECK(synthesize(longtext, pcm, ev, opt).has_value());
+ CHECK(pcm.size() > 16000 * 10); // 10 秒以上 (500 KB 超: 上の 400 KB 制限では収まらない長さ)
+}
+
+} // namespace
+
+int main(int argc, char** argv)
+{
+ test_split();
+ test_subtitle();
+ test_pcm_limit();
+ for (int i = 1; i < argc; ++i) test_hmm(argv[i]);
+ if (g_failures == 0) std::puts("test_hmm_chunk: all passed");
+ return g_failures == 0 ? 0 : 1;
+}
diff --git a/components/jtts/test/host/test_visemes.cpp b/components/jtts/test/host/test_visemes.cpp
new file mode 100644
index 0000000..f384f15
--- /dev/null
+++ b/components/jtts/test/host/test_visemes.cpp
@@ -0,0 +1,195 @@
+// SPDX-FileCopyrightText: 2026 Kenta IDA
+// SPDX-License-Identifier: BSL-1.0
+//
+// 口形イベント (VisemeEvent) の検証。
+// jtts_test_visemes [voice.htsvoice ...]
+// フォルマント エンジンのケースは常に実行する。.htsvoice を渡すと、それぞれを
+// ロードして HMM エンジンのケースも実行する (例: assets/voices/mei16.htsvoice)。
+#include
+#include
+#include
+#include
+#include
+#include
+#include
+#include
+
+#include "jtts/jtts.hpp"
+
+using namespace stackchan::jtts;
+
+namespace {
+
+int g_failures = 0;
+
+#define CHECK(cond) \
+ do { \
+ if (!(cond)) { \
+ std::fprintf(stderr, "FAIL %s:%d: %s\n", __FILE__, __LINE__, #cond); \
+ ++g_failures; \
+ } \
+ } while (0)
+
+char to_char(Vowel v)
+{
+ switch (v) {
+ case Vowel::A: return 'a';
+ case Vowel::I: return 'i';
+ case Vowel::U: return 'u';
+ case Vowel::E: return 'e';
+ case Vowel::O: return 'o';
+ case Vowel::None: break;
+ }
+ return '-';
+}
+
+// 母音列を "aiueo-" のような文字列にする (None は '-')。
+std::string shape_of(std::u32string_view kana, std::vector& pcm,
+ std::vector& ev, const Options& opt)
+{
+ auto r = synthesize(kana, pcm, ev, opt);
+ CHECK(r.has_value());
+ std::string s;
+ for (const auto& e : ev) s.push_back(to_char(e.vowel));
+ return s;
+}
+
+// 母音 (None を除く) だけを並べた文字列。
+std::string vowels_only(const std::vector& ev)
+{
+ std::string s;
+ for (const auto& e : ev) {
+ if (e.vowel != Vowel::None) s.push_back(to_char(e.vowel));
+ }
+ return s;
+}
+
+// [from_ms, to_ms) の PCM の RMS。
+double rms_between(const std::vector& pcm, std::uint32_t rate, double from_ms, double to_ms)
+{
+ const std::size_t a = static_cast(from_ms * rate / 1000.0);
+ const std::size_t b = std::min(pcm.size(), static_cast(to_ms * rate / 1000.0));
+ if (b <= a) return 0.0;
+ double acc = 0.0;
+ for (std::size_t i = a; i < b; ++i) acc += static_cast(pcm[i]) * pcm[i];
+ return std::sqrt(acc / static_cast(b - a));
+}
+
+void test_hmm(const char* voice_path)
+{
+ std::ifstream f(voice_path, std::ios::binary);
+ static std::vector blob; // set_hmm_voice は blob の寿命を要求する
+ blob.assign(std::istreambuf_iterator(f), std::istreambuf_iterator());
+ CHECK(!blob.empty());
+ CHECK(set_hmm_voice(blob));
+ std::printf("HMM voice: %s\n", voice_path);
+
+ Options opt;
+ opt.engine = Engine::Hmm;
+ std::vector pcm;
+ std::vector ev;
+
+ CHECK(synthesize(U"あいうえお", pcm, ev, opt).has_value());
+ CHECK(!pcm.empty());
+ CHECK(!ev.empty());
+ CHECK(vowels_only(ev) == "aiueo");
+
+ // 先頭は無音 (sil) で閉口、最後も閉口。時刻は昇順で隣接する母音は異なる。
+ CHECK(ev.front().start_ms == 0 && ev.front().vowel == Vowel::None);
+ CHECK(ev.back().vowel == Vowel::None);
+ for (std::size_t i = 1; i < ev.size(); ++i) {
+ CHECK(ev[i].start_ms > ev[i - 1].start_ms);
+ CHECK(ev[i].vowel != ev[i - 1].vowel);
+ }
+ const double pcm_ms = 1000.0 * static_cast(pcm.size()) / opt.sample_rate_hz;
+ CHECK(static_cast(ev.back().start_ms) <= pcm_ms + 1.0);
+ CHECK(pcm_ms - static_cast(ev.back().start_ms) < 1000.0);
+
+ // PCM との時間軸の一致: 先頭の無音区間は静かで、最初の母音区間は音が出ている。
+ for (std::size_t i = 0; i + 1 < ev.size(); ++i) {
+ if (ev[i].vowel == Vowel::None) continue;
+ const double v0 = ev[i].start_ms, v1 = ev[i + 1].start_ms;
+ const double sil = rms_between(pcm, opt.sample_rate_hz, 0.0, ev.front().start_ms + 0.5 * (ev[1].start_ms));
+ const double voiced = rms_between(pcm, opt.sample_rate_hz, v0 + 0.25 * (v1 - v0), v0 + 0.75 * (v1 - v0));
+ std::printf(" first vowel @%.0f-%.0f ms: rms(sil)=%.1f rms(vowel)=%.1f\n", v0, v1, sil, voiced);
+ CHECK(voiced > 4.0 * sil + 100.0);
+ break;
+ }
+
+ // 無声化母音は閉口: 「きした」で母音は た の a だけ。
+ CHECK(synthesize(U"きした", pcm, ev, opt).has_value());
+ CHECK(vowels_only(ev) == "a");
+
+ // 呼気段落境界の pau は閉口として挟まる。
+ CHECK(synthesize(U"あ、い", pcm, ev, opt).has_value());
+ CHECK(vowels_only(ev) == "ai");
+
+ // 話速 (mora_ms) を倍にすると口形の時間軸も約 2 倍に伸びる。
+ CHECK(synthesize(U"あいうえお", pcm, ev, opt).has_value());
+ const double t_normal = ev.back().start_ms;
+ Options slow = opt;
+ slow.mora_ms = 220.0f;
+ CHECK(synthesize(U"あいうえお", pcm, ev, slow).has_value());
+ CHECK(vowels_only(ev) == "aiueo");
+ CHECK(ev.back().start_ms > 1.5 * t_normal);
+
+ // Auto でも HMM が選ばれ口形が出る。
+ Options autoo;
+ CHECK(synthesize(U"あいうえお", pcm, ev, autoo).has_value());
+ CHECK(vowels_only(ev) == "aiueo");
+
+ set_hmm_voice({});
+}
+
+} // namespace
+
+int main(int argc, char** argv)
+{
+ Options opt;
+ opt.engine = Engine::Formant; // 口形を出せるのはフォルマントのみ
+
+ std::vector pcm;
+ std::vector ev;
+
+ // 母音そのまま: 母音ごとに 1 イベント + 終端の閉口。
+ CHECK(shape_of(U"あいうえお", pcm, ev, opt) == "aiueo-");
+
+ // 時刻は昇順、隣接イベントの母音は異なり、終端は PCM 長と一致する。
+ for (std::size_t i = 1; i < ev.size(); ++i) {
+ CHECK(ev[i].start_ms > ev[i - 1].start_ms);
+ CHECK(ev[i].vowel != ev[i - 1].vowel);
+ }
+ const double pcm_ms = 1000.0 * static_cast(pcm.size()) / opt.sample_rate_hz;
+ CHECK(std::abs(static_cast(ev.back().start_ms) - pcm_ms) < 5.0);
+ CHECK(ev.front().start_ms == 0);
+
+ // 両唇音は閉口 → 母音。それ以外の子音は母音の形を先取りする。
+ CHECK(shape_of(U"ま", pcm, ev, opt) == "-a-");
+ CHECK(shape_of(U"か", pcm, ev, opt) == "a-");
+ CHECK(ev.front().start_ms == 0);
+
+ // 促音・撥音は閉口、長音は直前の母音を保つ。
+ CHECK(shape_of(U"あっあ", pcm, ev, opt) == "a-a-");
+ CHECK(shape_of(U"あんあ", pcm, ev, opt) == "a-a-");
+ CHECK(shape_of(U"あーあ", pcm, ev, opt) == "a-");
+
+ // 無声化母音は閉口のまま。「きした」は き・し とも無声化される。
+ CHECK(shape_of(U"きした", pcm, ev, opt) == "-a-");
+ CHECK(shape_of(U"ひとつ", pcm, ev, opt) == "-o-");
+
+ // 空 / 不正な入力ではイベントも空。
+ auto bad = synthesize(U"", pcm, ev, opt);
+ CHECK(!bad.has_value());
+ CHECK(ev.empty());
+
+ // 既存の 3 引数 API と PCM が一致する (口形の収集が合成を変えない)。
+ std::vector pcm_plain;
+ CHECK(synthesize(U"こんにちは", pcm_plain, opt).has_value());
+ CHECK(synthesize(U"こんにちは", pcm, ev, opt).has_value());
+ CHECK(pcm == pcm_plain);
+
+ for (int i = 1; i < argc; ++i) test_hmm(argv[i]);
+
+ if (g_failures == 0) std::puts("test_visemes: all passed");
+ return g_failures == 0 ? 0 : 1;
+}
From 8d6cb861314603d894091dad0320e060350c4096 Mon Sep 17 00:00:00 2001
From: Takao Akaki
Date: Mon, 21 Sep 2026 18:02:33 +0900
Subject: [PATCH 07/10] =?UTF-8?q?feat(main):=20=E6=AF=8D=E9=9F=B3=E3=83=AA?=
=?UTF-8?q?=E3=83=83=E3=83=97=E3=82=B7=E3=83=B3=E3=82=AF=E3=81=A8=E3=83=81?=
=?UTF-8?q?=E3=83=A3=E3=83=B3=E3=82=AF=E5=8D=98=E4=BD=8D=E3=81=AE=E5=86=8D?=
=?UTF-8?q?=E7=94=9F=E3=83=BB=E5=90=B9=E3=81=8D=E5=87=BA=E3=81=97=E5=90=8C?=
=?UTF-8?q?=E6=9C=9F=E3=82=92=E8=BF=BD=E5=8A=A0=E3=81=99=E3=82=8B?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
jtts の口形イベント / ストリーミング合成 (前コミット) と、avatar の mouth_form
(mouth_form コミット) を発話に繋ぐ。Speech は上流の非同期化 (合成を別タスクで行い
demo_loop をブロックしない) の構造をそのまま使う。
■ 母音リップシンク
- Speech::current_mouth(): 母音ごとの口形 (開き, 幅) を返す。母音間は 60 ms で
補間。口形が無いエンジン (単位連結 / sanoTTS) は従来どおり音量エンベロープ。
- SharedState::face.mouth_form (既定 -1 = mouth_open に追従) → render_task →
Avatar::set_mouth_form。マイク・会話・BLE ストリームなど他の mouth_open の
書き手に切り替わるときは demo_loop が -1 に戻す。
■ チャンク単位の再生
- ChunkPlayer (新規): チャンクを M5.Speaker の 1 チャンネル (2 スロット) に順に
積み、再生し終えたバッファから解放する。スロットが空くまで待つので合成が
再生より先に進みすぎない。再生予定時刻も返す。stop() は別タスクから呼べ、
「取り消し確認 → 再生開始」と「取り消し → 停止」を同じミューテックスで直列化する
ので、stop() の後に遅れてチャンクが鳴り出すことは無い。
- speech_synth タスクの中で synthesize_stream を回し、1 チャンク目が出来次第
再生を始めて再生中に次を合成する (最初の音まで ≈ 18 秒 → ≈ 1.2 秒、146 文字の
フレーズ)。sanoTTS など 22.05 kHz のチャンクはチャンクのレートで鳴らす。
stop() は上流の世代カウンタ (gen_) で進行中の合成を捨てる。
- 口の更新は Speech 自身の 20 ms タイマー (esp_timer): demo_loop のポーリング
周期に依存せず、合成中もチャンクの継ぎ目で口が止まらない。
- say_worker (設定ページの発話テスト / MCP say / jtts-say) もストリーミング再生。
上流の PSRAM スタックとリーク対策 (処理本体を関数に切り出す) はそのまま。
■ 吹き出しのチャンク同期
- babble() は、表示テキストを SubtitleMapper で句ごとに対応付け、各チャンクの
再生開始時刻に、そのチャンクの部分をタイマーから吹き出しに出す (hold は
チャンクの音声長)。対応が取れないときは最初の音と同時に全文を出す。合成できな
かった / 開始できなかったときも全文を 1 回出す。従来は合成の完了後 (最初の音から
約 17 秒遅れ) に全文を出していた。
- 次の発話の開始は balloon_in_flight ではなく balloon_visible() で判定する。
Co-Authored-By: Claude Sonnet 5
---
main/CMakeLists.txt | 1 +
main/chunk_player.cpp | 121 +++++++++++++++
main/chunk_player.hpp | 70 +++++++++
main/demo_loop.cpp | 51 ++++--
main/render_task.cpp | 1 +
main/settings_sinks.cpp | 90 ++++++-----
main/shared_state.hpp | 5 +
main/speech.cpp | 333 ++++++++++++++++++++++++++++++++++------
main/speech.hpp | 119 +++++++++++---
9 files changed, 667 insertions(+), 124 deletions(-)
create mode 100644 main/chunk_player.cpp
create mode 100644 main/chunk_player.hpp
diff --git a/main/CMakeLists.txt b/main/CMakeLists.txt
index a2e25e5..f7c8081 100644
--- a/main/CMakeLists.txt
+++ b/main/CMakeLists.txt
@@ -45,6 +45,7 @@ set(_main_srcs
"dance_storage.cpp"
"dance_poc.cpp"
"captive_portal.cpp"
+ "chunk_player.cpp"
"demo_loop.cpp"
"device_ui.cpp"
"diag.cpp"
diff --git a/main/chunk_player.cpp b/main/chunk_player.cpp
new file mode 100644
index 0000000..a35fd2e
--- /dev/null
+++ b/main/chunk_player.cpp
@@ -0,0 +1,121 @@
+// SPDX-FileCopyrightText: 2026 Kenta IDA
+// SPDX-License-Identifier: BSL-1.0
+
+#include "chunk_player.hpp"
+
+#include
+
+#include
+#include
+#include
+#include
+
+namespace stackchan::app {
+
+namespace {
+
+// M5.Speaker の 1 チャンネルのスロット数 (再生中 + 次)。
+constexpr std::size_t kSlots = 2;
+
+// 折り返しを考慮して a が b より後か。
+bool after(std::uint32_t a, std::uint32_t b)
+{
+ return static_cast(a - b) > 0;
+}
+
+} // namespace
+
+std::uint32_t now_ms()
+{
+ return static_cast(esp_timer_get_time() / 1000);
+}
+
+void ChunkPlayer::begin()
+{
+ const auto ch = static_cast(channel_);
+ bool was_playing = false;
+ {
+ std::lock_guard lock(mtx_);
+ was_playing = end_ms_ != 0 && after(end_ms_, now_ms());
+ if (was_playing || M5.Speaker.isPlaying(ch) != 0) {
+ M5.Speaker.stop();
+ was_playing = true;
+ }
+ end_ms_ = 0;
+ }
+ if (was_playing) {
+ // スピーカー タスクが最後のブロックを読み終えるまで待ってから解放する。
+ vTaskDelay(pdMS_TO_TICKS(30));
+ }
+ release();
+}
+
+std::optional ChunkPlayer::enqueue(std::vector&& pcm, std::uint32_t sample_rate,
+ const std::function& cancelled)
+{
+ const auto is_cancelled = [&] { return cancelled && cancelled(); };
+ if (pcm.empty()) {
+ std::lock_guard lock(mtx_);
+ return std::max(now_ms(), end_ms_);
+ }
+ const auto ch = static_cast(channel_);
+
+ // 空きスロットが出るまで待つ。合成の方が再生より速いとここで待たされるが、
+ // その分メモリを溜め込まない (最大でも「再生中 + 次 + 合成中」の 3 チャンク)。
+ while (M5.Speaker.isPlaying(ch) >= kSlots) {
+ if (is_cancelled()) return std::nullopt;
+ vTaskDelay(pdMS_TO_TICKS(10));
+ }
+
+ // 「取り消し確認 → 再生開始」を stop() と直列化する。stop() はこのロックの下で
+ // 取り消しフラグを立ててスピーカーを止めるので、確認を通ったチャンクは必ず
+ // stop() より前に鳴り始めていて、stop() がそれも止める。
+ std::lock_guard lock(mtx_);
+ if (is_cancelled()) return std::nullopt;
+
+ bufs_.push_back(std::move(pcm));
+ const std::vector& buf = bufs_.back();
+ while (!M5.Speaker.playRaw(buf.data(), buf.size(), sample_rate, /*stereo=*/false,
+ /*repeat=*/1, channel_, /*stop_current_sound=*/false)) {
+ vTaskDelay(pdMS_TO_TICKS(10)); // スロットが埋まっていた (稀): 空くまで再試行
+ }
+
+ // 再生予定: 直前のチャンクがまだ鳴っていればその直後、途切れていれば今。
+ const std::uint32_t now = now_ms();
+ const std::uint32_t start = after(end_ms_, now) ? end_ms_ : now;
+ const auto dur_ms = static_cast