diff --git a/.github/workflows/host-tests.yml b/.github/workflows/host-tests.yml index 65e53f1..ff885ae 100644 --- a/.github/workflows/host-tests.yml +++ b/.github/workflows/host-tests.yml @@ -60,6 +60,15 @@ jobs: cmake --build build-san ctest --test-dir build-san --output-on-failure + # --- avatar: balloon scroll layout (pure functions, header only) ------ + - name: avatar host tests + working-directory: components/avatar/test/host + run: | + set -euo pipefail + cmake -S . -B build -G Ninja -DCMAKE_BUILD_TYPE=Debug + cmake --build build + ctest --test-dir build --output-on-failure + # --- flash_layout: ADR-001 exttab / bootctl format -------------------- # format.c is the exact source the custom bootloader compiles, so the # validation / A-B selection / boot-decision logic is proven here. @@ -119,3 +128,14 @@ jobs: cmake --build build ./build/jtts_test_vowels ./build/jtts_test_sano_ir + # sanoTTS clause-by-clause synthesis (clause split / edge-silence trim / + # inter-clause pause). The 1-clause synth is stubbed, so no weights needed. + ./build/jtts_test_clause_stream + # Vowel lip-sync events, long-text chunking, streaming synthesis and + # subtitle mapping. Without arguments only the formant / pure-logic + # cases run; with .htsvoice files the HMM cases run too (16 kHz voice + # = native rate, fast). + ./build/jtts_test_visemes + ./build/jtts_test_hmm_chunk + ./build/jtts_test_visemes ../../../../assets/voices/mei16.htsvoice + ./build/jtts_test_hmm_chunk ../../../../assets/voices/mei16.htsvoice diff --git a/assets/default_face.avdsl b/assets/default_face.avdsl index 670f5b0..bf28695 100644 --- a/assets/default_face.avdsl +++ b/assets/default_face.avdsl @@ -10,13 +10,17 @@ -- A `fn draw()` is mandatory (entry point, fn id 0). Helper functions follow. ---------------------------------------------------------------------- --- Mouth: single rectangle whose width shrinks and height grows with --- mouth_open. The group box covers the maximum extent so the direct --- canvas strategy clears stale frames correctly. +-- Mouth: single rectangle whose height grows with mouth_open and whose width +-- shrinks with mouth_form (0 = wide, 1 = narrow). Hosts that only drive +-- mouth_open (level meters) get mouth_form == mouth_open, i.e. the classic +-- "narrower as it opens" look; TTS vowel lip-sync sets the two independently +-- (い = wide + slightly open, う = narrow, あ = wide-ish + fully open). +-- The group box covers the maximum extent so the direct canvas strategy +-- clears stale frames correctly. ---------------------------------------------------------------------- fn mouth(cx, cy, min_w, max_w, min_h, max_h, bo) let h = min_h + (max_h - min_h) * mouth_open - let w = min_w + (max_w - min_w) * (1 - mouth_open) + let w = min_w + (max_w - min_w) * (1 - mouth_form) let x = cx - w / 2 let y = cy - h / 2 + breath * 2 + bo let gx = cx - max_w / 2 - 1 diff --git a/components/avatar/avatar.cpp b/components/avatar/avatar.cpp index 5e064ae..6dfc08d 100644 --- a/components/avatar/avatar.cpp +++ b/components/avatar/avatar.cpp @@ -134,6 +134,16 @@ void Avatar::set_mouth_open(float ratio) noexcept impl_->context().mouth_open_ratio = ratio; } +void Avatar::set_mouth_form(float ratio) noexcept +{ + if (ratio > 1.0f) { + ratio = 1.0f; + } else if (ratio < 0.0f) { + ratio = -1.0f; + } + impl_->context().mouth_form_ratio = ratio; +} + void Avatar::set_gaze(float horizontal, float vertical) noexcept { impl_->context().gaze_horizontal = horizontal; diff --git a/components/avatar/balloon.cpp b/components/avatar/balloon.cpp index d03e159..cd2537b 100644 --- a/components/avatar/balloon.cpp +++ b/components/avatar/balloon.cpp @@ -3,7 +3,7 @@ #include "balloon.hpp" -#include +#include "balloon_layout.hpp" namespace stackchan::avatar::internal { @@ -21,16 +21,6 @@ constexpr std::int16_t kSmallPanelHeightThreshold = 160; constexpr std::int16_t kBigPanelH = 40; // for 24-px font constexpr std::int16_t kSmallPanelH = 22; // for 12-px font -// Marquee tuning. -constexpr std::int32_t kScrollSpeedPxPerSec = 60; -// Gap (px) of "empty space" between the trailing edge of one pass and the -// leading edge of the next so the user perceives the message restarting. -constexpr std::int32_t kRepeatGapPx = 60; - -// Default minimum display time for short (non-scrolling) text. The application -// can override with `Avatar::set_balloon_text(text, hold_ms)`. -constexpr std::uint32_t kDefaultStaticHoldMs = 3000; - } // namespace void draw_balloon(RichCanvas& canvas, DrawContext& ctx) @@ -71,42 +61,14 @@ void draw_balloon(RichCanvas& canvas, DrawContext& ctx) const std::int32_t mid_y = panel_y + panel_h / 2; const std::uint32_t elapsed_ms = ctx.now_ms - ctx.balloon_set_ms; - if (text_w <= inner_w) { - // Text fits — static centered. Mark done after the configured hold. - canvas.setTextDatum(lgfx::textdatum_t::middle_center); - canvas.drawString(text.c_str(), panel_x + panel_w / 2, mid_y); - - const std::uint32_t hold_ms = - std::max(ctx.balloon_hold_ms, kDefaultStaticHoldMs); - if (elapsed_ms >= hold_ms) { - ctx.balloon_done = true; - } - canvas.end_group(); - return; - } - - // Marquee: text starts just past the right inner edge and scrolls left. - // A single "pass" travels `text_w + inner_w` pixels (entry + traverse + - // exit). One full cycle adds `kRepeatGapPx` so the message restarts with - // a perceivable gap. - const std::int32_t one_pass_px = text_w + inner_w; - const std::int32_t cycle_px = one_pass_px + kRepeatGapPx; - const std::int32_t offset_in_cycle = - static_cast(elapsed_ms) * kScrollSpeedPxPerSec / 1000 % cycle_px; - const std::int32_t x = inner_x + inner_w - offset_in_cycle; - + // Left-aligned; scrolls only when the text overflows the balloon (starts at + // the beginning of the text, then reveals the rest — see balloon_layout.hpp). + const BalloonScroll scroll = compute_balloon_scroll(text_w, inner_w, elapsed_ms, ctx.balloon_hold_ms); canvas.setClipRect(inner_x, panel_y, inner_w, panel_h); canvas.setTextDatum(lgfx::textdatum_t::middle_left); - canvas.drawString(text.c_str(), x, mid_y); + canvas.drawString(text.c_str(), inner_x - scroll.offset_px, mid_y); canvas.clearClipRect(); - - // Mark done once the message has scrolled across at least once - // (or the caller-requested hold time has elapsed, whichever is longer). - const std::uint32_t one_pass_ms = - static_cast(one_pass_px) * 1000u / - static_cast(kScrollSpeedPxPerSec); - const std::uint32_t complete_at = std::max(ctx.balloon_hold_ms, one_pass_ms); - if (elapsed_ms >= complete_at) { + if (scroll.done) { ctx.balloon_done = true; } canvas.end_group(); diff --git a/components/avatar/balloon_layout.hpp b/components/avatar/balloon_layout.hpp new file mode 100644 index 0000000..8fb50a3 --- /dev/null +++ b/components/avatar/balloon_layout.hpp @@ -0,0 +1,65 @@ +// SPDX-FileCopyrightText: 2026 Kenta IDA +// SPDX-License-Identifier: BSL-1.0 +// +// 吹き出しの文字列の表示位置 (スクロール) の計算。描画から切り離した純粋な関数で、 +// ホストでテストできる。 +// +// - 文字列が吹き出しに収まるなら、左詰めで静止表示する (スクロールしない)。 +// - 溢れるときだけスクロールする: 最初は左詰め (文頭が見える) で少し止まり、 +// 溢れた分だけ左へ動かして、文末が見えたところで止める。ループはしない。 +// - 表示時間 (hold_ms) が分かっているときは、その間に文末まで読めるよう速度を +// 合わせる。チャンクごとに切り替わる吹き出しなら、次のチャンクが始まるまでに +// 文末が見える。 +#pragma once + +#include +#include + +namespace stackchan::avatar::internal { + +// 収まる文字列を最低限見せておく時間 (hold_ms 未指定のとき)。 +constexpr std::uint32_t kBalloonStaticHoldMs = 3000; +// 溢れる文字列: スクロール前後の静止時間の上限と、既定 / 上限 / 下限のスクロール速度。 +constexpr std::uint32_t kBalloonEdgeHoldMs = 1000; +constexpr std::int32_t kBalloonScrollPxPerSec = 60; +constexpr std::int32_t kBalloonScrollMaxPxPerSec = 120; +constexpr std::int32_t kBalloonScrollMinPxPerSec = 30; + +struct BalloonScroll { + std::int32_t offset_px = 0; // 左へ動かした量 (0 = 左詰め、travel = 文末が右端に揃う) + bool done = false; // 表示を終えてよい (呼び出し側が吹き出しを閉じる) +}; + +// text_w: 文字列の幅、inner_w: 吹き出しの内側の幅 [px]。 +// elapsed_ms: 表示してからの時間、hold_ms: 呼び出し側の指定表示時間 (0 = 未指定)。 +constexpr BalloonScroll compute_balloon_scroll(std::int32_t text_w, std::int32_t inner_w, std::uint32_t elapsed_ms, + std::uint32_t hold_ms) +{ + if (text_w <= inner_w) { + return {0, elapsed_ms >= std::max(hold_ms, kBalloonStaticHoldMs)}; + } + const auto travel = static_cast(text_w - inner_w); + const std::uint32_t default_ms = travel * 1000u / kBalloonScrollPxPerSec; + const std::uint32_t min_ms = travel * 1000u / kBalloonScrollMaxPxPerSec; + const std::uint32_t max_ms = travel * 1000u / kBalloonScrollMinPxPerSec; + + std::uint32_t edge_ms = kBalloonEdgeHoldMs; // 動き出し前 / 動いた後の静止 + std::uint32_t move_ms = default_ms; + if (hold_ms > 0) { + // 指定時間の中で、前後の静止 (各 1/5、最大 1 秒) を除いた分を動かす。 + edge_ms = std::min(kBalloonEdgeHoldMs, hold_ms / 5); + const std::uint32_t avail = hold_ms > 2 * edge_ms ? hold_ms - 2 * edge_ms : 0; + move_ms = std::clamp(avail, min_ms, max_ms); + } + + BalloonScroll s; + if (elapsed_ms > edge_ms) { + const std::uint32_t t = elapsed_ms - edge_ms; + s.offset_px = t >= move_ms ? static_cast(travel) + : static_cast(static_cast(travel) * t / move_ms); + } + s.done = elapsed_ms >= std::max(hold_ms, edge_ms + move_ms + edge_ms); + return s; +} + +} // namespace stackchan::avatar::internal diff --git a/components/avatar/include/avatar/avatar.hpp b/components/avatar/include/avatar/avatar.hpp index d4722f3..82bbc6c 100644 --- a/components/avatar/include/avatar/avatar.hpp +++ b/components/avatar/include/avatar/avatar.hpp @@ -30,19 +30,24 @@ class Avatar { void set_expression(Expression expression) noexcept; void set_mouth_open(float ratio) noexcept; + // Mouth shape, 0 = wide .. 1 = narrow (see DrawContext::mouth_form_ratio). + // A negative value clears it: the shape then follows set_mouth_open. + void set_mouth_form(float ratio) noexcept; void set_gaze(float horizontal, float vertical) noexcept; void set_palette(const Palette& palette) noexcept; // Rebuild the face layout from user tuning (eye/eyebrow/mouth geometry) and // apply its face/background colours. Takes effect on the next tick(); safe // to call live (e.g. from the render task on a config change). void set_face_tuning(const FaceTuning& tuning) noexcept; - // Show `text` in the balloon. `hold_ms` overrides the default display - // time (0 = use balloon defaults: short text holds for a few seconds, - // long text plays one full marquee pass). + // Show `text` left-aligned in the balloon. Text that does not fit scrolls + // once (start of the text first, then the rest is revealed); text that fits + // never scrolls. `hold_ms` is the on-screen time: 0 = defaults (fitting text + // holds a few seconds, overflowing text scrolls at a fixed speed); when set, + // overflowing text is scrolled so its end is reached within that time. void set_balloon_text(std::string_view text, std::uint32_t hold_ms = 0); void clear_balloon() noexcept; // True once the current balloon has been fully displayed (hold elapsed - // or one marquee pass completed). Stays true until the next + // or the scroll finished). Stays true until the next // set_balloon_text / clear_balloon. bool is_balloon_done() const noexcept; diff --git a/components/avatar/test/host/CMakeLists.txt b/components/avatar/test/host/CMakeLists.txt new file mode 100644 index 0000000..796a3a8 --- /dev/null +++ b/components/avatar/test/host/CMakeLists.txt @@ -0,0 +1,18 @@ +# SPDX-FileCopyrightText: 2026 Kenta IDA +# SPDX-License-Identifier: BSL-1.0 +# +# Host-side (ESP-IDF independent) tests for the balloon scroll layout +# (components/avatar/balloon_layout.hpp — pure functions, header only). + +cmake_minimum_required(VERSION 3.16) +project(avatar_host_test CXX) + +set(CMAKE_CXX_STANDARD 20) +set(CMAKE_CXX_STANDARD_REQUIRED ON) +set(CMAKE_CXX_EXTENSIONS OFF) + +enable_testing() + +add_executable(avatar_test_balloon_layout test_balloon_layout.cpp) +target_compile_options(avatar_test_balloon_layout PRIVATE -Wall -Wextra) +add_test(NAME balloon_layout COMMAND avatar_test_balloon_layout) diff --git a/components/avatar/test/host/test_balloon_layout.cpp b/components/avatar/test/host/test_balloon_layout.cpp new file mode 100644 index 0000000..06971c8 --- /dev/null +++ b/components/avatar/test/host/test_balloon_layout.cpp @@ -0,0 +1,94 @@ +// SPDX-FileCopyrightText: 2026 Kenta IDA +// SPDX-License-Identifier: BSL-1.0 +// +// 吹き出しのスクロール計算 (balloon_layout.hpp) の検証。 +#include + +#include "../../balloon_layout.hpp" + +using namespace stackchan::avatar::internal; + +namespace { + +int g_failures = 0; + +#define CHECK(cond) \ + do { \ + if (!(cond)) { \ + std::fprintf(stderr, "FAIL %s:%d: %s\n", __FILE__, __LINE__, #cond); \ + ++g_failures; \ + } \ + } while (0) + +} // namespace + +int main() +{ + constexpr std::int32_t kInner = 288; // CoreS3 の吹き出し内側 (320 - 2*4 - 2*8) + + // 収まる文字列はスクロールしない (常に 0 = 左詰め)。既定では 3 秒見せて終了、 + // hold_ms があればそれまで。 + for (std::uint32_t t : {0u, 500u, 2999u}) { + const auto s = compute_balloon_scroll(200, kInner, t, 0); + CHECK(s.offset_px == 0 && !s.done); + } + CHECK(compute_balloon_scroll(200, kInner, 3000, 0).done); + CHECK(compute_balloon_scroll(kInner, kInner, 0, 0).offset_px == 0); // ちょうど収まる + CHECK(!compute_balloon_scroll(200, kInner, 3500, 5000).done); + CHECK(compute_balloon_scroll(200, kInner, 5000, 5000).done); + + // 溢れる文字列 (hold なし): 最初は左詰めで 1 秒止まり、60 px/s で溢れた分だけ動き、 + // 文末が右端に揃ったところで 1 秒止まって終了。ループしない。 + { + const std::int32_t text_w = kInner + 300; // 溢れは 300 px → 5 秒で動く + CHECK(compute_balloon_scroll(text_w, kInner, 0, 0).offset_px == 0); // 文頭が見えている + CHECK(compute_balloon_scroll(text_w, kInner, 1000, 0).offset_px == 0); // まだ動かない + CHECK(compute_balloon_scroll(text_w, kInner, 2000, 0).offset_px == 60); // 1 秒動いた = 60 px + CHECK(compute_balloon_scroll(text_w, kInner, 6000, 0).offset_px == 300); // 文末が右端 + CHECK(compute_balloon_scroll(text_w, kInner, 60000, 0).offset_px == 300); // それ以上は動かない + CHECK(!compute_balloon_scroll(text_w, kInner, 6999, 0).done); + CHECK(compute_balloon_scroll(text_w, kInner, 7000, 0).done); // 1 + 5 + 1 秒 + // 単調増加 (途中で戻らない)。 + std::int32_t prev = 0; + for (std::uint32_t t = 0; t <= 8000; t += 50) { + const auto s = compute_balloon_scroll(text_w, kInner, t, 0); + CHECK(s.offset_px >= prev && s.offset_px >= 0 && s.offset_px <= 300); + prev = s.offset_px; + } + } + + // 溢れる文字列 (hold あり): その時間内に文末まで動く。前後の静止は 1/5 ずつ。 + { + const std::int32_t text_w = kInner + 100; // 溢れ 100 px + const std::uint32_t hold = 3000; // 前後 600 ms、動くのは 1800 ms (≈ 55 px/s) + CHECK(compute_balloon_scroll(text_w, kInner, 0, hold).offset_px == 0); + CHECK(compute_balloon_scroll(text_w, kInner, 600, hold).offset_px == 0); + CHECK(compute_balloon_scroll(text_w, kInner, 2400, hold).offset_px == 100); // 動き終わり + CHECK(compute_balloon_scroll(text_w, kInner, hold - 1, hold).offset_px == 100); + CHECK(!compute_balloon_scroll(text_w, kInner, hold - 1, hold).done); + CHECK(compute_balloon_scroll(text_w, kInner, hold, hold).done); + } + // hold が短くて間に合わないときは最大速度 (120 px/s) で動く (それでも hold 内には終わらない)。 + { + const std::int32_t text_w = kInner + 600; // 溢れ 600 px → 最速 120 px/s でも 5 秒 + // hold=2000 → 前後の静止は各 400 ms。動き出して 800 ms 後 (elapsed=1200) は 120*0.8=96 px。 + CHECK(compute_balloon_scroll(text_w, kInner, 1200, 2000).offset_px == 96); + CHECK(compute_balloon_scroll(text_w, kInner, 2000, 2000).offset_px == 192); // 動き出して 1600 ms = 120*1.6。hold を過ぎても動き続ける + CHECK(!compute_balloon_scroll(text_w, kInner, 2000, 2000).done); + CHECK(compute_balloon_scroll(text_w, kInner, 5400, 2000).offset_px == 600); + CHECK(!compute_balloon_scroll(text_w, kInner, 5799, 2000).done); + CHECK(compute_balloon_scroll(text_w, kInner, 5800, 2000).done); // 0.4 + 5 + 0.4 秒 + } + // hold が長いときは最低速度 (30 px/s) まで落とす (それ以上は遅くしない)。 + { + const std::int32_t text_w = kInner + 60; // 溢れ 60 px → 最長 2 秒 + // hold = 30 秒: 前後は各 1 秒 (上限)、動くのは 2 秒 (30 px/s)、あとは静止のまま hold まで。 + CHECK(compute_balloon_scroll(text_w, kInner, 1000, 30000).offset_px == 0); + CHECK(compute_balloon_scroll(text_w, kInner, 3000, 30000).offset_px == 60); + CHECK(!compute_balloon_scroll(text_w, kInner, 29999, 30000).done); + CHECK(compute_balloon_scroll(text_w, kInner, 30000, 30000).done); + } + + if (g_failures == 0) std::puts("test_balloon_layout: all passed"); + return g_failures == 0 ? 0 : 1; +} diff --git a/components/avatar_vm/include/avatar/draw_context.hpp b/components/avatar_vm/include/avatar/draw_context.hpp index cd2b645..a50ef27 100644 --- a/components/avatar_vm/include/avatar/draw_context.hpp +++ b/components/avatar_vm/include/avatar/draw_context.hpp @@ -27,18 +27,24 @@ struct DrawContext { float gaze_saccade_v{0.0f}; float eye_open_ratio{1.0f}; float mouth_open_ratio{0.0f}; + // Mouth shape: 0 = wide, 1 = narrow (pursed). Set by Avatar::set_mouth_form + // (e.g. TTS vowel lip-sync: い is wide, う is narrow). Negative = "not set": + // Var::MouthForm then falls back to mouth_open_ratio so faces driven only + // by a level meter keep the classic "narrower as it opens" behaviour. + float mouth_form_ratio{-1.0f}; Palette palette{kDefaultPalette}; std::uint32_t rng_state{0xC0FFEEu}; std::optional balloon_text{}; - // Wall-clock used for time-based animation (e.g. balloon marquee). + // Wall-clock used for time-based animation (e.g. balloon scroll). std::uint32_t now_ms{0}; - // Set to `now_ms` whenever balloon_text changes — drives marquee phase. + // Set to `now_ms` whenever balloon_text changes — drives the scroll phase. std::uint32_t balloon_set_ms{0}; - // Minimum display time. 0 means "use balloon defaults" (short = a fixed - // hold, long = one marquee pass). + // Display time. 0 means "use balloon defaults" (fitting text = a fixed + // hold, overflowing text = one scroll at a fixed speed); otherwise the + // scroll of overflowing text is timed to finish within it. std::uint32_t balloon_hold_ms{0}; // Set by balloon rendering once the message has been displayed in full - // (i.e. hold time elapsed for short text, or one marquee cycle for long + // (i.e. hold time elapsed for short text, or the scroll finished for long // text). The render task polls this and notifies the application. bool balloon_done{false}; }; diff --git a/components/avatar_vm/include/avatar_vm/opcodes.hpp b/components/avatar_vm/include/avatar_vm/opcodes.hpp index 49062ed..505183d 100644 --- a/components/avatar_vm/include/avatar_vm/opcodes.hpp +++ b/components/avatar_vm/include/avatar_vm/opcodes.hpp @@ -103,6 +103,7 @@ enum class Var : std::uint8_t { CheekRadius = 0x1C, CheekOffX = 0x1D, CheekOffY = 0x1E, + MouthForm = 0x1F, Accessories = 0x20, // FaceTuning::accessories bitmask (0..255) Accessory0 = 0x21, // (accessories >> n) & 1, n = 0..7 Accessory1 = 0x22, diff --git a/components/avatar_vm/test/host/test_vm.cpp b/components/avatar_vm/test/host/test_vm.cpp index 1f64d04..2179fa9 100644 --- a/components/avatar_vm/test/host/test_vm.cpp +++ b/components/avatar_vm/test/host/test_vm.cpp @@ -49,6 +49,7 @@ struct RunResult { }; RunResult run_program(const std::vector& buf, + const stackchan::avatar::DrawContext& ctx = stackchan::avatar::DrawContext{}, const stackchan::avatar::FaceTuning& tuning = stackchan::avatar::FaceTuning{}) { RunResult rr; @@ -57,7 +58,6 @@ RunResult run_program(const std::vector& buf, rr.ran = tl::unexpected(rr.decoded.error()); return rr; } - stackchan::avatar::DrawContext ctx; Vm vm; rr.ran = vm.run(*rr.decoded, rr.canvas, ctx, tuning); return rr; @@ -187,6 +187,37 @@ int main() CHECK(rr.canvas.ops.size() == 1 && rr.canvas.ops[0].a == 320); } + // --- PushVar MouthForm: explicit value wins, unset follows MouthOpen --- + { + BytecodeBuilder b; + b.code(PUSH_VAR); + b.code(0x1F); // Var::MouthForm + b.code(PUSH_I8); + b.code(1); + b.code(PUSH_I8); + b.code(1); + b.code(PUSH_I8); + b.code(3); + b.code(FILL_CIRCLE); + b.code(RET); + b.add_fn(0, 0, 0); + const auto buf = b.build(0); + + stackchan::avatar::DrawContext ctx; + ctx.mouth_open_ratio = 1.0f; + auto rr = run_program(buf, ctx); // form unset (-1) → follows open + CHECK(rr.ran.has_value() && rr.canvas.ops.size() == 1 && rr.canvas.ops[0].a == 1); + + ctx.mouth_form_ratio = 0.0f; // explicitly wide, even though fully open + rr = run_program(buf, ctx); + CHECK(rr.ran.has_value() && rr.canvas.ops.size() == 1 && rr.canvas.ops[0].a == 0); + + ctx.mouth_open_ratio = 0.0f; + ctx.mouth_form_ratio = 1.0f; // explicitly narrow, even though closed + rr = run_program(buf, ctx); + CHECK(rr.ran.has_value() && rr.canvas.ops.size() == 1 && rr.canvas.ops[0].a == 1); + } + // --- PushVar Accessories / Accessory_n read FaceTuning::accessories --- { // fillCircle(accessories, accessory_0, accessory_3, 1) @@ -204,7 +235,7 @@ int main() b.add_fn(0, 0, 0); stackchan::avatar::FaceTuning tuning; tuning.accessories = 0x09; // slots 0 and 3 - auto rr = run_program(b.build(0), tuning); + auto rr = run_program(b.build(0), stackchan::avatar::DrawContext{}, tuning); CHECK(rr.ran.has_value()); CHECK(rr.canvas.ops.size() == 1 && rr.canvas.ops[0].a == 9 && rr.canvas.ops[0].b == 1 && rr.canvas.ops[0].c == 1); diff --git a/components/avatar_vm/vm.cpp b/components/avatar_vm/vm.cpp index 1867d44..2069afd 100644 --- a/components/avatar_vm/vm.cpp +++ b/components/avatar_vm/vm.cpp @@ -74,6 +74,7 @@ inline float read_var(Var v, const avatar::Canvas& canvas, const avatar::DrawCon case Var::GazeH: return ctx.gaze_horizontal + ctx.gaze_saccade_h; case Var::GazeV: return ctx.gaze_vertical + ctx.gaze_saccade_v; case Var::MouthOpen: return ctx.mouth_open_ratio; + case Var::MouthForm: return ctx.mouth_form_ratio >= 0.0f ? ctx.mouth_form_ratio : ctx.mouth_open_ratio; case Var::Expr: return expression_to_f(ctx.expression); case Var::Primary: return static_cast(ctx.palette.primary); case Var::Background: return static_cast(ctx.palette.background); diff --git a/components/hts_engine/lib/HTS_pstream.c b/components/hts_engine/lib/HTS_pstream.c index 3d52557..87bbf84 100644 --- a/components/hts_engine/lib/HTS_pstream.c +++ b/components/hts_engine/lib/HTS_pstream.c @@ -331,7 +331,7 @@ HTS_Boolean HTS_PStreamSet_create(HTS_PStreamSet * pss, HTS_SStreamSet * sss, fl /* copy dynamic window */ pst->win_l_width = (int *) HTS_calloc(pst->win_size, sizeof(int)); pst->win_r_width = (int *) HTS_calloc(pst->win_size, sizeof(int)); - pst->win_coefficient = (float **) HTS_calloc(pst->win_size, sizeof(float)); + pst->win_coefficient = (float **) HTS_calloc(pst->win_size, sizeof(float *)); for (j = 0; j < pst->win_size; j++) { pst->win_l_width[j] = HTS_SStreamSet_get_window_left_width(sss, i, j); pst->win_r_width[j] = HTS_SStreamSet_get_window_right_width(sss, i, j); diff --git a/components/hts_engine/lib/HTS_sstream.c b/components/hts_engine/lib/HTS_sstream.c index d1a5f09..9464ff1 100644 --- a/components/hts_engine/lib/HTS_sstream.c +++ b/components/hts_engine/lib/HTS_sstream.c @@ -305,7 +305,7 @@ HTS_Boolean HTS_SStreamSet_create(HTS_SStreamSet * sss, HTS_ModelSet * ms, HTS_L sst->win_max_width = HTS_ModelSet_get_window_max_width(ms, i); sst->win_l_width = (int *) HTS_calloc(sst->win_size, sizeof(int)); sst->win_r_width = (int *) HTS_calloc(sst->win_size, sizeof(int)); - sst->win_coefficient = (float **) HTS_calloc(sst->win_size, sizeof(float)); + sst->win_coefficient = (float **) HTS_calloc(sst->win_size, sizeof(float *)); for (j = 0; j < sst->win_size; j++) { sst->win_l_width[j] = HTS_ModelSet_get_window_left_width(ms, i, j); sst->win_r_width[j] = HTS_ModelSet_get_window_right_width(ms, i, j); diff --git a/components/jtts/CMakeLists.txt b/components/jtts/CMakeLists.txt index 55f8523..2584f3c 100644 --- a/components/jtts/CMakeLists.txt +++ b/components/jtts/CMakeLists.txt @@ -13,7 +13,11 @@ idf_component_register( "src/hts_label.cpp" "src/hmm_synth.cpp" "src/sano_ir.cpp" + "src/clause_stream.cpp" + "src/sano_visemes.cpp" "src/sano_synth.cpp" + "src/hmm_chunk.cpp" + "src/subtitle.cpp" INCLUDE_DIRS "include" PRIV_INCLUDE_DIRS "src" REQUIRES tl_expected diff --git a/components/jtts/include/jtts/jtts.hpp b/components/jtts/include/jtts/jtts.hpp index 204bfe4..1642402 100644 --- a/components/jtts/include/jtts/jtts.hpp +++ b/components/jtts/include/jtts/jtts.hpp @@ -3,12 +3,16 @@ #pragma once #include +#include #include +#include #include #include #include +#include "jtts/phoneme.hpp" + namespace stackchan::jtts { enum class Voice : std::uint8_t { @@ -90,6 +94,7 @@ struct Options { enum class Error { InvalidKana, OutOfMemory, + Cancelled, // synthesize_stream のシンクが false を返して中断した }; const char* to_string(Error e); @@ -107,6 +112,57 @@ tl::expected synthesize_ex(std::u32string_view kana, std::vector& out, const Options& opt = {}); +// リップシンク用の口形イベント。`start_ms` (発話先頭からの経過時間) から次の +// イベントまで、口は `vowel` の形を取る。Vowel::None は閉口。 +struct VisemeEvent { + std::uint32_t start_ms = 0; + Vowel vowel = Vowel::None; +}; + +// synthesize に加えて、PCM と時間軸が揃った口形イベント列を `visemes` に返す +// (時刻昇順、隣り合うイベントの vowel は異なる。最後は閉口で終わる)。 +// 口形を出せるのはフォルマント / HMM / sanoTTS (音素ごとの継続長から作る)。単位連結 +// エンジンで合成された場合と失敗時は空になるので、呼び出し側は音量エンベロープなどに +// フォールバックすること。 +tl::expected synthesize(std::u32string_view kana, + std::vector& out, + std::vector& visemes, + const Options& opt = {}); + +// ストリーミング合成の 1 チャンク分。 +struct SynthChunk { + // このチャンクが読む部分 (synthesize_stream に渡した読みの一部。HMM で強制分割した + // ときはアクセント記号が落ちる)。発話全体が 1 チャンクなら読み全体。 + // 吹き出しをチャンクに同期させるとき (jtts/subtitle.hpp) に使う。 + std::u32string text; + std::vector pcm; + // pcm のサンプルレート [Hz]。HMM / フォルマント / 単位連結は opt.sample_rate_hz、 + // sanoTTS は 22.05 kHz 固定 (synthesize_ex と同じ)。再生側がこのレートで鳴らすこと。 + std::uint32_t sample_rate = 0; + // このチャンク先頭からの口形イベント (synthesize と同じ規約)。空ならこの + // エンジンは口形を出せない (音量エンベロープなどにフォールバックすること)。 + std::vector visemes; +}; + +// チャンクの受け取り側。false を返すと合成を中断する (Error::Cancelled)。 +using ChunkSink = std::function; + +// synthesize と同じ合成を、チャンクごとに sink へ渡しながら行う。長い発話は +// HMM エンジンが句読点などで分割して順に合成するので、最初のチャンクが出来た +// 時点で再生を始め、再生中に次を合成できる (全体の合成完了を待たなくてよい)。 +// - チャンクの PCM は連続再生すればそのまま 1 本の発話になる (境界の無音は調整済み)。 +// - 最初のチャンクを小さく、以降を徐々に大きくして、再生が途切れにくくする。 +// - 単位連結 / フォルマントも sanoTTS と同じく句読点 (、。) ごとに 1 句ずつ合成して渡す。 +// 句読点が無い短文は 1 チャンク。 +// 句と句の間には HMM の pau と同じ長さの無音を明示的に挟む (句の後ろに足すので、 +// チャンクは「句 + 間」。話速に比例)。sanoTTS は synthesize_ex と同様に 22.05 kHz で +// 出力する (SynthChunk::sample_rate)。 +// - HMM で 1 つ以上渡した後にメモリ不足になったら Error::OutOfMemory (途中まで +// 渡した分は取り消せない)。何も渡す前なら他エンジンへフォールバックする。 +// sink はこの関数を呼んだスレッドで、合成の合間に呼ばれる。 +tl::expected synthesize_stream(std::u32string_view kana, const ChunkSink& sink, + const Options& opt = {}); + // 単位連結エンジン用の音声 DB (.jvox、codec=0 の生形式) を登録する。 // blob の寿命は呼び出し側が保証する (PSRAM バッファ / flash mmap)。 // パースに失敗すると false を返し、DB 未ロード状態のまま。空 span で解除。 diff --git a/components/jtts/include/jtts/subtitle.hpp b/components/jtts/include/jtts/subtitle.hpp new file mode 100644 index 0000000..890658b --- /dev/null +++ b/components/jtts/include/jtts/subtitle.hpp @@ -0,0 +1,45 @@ +// SPDX-FileCopyrightText: 2026 Kenta IDA +// SPDX-License-Identifier: BSL-1.0 +#pragma once + +#include +#include +#include +#include + +namespace stackchan::jtts { + +// 吹き出し (表示テキスト) を、合成チャンクの再生に同期させるための対応付け。 +// +// jtts は「読み」(かな) しか持たず、表示テキスト (漢字まじり) との対応は知らない。 +// しかしチャンクは句読点 (、。,,..) を境に分けられるので、表示テキストも同じ句読点で +// 区切れば「n 個目の句読点までの部分」が対応する。読みと表示の句読点の数が合わない +// ときは対応を諦め、最初のチャンクで全文を出す。 +// +// SubtitleMapper m(display_utf8, reading); +// synthesize_stream(reading, [&](SynthChunk&& c) { +// std::string text = m.next(c.text); // このチャンクの間に出す表示テキスト +// ... +// }); +class SubtitleMapper { +public: + SubtitleMapper(std::string_view display_utf8, std::u32string_view reading); + + // 句読点で対応が取れるか。false のときは next() が最初に全文を返し、以降は空を返す。 + bool mapped() const { return mapped_; } + + // 次のチャンク (その読み) の間に出す表示テキスト。チャンクは先頭から順に渡すこと。 + // 句読点で終わるチャンクはその句までを、途中で切れたチャンクは続きの句を受け持つ + // (強制分割で同じ句を 2 チャンクにまたがって出す場合は同じ文字列が返る)。 + // 出すものが無ければ空。 + std::string next(std::u32string_view chunk_reading); + +private: + std::string whole_; + std::vector segments_; // 表示テキストを句読点の直後で区切ったもの (句読点を含む) + bool mapped_ = false; + bool first_ = true; + std::size_t consumed_ = 0; // ここまでのチャンクが消費した句読点の数 +}; + +} // namespace stackchan::jtts diff --git a/components/jtts/src/clause_stream.cpp b/components/jtts/src/clause_stream.cpp new file mode 100644 index 0000000..98bde98 --- /dev/null +++ b/components/jtts/src/clause_stream.cpp @@ -0,0 +1,92 @@ +// SPDX-FileCopyrightText: 2026 Kenta IDA +// SPDX-License-Identifier: BSL-1.0 +// +// 句ごとの合成の制御 (sanoTTS / フォルマント / 単位連結で共通、エンジンには依存しない)。 +// 1 句の合成は関数で受け取るので、重み (非 MIT の blob) や音声 DB が無くてもホストで +// テストできる。設計は internal.hpp の「句ごとの合成」を参照。 +#include +#include +#include +#include +#include + +#include "internal.hpp" + +namespace stackchan::jtts::internal { + +SilenceTrim trim_silence(std::vector& pcm, std::uint32_t rate_hz, bool lead, bool trail, + std::uint32_t keep_ms) { + SilenceTrim removed; + if (pcm.empty() || (!lead && !trail)) return removed; + int peak = 0; + for (std::int16_t v : pcm) peak = std::max(peak, std::abs(static_cast(v))); + const int thr = std::max(48, peak / 100); + std::size_t first = 0; + while (first < pcm.size() && std::abs(static_cast(pcm[first])) < thr) ++first; + if (first == pcm.size()) return removed; // 全体が無音: 触らない + std::size_t last = pcm.size(); // 最後の音の次 + while (last > first && std::abs(static_cast(pcm[last - 1])) < thr) --last; + + const std::size_t keep = static_cast(keep_ms) * rate_hz / 1000u; + const std::size_t begin = lead ? (first > keep ? first - keep : 0) : 0; + const std::size_t end = trail ? std::min(pcm.size(), last + keep) : pcm.size(); + removed.back = pcm.size() - end; + removed.front = begin; + pcm.erase(pcm.begin() + static_cast(end), pcm.end()); + pcm.erase(pcm.begin(), pcm.begin() + static_cast(begin)); + return removed; +} + +void crop_spans(std::vector& spans, float drop_front_ms, float keep_ms) { + std::vector out; + float t = 0.0f; // 元の時間軸での現在位置 + const float keep_end = drop_front_ms + keep_ms; + for (const VisemeSpan& s : spans) { + const float b = std::max(t, drop_front_ms); + const float e = std::min(t + s.duration_ms, keep_end); + if (e > b) out.push_back({s.vowel, e - b}); + t += s.duration_ms; + } + spans = std::move(out); +} + +StreamOutcome stream_clauses(std::u32string_view text, const Options& opt, const ClauseSynth& synth_one, + const ClauseChunkFn& emit, bool trim_edges) { + std::vector clauses; + if (!split_clauses(text, clauses)) return StreamOutcome::NoOutput; + + const float pause_ms = clause_pause_ms(opt.mora_ms); + + std::size_t emitted = 0; + for (std::size_t k = 0; k < clauses.size(); ++k) { + std::vector pcm; + std::vector spans; + std::uint32_t rate = 0; + const ClauseResult r = synth_one(clauses[k].text, pcm, rate, spans); + if (r == ClauseResult::Skip) continue; + if (r == ClauseResult::Fail || pcm.empty() || rate == 0) { + return emitted > 0 ? StreamOutcome::Aborted : StreamOutcome::NoOutput; + } + const bool has_prev = emitted > 0; + const bool has_next = k + 1 < clauses.size(); + // 内側の境界: モデルが付けた前後の無音は切り詰めて、こちらの無音に置き換える + // (trim_edges のとき)。発話の先頭 / 末尾は切り詰めない。口形の区間も同じだけ切り詰める。 + const SilenceTrim cut = trim_edges ? trim_silence(pcm, rate, /*lead=*/has_prev, /*trail=*/has_next) + : SilenceTrim{}; + const float ms_per_sample = 1000.0f / static_cast(rate); + if (!spans.empty() && (cut.front != 0 || cut.back != 0)) { + crop_spans(spans, static_cast(cut.front) * ms_per_sample, + static_cast(pcm.size()) * ms_per_sample); + } + if (has_next) { + const auto pause_samples = static_cast(pause_ms * static_cast(rate) / 1000.0f); + pcm.resize(pcm.size() + pause_samples, 0); + if (!spans.empty()) spans.push_back({Vowel::None, static_cast(pause_samples) * ms_per_sample}); + } + ++emitted; + if (!emit(std::move(pcm), rate, std::move(spans), clauses[k].text)) return StreamOutcome::Cancelled; + } + return emitted > 0 ? StreamOutcome::Ok : StreamOutcome::NoOutput; +} + +} // namespace stackchan::jtts::internal diff --git a/components/jtts/src/hmm_chunk.cpp b/components/jtts/src/hmm_chunk.cpp new file mode 100644 index 0000000..dd8a122 --- /dev/null +++ b/components/jtts/src/hmm_chunk.cpp @@ -0,0 +1,218 @@ +// SPDX-FileCopyrightText: 2026 Kenta IDA +// SPDX-License-Identifier: BSL-1.0 +// +// HMM エンジン用のテキスト分割。hts_engine は 1 発話の合成中、フレーム数に +// 比例した量 (≈ 2 KB / 5 ms フレーム) の作業メモリを確保し続けるため、長い +// フレーズをそのまま合成すると PSRAM を使い切って確保失敗 → クラッシュする。 +// そこで発話を、メモリ予算に収まるモーラ数のチャンクに分けて順に合成する。 +// +// 分割位置の優先順位: +// 1. 句読点 (、。,,..) … アクセント句 + 呼気段落の境界。元々ポーズが入る所 +// 2. アクセント句境界 (/) … ポーズなし +// 3. 最後の手段: モーラ境界 (1 つの句がそれ単体で予算を超える場合)。 +// 拗音・長音・アクセント核の直前では切らない。 +#include +#include +#include + +#include "internal.hpp" + +#if defined(ESP_PLATFORM) +#include "esp_heap_caps.h" +#endif + +namespace stackchan::jtts::internal { + +namespace { + +bool is_pause_char(char32_t c) { + return c == U'、' || c == U'。' || c == U',' || c == U',' || c == U'.' || c == U'.'; +} + +bool is_accent_mark(char32_t c) { + return c == U'\'' || c == U'’'; +} + +// 直前のモーラと一体になる文字 (この直前では切らない)。 +bool is_attached_to_prev(char32_t c) { + switch (c) { + case U'ゃ': case U'ゅ': case U'ょ': case U'ぁ': case U'ぃ': case U'ぅ': case U'ぇ': case U'ぉ': + case U'ゎ': case U'ャ': case U'ュ': case U'ョ': case U'ァ': case U'ィ': case U'ゥ': case U'ェ': + case U'ォ': case U'ヮ': case U'ー': case U'/': + return true; + default: + return is_accent_mark(c); + } +} + +std::size_t count_moras(std::u32string_view s) { + std::vector moras; + return parse_kana(s, moras) ? moras.size() : 0; +} + +struct Unit { + std::u32string text; + std::size_t moras = 0; + bool pause = false; // 末尾が句読点 (= 直後にポーズが入る) +}; + +// 1 つの句が max_moras を超えるとき、モーラ境界で切って units に積む。 +// 切った断片ではアクセント核 (') の位置が意味を失うので取り除く (平板になる)。 +void push_split_unit(const Unit& u, std::size_t max_moras, std::vector& units) { + std::u32string rest; + for (char32_t c : u.text) { + if (!is_accent_mark(c)) rest.push_back(c); + } + while (count_moras(rest) > max_moras) { + // max_moras 以内に収まる最長の接頭辞のうち、切ってよい位置を探す。 + std::size_t cut = 0; + std::size_t fallback = 0; + for (std::size_t i = 1; i < rest.size(); ++i) { + if (count_moras(std::u32string_view(rest).substr(0, i)) > max_moras) break; + fallback = i; + if (!is_attached_to_prev(rest[i])) cut = i; + } + if (cut == 0) cut = fallback; + if (cut == 0) break; // 1 モーラで既に予算超過: これ以上は割れない + Unit piece; + piece.text = rest.substr(0, cut); + piece.moras = count_moras(piece.text); + units.push_back(std::move(piece)); + rest.erase(0, cut); + } + Unit last; + last.text = std::move(rest); + last.moras = count_moras(last.text); + last.pause = u.pause; + units.push_back(std::move(last)); +} + +// テスト用: 0 以外なら「PCM に使える PSRAM」をこの値 [byte] とみなす。 +std::size_t g_pcm_free_override = 0; + +} // namespace + +float estimate_utterance_ms(std::u32string_view text, float mora_ms) { + // 実測: 先頭 + 末尾の sil ≈ 0.65 s、句読点の pau ≈ 0.46 s、1 モーラ ≈ 1.25 × mora_ms + // (HMM: 等速時 0.138〜0.150 s、フォルマントは mora_ms + 子音分)。上限寄りに見積もる。 + std::size_t pauses = 0; + for (char32_t c : text) { + if (is_pause_char(c)) ++pauses; + } + return 700.0f + 500.0f * static_cast(pauses) + + 1.3f * mora_ms * static_cast(count_moras(text)); +} + +bool pcm_fits_in_memory(std::size_t samples) { + const std::size_t bytes = samples * sizeof(std::int16_t); + if (g_pcm_free_override != 0) return bytes * 2 + 256 * 1024 <= g_pcm_free_override; +#if defined(ESP_PLATFORM) + // PCM は 1 つの連続ブロック (std::vector) で PSRAM に置く。合成中の他の確保 + // (hts_engine の作業メモリ、再生バッファ) と一時的な倍増に備えて 2 倍 + 256 KB を要求する。 + const std::size_t free_psram = heap_caps_get_free_size(MALLOC_CAP_SPIRAM); + const std::size_t largest = heap_caps_get_largest_free_block(MALLOC_CAP_SPIRAM); + return bytes <= largest && bytes * 2 + 256 * 1024 <= free_psram; +#else + return true; +#endif +} + +void set_pcm_memory_limit_for_test(std::size_t bytes) { g_pcm_free_override = bytes; } + +bool split_hmm_text(std::u32string_view text, std::size_t max_moras, std::vector& out, + std::size_t first_moras) { + out.clear(); + if (max_moras == 0) return false; + const bool ramp = first_moras > 0; + + // 全体が収まるなら手を加えず 1 チャンク (従来と完全に同じ入力で合成する)。 + // 低遅延モードでは句読点で分けて先頭を早く出したいので、この近道は使わない + // (句の途中では切らないので、句読点の無い短文は結局 1 チャンクになる)。 + const std::size_t total = count_moras(text); + if (total == 0) return false; + if (!ramp && total <= max_moras) { + out.push_back({std::u32string(text), false, total}); + return true; + } + + // 句読点 / アクセント句境界で「句」に分ける。区切り文字は前の句に含める。 + std::vector units; + Unit cur; + auto flush_unit = [&](bool pause) { + if (cur.text.empty()) return; + cur.moras = count_moras(cur.text); + cur.pause = pause; + if (cur.moras > max_moras) { + push_split_unit(cur, max_moras, units); + } else { + units.push_back(std::move(cur)); + } + cur = Unit{}; + }; + for (char32_t c : text) { + cur.text.push_back(c); + if (is_pause_char(c)) { + flush_unit(true); + } else if (c == U'/') { + flush_unit(false); + } + } + flush_unit(false); + + // 予算に収まるまで句を貪欲に詰める。低遅延モードでは上限を first_moras から + // 始め、チャンクを閉じるたびに直前のチャンクの 1.3 倍へ引き上げる。 + std::size_t cap = ramp ? std::min(first_moras, max_moras) : max_moras; + HmmChunk chunk; + auto flush_chunk = [&] { + const std::size_t done = chunk.moras; + if (chunk.moras > 0) out.push_back(std::move(chunk)); + chunk = HmmChunk{}; + if (ramp && done > 0) { + const std::size_t grown = (done * 13 + 9) / 10; // ceil(done * 1.3) + cap = std::min(max_moras, std::max(first_moras, grown)); + } + }; + for (auto& u : units) { + if (chunk.moras > 0 && chunk.moras + u.moras > cap) flush_chunk(); + chunk.text += u.text; + chunk.moras += u.moras; + chunk.pause_after = u.pause; + } + flush_chunk(); + return !out.empty(); +} + +bool split_clauses(std::u32string_view text, std::vector& out) { + out.clear(); + std::u32string cur; + auto flush = [&](bool pause) { + if (cur.empty()) return; + const std::size_t moras = count_moras(cur); + if (moras == 0) { + // 読める内容が無い断片 (記号だけなど) は前の句に含める。先頭なら次の句の前に残す。 + if (!out.empty()) { + out.back().text += cur; + out.back().pause_after = out.back().pause_after || pause; + cur.clear(); + } + return; + } + HmmChunk c; + c.text = std::move(cur); + c.moras = moras; + c.pause_after = pause; + out.push_back(std::move(c)); + cur.clear(); + }; + for (std::size_t i = 0; i < text.size(); ++i) { + cur.push_back(text[i]); + if (!is_pause_char(text[i])) continue; + // 句読点が続く場合 (、、 や 。。。) は同じ句にまとめる。 + while (i + 1 < text.size() && is_pause_char(text[i + 1])) cur.push_back(text[++i]); + flush(true); + } + flush(false); + return !out.empty(); +} + +} // namespace stackchan::jtts::internal diff --git a/components/jtts/src/hmm_synth.cpp b/components/jtts/src/hmm_synth.cpp index c25eb90..4578b3f 100644 --- a/components/jtts/src/hmm_synth.cpp +++ b/components/jtts/src/hmm_synth.cpp @@ -11,6 +11,7 @@ #include "sdkconfig.h" #endif +#include #include #include #include @@ -31,10 +32,12 @@ #include #include #include +#include #include "HTS_engine.h" #if defined(ESP_PLATFORM) +#include "esp_heap_caps.h" #include "esp_log.h" #endif @@ -99,26 +102,125 @@ const std::array& decim_coeffs() { return coeffs; } -} // namespace - -bool render_hmm(std::u32string_view text, std::vector& out, const Options& opt) { - std::lock_guard lock(g_engine_mutex); - if (!g_loaded) return false; +// full-context ラベル "p1^p2-p3+p4=p5/A:..." から現在音素 p3 を取り出す。 +std::string_view phoneme_of_label(std::string_view label) { + const std::size_t dash = label.find('-'); + if (dash == std::string_view::npos) return {}; + const std::size_t plus = label.find('+', dash + 1); + if (plus == std::string_view::npos) return {}; + return label.substr(dash + 1, plus - dash - 1); +} - // ボイスはネイティブ レート (48 kHz) のまま合成し、出力レート (16 kHz) へは - // FIR 1/3 デシメーションで落とす。ボコーダを 16 kHz で直接回す (α 再設定) - // 近似も試したが、メルケプの周波数軸はどの α でも 48 kHz 分析軸と一致せず - // フォルマントが下方に歪む (声が暗く低く聞こえる) ため不採用。 - const std::size_t voice_rate = g_engine.ms.sampling_frequency; - std::size_t decim; - if (opt.sample_rate_hz == voice_rate) { - decim = 1; - } else if (voice_rate == 3 * opt.sample_rate_hz) { - decim = 3; - } else { - return false; // 対応外レート → フォールバック +// 音素名 → 口形。フォルマント エンジン (build_segments) と同じ規則: +// 母音 a i u e o … その母音の形 +// 無声化母音 A I U E O・撥音 N・促音 cl・無音 sil / pau … 閉口 +// 両唇音 m b p (拗音含む) … 閉口 (唇を閉じる) +// その他の子音 … 後続母音の形を先取り (後続が無声化母音なら閉口) +Vowel viseme_of_phoneme(std::string_view ph, std::string_view next) { + if (ph.size() == 1) { + switch (ph[0]) { + case 'a': return Vowel::A; + case 'i': return Vowel::I; + case 'u': return Vowel::U; + case 'e': return Vowel::E; + case 'o': return Vowel::O; + case 'A': case 'I': case 'U': case 'E': case 'O': case 'N': + return Vowel::None; + default: break; + } + if (ph[0] == 'm' || ph[0] == 'b' || ph[0] == 'p') return Vowel::None; + } else if (ph == "sil" || ph == "pau" || ph == "cl" || ph == "my" || ph == "by" || ph == "py") { + return Vowel::None; } + // 子音: 後続が有声母音ならその形。 + if (next.size() == 1) { + switch (next[0]) { + case 'a': return Vowel::A; + case 'i': return Vowel::I; + case 'u': return Vowel::U; + case 'e': return Vowel::E; + case 'o': return Vowel::O; + default: break; + } + } + return Vowel::None; +} +// ---- メモリ予算 / 見積り ------------------------------------------------- +// +// hts_engine は 1 回の合成中、フレーム数に比例した作業メモリを確保し続け +// (mean / ivar / wuw / par 行列で ≈ 2 KB / フレーム + 波形 4 B / サンプル)、 +// HTS_Engine_refresh まで解放しない。確保に失敗すると hts_engine は復帰できず +// クラッシュするため、合成前に「収まる長さ」を見積もり、超えるなら分割する。 +// 実測 (mei16、5 ms フレーム): ≈ 470 KB / 発話秒、≈ 75 KB / 文字。 + +// テスト用: 0 以外ならこの値 [byte] を予算として使う。 +std::size_t g_budget_override = 0; + +constexpr float kBytesPerFrame = 2300.0f; // 実測 ≈ 2.05 KB + ヒープ ヘッダ / 余裕 +constexpr float kEdgeSilSec = 0.85f; // 先頭 + 末尾の sil (各 ≈ 0.32 s) + 1 合成ごとの固定分 (≈ 100〜140 KB) +constexpr float kSecPerMora = 0.15f; // 等速時の 1 モーラあたり秒 (ポーズ込み、実測 0.138〜0.150) + +// 今 1 回の合成に使ってよいバイト数。他タスクの分 (128 KB) を残し、残りの 3/4 を +// 上限とする (見積りの誤差と断片化の余裕)。 +std::size_t memory_budget() { + if (g_budget_override != 0) return g_budget_override; +#if defined(ESP_PLATFORM) + const std::size_t free_psram = heap_caps_get_free_size(MALLOC_CAP_SPIRAM); + const std::size_t reserve = 128 * 1024; + return free_psram > reserve ? (free_psram - reserve) / 4 * 3 : 0; +#else + return SIZE_MAX; // ホストでは無制限 +#endif +} + +// 直前までの分割計画が既に安全率を見込んでいるので、2 チャンク目以降の再確認は +// 余裕 (3/4) を掛けずに「実際に足りるか」だけを見る。 +std::size_t memory_free_hard() { + if (g_budget_override != 0) return g_budget_override; +#if defined(ESP_PLATFORM) + const std::size_t free_psram = heap_caps_get_free_size(MALLOC_CAP_SPIRAM); + return free_psram > 128 * 1024 ? free_psram - 128 * 1024 : 0; +#else + return SIZE_MAX; +#endif +} + +// 予算に収まる 1 チャンクの最大モーラ数 (0 = 1 モーラも収まらない)。 +std::size_t max_moras_for_budget(std::size_t budget, std::size_t voice_rate, std::size_t fperiod, float speed) { + if (budget == SIZE_MAX) return SIZE_MAX; + const float frames_per_sec = fperiod > 0 ? static_cast(voice_rate) / static_cast(fperiod) : 200.0f; + const float bytes_per_sec = frames_per_sec * kBytesPerFrame + static_cast(voice_rate) * sizeof(float); + const float sec = static_cast(budget) / bytes_per_sec - kEdgeSilSec; + if (sec <= 0.0f) return 0; + return static_cast(sec * speed / kSecPerMora); +} + +// ---- チャンク境界の無音 --------------------------------------------------- +// +// チャンクごとの合成は前後に sil (≈ 0.32 s) を持つので、そのままつなぐと +// 間延びする。境界では両側の sil を切り詰め、句読点なら元の pau (≈ 0.46 s) と +// 同じ長さ (clause_pause_ms、sanoTTS / フォルマントの句間の無音と共通)、それ以外 +// (アクセント句境界 / 強制分割) ならごく短い無音だけ残す。 +constexpr std::size_t kSoftKeepFrames = 2; // その他の境界で片側に残す無音 (≈ 10 ms) + +// half_pause_ms: 句読点の境界で片側に残す無音 [ms] (= clause_pause_ms / 2)。 +std::size_t boundary_keep_frames(bool pause, float ms_per_frame, float half_pause_ms) { + if (!pause) return kSoftKeepFrames; + const auto frames = static_cast(half_pause_ms / ms_per_frame + 0.5f); + return frames > kSoftKeepFrames ? frames : kSoftKeepFrames; +} + +// 低遅延モードの最初のチャンクの目安モーラ数 (≈ 2 秒の音声 = 合成 ≈ 1.5 秒)。 +constexpr std::size_t kStreamFirstMoras = 14; + +// 1 チャンクを合成して pcm / spans に出力する。 +// lead_cap / trail_cap: 残す先頭 / 末尾 sil の最大フレーム数 (SIZE_MAX = 全部残す)。 +// need_frames: 継続長が取れないときは失敗にする (複数チャンクでは切り詰めに必須)。 +// 失敗時は false (pcm / spans は不定)。 +bool synth_chunk(const std::u32string& text, const Options& opt, std::size_t decim, std::size_t voice_rate, + std::size_t fperiod, float speed, bool need_frames, std::size_t lead_cap, std::size_t trail_cap, + std::vector& pcm, std::vector& spans) { std::vector labels; if (!build_hts_labels(text, labels)) return false; @@ -126,10 +228,6 @@ bool render_hmm(std::u32string_view text, std::vector& out, const lines.reserve(labels.size()); for (auto& l : labels) lines.push_back(l.data()); - // mora_ms は「1 モーラの長さ」なので speed は逆比。既定 110 ms = 等速。 - float speed = 110.0f / opt.mora_ms; - if (speed < 0.5f) speed = 0.5f; - if (speed > 2.0f) speed = 2.0f; HTS_Engine_set_speed(&g_engine, speed); HTS_Engine_add_half_tone(&g_engine, opt.hmm_half_tone); @@ -141,25 +239,61 @@ bool render_hmm(std::u32string_view text, std::vector& out, const const auto t1 = std::chrono::steady_clock::now(); const std::size_t nsamples = HTS_Engine_get_nsamples(&g_engine); + + // ラベル (音素) ごとの継続長 [フレーム]。ラベル 1 行 = 1 音素 = nstate 状態で、 + // 総和 × フレーム周期 = 波形長。噛み合わなければ口形も切り詰めも諦める。 + const std::size_t nstate = HTS_Engine_get_nstate(&g_engine); + std::vector frames; + if (nstate > 0 && fperiod > 0 && labels.size() >= 2 && + HTS_Engine_get_total_state(&g_engine) == labels.size() * nstate && + HTS_Engine_get_total_frame(&g_engine) * fperiod == nsamples) { + frames.assign(labels.size(), 0); + for (std::size_t i = 0; i < labels.size(); ++i) { + for (std::size_t s = 0; s < nstate; ++s) { + frames[i] += HTS_Engine_get_state_duration(&g_engine, i * nstate + s); + } + } + } else if (need_frames) { + HTS_Engine_refresh(&g_engine); + return false; + } + + // 先頭 / 末尾の sil を切り詰める (波形は [begin, end) だけ出力する)。 + std::size_t begin = 0; + std::size_t end = nsamples; + if (!frames.empty()) { + const std::size_t lead = frames.front(); + const std::size_t trail = frames.back(); + const std::size_t keep_lead = lead < lead_cap ? lead : lead_cap; + const std::size_t keep_trail = trail < trail_cap ? trail : trail_cap; + begin = (lead - keep_lead) * fperiod; + end = nsamples - (trail - keep_trail) * fperiod; + frames.front() = keep_lead; + frames.back() = keep_trail; + } + + pcm.clear(); const float* speech = g_engine.gss.gspeech; // per-sample getter は高いので直接参照 const float gain = opt.gain; if (decim == 1) { - out.reserve(out.size() + nsamples); - for (std::size_t i = 0; i < nsamples; ++i) { + pcm.reserve(end - begin); + for (std::size_t i = begin; i < end; ++i) { float v = speech[i] * gain; if (v > 32767.0f) v = 32767.0f; if (v < -32768.0f) v = -32768.0f; - out.push_back(static_cast(v)); + pcm.push_back(static_cast(v)); } } else { // 1/3 ポリフェーズ デシメーション (45-tap Hamming sinc、fc=7.2 kHz)。 // 出力サンプルあたり実質 15 MAC なので合成コストに対して無視できる。 + // フィルタは切り詰め前の全波形を参照するので、境界にもエッジ アーチファクトは出ない。 const auto& h = decim_coeffs(); constexpr int mid = static_cast(kDecimTaps) / 2; - const std::size_t nout = nsamples / 3; - out.reserve(out.size() + nout); - for (std::size_t n = 0; n < nout; ++n) { - const long center = static_cast(n) * 3; + const std::size_t n_begin = begin / decim; + const std::size_t n_end = end / decim; + pcm.reserve(n_end - n_begin); + for (std::size_t n = n_begin; n < n_end; ++n) { + const long center = static_cast(n) * static_cast(decim); long lo = center - mid; long hi = center + mid; // inclusive int skip = 0; @@ -175,11 +309,23 @@ bool render_hmm(std::u32string_view text, std::vector& out, const acc *= gain; if (acc > 32767.0f) acc = 32767.0f; if (acc < -32768.0f) acc = -32768.0f; - out.push_back(static_cast(acc)); + pcm.push_back(static_cast(acc)); } } const auto t2 = std::chrono::steady_clock::now(); + spans.clear(); + if (!frames.empty()) { + const float ms_per_frame = 1000.0f * static_cast(fperiod) / static_cast(voice_rate); + spans.reserve(labels.size()); + for (std::size_t i = 0; i < labels.size(); ++i) { + const std::string_view next = + i + 1 < labels.size() ? phoneme_of_label(labels[i + 1]) : std::string_view{}; + spans.push_back({viseme_of_phoneme(phoneme_of_label(labels[i]), next), + static_cast(frames[i]) * ms_per_frame}); + } + } + #if defined(ESP_PLATFORM) ESP_LOGI("jtts-hmm", "synth %u ms + copy %u ms for %u samples @%u Hz", static_cast( @@ -187,12 +333,101 @@ bool render_hmm(std::u32string_view text, std::vector& out, const static_cast( std::chrono::duration_cast(t2 - t1).count()), static_cast(nsamples), static_cast(voice_rate)); +#else + (void)t0; + (void)t1; + (void)t2; #endif HTS_Engine_refresh(&g_engine); return true; } +} // namespace + +void set_hmm_memory_budget_for_test(std::size_t bytes) { g_budget_override = bytes; } + +StreamOutcome render_hmm_stream(std::u32string_view text, const Options& opt, const ChunkFn& emit, bool stream) { + std::lock_guard lock(g_engine_mutex); + if (!g_loaded) return StreamOutcome::NoOutput; + + // ボイスはネイティブ レート (48 kHz) のまま合成し、出力レート (16 kHz) へは + // FIR 1/3 デシメーションで落とす。ボコーダを 16 kHz で直接回す (α 再設定) + // 近似も試したが、メルケプの周波数軸はどの α でも 48 kHz 分析軸と一致せず + // フォルマントが下方に歪む (声が暗く低く聞こえる) ため不採用。 + const std::size_t voice_rate = g_engine.ms.sampling_frequency; + std::size_t decim; + if (opt.sample_rate_hz == voice_rate) { + decim = 1; + } else if (voice_rate == 3 * opt.sample_rate_hz) { + decim = 3; + } else { + return StreamOutcome::NoOutput; // 対応外レート → フォールバック + } + const std::size_t fperiod = HTS_Engine_get_fperiod(&g_engine); + + // mora_ms は「1 モーラの長さ」なので speed は逆比。既定 110 ms = 等速。 + float speed = 110.0f / opt.mora_ms; + if (speed < 0.5f) speed = 0.5f; + if (speed > 2.0f) speed = 2.0f; + + // 長い発話はメモリ予算に収まるチャンクに分けて順に合成する。どうしても + // 収まらない (空きメモリが足りない) ときは諦めて呼び出し側にフォールバック + // (フォルマント合成) させる — 確保失敗は hts_engine 内でクラッシュになる。 + const std::size_t max_moras = max_moras_for_budget(memory_budget(), voice_rate, fperiod, speed); + std::vector chunks; + if (!split_hmm_text(text, max_moras, chunks, stream ? kStreamFirstMoras : 0)) { +#if defined(ESP_PLATFORM) + ESP_LOGW("jtts-hmm", "no memory for HMM synthesis (budget allows %u moras) → fallback", + static_cast(max_moras)); +#endif + return StreamOutcome::NoOutput; + } + const bool multi = chunks.size() > 1; + // 切り詰めは fperiod 単位 (デシメーション後も整数サンプル) で行う。 + if (multi && (fperiod == 0 || fperiod % decim != 0)) return StreamOutcome::NoOutput; +#if defined(ESP_PLATFORM) + if (multi) { + ESP_LOGI("jtts-hmm", "long text: %u chunks (max %u moras each, PSRAM free %u KB)", + static_cast(chunks.size()), static_cast(max_moras), + static_cast(heap_caps_get_free_size(MALLOC_CAP_SPIRAM) / 1024)); + } +#endif + + const float ms_per_frame = + fperiod > 0 ? 1000.0f * static_cast(fperiod) / static_cast(voice_rate) : 5.0f; + const float half_pause_ms = 0.5f * clause_pause_ms(opt.mora_ms); + std::size_t emitted = 0; + const auto failed = [&] { return emitted > 0 ? StreamOutcome::Aborted : StreamOutcome::NoOutput; }; + + std::vector pcm; + std::vector spans; + for (std::size_t k = 0; k < chunks.size(); ++k) { + // 2 チャンク目以降は、その間に空きメモリが減っている (再生待ちの PCM など) ので + // 収まるか確かめ直す。ストリーミングでは分割計画が安全率を見込み済みなので + // 「実際に足りるか」だけを見る。 + if (k > 0) { + const std::size_t budget = stream ? memory_free_hard() : memory_budget(); + if (max_moras_for_budget(budget, voice_rate, fperiod, speed) < chunks[k].moras) return failed(); + } + const bool first = (k == 0); + const bool last = (k + 1 == chunks.size()); + const std::size_t lead_cap = + first ? SIZE_MAX : boundary_keep_frames(chunks[k - 1].pause_after, ms_per_frame, half_pause_ms); + const std::size_t trail_cap = + last ? SIZE_MAX : boundary_keep_frames(chunks[k].pause_after, ms_per_frame, half_pause_ms); + if (!synth_chunk(chunks[k].text, opt, decim, voice_rate, fperiod, speed, multi, lead_cap, trail_cap, pcm, + spans)) { + return failed(); + } + ++emitted; + if (!emit(std::move(pcm), std::move(spans), chunks[k].text)) return StreamOutcome::Cancelled; + pcm = {}; + spans = {}; + } + return StreamOutcome::Ok; +} + } // namespace internal } // namespace stackchan::jtts @@ -204,7 +439,10 @@ bool set_hmm_voice(std::span) { return false; } bool hmm_voice_loaded() { return false; } namespace internal { -bool render_hmm(std::u32string_view, std::vector&, const Options&) { return false; } +StreamOutcome render_hmm_stream(std::u32string_view, const Options&, const ChunkFn&, bool) { + return StreamOutcome::NoOutput; +} +void set_hmm_memory_budget_for_test(std::size_t) {} } // namespace internal } // namespace stackchan::jtts diff --git a/components/jtts/src/internal.hpp b/components/jtts/src/internal.hpp index 09afb88..10e3a44 100644 --- a/components/jtts/src/internal.hpp +++ b/components/jtts/src/internal.hpp @@ -2,8 +2,11 @@ // SPDX-License-Identifier: BSL-1.0 #pragma once +#include #include +#include #include +#include #include #include @@ -52,6 +55,9 @@ struct Segment { FormantFrame start; FormantFrame end; float duration_ms = 0.0f; + // リップシンク用: この区間で口が取る母音形。None = 閉口 (無音・「ん」・ + // 両唇子音の閉鎖・無声化母音)。音の合成には使わない。 + Vowel vowel = Vowel::None; }; bool parse_kana(std::u32string_view kana, std::vector& out); @@ -90,6 +96,22 @@ void render_segments(std::span segs, std::vector& o void render_segments_classic(std::span segs, std::vector& out, const Options& opt); +// 口形イベント列の組み立て (jtts.cpp)。区間を時間順に add() していくと、 +// 口形が変わる所だけがイベントになる。finish() で最後が閉口でなければ +// 終端に閉口イベントを付ける。 +class VisemeBuilder { +public: + explicit VisemeBuilder(std::vector& out) : out_(out) {} + void add(Vowel v, float duration_ms); + void finish(); + +private: + std::vector& out_; + float t_ms_ = 0.0f; + Vowel last_ = Vowel::None; + bool have_last_ = false; +}; + } // namespace stackchan::jtts::internal namespace stackchan::jtts::jvox { @@ -112,9 +134,67 @@ bool render_units(std::span moras, const jvox::Db& db, // 検証リファレンス: tools/jvox/hts_label_kana.py bool build_hts_labels(std::u32string_view text, std::vector& labels); -// HMM エンジン本体 (hmm_synth.cpp)。ボイス未ロード・ラベル生成失敗・ -// レート非対応時は out を触らず false (呼び出し側がフォールバック)。 -bool render_hmm(std::u32string_view text, std::vector& out, const Options& opt); +// 口形の 1 区間 (口形 + 継続時間)。HMM は音素ごとの継続長からこれを作る。 +struct VisemeSpan { + Vowel vowel = Vowel::None; + float duration_ms = 0.0f; +}; + +// spans → 口形イベント列 (先頭を 0 ms とし、同じ口形は連結、最後は閉口で終わる)。 +void spans_to_events(std::span spans, std::vector& out); + +// HMM の 1 チャンク分の受け取り側 (PCM と口形区間)。false で中断。 +using ChunkFn = + std::function&&, std::vector&&, const std::u32string& text)>; + +// チャンク単位のストリーミング合成 (HMM / sanoTTS) の結果。 +enum class StreamOutcome { + Ok, // 全チャンクを emit した + NoOutput, // 何も emit せずに諦めた (ボイス未ロード / メモリ不足など): 呼び出し側がフォールバック + Aborted, // 1 つ以上 emit した後に失敗した (メモリ不足など) + Cancelled, // emit が false を返した +}; + +// HMM エンジン本体 (hmm_synth.cpp)。長い発話はメモリ予算に収まるチャンクに分けて +// 順に合成し、チャンクごとに emit する。 +// stream = false: 一括 (synthesize 用)。チャンクは予算いっぱいまで詰める。 +// stream = true : 低遅延 (synthesize_stream 用)。最初のチャンクを小さく、以降を +// 徐々に大きくして、再生しながら次を合成しても途切れにくくする。 +StreamOutcome render_hmm_stream(std::u32string_view text, const Options& opt, const ChunkFn& emit, bool stream); + +// HMM 合成 1 回分のテキスト チャンク (hmm_chunk.cpp)。 +struct HmmChunk { + std::u32string text; + bool pause_after = false; // 末尾が句読点 (次のチャンクとの間に本来ポーズが入る) + std::size_t moras = 0; +}; + +// text を、各チャンクが max_moras 以下になるよう句読点 / アクセント句境界 +// (最後の手段でモーラ境界) で分割する。全体が収まるなら text をそのまま +// 1 チャンクにする。max_moras == 0 や発声できる内容が無いときは false。 +// +// first_moras > 0 のときは低遅延モード: 最初のチャンクを first_moras 程度に抑え、 +// 以降は直前のチャンクの 1.3 倍まで (max_moras を上限に) 徐々に大きくする。合成時間は +// 音声長の約 0.72 倍なので、次のチャンクの合成が前のチャンクの再生中に終わる。分割は +// 句読点 / アクセント句境界だけで行い、ここでは句の途中では切らない。 +bool split_hmm_text(std::u32string_view text, std::size_t max_moras, std::vector& out, + std::size_t first_moras = 0); + +// かな文字列 → 発話長 [ms] の粗い上限見積り (hmm_chunk.cpp)。PCM バッファの +// 確保量の事前見積りに使う。 +float estimate_utterance_ms(std::u32string_view text, float mora_ms); + +// samples 個の int16 PCM (+ 合成中の作業余裕) が空きメモリに収まるか。ESP では +// 空き PSRAM を見る。ホストでは常に true。収まらない発話は合成前に断り、 +// std::vector の確保失敗 (例外無効なので abort) を避ける。 +bool pcm_fits_in_memory(std::size_t samples); + +// テスト用: PCM に使える空きメモリ [byte] を固定する (0 で実機同様に自動判定)。 +void set_pcm_memory_limit_for_test(std::size_t bytes); + +// テスト用: HMM 合成のメモリ予算 [byte] を固定する (0 で実機同様に自動算出 / +// ホストでは無制限)。分割合成をホストで検証するために使う。 +void set_hmm_memory_budget_for_test(std::size_t bytes); // ---- sanoTTS-jp エンジン ---- @@ -126,10 +206,90 @@ bool render_hmm(std::u32string_view text, std::vector& out, const // 有効なモーラが 1 つも無ければ false。 bool build_sano_ir(std::u32string_view text, std::string& ir_utf8); -// sanoTTS エンジン本体 (sano_synth.cpp)。重み未ロード・IR 変換失敗・G2P 失敗・ -// arena 不足時は out を触らず false。成功時 out は 22.05 kHz mono int16 で、 -// *out_rate_hz に SAAN_SR を書く。 +// ---- 句ごとの合成 (sanoTTS / フォルマント / 単位連結で共通) -------------------- +// +// 発話全体を 1 回で合成すると、「、」「。」で間が入らない (フォルマント / 単位連結は +// 句読点を読み飛ばす) か、モデル任せの短いポーズになる (sanoTTS) ので、句の区切りが +// 弱い。HMM と同じく句読点 (、。,,..) ごとに 1 句ずつ合成し、句と句の間に HMM の +// pau と同じ長さの無音を明示的に挟む。HMM はモデルが句をまとめて合成して pau を +// 生成する (その方が速く自然) ので、この経路は通さないが、区切りの定義と間の長さは +// 共有する (下の clause_pause_ms / split_clauses)。 + +// 句と句の間の無音 [ms]。等速 (mora_ms = 110) で 420 ms、話速に比例する (HMM の pau と +// 同じ: 両側の sil の残り 210 ms ずつ、hmm_synth.cpp)。sanoTTS の s_v / HMM の speed と +// 同じ写像で、mora_ms は [55, 220] に丸める。 +inline float clause_pause_ms(float mora_ms) { + const float scale = mora_ms / 110.0f; + return 420.0f * (scale < 0.5f ? 0.5f : (scale > 2.0f ? 2.0f : scale)); +} + +// 句 (先頭から、直後の句読点までを 1 つ。続く句読点はまとめる) に分ける。区切りは +// HMM と同じ (、。,,..)。モーラを持たない断片は前の句に含める。HmmChunk::pause_after +// は句読点で終わるか。全体にモーラが無ければ false。 +bool split_clauses(std::u32string_view text, std::vector& out); + +// 1 句の合成結果 (テスト用のシームでもある)。 +enum class ClauseResult { + Ok, // pcm / rate_hz / spans を返した + Skip, // 読める内容が無い (この句は飛ばす) + Fail, // 合成失敗 (重み未ロード / G2P / arena 不足など) +}; +// 1 句を合成する。spans は pcm 全体を覆う口形の区間 (母音 / 閉口 + 継続時間)。作れなければ空。 +using ClauseSynth = std::function& pcm, + std::uint32_t& rate_hz, std::vector& spans)>; + +// 句ごとの合成の 1 チャンク (句の音声 + 句間の無音) の受け取り側。spans は pcm 全体を覆う +// 口形の区間 (句間の無音は閉口)。false で中断。 +using ClauseChunkFn = std::function&& pcm, std::uint32_t rate_hz, + std::vector&& spans, const std::u32string& text)>; + +// 句ごとに synth_one で合成し、句と句の間の境界を整えて emit する: +// - 内側の境界では、句の後ろに句間の無音 (clause_pause_ms) を足す。口形の区間にも +// 閉口を足す (音と口形の時刻が揃ったまま)。 +// - trim_edges = true (sanoTTS): 内側の境界で、各句の前後の無音 (モデルが付ける) を +// 先に切り詰めて (口形の区間も同じだけ)、間の長さがモデル任せにならないようにする。 +// false (フォルマント / 単位連結): 句の音声はそのまま。 +// - 発話の先頭 / 末尾は切り詰めない。 +// 最初の句が Fail なら何も出さず NoOutput (呼び出し側が他エンジンへフォールバック)、 +// 途中の Fail は Aborted、Skip は飛ばす。句が 1 つだけなら加工せず、全体をそのまま渡す。 +StreamOutcome stream_clauses(std::u32string_view text, const Options& opt, const ClauseSynth& synth_one, + const ClauseChunkFn& emit, bool trim_edges); + +// trim_silence が削った量 [サンプル]。 +struct SilenceTrim { + std::size_t front = 0; + std::size_t back = 0; +}; + +// pcm の前 (lead) / 後ろ (trail) の無音を切り詰める。無音 = 絶対値がピークの約 1 % (最低 48) +// 未満。音の手前 / 奥に keep_ms の余白は残す (立ち上がり / 減衰を削らないため)。 +// 全体が無音なら何もしない。テスト用に公開。 +SilenceTrim trim_silence(std::vector& pcm, std::uint32_t rate_hz, bool lead, bool trail, + std::uint32_t keep_ms = 10); + +// 口形の区間列から、先頭の drop_front_ms を捨て、その後 keep_ms だけ残す (残りは捨てる)。 +void crop_spans(std::vector& spans, float drop_front_ms, float keep_ms); + +// sanoTTS の音素 ID 列 (saan_g2p の出力: ^ PAD 音素 PAD 音素 ... $) と、音素ごとの継続長 +// d_hat [フレーム] から、口形の区間列を作る (HMM と同じ規則): +// 母音 あ/い/う/え/お … その母音の形 +// 無声化母音・ん・っ・ポーズ … 閉口 (ポーズ = PAD が 2 つ以上続いたときの 2 つ目以降) +// 両唇音 m b p (拗音含む) … 閉口 +// その他の子音 … 直後の母音の形を先取り +// PAD / 記号 [ ] # ? … 直前の音素の形を保つ (音素の間の余白)。先頭の ^ と末尾の $ は閉口。 +// ms_per_frame は 1 フレームの長さ [ms]。区間の合計 = n × ms_per_frame。 +void sano_ids_to_spans(const std::int32_t* ids, const std::int32_t* d_hat, std::int32_t n, float ms_per_frame, + std::vector& out); + +// sanoTTS エンジン本体 (sano_synth.cpp)。stream_clauses (trim_edges) で句ごとに合成して emit する。重み未ロード +// なら NoOutput。出力は 22.05 kHz mono int16。 +StreamOutcome render_sano_stream(std::u32string_view text, const Options& opt, const ClauseChunkFn& emit); + +// render_sano_stream の全チャンクを 1 本に連結する (synthesize / synthesize_ex 用)。 +// 重み未ロード・IR 変換失敗・G2P 失敗・arena 不足時は out を空にして false。 +// 成功時 out は 22.05 kHz mono int16 で、*out_rate_hz に SAAN_SR を書く。 +// visemes が非 null なら、口形イベントを追記する (out と時間軸が揃う)。 bool render_sano(std::u32string_view text, std::vector& out, const Options& opt, - std::uint32_t& out_rate_hz); + std::uint32_t& out_rate_hz, std::vector* visemes = nullptr); } // namespace stackchan::jtts::internal diff --git a/components/jtts/src/jtts.cpp b/components/jtts/src/jtts.cpp index cdfebd9..eac88b7 100644 --- a/components/jtts/src/jtts.cpp +++ b/components/jtts/src/jtts.cpp @@ -2,6 +2,7 @@ // SPDX-License-Identifier: BSL-1.0 #include "jtts/jtts.hpp" +#include #include #include #include @@ -41,6 +42,7 @@ const char* to_string(Error e) { switch (e) { case Error::InvalidKana: return "InvalidKana"; case Error::OutOfMemory: return "OutOfMemory"; + case Error::Cancelled: return "Cancelled"; } return "Unknown"; } @@ -76,84 +78,271 @@ void apply_formant_scale(std::vector& segs, float scale) { } } -} // namespace - -namespace { - -tl::expected synthesize_impl(std::u32string_view kana, - std::vector& out, - const Options& opt_in, bool allow_native_rate) { - out.clear(); - Options opt = resolve_defaults(opt_in); - - // sanoTTS エンジン: 重みがロード済みなら最優先。出力は 22.05 kHz 固定なので、 - // 呼び出し側がレートを受け取れる (synthesize_ex) か、要求レートが一致する - // ときだけ使う。 - if (opt.engine == Engine::Auto || opt.engine == Engine::Sano) { - if (allow_native_rate || opt.sample_rate_hz == 22050u) { - std::uint32_t rate = 0; - if (internal::render_sano(kana, out, opt, rate)) { - return rate; - } - } - } +// sanoTTS: Auto では最優先、Sano 指定では必須。出力は 22.05 kHz 固定なので、呼び出し側が +// レートを受け取れる (synthesize_ex / synthesize_stream) か、要求レートが一致するとき +// だけ使う。 +bool wants_sano(const Options& opt) { + return opt.engine == Engine::Auto || opt.engine == Engine::Sano; +} - // HMM エンジン: ボイスがロード済みなら最優先 (品質最良)。 - // アクセント記号 (' と /) は HMM のみ解釈し、他エンジンでは - // parse_kana が読み飛ばす。 - if (opt.engine == Engine::Auto || opt.engine == Engine::Hmm || opt.engine == Engine::Sano) { - if (internal::render_hmm(kana, out, opt)) { - return opt.sample_rate_hz; - } - } +// HMM: Auto / Hmm、および Sano 指定 (重み未ロード時のフォールバック先)。 +bool wants_hmm(const Options& opt) { + return opt.engine == Engine::Auto || opt.engine == Engine::Hmm || opt.engine == Engine::Sano; +} +// 1 句を単位連結 → フォルマントで合成する (句ごとの合成の 1 句分)。口形はフォルマントの +// セグメントから作る (単位連結は口形を出せない = 空)。 +internal::ClauseResult render_clause_local(const std::u32string& clause, const Options& opt, + std::vector& pcm, std::uint32_t& rate_hz, + std::vector& spans) { std::vector moras; - if (!internal::parse_kana(kana, moras)) { - return tl::make_unexpected(Error::InvalidKana); - } + if (!internal::parse_kana(clause, moras)) return internal::ClauseResult::Skip; // 読める内容が無い internal::apply_devoicing(moras); + rate_hz = opt.sample_rate_hz; // 単位連結エンジン: DB があり、必要な単位が全部揃っていれば - // render_units が out を埋めて true。欠け/未ロード/サンプルレート不一致は + // render_units が pcm を埋めて true。欠け/未ロード/サンプルレート不一致は // フォルマントへ。 if (opt.engine != Engine::Formant) { auto db = g_voice_db.load(); if (db && db->sample_rate() == opt.sample_rate_hz && - internal::render_units(moras, *db, out, opt)) { - return opt.sample_rate_hz; + internal::render_units(moras, *db, pcm, opt)) { + return internal::ClauseResult::Ok; } } std::vector segs; internal::build_segments(moras, segs, opt); - if (segs.empty()) { - return tl::make_unexpected(Error::InvalidKana); - } + if (segs.empty()) return internal::ClauseResult::Skip; apply_formant_scale(segs, opt.formant_scale); internal::apply_prosody(segs, opt); std::size_t estimated_samples = 0; for (const auto& s : segs) { estimated_samples += static_cast(s.duration_ms * 0.001f * opt.sample_rate_hz) + 16; + spans.push_back({s.vowel, s.duration_ms}); + } + pcm.reserve(estimated_samples); + + internal::render_segments(segs, pcm, opt); + return internal::ClauseResult::Ok; +} + +// 単位連結 → フォルマントを句ごとに合成して emit する。sanoTTS と同じ制御 (stream_clauses: +// 句読点 (、。) ごとに 1 句ずつ、句間に HMM の pau と同じ長さの無音) を使う。この 2 つは句読点 +// を読み飛ばすので、従来は句の間に間が全く入らなかった。句が 1 つだけなら従来と同じ出力。 +internal::StreamOutcome render_local_stream(std::u32string_view kana, const Options& opt, + const internal::ClauseChunkFn& emit) { + return internal::stream_clauses( + kana, opt, + [&](const std::u32string& clause, std::vector& pcm, std::uint32_t& rate, + std::vector& spans) { return render_clause_local(clause, opt, pcm, rate, spans); }, + emit, /*trim_edges=*/false); +} + +// 一括版: 全チャンクを 1 本の PCM に連結する。全体を 1 つの std::vector に作るので、空きメモリに +// 収まらない長さは合成前に断る (std::vector の確保失敗は例外無効ビルドでは abort = 再起動になる)。 +tl::expected render_local(std::u32string_view kana, const Options& opt, + std::vector& out, std::vector* visemes) { + const auto est_samples = static_cast( + internal::estimate_utterance_ms(kana, opt.mora_ms) * static_cast(opt.sample_rate_hz) / 1000.0f); + if (!internal::pcm_fits_in_memory(est_samples)) { + return tl::make_unexpected(Error::OutOfMemory); + } + out.reserve(est_samples); + std::vector scratch; + internal::VisemeBuilder builder(visemes ? *visemes : scratch); + const auto outcome = render_local_stream( + kana, opt, + [&](std::vector&& pcm, std::uint32_t, std::vector&& spans, + const std::u32string&) { + out.insert(out.end(), pcm.begin(), pcm.end()); + for (const auto& sp : spans) builder.add(sp.vowel, sp.duration_ms); + return true; + }); + if (outcome != internal::StreamOutcome::Ok || out.empty()) { + out.clear(); + if (visemes) visemes->clear(); + return tl::make_unexpected(Error::InvalidKana); + } + builder.finish(); + if (out.capacity() - out.size() > 32 * 1024) out.shrink_to_fit(); + return {}; +} + +// 一括合成。戻り値は出力 PCM のサンプルレート (sanoTTS は 22.05 kHz、他は opt.sample_rate_hz)。 +tl::expected synthesize_impl(std::u32string_view kana, std::vector& out, + std::vector* visemes, const Options& opt_in, + bool allow_native_rate) { + out.clear(); + if (visemes) visemes->clear(); + Options opt = resolve_defaults(opt_in); + + // 発話が長すぎて PCM が空きメモリに収まらないなら、どのエンジンでも合成せず断る。 + const std::uint32_t est_rate = + wants_sano(opt) && sano_weights_loaded() ? std::max(opt.sample_rate_hz, 22050u) : opt.sample_rate_hz; + const auto est_samples = static_cast( + internal::estimate_utterance_ms(kana, opt.mora_ms) * static_cast(est_rate) / 1000.0f); + if (!internal::pcm_fits_in_memory(est_samples)) { + return tl::make_unexpected(Error::OutOfMemory); + } + + // sanoTTS エンジン: 重みがロード済みなら最優先。口形は音素の継続長から作る。 + if (wants_sano(opt) && (allow_native_rate || opt.sample_rate_hz == 22050u)) { + std::uint32_t rate = 0; + if (internal::render_sano(kana, out, opt, rate, visemes)) { + return rate; + } + out.clear(); + if (visemes) visemes->clear(); + } + + // HMM エンジン: ボイスがロード済みなら次に優先 (品質最良)。 + // アクセント記号 (' と /) は HMM のみ解釈し、他エンジンでは + // parse_kana が読み飛ばす。全チャンクを 1 本の PCM に連結する。 + if (wants_hmm(opt) && hmm_voice_loaded()) { + // PCM は最初に 1 回だけ確保し (途中の再確保 = 旧 + 新の一時倍増を避ける)、 + // 合成後に余りを返す。 + out.reserve(est_samples); + std::vector scratch; + internal::VisemeBuilder builder(visemes ? *visemes : scratch); + const auto outcome = internal::render_hmm_stream( + kana, opt, + [&](std::vector&& pcm, std::vector&& spans, const std::u32string&) { + out.insert(out.end(), pcm.begin(), pcm.end()); + for (const auto& sp : spans) builder.add(sp.vowel, sp.duration_ms); + return true; + }, + /*stream=*/false); + if (outcome == internal::StreamOutcome::Ok) { + builder.finish(); + if (out.capacity() - out.size() > 32 * 1024) out.shrink_to_fit(); + return opt.sample_rate_hz; + } + // 諦めた (メモリ不足など): 途中まで作った分は捨てて他エンジンへ。 + out.clear(); + if (visemes) visemes->clear(); } - out.reserve(estimated_samples); - internal::render_segments(segs, out, opt); + if (auto r = render_local(kana, opt, out, visemes); !r) return tl::make_unexpected(r.error()); return opt.sample_rate_hz; } } // namespace -tl::expected synthesize(std::u32string_view kana, - std::vector& out, const Options& opt) { - auto r = synthesize_impl(kana, out, opt, /*allow_native_rate=*/false); +namespace internal { + +void VisemeBuilder::add(Vowel v, float duration_ms) { + if (duration_ms <= 0.0f) return; + if (!have_last_ || v != last_) { + out_.push_back({static_cast(t_ms_ + 0.5f), v}); + last_ = v; + have_last_ = true; + } + t_ms_ += duration_ms; +} + +void spans_to_events(std::span spans, std::vector& out) { + out.clear(); + VisemeBuilder b(out); + for (const auto& sp : spans) b.add(sp.vowel, sp.duration_ms); + b.finish(); +} + +void VisemeBuilder::finish() { + if (have_last_ && last_ != Vowel::None) { + out_.push_back({static_cast(t_ms_ + 0.5f), Vowel::None}); + last_ = Vowel::None; + } +} + +} // namespace internal + +tl::expected synthesize(std::u32string_view kana, std::vector& out, + const Options& opt) { + auto r = synthesize_impl(kana, out, nullptr, opt, /*allow_native_rate=*/false); + if (!r) return tl::make_unexpected(r.error()); + return {}; +} + +tl::expected synthesize(std::u32string_view kana, std::vector& out, + std::vector& visemes, const Options& opt) { + auto r = synthesize_impl(kana, out, &visemes, opt, /*allow_native_rate=*/false); if (!r) return tl::make_unexpected(r.error()); return {}; } tl::expected synthesize_ex(std::u32string_view kana, std::vector& out, const Options& opt) { - return synthesize_impl(kana, out, opt, /*allow_native_rate=*/true); + return synthesize_impl(kana, out, nullptr, opt, /*allow_native_rate=*/true); +} + +tl::expected synthesize_stream(std::u32string_view kana, const ChunkSink& sink, const Options& opt_in) { + const Options opt = resolve_defaults(opt_in); + + // sanoTTS: 句 (、。) ごとに合成し、句間に HMM と同じ長さの無音を入れて 1 句ずつ渡す + // (22.05 kHz)。重み未ロード / 最初の句の失敗なら何も渡さず次のエンジンへ。 + if (wants_sano(opt)) { + const auto outcome = internal::render_sano_stream( + kana, opt, + [&](std::vector&& pcm, std::uint32_t rate_hz, std::vector&& spans, + const std::u32string& text) { + SynthChunk chunk; + chunk.text = text; + chunk.pcm = std::move(pcm); + chunk.sample_rate = rate_hz; + internal::spans_to_events(spans, chunk.visemes); + return sink(std::move(chunk)); + }); + switch (outcome) { + case internal::StreamOutcome::Ok: return {}; + case internal::StreamOutcome::Cancelled: return tl::make_unexpected(Error::Cancelled); + // 途中まで渡してしまった分は取り消せないので、他エンジンでやり直さない。 + case internal::StreamOutcome::Aborted: return tl::make_unexpected(Error::OutOfMemory); + case internal::StreamOutcome::NoOutput: break; // 何も渡していない → 他エンジンへ + } + } + + if (wants_hmm(opt) && hmm_voice_loaded()) { + const auto outcome = internal::render_hmm_stream( + kana, opt, + [&](std::vector&& pcm, std::vector&& spans, const std::u32string& text) { + SynthChunk chunk; + chunk.text = text; + chunk.pcm = std::move(pcm); + chunk.sample_rate = opt.sample_rate_hz; + internal::spans_to_events(spans, chunk.visemes); + return sink(std::move(chunk)); + }, + /*stream=*/true); + switch (outcome) { + case internal::StreamOutcome::Ok: return {}; + case internal::StreamOutcome::Cancelled: return tl::make_unexpected(Error::Cancelled); + // 途中まで渡してしまった分は取り消せないので、他エンジンでやり直さない。 + case internal::StreamOutcome::Aborted: return tl::make_unexpected(Error::OutOfMemory); + case internal::StreamOutcome::NoOutput: break; // 何も渡していない → 他エンジンへ + } + } + + // HMM 以外 / HMM を使えなかった (単位連結 → フォルマント): 句ごとに 1 チャンク。 + const auto outcome = render_local_stream( + kana, opt, + [&](std::vector&& pcm, std::uint32_t rate_hz, std::vector&& spans, + const std::u32string& text) { + SynthChunk chunk; + chunk.text = text; + chunk.pcm = std::move(pcm); + chunk.sample_rate = rate_hz; + internal::spans_to_events(spans, chunk.visemes); + return sink(std::move(chunk)); + }); + switch (outcome) { + case internal::StreamOutcome::Ok: return {}; + case internal::StreamOutcome::Cancelled: return tl::make_unexpected(Error::Cancelled); + case internal::StreamOutcome::Aborted: return tl::make_unexpected(Error::OutOfMemory); + case internal::StreamOutcome::NoOutput: return tl::make_unexpected(Error::InvalidKana); + } + return tl::make_unexpected(Error::InvalidKana); } } // namespace stackchan::jtts diff --git a/components/jtts/src/phoneme.cpp b/components/jtts/src/phoneme.cpp index 9bb87b3..157cda2 100644 --- a/components/jtts/src/phoneme.cpp +++ b/components/jtts/src/phoneme.cpp @@ -20,7 +20,8 @@ FormantFrame silent_frame(float f0) { // prenasal_zero_hz > 0 のとき、後続モーラが「ん」なので母音末尾 ~30 ms で // nasal を 0→0.5 に立ち上げる (先行母音の鼻音化。ん への移行でスペクトルが // 急変するのを防ぎ、自然な渡りになる)。値は後続の鼻音ゼロ周波数。 -void add_cv_segments(Consonant c, Vowel v, bool palatalized, bool devoiced, float mora_ms, +// 戻り値は out 内で母音本体 (push_vowel_tail が積む区間) が始まる添字。 +std::size_t add_cv_segments(Consonant c, Vowel v, bool palatalized, bool devoiced, float mora_ms, float f0, float prenasal_zero_hz, std::vector& out) { FormantFrame vowel = vowel_frame(v, palatalized); vowel.f0_hz = f0; @@ -37,7 +38,13 @@ void add_cv_segments(Consonant c, Vowel v, bool palatalized, bool devoiced, floa // 母音末尾セグメントを積む。prenasal (次モーラが「ん」) なら末尾 ~30 ms // で nasal を 0→0.5 に上げて先行母音を鼻音化する。 + std::size_t tail_begin = out.size(); + bool tail_marked = false; auto push_vowel_tail = [&](float consumed) { + if (!tail_marked) { + tail_begin = out.size(); + tail_marked = true; + } float v_ms = std::max(20.0f, mora_ms - consumed); if (prenasal_zero_hz > 0.0f) { FormantFrame nasalized = vowel; @@ -56,7 +63,7 @@ void add_cv_segments(Consonant c, Vowel v, bool palatalized, bool devoiced, floa if (c == Consonant::None) { push_vowel_tail(0.0f); - return; + return tail_begin; } FormantFrame burst = consonant_burst(c, v); @@ -140,6 +147,7 @@ void add_cv_segments(Consonant c, Vowel v, bool palatalized, bool devoiced, floa } else { out.push_back({vowel, vowel, mora_ms}); } + return tail_begin; } } // namespace @@ -179,8 +187,17 @@ void build_segments(std::span moras, std::vector& out, cons if (i + 1 < moras.size() && moras[i + 1].kind == MoraKind::MoraicN) { prenasal_zero_hz = moraic_n_frame(i + 1).nasal_zero_hz; } - add_cv_segments(m.c, m.v, m.palatalized, m.devoiced, mora_ms, f0, - prenasal_zero_hz, out); + const std::size_t first = out.size(); + const std::size_t tail = add_cv_segments(m.c, m.v, m.palatalized, m.devoiced, mora_ms, f0, + prenasal_zero_hz, out); + // 口形: 無声化母音は閉口のまま。子音区間は後続母音の形を先取りするが、 + // 両唇音 (m b p) の閉鎖だけは唇を閉じる。 + if (!m.devoiced) { + const bool bilabial = m.c == Consonant::M || m.c == Consonant::B || m.c == Consonant::P; + for (std::size_t k = first; k < out.size(); ++k) { + out[k].vowel = (bilabial && k < tail) ? Vowel::None : m.v; + } + } break; } case MoraKind::MoraicN: { @@ -196,7 +213,9 @@ void build_segments(std::span moras, std::vector& out, cons case MoraKind::Chouon: { if (!out.empty()) { FormantFrame ref = out.back().end; - out.push_back({ref, ref, mora_ms}); + Segment held{ref, ref, mora_ms}; + held.vowel = out.back().vowel; // 長音は直前の口形を保つ + out.push_back(held); } break; } diff --git a/components/jtts/src/sano_synth.cpp b/components/jtts/src/sano_synth.cpp index 94f03ba..b6f7583 100644 --- a/components/jtts/src/sano_synth.cpp +++ b/components/jtts/src/sano_synth.cpp @@ -149,13 +149,15 @@ void set_sano_arena(void* buf, std::size_t size) { namespace internal { -bool render_sano(std::u32string_view text, std::vector& out, const Options& opt, - std::uint32_t& out_rate_hz) { - std::lock_guard lock(g_mutex); - if (!g_loaded) return false; +namespace { + +// 1 句を合成して out に *追記*せず、pcm に作る。g_mutex は呼び出し側 (1 句ごと) が取る。 +ClauseResult synth_clause_locked(const std::u32string& clause, std::vector& out, const Options& opt, + std::uint32_t& out_rate_hz, std::vector& spans) { + if (!g_loaded) return ClauseResult::Fail; std::string ir; - if (!build_sano_ir(text, ir)) return false; + if (!build_sano_ir(clause, ir)) return ClauseResult::Skip; // 読める内容が無い const std::int32_t cap = saan_g2p_capacity(ir.size()); std::vector ids(static_cast(cap)); @@ -165,9 +167,9 @@ bool render_sano(std::u32string_view text, std::vector& out, const if (gs != SAAN_G2P_OK) { SANO_LOGW("g2p: %s at byte %d (ir='%s')", saan_g2p_strerror(gs), static_cast(info.err_byte), ir.c_str()); - return false; + return ClauseResult::Fail; } - if (!ensure_arena()) return false; + if (!ensure_arena()) return ClauseResult::Fail; // saan_stream_arena_needed() は上限式で実使用 (arena.peak) より大きく出るので、 // 事前判定には使わず init の SAAN_ERR_ARENA に任せる (上流の雛形と同じ)。 @@ -185,9 +187,15 @@ bool render_sano(std::u32string_view text, std::vector& out, const if (s != SAAN_OK) { SANO_LOGW("stream_init: %s (%d ids, arena %u B, needed<=%u B)", saan_strerror(s), static_cast(n_ids), static_cast(g_arena_size), static_cast(saan_stream_arena_needed(n_ids))); - return false; + return ClauseResult::Fail; } + // init が音素ごとの継続長 d_hat [フレーム] を確定させている (Σ d_hat = n_frames)。 + // arena 上にあるので、pull を始める前に口形の区間へ写し取る。 + const std::int32_t n_frames = st.n_frames; + std::vector durations(st.d_hat, st.d_hat + n_ids); + + out.clear(); std::vector chunk(static_cast(SAAN_CHUNK) * SAAN_HOP); const float gain = opt.gain * 32767.0f; for (;;) { @@ -196,7 +204,7 @@ bool render_sano(std::u32string_view text, std::vector& out, const if (s != SAAN_OK) { SANO_LOGW("stream_pull: %s", saan_strerror(s)); out.clear(); - return false; + return ClauseResult::Fail; } if (n_out == 0) break; #if defined(ESP_PLATFORM) @@ -219,6 +227,60 @@ bool render_sano(std::u32string_view text, std::vector& out, const static_cast(out.size()), audio_ms, static_cast(ms), audio_ms > 0 ? static_cast(ms) / audio_ms : 0.0f, static_cast(arena.peak)); out_rate_hz = SAAN_SR; + + // 口形: 音素 ID + 継続長 → 母音の区間。1 フレームの長さは実際の出力長から求めて、 + // 区間の合計を音声の長さにぴったり合わせる (出力が n_frames × hop とずれても時刻が狂わない)。 + if (n_frames > 0 && !out.empty()) { + if (out.size() != static_cast(n_frames) * SAAN_HOP) { + SANO_LOGW("samples %u != n_frames %d x hop %d (visemes rescaled)", static_cast(out.size()), + static_cast(n_frames), static_cast(SAAN_HOP)); + } + const float ms_per_frame = audio_ms / static_cast(n_frames); + sano_ids_to_spans(ids.data(), durations.data(), n_ids, ms_per_frame, spans); + } + return ClauseResult::Ok; +} + +} // namespace + +StreamOutcome render_sano_stream(std::u32string_view text, const Options& opt, const ClauseChunkFn& emit) { + // 1 句ごとにロックする: emit (再生キューの空き待ちなどで長く止まりうる) の間は + // 重みの差し替え (set_sano_weights) やロード状態の問い合わせを塞がない。 + const auto synth_one = [&](const std::u32string& clause, std::vector& pcm, std::uint32_t& rate, + std::vector& spans) { + std::lock_guard lock(g_mutex); + return synth_clause_locked(clause, pcm, opt, rate, spans); + }; + { + std::lock_guard lock(g_mutex); + if (!g_loaded) return StreamOutcome::NoOutput; + } + return stream_clauses(text, opt, synth_one, emit, /*trim_edges=*/true); +} + +bool render_sano(std::u32string_view text, std::vector& out, const Options& opt, + std::uint32_t& out_rate_hz, std::vector* visemes) { + out.clear(); + if (visemes) visemes->clear(); + std::uint32_t rate = 0; + std::vector scratch; + VisemeBuilder builder(visemes ? *visemes : scratch); + const StreamOutcome r = render_sano_stream( + text, opt, + [&](std::vector&& pcm, std::uint32_t rate_hz, std::vector&& spans, + const std::u32string&) { + out.insert(out.end(), pcm.begin(), pcm.end()); + rate = rate_hz; + for (const auto& sp : spans) builder.add(sp.vowel, sp.duration_ms); + return true; + }); + if (r != StreamOutcome::Ok || out.empty()) { + out.clear(); + if (visemes) visemes->clear(); + return false; + } + builder.finish(); + out_rate_hz = rate; return true; } @@ -232,7 +294,11 @@ namespace stackchan::jtts { bool set_sano_weights(std::span) { return false; } bool sano_weights_loaded() { return false; } namespace internal { -bool render_sano(std::u32string_view, std::vector&, const Options&, std::uint32_t&) { +StreamOutcome render_sano_stream(std::u32string_view, const Options&, const ClauseChunkFn&) { + return StreamOutcome::NoOutput; +} +bool render_sano(std::u32string_view, std::vector&, const Options&, std::uint32_t&, + std::vector*) { return false; } } // namespace internal diff --git a/components/jtts/src/sano_visemes.cpp b/components/jtts/src/sano_visemes.cpp new file mode 100644 index 0000000..771622a --- /dev/null +++ b/components/jtts/src/sano_visemes.cpp @@ -0,0 +1,91 @@ +// SPDX-FileCopyrightText: 2026 Kenta IDA +// SPDX-License-Identifier: BSL-1.0 +// +// sanoTTS の音素 ID (saan_g2p の出力) と音素ごとの継続長 (d_hat) から、口形の区間列を作る +// (sanoTTS のコアには依存しない: 語彙は上流 g2p_table.h の生徒インデックス)。 +#include +#include + +#include "internal.hpp" + +namespace stackchan::jtts::internal { + +// --- 音素 ID → 口形 (語彙は上流 g2p_table.h の生徒インデックス。仮名表との一致は +// test_sano_clause が kSaanG2pMora で検証する) --- +namespace { + +constexpr int kIdPad = 0; +constexpr int kIdBos = 1; +constexpr int kIdEos = 2; +constexpr int kIdVowelLo = 10; // a i u e o = 10..14 +constexpr int kIdVowelHi = 14; + +// マーク ? # [ ] (アクセント / 疑問の記号。音素の間に入る余白として扱う)。 +bool is_mark(int id) { return id == 3 || (id >= 7 && id <= 9); } + +// 両唇音 (唇を閉じる): p py b by m my。 +bool is_bilabial(int id) { return id == 35 || id == 36 || id == 37 || id == 38 || id == 51 || id == 52; } + +Vowel vowel_of_id(int id) { + switch (id) { + case 10: return Vowel::A; + case 11: return Vowel::I; + case 12: return Vowel::U; + case 13: return Vowel::E; + case 14: return Vowel::O; + default: return Vowel::None; + } +} + +} // namespace + +void sano_ids_to_spans(const std::int32_t* ids, const std::int32_t* d_hat, std::int32_t n, float ms_per_frame, + std::vector& out) { + out.clear(); + if (ids == nullptr || d_hat == nullptr || n <= 0) return; + + // 音素 (PAD / マーク以外) の位置を先に拾い、直後の音素の母音を引けるようにする。 + auto is_blank = [&](std::int32_t i) { return ids[i] == kIdPad || is_mark(ids[i]); }; + Vowel held = Vowel::None; // 直前の音素の形 (PAD / マークの間はこれを保つ) + bool prev_pad = false; // 直前のトークンが PAD (連続した PAD の 2 つ目以降 = ポーズ) + for (std::int32_t i = 0; i < n; ++i) { + const int id = ids[i]; + Vowel v; + if (id == kIdBos || id == kIdEos) { + v = Vowel::None; + held = Vowel::None; + prev_pad = false; + } else if (is_blank(i)) { + const bool pause = id == kIdPad && prev_pad; + v = pause ? Vowel::None : held; + prev_pad = id == kIdPad; + } else { + prev_pad = false; + if (id >= kIdVowelLo && id <= kIdVowelHi) { + v = vowel_of_id(id); + } else if (is_bilabial(id)) { + v = Vowel::None; + } else if (id >= 15 && id <= 24) { + v = Vowel::None; // 無声化母音 (15..19)・ん (20..23)・っ (24) + } else { + // その他の子音: 直後の (PAD / マークを飛ばした) 音素が平母音ならその形を先取り。 + v = Vowel::None; + for (std::int32_t j = i + 1; j < n; ++j) { + if (is_blank(j)) continue; + if (ids[j] >= kIdVowelLo && ids[j] <= kIdVowelHi) v = vowel_of_id(ids[j]); + break; + } + } + held = v; + } + const float ms = static_cast(d_hat[i]) * ms_per_frame; + if (ms <= 0.0f) continue; + if (!out.empty() && out.back().vowel == v) { + out.back().duration_ms += ms; + } else { + out.push_back({v, ms}); + } + } +} + +} // namespace stackchan::jtts::internal diff --git a/components/jtts/src/subtitle.cpp b/components/jtts/src/subtitle.cpp new file mode 100644 index 0000000..05b8e35 --- /dev/null +++ b/components/jtts/src/subtitle.cpp @@ -0,0 +1,82 @@ +// SPDX-FileCopyrightText: 2026 Kenta IDA +// SPDX-License-Identifier: BSL-1.0 +#include "jtts/subtitle.hpp" + +#include +#include + +namespace stackchan::jtts { + +namespace { + +bool is_pause_char(char32_t c) { + return c == U'、' || c == U'。' || c == U',' || c == U',' || c == U'.' || c == U'.'; +} + +std::size_t count_pauses(std::u32string_view s) { + return static_cast(std::count_if(s.begin(), s.end(), is_pause_char)); +} + +// チャンクの読みが句読点で終わるか (末尾の空白・アクセント記号・/ は無視)。 +bool ends_with_pause(std::u32string_view s) { + for (std::size_t i = s.size(); i > 0; --i) { + const char32_t c = s[i - 1]; + if (c == U' ' || c == U' ' || c == U'\n' || c == U'\'' || c == U'’' || c == U'/') continue; + return is_pause_char(c); + } + return false; +} + +// UTF-8 で s[i] から始まる句読点の長さ (バイト)。句読点でなければ 0。 +std::size_t pause_len_utf8(std::string_view s, std::size_t i) { + const auto b = [&](std::size_t k) { return i + k < s.size() ? static_cast(s[i + k]) : 0u; }; + if (b(0) == ',' || b(0) == '.') return 1; + if (b(0) == 0xE3 && b(1) == 0x80 && (b(2) == 0x81 || b(2) == 0x82)) return 3; // 、 。 + if (b(0) == 0xEF && b(1) == 0xBC && (b(2) == 0x8C || b(2) == 0x8E)) return 3; // , . + return 0; +} + +} // namespace + +SubtitleMapper::SubtitleMapper(std::string_view display_utf8, std::u32string_view reading) : whole_(display_utf8) { + // 表示テキストを句読点の直後で区切る (句読点は前の句に含める)。 + std::string cur; + std::size_t pauses = 0; + for (std::size_t i = 0; i < display_utf8.size();) { + const std::size_t n = pause_len_utf8(display_utf8, i); + if (n > 0) { + cur.append(display_utf8.substr(i, n)); + segments_.push_back(std::move(cur)); + cur.clear(); + ++pauses; + i += n; + } else { + cur.push_back(display_utf8[i]); + ++i; + } + } + segments_.push_back(std::move(cur)); // 最後の句読点より後ろ (空のこともある) + mapped_ = !display_utf8.empty() && pauses == count_pauses(reading); +} + +std::string SubtitleMapper::next(std::u32string_view chunk_reading) { + const bool first = first_; + first_ = false; + if (!mapped_) { + return first ? whole_ : std::string{}; + } + const std::size_t pauses = count_pauses(chunk_reading); + const std::size_t begin = consumed_; + const std::size_t after = consumed_ + pauses; + consumed_ = after; + // 句読点で終わるチャンクは、その句読点を含む句までが担当 (次のチャンクは次の句から)。 + // 途中で切れたチャンクは、まだ句読点に届いていない句 (after) も担当する。 + const std::size_t last = segments_.size() - 1; + const std::size_t a = std::min(begin, last); + const std::size_t b = std::min(pauses > 0 && ends_with_pause(chunk_reading) ? after - 1 : after, last); + std::string out; + for (std::size_t k = a; k <= b; ++k) out += segments_[k]; + return out; +} + +} // namespace stackchan::jtts diff --git a/components/jtts/test/host/CMakeLists.txt b/components/jtts/test/host/CMakeLists.txt index 8914eb6..9367629 100644 --- a/components/jtts/test/host/CMakeLists.txt +++ b/components/jtts/test/host/CMakeLists.txt @@ -33,7 +33,11 @@ add_library(jtts STATIC ${JTTS_ROOT}/src/hts_label.cpp ${JTTS_ROOT}/src/hmm_synth.cpp ${JTTS_ROOT}/src/sano_ir.cpp + ${JTTS_ROOT}/src/clause_stream.cpp + ${JTTS_ROOT}/src/sano_visemes.cpp ${JTTS_ROOT}/src/sano_synth.cpp + ${JTTS_ROOT}/src/hmm_chunk.cpp + ${JTTS_ROOT}/src/subtitle.cpp ) target_include_directories(jtts PUBLIC ${JTTS_ROOT}/include ${EXPECTED_INC} @@ -97,3 +101,20 @@ target_compile_options(jtts_test_sano_ir PRIVATE -Wall -Wextra) add_executable(jtts_sano_demo sano_demo.cpp wav_writer.cpp) target_link_libraries(jtts_sano_demo PRIVATE jtts) target_compile_options(jtts_sano_demo PRIVATE -Wall -Wextra) + +# フォルマント エンジンの口形イベント (リップシンク用) +add_executable(jtts_test_visemes test_visemes.cpp) +target_link_libraries(jtts_test_visemes PRIVATE jtts) +target_compile_options(jtts_test_visemes PRIVATE -Wall -Wextra) + +# HMM 長文分割 (メモリ予算に応じたチャンク分割合成) +add_executable(jtts_test_hmm_chunk test_hmm_chunk.cpp) +target_link_libraries(jtts_test_hmm_chunk PRIVATE jtts) +target_include_directories(jtts_test_hmm_chunk PRIVATE ${JTTS_ROOT}/src) +target_compile_options(jtts_test_hmm_chunk PRIVATE -Wall -Wextra) + +# 句ごとの合成 (句の分割 / 前後の無音の切り詰め / 句間の無音 / フォルマント / sanoTTS の口形)。重み不要。 +add_executable(jtts_test_clause_stream test_clause_stream.cpp) +target_link_libraries(jtts_test_clause_stream PRIVATE jtts saanotts) +target_include_directories(jtts_test_clause_stream PRIVATE ${JTTS_ROOT}/src) +target_compile_options(jtts_test_clause_stream PRIVATE -Wall -Wextra) diff --git a/components/jtts/test/host/test_clause_stream.cpp b/components/jtts/test/host/test_clause_stream.cpp new file mode 100644 index 0000000..9c0acf5 --- /dev/null +++ b/components/jtts/test/host/test_clause_stream.cpp @@ -0,0 +1,501 @@ +// SPDX-FileCopyrightText: 2026 Kenta IDA +// SPDX-License-Identifier: BSL-1.0 +// +// 句ごとの合成 (split_clauses / trim_silence / stream_clauses = sanoTTS・フォルマント・単位連結で共通) の検証。 +// 重み (非 MIT の blob) が無くても動くよう、1 句の合成は「前後に無音のある合成音」の +// スタブに差し替える。 +#include +#include +#include +#include +#include +#include + +#include "internal.hpp" +#include "jtts/jtts.hpp" + +extern "C" { +#include "g2p_table.h" // 上流の仮名表 (kSaanG2pMora): 音素 ID の語彙と一致するか検証する +} + +using namespace stackchan::jtts; +using namespace stackchan::jtts::internal; + +namespace { + +int g_failures = 0; + +#define CHECK(cond) \ + do { \ + if (!(cond)) { \ + std::fprintf(stderr, "FAIL %s:%d: %s\n", __FILE__, __LINE__, #cond); \ + ++g_failures; \ + } \ + } while (0) + +constexpr std::uint32_t kRate = 22050; + +// ms → サンプル数 +std::size_t samples(double ms) { return static_cast(ms * kRate / 1000.0 + 0.5); } + +// 前に lead_ms、後ろに trail_ms の無音がある、body_ms の音 (振幅 8000 の矩形波) を作る。 +std::vector tone(double lead_ms, double body_ms, double trail_ms) +{ + std::vector v(samples(lead_ms), 0); + for (std::size_t i = 0; i < samples(body_ms); ++i) v.push_back((i / 20) % 2 ? 8000 : -8000); + v.insert(v.end(), samples(trail_ms), 0); + return v; +} + +// 先頭 / 末尾の無音のサンプル数 (絶対値 < 100 を無音とみなす)。 +std::size_t lead_silence(const std::vector& v) +{ + std::size_t n = 0; + while (n < v.size() && std::abs(v[n]) < 100) ++n; + return n; +} +std::size_t trail_silence(const std::vector& v) +{ + std::size_t n = 0; + while (n < v.size() && std::abs(v[v.size() - 1 - n]) < 100) ++n; + return n; +} + +void test_split_clauses() +{ + std::vector c; + CHECK(!split_clauses(U"", c)); + CHECK(!split_clauses(U"、。", c)); + + // 句読点の直後で切る。区切りは前の句に含め、続く句読点はまとめる。 + CHECK(split_clauses(U"あいう、えお。かき", c)); + CHECK(c.size() == 3); + CHECK(c[0].text == U"あいう、" && c[0].moras == 3 && c[0].pause_after); + CHECK(c[1].text == U"えお。" && c[1].moras == 2 && c[1].pause_after); + CHECK(c[2].text == U"かき" && c[2].moras == 2 && !c[2].pause_after); + + CHECK(split_clauses(U"あ、、い。。。う", c)); + CHECK(c.size() == 3 && c[0].text == U"あ、、" && c[1].text == U"い。。。" && c[2].text == U"う"); + + // 句読点が無ければ 1 句。アクセント記号 / アクセント句境界はそのまま句の中に残る。 + CHECK(split_clauses(U"こ'んにちは/ありが'とう", c)); + CHECK(c.size() == 1 && c[0].text == U"こ'んにちは/ありが'とう" && !c[0].pause_after); + + // 先頭の句読点は次の句の前に残り、末尾の記号は前の句に含まれる。全体の文字が失われない。 + CHECK(split_clauses(U"、あい、う。、", c)); + std::u32string joined; + for (const auto& x : c) joined += x.text; + CHECK(joined == U"、あい、う。、"); + CHECK(c.size() == 2 && c[0].text == U"、あい、" && c[1].text == U"う。、"); +} + +void test_trim_silence() +{ + auto v = tone(300, 200, 400); + trim_silence(v, kRate, /*lead=*/true, /*trail=*/true, /*keep_ms=*/10); + // 音の前後に 10 ms の余白だけ残す。 + CHECK(std::abs(static_cast(lead_silence(v)) - static_cast(samples(10))) <= 2); + CHECK(std::abs(static_cast(trail_silence(v)) - static_cast(samples(10))) <= 2); + CHECK(std::abs(static_cast(v.size()) - static_cast(samples(220))) <= 4); + + // 削った量が戻る (口形の切り詰めに使う)。 + { + auto w = tone(300, 200, 400); + const std::size_t before = w.size(); + const SilenceTrim cut = trim_silence(w, kRate, true, true, 10); + CHECK(cut.front + cut.back + w.size() == before); + CHECK(std::abs(static_cast(cut.front) - static_cast(samples(290))) <= 2); + CHECK(std::abs(static_cast(cut.back) - static_cast(samples(390))) <= 2); + } + + // lead / trail は独立。 + auto a = tone(300, 200, 400); + trim_silence(a, kRate, true, false); + CHECK(lead_silence(a) <= samples(10) + 2 && trail_silence(a) == samples(400)); + auto b = tone(300, 200, 400); + trim_silence(b, kRate, false, true); + CHECK(lead_silence(b) == samples(300) && trail_silence(b) <= samples(10) + 2); + + // 無音が余白より短ければ何もしない / 全体が無音なら触らない / 空も安全。 + auto c = tone(5, 100, 5); + const auto c0 = c; + trim_silence(c, kRate, true, true); + CHECK(c == c0); + std::vector z(1000, 0); + trim_silence(z, kRate, true, true); + CHECK(z.size() == 1000); + std::vector e; + trim_silence(e, kRate, true, true); + CHECK(e.empty()); + + // 小さな雑音は無音扱い (ピークの約 1 %)。 + auto n = tone(0, 100, 0); + n.insert(n.begin(), 300, 20); + n.insert(n.end(), 300, -20); + trim_silence(n, kRate, true, true, 0); + CHECK(n.size() == samples(100)); +} + +struct Got { + std::vector pcm; + std::uint32_t rate; + std::vector spans; + std::u32string text; +}; + +// 口形の区間の合計 [ms]。 +double total_ms(const std::vector& sp) +{ + double t = 0; + for (const auto& x : sp) t += x.duration_ms; + return t; +} + +void test_stream() +{ + Options opt; + opt.mora_ms = 110.0f; // 等速: 句間の無音は 420 ms + + // 各句は「前 300 ms / 音 200 ms / 後ろ 400 ms」を返すスタブ。 + int calls = 0; + // 口形は「前 300 ms 閉口 / 音 200 ms は あ / 後ろ 400 ms 閉口」。 + const ClauseSynth synth = [&](const std::u32string&, std::vector& pcm, std::uint32_t& rate, + std::vector& spans) { + ++calls; + pcm = tone(300, 200, 400); + rate = kRate; + spans = {{Vowel::None, 300.0f}, {Vowel::A, 200.0f}, {Vowel::None, 400.0f}}; + return ClauseResult::Ok; + }; + std::vector got; + const ClauseChunkFn collect = [&](std::vector&& pcm, std::uint32_t rate, + std::vector&& spans, const std::u32string& text) { + got.push_back({std::move(pcm), rate, std::move(spans), text}); + return true; + }; + + CHECK(stream_clauses(U"あいう、えお。かき", opt, synth, collect, true) == StreamOutcome::Ok); + CHECK(calls == 3 && got.size() == 3); + std::u32string joined; + for (const auto& g : got) joined += g.text; + CHECK(joined == U"あいう、えお。かき"); // 句の読みが欠けずに順に渡る + for (const auto& g : got) CHECK(g.rate == kRate); + + // 先頭の句: 前の無音 (300 ms) は残し、後ろは切り詰めて 420 ms の無音に置き換える。 + CHECK(lead_silence(got[0].pcm) == samples(300)); + CHECK(std::abs(static_cast(trail_silence(got[0].pcm)) - static_cast(samples(420 + 10))) <= 4); + // 中の句: 前後とも切り詰め (前は 10 ms の余白だけ)、後ろに 420 ms。 + CHECK(lead_silence(got[1].pcm) <= samples(10) + 2); + CHECK(std::abs(static_cast(trail_silence(got[1].pcm)) - static_cast(samples(420 + 10))) <= 4); + // 最後の句: 前は切り詰め、後ろは切り詰めない (従来どおり) & 無音も足さない。 + CHECK(lead_silence(got[2].pcm) <= samples(10) + 2); + CHECK(trail_silence(got[2].pcm) == samples(400)); + // 句の間の実際の無音 (前の句の後ろ + 次の句の前) は HMM の pau に近い ≈ 0.42〜0.45 s。 + const double gap_ms = 1000.0 * static_cast(trail_silence(got[0].pcm) + lead_silence(got[1].pcm)) / kRate; + CHECK(gap_ms > 430 && gap_ms < 460); + + // 話速 (mora_ms) に比例して句間の無音が伸び縮みする (HMM の pau と同じ写像)。 + Options slow = opt; + slow.mora_ms = 220.0f; + got.clear(); + CHECK(stream_clauses(U"あ、い", slow, synth, collect, true) == StreamOutcome::Ok); + CHECK(std::abs(static_cast(trail_silence(got[0].pcm)) - static_cast(samples(840 + 10))) <= 4); + Options fast = opt; + fast.mora_ms = 55.0f; + got.clear(); + CHECK(stream_clauses(U"あ、い", fast, synth, collect, true) == StreamOutcome::Ok); + CHECK(std::abs(static_cast(trail_silence(got[0].pcm)) - static_cast(samples(210 + 10))) <= 4); + + // 句が 1 つだけ (句読点なし) なら何も切り詰めず、無音も足さない = 従来どおりの 1 チャンク。 + got.clear(); + CHECK(stream_clauses(U"こんにちは", opt, synth, collect, true) == StreamOutcome::Ok); + CHECK(got.size() == 1 && got[0].pcm == tone(300, 200, 400)); + + // trim_edges=false (フォルマント / 単位連結): 各句の前後の無音はそのまま、句の後ろに間だけ足す。 + got.clear(); + CHECK(stream_clauses(U"あ、い、う", opt, synth, collect, /*trim_edges=*/false) == StreamOutcome::Ok); + CHECK(got.size() == 3); + CHECK(lead_silence(got[0].pcm) == samples(300) && lead_silence(got[1].pcm) == samples(300)); + CHECK(std::abs(static_cast(trail_silence(got[0].pcm)) - static_cast(samples(400 + 420))) <= 4); + CHECK(trail_silence(got[2].pcm) == samples(400)); // 最後の句には足さない + CHECK(std::abs(total_ms(got[0].spans) - (300.0 + 200.0 + 400.0 + 420.0)) < 1e-3); // 口形は PCM と同じ長さ + CHECK(got[0].spans.back().vowel == Vowel::None); + + // 中断: emit が false を返したらそこで止まる。 + got.clear(); + calls = 0; + int n = 0; + const ClauseChunkFn stop_after_first = [&](std::vector&&, std::uint32_t, + std::vector&&, const std::u32string&) { return ++n < 1; }; + CHECK(stream_clauses(U"あ、い、う", opt, synth, stop_after_first, true) == StreamOutcome::Cancelled); + CHECK(calls == 1); + + // 最初の句の失敗 = 何も出さず NoOutput (呼び出し側が他エンジンへ)。途中の失敗 = Aborted。 + const ClauseSynth fail_first = [&](const std::u32string&, std::vector&, std::uint32_t&, + std::vector&) { return ClauseResult::Fail; }; + got.clear(); + CHECK(stream_clauses(U"あ、い", opt, fail_first, collect, true) == StreamOutcome::NoOutput && got.empty()); + int k = 0; + const ClauseSynth fail_second = [&](const std::u32string& c, std::vector& pcm, + std::uint32_t& rate, std::vector& sp) { + return ++k == 2 ? ClauseResult::Fail : synth(c, pcm, rate, sp); + }; + got.clear(); + CHECK(stream_clauses(U"あ、い、う", opt, fail_second, collect, true) == StreamOutcome::Aborted && got.size() == 1); + + // Skip (読める内容が無い句) は飛ばして続ける。全部 Skip なら NoOutput。 + k = 0; + const ClauseSynth skip_second = [&](const std::u32string& c, std::vector& pcm, + std::uint32_t& rate, std::vector& sp) { + return ++k == 2 ? ClauseResult::Skip : synth(c, pcm, rate, sp); + }; + got.clear(); + CHECK(stream_clauses(U"あ、い、う", opt, skip_second, collect, true) == StreamOutcome::Ok && got.size() == 2); + const ClauseSynth skip_all = [&](const std::u32string&, std::vector&, std::uint32_t&, + std::vector&) { return ClauseResult::Skip; }; + got.clear(); + CHECK(stream_clauses(U"あ、い", opt, skip_all, collect, true) == StreamOutcome::NoOutput && got.empty()); + + // 読める内容が無い入力は NoOutput。 + CHECK(stream_clauses(U"、。", opt, synth, collect, true) == StreamOutcome::NoOutput); + + // 重みが無い環境 (このテスト) では、本物の render_sano_stream は NoOutput でフォールバックさせる。 + got.clear(); + CHECK(render_sano_stream(U"あ、い", opt, collect) == StreamOutcome::NoOutput && got.empty()); + std::vector out; + std::uint32_t rate = 0; + CHECK(!render_sano(U"あ、い", out, opt, rate) && out.empty()); +} + +void test_crop_spans() +{ + std::vector sp = {{Vowel::None, 100.0f}, {Vowel::A, 200.0f}, {Vowel::None, 300.0f}}; + auto a = sp; + crop_spans(a, 0.0f, 600.0f); // 何も削らない + CHECK(a.size() == 3 && total_ms(a) == 600.0); + a = sp; + crop_spans(a, 150.0f, 100.0f); // あ の途中から 100 ms + CHECK(a.size() == 1 && a[0].vowel == Vowel::A && std::abs(a[0].duration_ms - 100.0f) < 1e-3); + a = sp; + crop_spans(a, 250.0f, 1000.0f); // 先頭を落とし、後ろは残り全部 + CHECK(a.size() == 2 && a[0].vowel == Vowel::A && std::abs(a[0].duration_ms - 50.0f) < 1e-3 && + std::abs(total_ms(a) - 350.0) < 1e-3); + a = sp; + crop_spans(a, 50.0f, 300.0f); // 閉口の一部 + あ + 閉口の一部 + CHECK(a.size() == 3 && std::abs(a[0].duration_ms - 50.0f) < 1e-3 && std::abs(a[2].duration_ms - 50.0f) < 1e-3); +} + +// 音素 ID (saan_g2p の出力: ^ PAD 音素 PAD 音素 ... $) → 口形。1 フレーム = 10 ms として、d_hat を並べる。 +std::vector spans_of(const std::vector& ids, const std::vector& d) +{ + std::vector out; + std::vector i32(ids.begin(), ids.end()), d32(d.begin(), d.end()); + sano_ids_to_spans(i32.data(), d32.data(), static_cast(ids.size()), 10.0f, out); + return out; +} + +std::string shape(const std::vector& sp) +{ + std::string s; + for (const auto& x : sp) { + const char* n = "-aiueo"; + s += n[static_cast(x.vowel)]; + } + return s; +} + +void test_ids_to_spans() +{ + constexpr int PAD = 0, BOS = 1, EOS = 2, A = 10, I = 11, U = 12, E = 13, O = 14; + constexpr int K = 25, M = 51, B = 37, N = 20, CL = 24, S = 41, Y = 56; + + // 「あいうえお」: ^ _ a _ i _ u _ e _ o _ $ + auto v = spans_of({BOS, PAD, A, PAD, I, PAD, U, PAD, E, PAD, O, PAD, EOS}, {2, 3, 10, 2, 10, 2, 10, 2, 10, 2, 10, 2, 5}); + CHECK(shape(v) == "-aiueo-"); + // 音素の後ろの PAD (余白) は直前の母音に含まれ、先頭の ^ _ と末尾の $ は閉口。 + CHECK(std::abs(v[0].duration_ms - 50.0f) < 1e-3); // (2 + 3) × 10 + CHECK(std::abs(v[1].duration_ms - 120.0f) < 1e-3); // a: 10 + 2 (余白) + CHECK(std::abs(v.back().duration_ms - 50.0f) < 1e-3); // 末尾の $ (5 フレーム) + CHECK(std::abs(total_ms(v) - 10.0 * (2 + 3 + 10 + 2 + 10 + 2 + 10 + 2 + 10 + 2 + 10 + 2 + 5)) < 1e-3); + + // 子音は直後の母音の形を先取り (か: k a)。 + v = spans_of({BOS, PAD, K, PAD, A, PAD, EOS}, {2, 2, 5, 1, 10, 1, 4}); + CHECK(shape(v) == "-a-"); + CHECK(std::abs(v[1].duration_ms - 170.0f) < 1e-3); // k 5 + PAD 1 + a 10 + PAD 1 = 17 フレーム + + // 両唇音 (ま: m a / ば: b a) は閉口 → 母音。 + v = spans_of({BOS, PAD, M, PAD, A, PAD, EOS}, {2, 2, 5, 1, 10, 1, 4}); + CHECK(shape(v) == "-a-"); + v = spans_of({BOS, PAD, B, PAD, I, PAD, EOS}, {2, 2, 5, 1, 10, 1, 4}); + CHECK(shape(v) == "-i-"); + CHECK(std::abs(v[0].duration_ms - (2 + 2 + 5 + 1) * 10.0f) < 1e-3); // 両唇音 (と余白) までは閉口 + + // 拗音 (きゃ = ky a) は a の形、よ (y o) は o。y + 母音。 + v = spans_of({BOS, PAD, Y, PAD, O, PAD, EOS}, {2, 2, 4, 1, 10, 1, 4}); + CHECK(shape(v) == "-o-"); + + // ん / っ は閉口 (前後の母音の間で口が閉じる)。 + v = spans_of({BOS, PAD, A, PAD, N, PAD, A, PAD, EOS}, {2, 2, 8, 1, 6, 1, 8, 1, 4}); + CHECK(shape(v) == "-a-a-"); + v = spans_of({BOS, PAD, A, PAD, CL, PAD, K, PAD, A, PAD, EOS}, {2, 2, 8, 1, 6, 1, 4, 1, 8, 1, 4}); + CHECK(shape(v) == "-a-a-"); + + // ポーズ (PAD が 2 つ続く 2 つ目以降) は閉口。 + v = spans_of({BOS, PAD, A, PAD, PAD, S, PAD, U, PAD, EOS}, {2, 2, 8, 1, 20, 6, 1, 8, 1, 4}); + CHECK(shape(v) == "-a-u-"); + CHECK(std::abs(v[2].duration_ms - 200.0f) < 1e-3); // ポーズの 20 フレームだけが閉口 + + // 無声化母音 (15..19) は閉口。 + v = spans_of({BOS, PAD, 15, PAD, EOS}, {2, 2, 8, 1, 4}); + CHECK(shape(v) == "-"); + + // マーク [ ] # ? は余白として直前の形を保つ (a [ ] → a のまま)。 + v = spans_of({BOS, PAD, A, PAD, 8, PAD, 9, PAD, I, PAD, EOS}, {2, 2, 6, 1, 1, 1, 1, 1, 6, 1, 4}); + CHECK(shape(v) == "-ai-"); + CHECK(std::abs(v[1].duration_ms - (6 + 1 + 1 + 1 + 1 + 1) * 10.0f) < 1e-3); + + // 空 / null は空。 + CHECK(spans_of({}, {}).empty()); + std::vector none; + sano_ids_to_spans(nullptr, nullptr, 3, 10.0f, none); + CHECK(none.empty()); +} + +// 口形の規則が使う音素 ID (母音 / ん / っ / 両唇音) が、上流の仮名表 (語彙) と一致する。 +// 上流の語彙が変わると (g2p_table.h の更新) ここが落ちる。 +void test_vocab_matches_upstream_table() +{ + const auto lookup = [](char32_t a, char32_t b = 0) -> const saan_g2p_mora* { + const auto c1 = static_cast(a - SAAN_G2P_KANA_BASE); + const auto c2 = static_cast(b ? b - SAAN_G2P_KANA_BASE : 0); + for (const auto& m : kSaanG2pMora) { + if (m.c1 == c1 && m.c2 == c2) return &m; + } + return nullptr; + }; + // 母音: あいうえお = 10..14 (無声化 +5 = 15..19)。 + constexpr char32_t kVowels[] = {U'あ', U'い', U'う', U'え', U'お'}; + for (int i = 0; i < 5; ++i) { + const auto* m = lookup(kVowels[i]); + CHECK(m != nullptr && m->p0 == 10 + i && m->p1 < 0); + } + CHECK(SAAN_G2P_VOWEL_LO == 10 && SAAN_G2P_VOWEL_HI == 14 && SAAN_G2P_DEVOICE_STEP == 5); + CHECK(SAAN_G2P_ID_PAD == 0 && SAAN_G2P_ID_BOS == 1 && SAAN_G2P_ID_EOS == 2); + + // 両唇音 (ま行 / ば行 / ぱ行、拗音を含む) の第 1 音素は、口形の規則で閉口 (35 36 37 38 51 52)。 + const std::vector bilabial = {35, 36, 37, 38, 51, 52}; + const auto is_bilabial = [&](int id) { return std::find(bilabial.begin(), bilabial.end(), id) != bilabial.end(); }; + constexpr char32_t kBilabialKana[] = {U'ま', U'み', U'む', U'め', U'も', U'ば', U'び', U'ぶ', U'べ', U'ぼ', + U'ぱ', U'ぴ', U'ぷ', U'ぺ', U'ぽ'}; + for (char32_t k : kBilabialKana) { + const auto* m = lookup(k); + CHECK(m != nullptr && is_bilabial(m->p0)); + } + for (char32_t k : {U'み', U'び', U'ぴ'}) { + for (char32_t y : {U'ゃ', U'ゅ', U'ょ'}) { + const auto* m = lookup(k, y); + CHECK(m != nullptr && is_bilabial(m->p0)); + } + } + // 逆に、両唇音でない子音 (か さ た な は ら わ や) は閉口扱いにならない。 + for (char32_t k : {U'か', U'さ', U'た', U'な', U'は', U'ら', U'わ', U'や', U'ふ', U'が', U'ざ', U'だ'}) { + const auto* m = lookup(k); + CHECK(m != nullptr && !is_bilabial(m->p0)); + } + // っ = 24 (閉口)、ん = 20..23 の異音 (N_m = 20)。 + const auto* cl = lookup(U'っ'); + CHECK(cl != nullptr && cl->p0 == 24 && cl->p1 < 0); + const auto* nn = lookup(U'ん'); + CHECK(nn != nullptr && nn->p0 == 20); + // 「ん」の異音表 (後続の音素ごと) の値は全て 20..23 (口形の規則では ん = 閉口)。 + for (const auto id : kSaanG2pNAllophone) CHECK(id >= 20 && id <= 23); +} + +// フォルマント合成 (実物): 句読点で句に分かれ、句の間に HMM の pau と同じ長さの無音が入る。 +void test_formant_clauses() +{ + Options opt; + opt.engine = Engine::Formant; + opt.mora_ms = 110.0f; // 句間の無音は 420 ms + const auto rate = opt.sample_rate_hz; + + // 句が 1 つ: 従来どおり 1 チャンク・間なし。一括版と同じ PCM。 + std::vector one; + auto r = synthesize_stream( + U"こんにちは", [&](SynthChunk&& c) { one.push_back(std::move(c)); return true; }, opt); + CHECK(r.has_value() && one.size() == 1 && !one[0].pcm.empty() && !one[0].visemes.empty()); + CHECK(one[0].text == U"こんにちは" && one[0].sample_rate == rate); + std::vector whole; + std::vector whole_ev; + CHECK(synthesize(U"こんにちは", whole, whole_ev, opt).has_value()); + CHECK(whole == one[0].pcm); + + // 句が 3 つ: 3 チャンク。間は最後を除く各チャンクの末尾に入る。 + const std::u32string text = U"あいうえお、かきくけこ。さしすせそ"; + std::vector ch; + r = synthesize_stream(text, [&](SynthChunk&& c) { ch.push_back(std::move(c)); return true; }, opt); + CHECK(r.has_value() && ch.size() == 3); + std::u32string joined; + for (const auto& c : ch) joined += c.text; + CHECK(joined == text); + const double pause = clause_pause_ms(opt.mora_ms); + for (std::size_t i = 0; i < ch.size(); ++i) { + const double tail_ms = 1000.0 * static_cast(trail_silence(ch[i].pcm)) / rate; + if (i + 1 < ch.size()) { + CHECK(tail_ms >= pause - 1 && tail_ms < pause + 60); + } else { + CHECK(tail_ms < 60); // 最後の句の後ろには足さない + } + // 口形の時間軸は PCM と一致し、間の部分は閉口。 + CHECK(!ch[i].visemes.empty()); + } + CHECK(!ch[0].visemes.empty() && ch[0].visemes.back().vowel == Vowel::None); + + // 一括版 = チャンクの連結 (口形は各チャンクの先頭に時間をずらして並ぶ)。 + std::vector cat; + for (const auto& c : ch) cat.insert(cat.end(), c.pcm.begin(), c.pcm.end()); + std::vector ev; + CHECK(synthesize(text, whole, ev, opt).has_value()); + CHECK(whole == cat); + CHECK(!ev.empty() && ev.front().start_ms == 0); + // 口形の時間軸は句間の無音を含めた PCM 全体と一致する (末尾は発話の終わりに置く閉口イベント)。 + const double total = 1000.0 * static_cast(whole.size()) / rate; + CHECK(ev.back().vowel == Vowel::None && std::abs(static_cast(ev.back().start_ms) - total) < 2.0); + + // 話速が遅いと間も伸びる (HMM の pau と同じ写像)。 + Options slow = opt; + slow.mora_ms = 220.0f; + std::vector sc; + CHECK(synthesize_stream(U"あ、い", [&](SynthChunk&& c) { sc.push_back(std::move(c)); return true; }, slow) + .has_value()); + CHECK(sc.size() == 2); + CHECK(1000.0 * static_cast(trail_silence(sc[0].pcm)) / rate >= clause_pause_ms(220.0f) - 1); + + // 読めない句 (かな以外だけ) は飛ばし、句読点だけの入力は InvalidKana。 + std::vector sk; + CHECK(synthesize_stream(U"あ、xyz、い", [&](SynthChunk&& c) { sk.push_back(std::move(c)); return true; }, opt) + .has_value()); + CHECK(sk.size() == 2); + auto bad = synthesize_stream(U"、。", [&](SynthChunk&&) { return true; }, opt); + CHECK(!bad.has_value() && bad.error() == Error::InvalidKana); + + // 中断。 + std::size_t n = 0; + auto cancelled = synthesize_stream(text, [&](SynthChunk&&) { return ++n < 2; }, opt); + CHECK(!cancelled.has_value() && cancelled.error() == Error::Cancelled && n == 2); +} + +} // namespace + +int main() +{ + test_split_clauses(); + test_trim_silence(); + test_crop_spans(); + test_ids_to_spans(); + test_vocab_matches_upstream_table(); + test_stream(); + test_formant_clauses(); + if (g_failures == 0) std::puts("test_clause_stream: all passed"); + return g_failures == 0 ? 0 : 1; +} diff --git a/components/jtts/test/host/test_hmm_chunk.cpp b/components/jtts/test/host/test_hmm_chunk.cpp new file mode 100644 index 0000000..839686a --- /dev/null +++ b/components/jtts/test/host/test_hmm_chunk.cpp @@ -0,0 +1,489 @@ +// SPDX-FileCopyrightText: 2026 Kenta IDA +// SPDX-License-Identifier: BSL-1.0 +// +// HMM 合成の長文分割 (split_hmm_text / 分割合成 / メモリ不足フォールバック) の検証。 +// jtts_test_hmm_chunk [voice.htsvoice ...] +// 分割ロジックは常に検証する。.htsvoice を渡すと、それぞれをロードして、メモリ予算を +// 絞った分割合成も検証する (例: assets/voices/mei16.htsvoice)。 +#include +#include +#include +#include +#include +#include +#include +#include + +#include "internal.hpp" +#include "jtts/jtts.hpp" +#include "jtts/subtitle.hpp" + +using namespace stackchan::jtts; + +namespace { + +int g_failures = 0; + +#define CHECK(cond) \ + do { \ + if (!(cond)) { \ + std::fprintf(stderr, "FAIL %s:%d: %s\n", __FILE__, __LINE__, #cond); \ + ++g_failures; \ + } \ + } while (0) + +std::u32string join(const std::vector& chunks) +{ + std::u32string s; + for (const auto& c : chunks) s += c.text; + return s; +} + +std::u32string strip_accent(std::u32string s) +{ + s.erase(std::remove(s.begin(), s.end(), U'\''), s.end()); + return s; +} + +void test_split() +{ + using internal::HmmChunk; + using internal::split_hmm_text; + std::vector ch; + + // 収まるなら手を加えず 1 チャンク (アクセント記号もそのまま)。 + CHECK(split_hmm_text(U"こ'んにちは", 100, ch)); + CHECK(ch.size() == 1 && ch[0].text == U"こ'んにちは" && ch[0].moras == 5); + + // 不正入力。 + CHECK(!split_hmm_text(U"こんにちは", 0, ch)); + CHECK(!split_hmm_text(U"、。", 10, ch)); + CHECK(!split_hmm_text(U"", 10, ch)); + + // 句読点で分け、予算に収まる範囲で貪欲に詰める。区切りは前のチャンクに残る。 + CHECK(split_hmm_text(U"あいう、えお。かき", 5, ch)); + CHECK(ch.size() == 2); + CHECK(ch[0].text == U"あいう、えお。" && ch[0].moras == 5 && ch[0].pause_after); + CHECK(ch[1].text == U"かき" && ch[1].moras == 2 && !ch[1].pause_after); + CHECK(join(ch) == U"あいう、えお。かき"); + + CHECK(split_hmm_text(U"あいう、えお。かき", 3, ch)); + CHECK(ch.size() == 3); + CHECK(ch[0].text == U"あいう、" && ch[0].pause_after); + CHECK(ch[1].text == U"えお。" && ch[1].pause_after); + CHECK(ch[2].text == U"かき"); + + // アクセント句境界 (/) ではポーズなし。 + CHECK(split_hmm_text(U"あいう/えお", 3, ch)); + CHECK(ch.size() == 2 && ch[0].text == U"あいう/" && !ch[0].pause_after && ch[1].text == U"えお"); + + // 句読点だけの句は前後に吸収され、モーラを持たないチャンクは作らない。 + CHECK(split_hmm_text(U"あい、、、うえ", 2, ch)); + for (const auto& c : ch) CHECK(c.moras > 0 && c.moras <= 2); + CHECK(join(ch) == U"あい、、、うえ"); + + // 句が単体で予算を超えるとモーラ境界で強制分割する。各チャンクは予算内。 + const std::u32string flat = U"あいうえおかきくけこさしすせそたちつてとなにぬねの"; + CHECK(split_hmm_text(flat, 8, ch)); + CHECK(ch.size() == 4); + for (const auto& c : ch) CHECK(c.moras <= 8 && c.moras > 0); + CHECK(join(ch) == flat); + + // 拗音・長音・アクセント核の直前では切らない。強制分割ではアクセント核を落とす。 + const std::u32string youon = U"きゃきゅきょきゃきゅきょきゃきゅきょ"; + CHECK(split_hmm_text(youon, 4, ch)); + for (const auto& c : ch) { + CHECK(c.moras <= 4); + CHECK(c.text.front() != U'ゃ' && c.text.front() != U'ゅ' && c.text.front() != U'ょ'); + } + CHECK(join(ch) == youon); + CHECK(split_hmm_text(U"あ'いう'えおかきくけこ", 4, ch)); + for (const auto& c : ch) CHECK(c.text.find(U'\'') == std::u32string::npos); + CHECK(join(ch) == strip_accent(U"あ'いう'えおかきくけこ")); + + // 実際の長文 (146 文字) が予算 20 モーラで全て収まる。 + const std::u32string longtext = + U"すたっくちゃんは、ちいさくてかわいい、てのひらさいずのろぼっとです。" + U"かおのひょうじをかえたり、くびをうごかしたり、おしゃべりしたりできます。" + U"じぶんでぷろぐらむをつくって、いろいろなことをさせられるのも、たのしいところです。" + U"つくるひとによって、いろいろなこせいがうまれる、たのしいろぼっとです。"; + CHECK(split_hmm_text(longtext, 20, ch)); + CHECK(ch.size() > 4); + for (const auto& c : ch) CHECK(c.moras <= 20 && c.moras > 0); + CHECK(join(ch) == longtext); + // 句読点で終わるチャンクが大半 (強制分割は不要な長さの句ばかり)。 + std::size_t pauses = 0; + for (const auto& c : ch) pauses += c.pause_after ? 1 : 0; + CHECK(pauses + 1 >= ch.size()); + + // 低遅延モード (first_moras > 0): 先頭を小さく出し、句の途中では切らない。 + CHECK(split_hmm_text(longtext, 23, ch, 14)); + CHECK(join(ch) == longtext); + CHECK(ch.front().moras <= 14); + CHECK(ch.size() >= 8); + for (std::size_t i = 0; i < ch.size(); ++i) { + CHECK(ch[i].moras > 0 && ch[i].moras <= 23); + CHECK(ch[i].pause_after); // 全て句読点で終わる = 句の途中では切っていない + // 直前のチャンクの 1.3 倍を超えて急に大きくならない (先頭 14 モーラ以内は除く)。 + if (i > 0) CHECK(ch[i].moras <= std::max(14, (ch[i - 1].moras * 13 + 9) / 10) || + ch[i].moras <= ch[i - 1].moras + 4); + } + // 句読点の無い短文は 1 チャンクのまま (原文どおり)。 + CHECK(split_hmm_text(U"こんにちは、", 23, ch, 14)); + CHECK(ch.size() == 1 && ch[0].text == U"こんにちは、"); + CHECK(split_hmm_text(U"こ'んにちはありが'とうございま'す", 23, ch, 14)); + CHECK(ch.size() == 1 && ch[0].text == U"こ'んにちはありが'とうございま'す"); + // 全体が予算内でも、句読点があれば先頭を早く出すために分ける。 + CHECK(split_hmm_text(U"あいうえお、かきくけこさし、たちつてと", 23, ch, 14)); + CHECK(ch.size() == 2 && ch[0].moras == 12 && ch[1].moras == 5); + // 非低遅延 (first_moras = 0) では従来どおり 1 チャンク。 + CHECK(split_hmm_text(U"あいうえお、かきくけこさし、たちつてと", 23, ch)); + CHECK(ch.size() == 1); +} + +// 吹き出しをチャンクに対応させる。 +void test_subtitle() +{ + // 句読点ごとに対応する。 + { + SubtitleMapper m("スタックチャンは、小さくて、かわいい。ロボットです。", + U"すたっくちゃんは、ちいさくて、かわいい。ろぼっとです。"); + CHECK(m.mapped()); + CHECK(m.next(U"すたっくちゃんは、") == "スタックチャンは、"); + CHECK(m.next(U"ちいさくて、かわいい。") == "小さくて、かわいい。"); + CHECK(m.next(U"ろぼっとです。") == "ロボットです。"); + } + // 1 チャンクが複数の句を受け持つ。 + { + SubtitleMapper m("A、B、C。", U"あ、い、う。"); + CHECK(m.mapped()); + CHECK(m.next(U"あ、い、") == "A、B、"); + CHECK(m.next(U"う。") == "C。"); + } + // 句の途中で切れたチャンク (強制分割) は、同じ句を続けて返す。 + { + SubtitleMapper m("あいうえおかきくけこ、さしすせそ。", U"あいうえおかきくけこ、さしすせそ。"); + CHECK(m.next(U"あいうえお") == "あいうえおかきくけこ、"); + CHECK(m.next(U"かきくけこ、") == "あいうえおかきくけこ、"); + CHECK(m.next(U"さしすせそ。") == "さしすせそ。"); + } + // アクセント記号・空白が付いた読み。 + { + SubtitleMapper m("今日は。天気。", U"きょ'うは。 てんき。"); + CHECK(m.next(U"きょ'うは。 ") == "今日は。"); + CHECK(m.next(U"てんき。") == "天気。"); + } + // 句読点が無ければ全体を 1 つの句として返す。 + { + SubtitleMapper m("こんにちは", U"こんにちは"); + CHECK(m.mapped() && m.next(U"こんにちは") == "こんにちは"); + } + // 対応が取れない (句読点の数が違う / 表示が空): 最初に全文、以降は空。 + for (const auto* d : {"こんにちは!げんき?", ""}) { + SubtitleMapper m(d, U"こんにちは、げんき?"); + CHECK(!m.mapped()); + CHECK(m.next(U"こんにちは、") == d); + CHECK(m.next(U"げんき?").empty()); + } +} + +std::string vowels_only(const std::vector& ev) +{ + std::string s; + for (const auto& e : ev) { + switch (e.vowel) { + case Vowel::A: s.push_back('a'); break; + case Vowel::I: s.push_back('i'); break; + case Vowel::U: s.push_back('u'); break; + case Vowel::E: s.push_back('e'); break; + case Vowel::O: s.push_back('o'); break; + case Vowel::None: break; + } + } + return s; +} + +// 母音列の編集距離 (Levenshtein)。 +std::size_t edit_distance(const std::string& a, const std::string& b) +{ + std::vector prev(b.size() + 1), cur(b.size() + 1); + for (std::size_t j = 0; j <= b.size(); ++j) prev[j] = j; + for (std::size_t i = 1; i <= a.size(); ++i) { + cur[0] = i; + for (std::size_t j = 1; j <= b.size(); ++j) { + cur[j] = std::min({prev[j] + 1, cur[j - 1] + 1, prev[j - 1] + (a[i - 1] != b[j - 1] ? 1u : 0u)}); + } + std::swap(prev, cur); + } + return prev[b.size()]; +} + +// synthesize_stream: チャンクごとに渡され、最初のチャンクが早く出て、連結すると +// 一括合成と同じ発話になる。 +void test_stream(const std::u32string& longtext, const Options& opt, const std::string& ref_vowels, double ref_ms) +{ + internal::set_hmm_memory_budget_for_test(0); + std::vector got; + auto r = synthesize_stream( + longtext, + [&](SynthChunk&& c) { + got.push_back(std::move(c)); + return true; + }, + opt); + CHECK(r.has_value()); + std::printf(" stream: %zu chunks", got.size()); + CHECK(got.size() >= 6); + + // 連結: PCM の長さと母音列 (チャンク先頭からの口形を積算して並べ直す)。 + std::size_t total_samples = 0; + std::string vowels; + for (const auto& c : got) { + CHECK(!c.pcm.empty()); + total_samples += c.pcm.size(); + CHECK(!c.visemes.empty() && c.visemes.front().start_ms == 0); + CHECK(c.visemes.back().vowel == Vowel::None); + for (std::size_t i = 1; i < c.visemes.size(); ++i) CHECK(c.visemes[i].start_ms > c.visemes[i - 1].start_ms); + vowels += vowels_only(c.visemes); + } + const double ms = 1000.0 * static_cast(total_samples) / opt.sample_rate_hz; + const double first_ms = 1000.0 * static_cast(got.front().pcm.size()) / opt.sample_rate_hz; + std::printf(", total %.0f ms (%+.1f%%), first chunk %.0f ms\n", ms, (ms - ref_ms) * 100.0 / ref_ms, first_ms); + CHECK(vowels == ref_vowels); + CHECK(std::abs(ms - ref_ms) < 0.05 * ref_ms); + CHECK(first_ms < 0.2 * ms && first_ms < 3000.0); // 低遅延: 先頭は全体の 1/5 未満・3 秒未満 + // 各チャンクの口形は自分の PCM の長さに収まる。 + for (const auto& c : got) { + const double len = 1000.0 * static_cast(c.pcm.size()) / opt.sample_rate_hz; + CHECK(static_cast(c.visemes.back().start_ms) <= len + 1.0); + } + + // 各チャンクの読み (text) を連結すると元の読みになり、それに合わせて表示テキストを + // 切り出すと、全チャンクで空でなく、連結すると表示テキスト全体になる。 + { + const std::string display = + "スタックチャンは、小さくてかわいい、手のひらサイズのロボットです。" + "顔の表情を変えたり、首を動かしたり、おしゃべりしたりできます。" + "自分でプログラムを作って、いろいろなことをさせられるのも、楽しいところです。" + "作る人によって、いろいろな個性が生まれる、楽しいロボットです。"; + SubtitleMapper m(display, longtext); + CHECK(m.mapped()); + std::u32string joined; + std::string shown; + for (const auto& c : got) { + joined += c.text; + const std::string t = m.next(c.text); + CHECK(!t.empty()); + shown += t; + } + CHECK(joined == longtext); + CHECK(shown == display); + } + + // sink が false を返すと中断: それ以降のチャンクは合成されない。 + int calls = 0; + r = synthesize_stream( + longtext, + [&](SynthChunk&&) { return ++calls < 2; }, + opt); + CHECK(!r.has_value() && r.error() == Error::Cancelled); + CHECK(calls == 2); + + // 全体の PCM が収まらない空き (400 KB) でも、HMM はチャンクごとなので発話できる。 + internal::set_pcm_memory_limit_for_test(400 * 1024); + std::size_t n = 0; + r = synthesize_stream( + longtext, + [&](SynthChunk&& c) { + ++n; + CHECK(c.pcm.size() * 2 < 400 * 1024); // 1 チャンクなら十分小さい + return true; + }, + opt); + CHECK(r.has_value() && n >= 6); + // 一括版は同じ条件だと断る (全体を 1 本の PCM に持つため)。 + std::vector whole; + auto rw = synthesize(longtext, whole, opt); + CHECK(!rw.has_value() && rw.error() == Error::OutOfMemory); + internal::set_pcm_memory_limit_for_test(0); + + // 予算を絞っても (チャンクがさらに小さくなるだけで) 最後まで発話できる。 + internal::set_hmm_memory_budget_for_test(std::size_t{1500} * 1024); + n = 0; + r = synthesize_stream( + longtext, + [&](SynthChunk&&) { + ++n; + return true; + }, + opt); + CHECK(r.has_value() && n >= 6); + + // 1 モーラも収まらない予算: 何も渡す前なので他エンジン (フォルマント) にフォールバック (句ごとのチャンク)。 + internal::set_hmm_memory_budget_for_test(64 * 1024); + got.clear(); + r = synthesize_stream( + longtext, + [&](SynthChunk&& c) { + got.push_back(std::move(c)); + return true; + }, + opt); + CHECK(r.has_value() && got.size() == 12 && !got[0].pcm.empty()); + internal::set_hmm_memory_budget_for_test(0); +} + +void test_hmm(const char* voice_path) +{ + std::ifstream f(voice_path, std::ios::binary); + static std::vector blob; // set_hmm_voice は blob の寿命を要求する + blob.assign(std::istreambuf_iterator(f), std::istreambuf_iterator()); + CHECK(set_hmm_voice(blob)); + std::printf("HMM voice: %s\n", voice_path); + + Options opt; + opt.engine = Engine::Hmm; + opt.mora_ms = 120.0f; // 実機の既定 + const std::u32string longtext = + U"すたっくちゃんは、ちいさくてかわいい、てのひらさいずのろぼっとです。" + U"かおのひょうじをかえたり、くびをうごかしたり、おしゃべりしたりできます。" + U"じぶんでぷろぐらむをつくって、いろいろなことをさせられるのも、たのしいところです。" + U"つくるひとによって、いろいろなこせいがうまれる、たのしいろぼっとです。"; + + // 無制限 (1 チャンク) の基準。 + internal::set_hmm_memory_budget_for_test(0); + std::vector ref_pcm; + std::vector ref_ev; + CHECK(synthesize(longtext, ref_pcm, ref_ev, opt).has_value()); + const std::string ref_vowels = vowels_only(ref_ev); + const double ref_ms = 1000.0 * static_cast(ref_pcm.size()) / opt.sample_rate_hz; + std::printf(" unchunked: %.0f ms, %zu vowels\n", ref_ms, ref_vowels.size()); + CHECK(!ref_vowels.empty()); + + // 予算を絞って分割合成: 落ちずに合成でき、母音列も長さも基準に近い。 + // 実機で想定する予算 (2 MB 前後) では句読点でだけ分かれ、母音列は完全に一致する。 + // 極端に小さい予算 (句の途中で強制分割される) では、切れ目で無声化の文脈が + // 失われて母音が数個ずれるのを許容する。 + struct Case { + std::size_t budget_kb; + std::size_t max_edit; // 母音列の許容編集距離 + double tol; // 長さの許容誤差 (割合) + }; + for (const Case c : {Case{2000, 0, 0.05}, Case{1500, 1, 0.05}, Case{1000, 4, 0.20}}) { + internal::set_hmm_memory_budget_for_test(c.budget_kb * 1024); + std::vector pcm; + std::vector ev; + CHECK(synthesize(longtext, pcm, ev, opt).has_value()); + const double ms = 1000.0 * static_cast(pcm.size()) / opt.sample_rate_hz; + const std::size_t dist = edit_distance(vowels_only(ev), ref_vowels); + std::printf(" budget %zu KB: %.0f ms (%+.1f%%), vowel edit distance %zu\n", c.budget_kb, ms, + (ms - ref_ms) * 100.0 / ref_ms, dist); + CHECK(dist <= c.max_edit); + CHECK(std::abs(ms - ref_ms) < c.tol * ref_ms); + + // 口形イベントは PCM と時間軸が揃う: 昇順、隣接は異なる、最後は PCM 長以内で閉口。 + CHECK(!ev.empty() && ev.front().start_ms == 0); + for (std::size_t i = 1; i < ev.size(); ++i) { + CHECK(ev[i].start_ms > ev[i - 1].start_ms); + CHECK(ev[i].vowel != ev[i - 1].vowel); + } + CHECK(ev.back().vowel == Vowel::None); + CHECK(static_cast(ev.back().start_ms) <= ms + 1.0); + CHECK(ms - static_cast(ev.back().start_ms) < 1000.0); + } + + // 母音の位置と実際の音: 分割合成でも各母音区間に音が出ている (無音のまま + // 母音イベントが立っていない)。全母音区間の RMS が背景 (先頭 sil) を上回る割合を見る。 + { + internal::set_hmm_memory_budget_for_test(std::size_t{800} * 1024); + std::vector pcm; + std::vector ev; + CHECK(synthesize(longtext, pcm, ev, opt).has_value()); + std::size_t voiced = 0, total = 0; + for (std::size_t i = 0; i + 1 < ev.size(); ++i) { + if (ev[i].vowel == Vowel::None) continue; + const double len = ev[i + 1].start_ms - ev[i].start_ms; + if (len < 40.0) continue; + const std::size_t a = static_cast((ev[i].start_ms + 0.25 * len) * 16.0); + const std::size_t b = std::min(pcm.size(), static_cast((ev[i].start_ms + 0.75 * len) * 16.0)); + double acc = 0; + for (std::size_t k = a; k < b; ++k) acc += static_cast(pcm[k]) * pcm[k]; + const double rms = std::sqrt(acc / static_cast(std::max(1, b - a))); + ++total; + if (rms > 300.0) ++voiced; + } + std::printf(" vowel spans with sound: %zu / %zu\n", voiced, total); + CHECK(total > 20 && voiced * 10 >= total * 9); + } + + test_stream(longtext, opt, ref_vowels, ref_ms); + + // 1 モーラも収まらない予算: HMM を諦めて (クラッシュせず) フォールバックする。 + internal::set_hmm_memory_budget_for_test(64 * 1024); + { + std::vector pcm; + std::vector ev; + CHECK(synthesize(longtext, pcm, ev, opt).has_value()); + CHECK(!pcm.empty()); // フォルマント合成で音が出る + } + internal::set_hmm_memory_budget_for_test(0); + set_hmm_voice({}); +} + +// 発話が長すぎて PCM が空きメモリに収まらないときは、どのエンジンでも合成せず +// OutOfMemory を返す (確保失敗によるクラッシュを避ける)。 +void test_pcm_limit() +{ + Options opt; + opt.engine = Engine::Formant; + std::vector pcm; + std::vector ev; + const std::u32string longtext = + U"すたっくちゃんは、ちいさくてかわいい、てのひらさいずのろぼっとです。" + U"かおのひょうじをかえたり、くびをうごかしたり、おしゃべりしたりできます。" + U"じぶんでぷろぐらむをつくって、いろいろなことをさせられるのも、たのしいところです。" + U"つくるひとによって、いろいろなこせいがうまれる、たのしいろぼっとです。"; + + // 空きが 400 KB: 短い発話は通り、長文 (PCM ≈ 1 MB) は断る。 + internal::set_pcm_memory_limit_for_test(400 * 1024); + CHECK(synthesize(U"こんにちは", pcm, ev, opt).has_value()); + CHECK(!pcm.empty()); + auto r = synthesize(longtext, pcm, ev, opt); + CHECK(!r.has_value() && r.error() == Error::OutOfMemory); + CHECK(pcm.empty() && ev.empty()); + // 3 引数版も同じ。 + r = synthesize(longtext, pcm, opt); + CHECK(!r.has_value() && r.error() == Error::OutOfMemory); + + // ストリーミングは句 (、。) ごとに 1 チャンク (この長文は 3 句 × 4 文 = 12) ずつ渡すので、 + // 長文でも 1 句が収まれば通る (一括版は全体を 1 本にするので断られる)。 + std::size_t chunks = 0; + std::size_t max_chunk_bytes = 0; + auto rs = synthesize_stream( + longtext, + [&](SynthChunk&& c) { + ++chunks; + max_chunk_bytes = std::max(max_chunk_bytes, c.pcm.size() * sizeof(std::int16_t)); + CHECK(!c.pcm.empty() && !c.visemes.empty() && c.visemes.front().start_ms == 0); + return true; + }, + opt); + CHECK(rs.has_value() && chunks == 12 && max_chunk_bytes < 400 * 1024); + + // 上限なし (ホスト既定) なら長文も一括で合成できる。 + internal::set_pcm_memory_limit_for_test(0); + CHECK(synthesize(longtext, pcm, ev, opt).has_value()); + CHECK(pcm.size() > 16000 * 10); // 10 秒以上 (500 KB 超: 上の 400 KB 制限では収まらない長さ) +} + +} // namespace + +int main(int argc, char** argv) +{ + test_split(); + test_subtitle(); + test_pcm_limit(); + for (int i = 1; i < argc; ++i) test_hmm(argv[i]); + if (g_failures == 0) std::puts("test_hmm_chunk: all passed"); + return g_failures == 0 ? 0 : 1; +} diff --git a/components/jtts/test/host/test_visemes.cpp b/components/jtts/test/host/test_visemes.cpp new file mode 100644 index 0000000..f384f15 --- /dev/null +++ b/components/jtts/test/host/test_visemes.cpp @@ -0,0 +1,195 @@ +// SPDX-FileCopyrightText: 2026 Kenta IDA +// SPDX-License-Identifier: BSL-1.0 +// +// 口形イベント (VisemeEvent) の検証。 +// jtts_test_visemes [voice.htsvoice ...] +// フォルマント エンジンのケースは常に実行する。.htsvoice を渡すと、それぞれを +// ロードして HMM エンジンのケースも実行する (例: assets/voices/mei16.htsvoice)。 +#include +#include +#include +#include +#include +#include +#include +#include + +#include "jtts/jtts.hpp" + +using namespace stackchan::jtts; + +namespace { + +int g_failures = 0; + +#define CHECK(cond) \ + do { \ + if (!(cond)) { \ + std::fprintf(stderr, "FAIL %s:%d: %s\n", __FILE__, __LINE__, #cond); \ + ++g_failures; \ + } \ + } while (0) + +char to_char(Vowel v) +{ + switch (v) { + case Vowel::A: return 'a'; + case Vowel::I: return 'i'; + case Vowel::U: return 'u'; + case Vowel::E: return 'e'; + case Vowel::O: return 'o'; + case Vowel::None: break; + } + return '-'; +} + +// 母音列を "aiueo-" のような文字列にする (None は '-')。 +std::string shape_of(std::u32string_view kana, std::vector& pcm, + std::vector& ev, const Options& opt) +{ + auto r = synthesize(kana, pcm, ev, opt); + CHECK(r.has_value()); + std::string s; + for (const auto& e : ev) s.push_back(to_char(e.vowel)); + return s; +} + +// 母音 (None を除く) だけを並べた文字列。 +std::string vowels_only(const std::vector& ev) +{ + std::string s; + for (const auto& e : ev) { + if (e.vowel != Vowel::None) s.push_back(to_char(e.vowel)); + } + return s; +} + +// [from_ms, to_ms) の PCM の RMS。 +double rms_between(const std::vector& pcm, std::uint32_t rate, double from_ms, double to_ms) +{ + const std::size_t a = static_cast(from_ms * rate / 1000.0); + const std::size_t b = std::min(pcm.size(), static_cast(to_ms * rate / 1000.0)); + if (b <= a) return 0.0; + double acc = 0.0; + for (std::size_t i = a; i < b; ++i) acc += static_cast(pcm[i]) * pcm[i]; + return std::sqrt(acc / static_cast(b - a)); +} + +void test_hmm(const char* voice_path) +{ + std::ifstream f(voice_path, std::ios::binary); + static std::vector blob; // set_hmm_voice は blob の寿命を要求する + blob.assign(std::istreambuf_iterator(f), std::istreambuf_iterator()); + CHECK(!blob.empty()); + CHECK(set_hmm_voice(blob)); + std::printf("HMM voice: %s\n", voice_path); + + Options opt; + opt.engine = Engine::Hmm; + std::vector pcm; + std::vector ev; + + CHECK(synthesize(U"あいうえお", pcm, ev, opt).has_value()); + CHECK(!pcm.empty()); + CHECK(!ev.empty()); + CHECK(vowels_only(ev) == "aiueo"); + + // 先頭は無音 (sil) で閉口、最後も閉口。時刻は昇順で隣接する母音は異なる。 + CHECK(ev.front().start_ms == 0 && ev.front().vowel == Vowel::None); + CHECK(ev.back().vowel == Vowel::None); + for (std::size_t i = 1; i < ev.size(); ++i) { + CHECK(ev[i].start_ms > ev[i - 1].start_ms); + CHECK(ev[i].vowel != ev[i - 1].vowel); + } + const double pcm_ms = 1000.0 * static_cast(pcm.size()) / opt.sample_rate_hz; + CHECK(static_cast(ev.back().start_ms) <= pcm_ms + 1.0); + CHECK(pcm_ms - static_cast(ev.back().start_ms) < 1000.0); + + // PCM との時間軸の一致: 先頭の無音区間は静かで、最初の母音区間は音が出ている。 + for (std::size_t i = 0; i + 1 < ev.size(); ++i) { + if (ev[i].vowel == Vowel::None) continue; + const double v0 = ev[i].start_ms, v1 = ev[i + 1].start_ms; + const double sil = rms_between(pcm, opt.sample_rate_hz, 0.0, ev.front().start_ms + 0.5 * (ev[1].start_ms)); + const double voiced = rms_between(pcm, opt.sample_rate_hz, v0 + 0.25 * (v1 - v0), v0 + 0.75 * (v1 - v0)); + std::printf(" first vowel @%.0f-%.0f ms: rms(sil)=%.1f rms(vowel)=%.1f\n", v0, v1, sil, voiced); + CHECK(voiced > 4.0 * sil + 100.0); + break; + } + + // 無声化母音は閉口: 「きした」で母音は た の a だけ。 + CHECK(synthesize(U"きした", pcm, ev, opt).has_value()); + CHECK(vowels_only(ev) == "a"); + + // 呼気段落境界の pau は閉口として挟まる。 + CHECK(synthesize(U"あ、い", pcm, ev, opt).has_value()); + CHECK(vowels_only(ev) == "ai"); + + // 話速 (mora_ms) を倍にすると口形の時間軸も約 2 倍に伸びる。 + CHECK(synthesize(U"あいうえお", pcm, ev, opt).has_value()); + const double t_normal = ev.back().start_ms; + Options slow = opt; + slow.mora_ms = 220.0f; + CHECK(synthesize(U"あいうえお", pcm, ev, slow).has_value()); + CHECK(vowels_only(ev) == "aiueo"); + CHECK(ev.back().start_ms > 1.5 * t_normal); + + // Auto でも HMM が選ばれ口形が出る。 + Options autoo; + CHECK(synthesize(U"あいうえお", pcm, ev, autoo).has_value()); + CHECK(vowels_only(ev) == "aiueo"); + + set_hmm_voice({}); +} + +} // namespace + +int main(int argc, char** argv) +{ + Options opt; + opt.engine = Engine::Formant; // 口形を出せるのはフォルマントのみ + + std::vector pcm; + std::vector ev; + + // 母音そのまま: 母音ごとに 1 イベント + 終端の閉口。 + CHECK(shape_of(U"あいうえお", pcm, ev, opt) == "aiueo-"); + + // 時刻は昇順、隣接イベントの母音は異なり、終端は PCM 長と一致する。 + for (std::size_t i = 1; i < ev.size(); ++i) { + CHECK(ev[i].start_ms > ev[i - 1].start_ms); + CHECK(ev[i].vowel != ev[i - 1].vowel); + } + const double pcm_ms = 1000.0 * static_cast(pcm.size()) / opt.sample_rate_hz; + CHECK(std::abs(static_cast(ev.back().start_ms) - pcm_ms) < 5.0); + CHECK(ev.front().start_ms == 0); + + // 両唇音は閉口 → 母音。それ以外の子音は母音の形を先取りする。 + CHECK(shape_of(U"ま", pcm, ev, opt) == "-a-"); + CHECK(shape_of(U"か", pcm, ev, opt) == "a-"); + CHECK(ev.front().start_ms == 0); + + // 促音・撥音は閉口、長音は直前の母音を保つ。 + CHECK(shape_of(U"あっあ", pcm, ev, opt) == "a-a-"); + CHECK(shape_of(U"あんあ", pcm, ev, opt) == "a-a-"); + CHECK(shape_of(U"あーあ", pcm, ev, opt) == "a-"); + + // 無声化母音は閉口のまま。「きした」は き・し とも無声化される。 + CHECK(shape_of(U"きした", pcm, ev, opt) == "-a-"); + CHECK(shape_of(U"ひとつ", pcm, ev, opt) == "-o-"); + + // 空 / 不正な入力ではイベントも空。 + auto bad = synthesize(U"", pcm, ev, opt); + CHECK(!bad.has_value()); + CHECK(ev.empty()); + + // 既存の 3 引数 API と PCM が一致する (口形の収集が合成を変えない)。 + std::vector pcm_plain; + CHECK(synthesize(U"こんにちは", pcm_plain, opt).has_value()); + CHECK(synthesize(U"こんにちは", pcm, ev, opt).has_value()); + CHECK(pcm == pcm_plain); + + for (int i = 1; i < argc; ++i) test_hmm(argv[i]); + + if (g_failures == 0) std::puts("test_visemes: all passed"); + return g_failures == 0 ? 0 : 1; +} diff --git a/components/wifi_config_service/web/settings_wifi.html b/components/wifi_config_service/web/settings_wifi.html index bf65ca7..78c6140 100644 --- a/components/wifi_config_service/web/settings_wifi.html +++ b/components/wifi_config_service/web/settings_wifi.html @@ -716,6 +716,11 @@

単位連結 音声 DB (.jvox) 無視されます)。| を省くと表示と読みが同じになります。 例: こんにちは | こんにちわ

+ +

LT タイムキーパー

@@ -1142,6 +1147,7 @@

// `表示 | 読み` pair support shared with the BLE page (settings_common.js). const phrases = StackchanSettings.jttsPhrasesFromText($('jtts-phrases').value); if (phrases.length) obj.phrases = phrases; + if ($('jtts-phrase-order').value === 'sequential') obj.phrase_order = 'sequential'; return Object.keys(obj).length ? JSON.stringify(obj) : ''; } @@ -1160,6 +1166,7 @@

if (Array.isArray(obj.phrases)) { $('jtts-phrases').value = StackchanSettings.jttsPhrasesToText(obj.phrases); } + $('jtts-phrase-order').value = obj.phrase_order === 'sequential' ? 'sequential' : 'random'; } // --- Servo limits + range-setting calibration --- diff --git a/docs/avatar_dsl.md b/docs/avatar_dsl.md index dffce8b..784394b 100644 --- a/docs/avatar_dsl.md +++ b/docs/avatar_dsl.md @@ -213,6 +213,7 @@ end_group() | `eye_open` | float | 0..1 | まばたき (0 = 閉) | | `gaze_h`, `gaze_v` | float | -1..+1 | 視線サッカード | | `mouth_open` | float | 0..1 | 口の開き | +| `mouth_form` | float | 0..1 | 口の形 (0 = 横広、1 = すぼめ)。ホストが設定しない場合は `mouth_open` と同値 | | `expr` | enum | 0..5 | 表情 (下記定数で名前指定可) | | `primary` | u16 → float | RGB565 | 前景色 (デフォルト 白 `0xFFFF`) | | `background` | u16 → float | RGB565 | 背景色 (デフォルト 黒 `0x0000`) | @@ -440,10 +441,9 @@ end | 0x08 | `mouth_open` | 0x12 | `brow_off_x` | 0x1C | `cheek_radius` | | 0x09 | `expr` | 0x13 | `brow_off_y` | 0x1D | `cheek_off_x` | | | | | | 0x1E | `cheek_off_y` | +| | | | | 0x1F | `mouth_form` | | 0x20 | `accessories` | 0x21..0x28 | `accessory_0`..`accessory_7` | | | -`0x1F` は予約 (PR #7 の `mouth_form` 用に空けてある)。 - > 真実源: [components/avatar_vm/include/avatar_vm/opcodes.hpp](https://github.com/ciniml/stackchan-idf/blob/main/components/avatar_vm/include/avatar_vm/opcodes.hpp) > (C++ 側) / [tools/avatar_dsl/opcodes.js](https://github.com/ciniml/stackchan-idf/blob/main/tools/avatar_dsl/opcodes.js) (JS 側 ミラー) diff --git a/main/CMakeLists.txt b/main/CMakeLists.txt index a2e25e5..f7c8081 100644 --- a/main/CMakeLists.txt +++ b/main/CMakeLists.txt @@ -45,6 +45,7 @@ set(_main_srcs "dance_storage.cpp" "dance_poc.cpp" "captive_portal.cpp" + "chunk_player.cpp" "demo_loop.cpp" "device_ui.cpp" "diag.cpp" diff --git a/main/chunk_player.cpp b/main/chunk_player.cpp new file mode 100644 index 0000000..5dcf091 --- /dev/null +++ b/main/chunk_player.cpp @@ -0,0 +1,122 @@ +// SPDX-FileCopyrightText: 2026 Kenta IDA +// SPDX-License-Identifier: BSL-1.0 + +#include "chunk_player.hpp" + +#include + +#include +#include +#include +#include + +namespace stackchan::app { + +namespace { + +// M5.Speaker の 1 チャンネルのスロット数 (再生中 + 次)。 +constexpr std::size_t kSlots = 2; + +// 折り返しを考慮して a が b より後か。 +bool after(std::uint32_t a, std::uint32_t b) +{ + return static_cast(a - b) > 0; +} + +} // namespace + +std::uint32_t now_ms() +{ + return static_cast(esp_timer_get_time() / 1000); +} + +void ChunkPlayer::begin() +{ + const auto ch = static_cast(channel_); + bool was_playing = false; + { + std::lock_guard lock(mtx_); + was_playing = end_ms_ != 0 && after(end_ms_, now_ms()); + if (was_playing || M5.Speaker.isPlaying(ch) != 0) { + M5.Speaker.stop(ch); + was_playing = true; + } + end_ms_ = 0; + } + if (was_playing) { + // スピーカー タスクが最後のブロックを読み終えるまで待ってから解放する。 + vTaskDelay(pdMS_TO_TICKS(30)); + } + release(); +} + +std::optional ChunkPlayer::enqueue(std::vector&& pcm, std::uint32_t sample_rate, + const std::function& cancelled) +{ + const auto is_cancelled = [&] { return cancelled && cancelled(); }; + if (pcm.empty()) { + std::lock_guard lock(mtx_); + return std::max(now_ms(), end_ms_); + } + const auto ch = static_cast(channel_); + + // 空きスロットが出るまで待つ。合成の方が再生より速いとここで待たされるが、 + // その分メモリを溜め込まない (最大でも「再生中 + 次 + 合成中」の 3 チャンク)。 + while (M5.Speaker.isPlaying(ch) >= kSlots) { + if (is_cancelled()) return std::nullopt; + vTaskDelay(pdMS_TO_TICKS(10)); + } + + // 「取り消し確認 → 再生開始」を stop() と直列化する。stop() はこのロックの下で + // 取り消しフラグを立ててスピーカーを止めるので、確認を通ったチャンクは必ず + // stop() より前に鳴り始めていて、stop() がそれも止める。 + std::lock_guard lock(mtx_); + if (is_cancelled()) return std::nullopt; + + bufs_.push_back(std::move(pcm)); + const std::vector& buf = bufs_.back(); + while (!M5.Speaker.playRaw(buf.data(), buf.size(), sample_rate, /*stereo=*/false, + /*repeat=*/1, channel_, /*stop_current_sound=*/false)) { + vTaskDelay(pdMS_TO_TICKS(10)); // スロットが埋まっていた (稀): 空くまで再試行 + } + + // 再生予定: 直前のチャンクがまだ鳴っていればその直後、途切れていれば今。 + const std::uint32_t now = now_ms(); + const std::uint32_t start = after(end_ms_, now) ? end_ms_ : now; + const auto dur_ms = static_cast(static_cast(buf.size()) * 1000u / sample_rate); + end_ms_ = start + dur_ms; + + // 鳴らし終えたバッファを解放する。スピーカーがまだ参照しうる分 (再生中 + 次) に + // 1 つ余裕を足して残す。 + const std::size_t pending = std::max(M5.Speaker.isPlaying(ch), 1); + while (bufs_.size() > pending + 1) { + bufs_.pop_front(); + } + return start; +} + +bool ChunkPlayer::finished() const +{ + std::lock_guard lock(mtx_); + return end_ms_ == 0 || !after(end_ms_, now_ms()); +} + +void ChunkPlayer::stop(const std::function& before) +{ + std::lock_guard lock(mtx_); + if (before) before(); + const auto ch = static_cast(channel_); + if (M5.Speaker.isPlaying(ch) != 0) { + M5.Speaker.stop(ch); + } + end_ms_ = 0; +} + +void ChunkPlayer::release() +{ + std::lock_guard lock(mtx_); + bufs_.clear(); + end_ms_ = 0; +} + +} // namespace stackchan::app diff --git a/main/chunk_player.hpp b/main/chunk_player.hpp new file mode 100644 index 0000000..5448a74 --- /dev/null +++ b/main/chunk_player.hpp @@ -0,0 +1,70 @@ +// SPDX-FileCopyrightText: 2026 Kenta IDA +// SPDX-License-Identifier: BSL-1.0 +// +// 合成済み PCM を「チャンク単位で」M5.Speaker の 1 チャンネルに順に積んで再生する。 +// jtts::synthesize_stream のシンクから enqueue() を呼ぶと、最初のチャンクは即座に +// 鳴り始め、以降は再生中に次を合成して積める (隙間なく連続再生される)。 +// +// M5.Speaker の 1 チャンネルは「再生中 + 次」の 2 スロットを持ち、playRaw は渡された +// バッファを *コピーせず* 再生が終わるまで参照する。そのため enqueue() は +// - スロットが空くまで待つ (= 合成が再生より先に進みすぎない。メモリも抑えられる) +// - PCM を保持し、再生し終えたものから順に解放する +// を担う。再生予定の時刻も管理するので、口形イベントの時刻合わせに使える。 +// +// スレッド: enqueue() / begin() / release() は 1 つのタスク (合成タスク) からだけ呼ぶ。 +// stop() だけは別のタスクから呼んでよい (発話の取り消し)。stop() の後に遅れて +// チャンクが鳴り出さないよう、enqueue() の「取り消し確認 → 再生開始」は stop() と +// 同じミューテックスで直列化している。 +#pragma once + +#include +#include +#include +#include +#include +#include + +namespace stackchan::app { + +class ChunkPlayer { +public: + // 使うスピーカー チャンネル (会話の再生 = 0 とは別にして互いの再生状態に干渉しない)。 + explicit ChunkPlayer(int channel) : channel_(channel) {} + + // 新しい発話を始める: 前の発話がまだ鳴っていれば止め、保持しているバッファを全て解放する。 + void begin(); + + // pcm をキューの末尾に積む。空きスロットが出るまで待つ (最大数百 ms〜数秒)。 + // 戻り値: この PCM の再生開始予定時刻 [ms、esp_timer 基準]。直前のチャンクが + // まだ鳴っていればその終了時刻、途切れていれば今。 + // cancelled が true を返したら (待っている間も、再生を始める直前も確認する) + // 積まずに nullopt を返す。 + std::optional enqueue(std::vector&& pcm, std::uint32_t sample_rate, + const std::function& cancelled = {}); + + // 積んだ音の再生が終わる予定時刻 [ms]。何も積んでいなければ 0。 + std::uint32_t end_ms() const { return end_ms_; } + + // 再生予定時刻を過ぎたか (積んだ音を全て鳴らし終えたか)。何も積んでいなければ true。 + bool finished() const; + + // 再生を止める (バッファは次の begin() / release() まで保持: スピーカー タスクが + // 停止直後に読み終えるまで余裕を持たせるため)。別のタスクから呼んでよい。 + // `before` はミューテックスを持った状態で、スピーカーを止める前に呼ばれる + // (取り消しフラグを立てるのに使う: enqueue() の cancelled と同時に成り立つ)。 + void stop(const std::function& before = {}); + + // 保持しているバッファを全て解放する。再生が止まっていること。 + void release(); + +private: + int channel_; + mutable std::mutex mtx_; // bufs_ / end_ms_ と、再生開始 / 停止の直列化 + std::deque> bufs_; // 古い順。末尾が最後に積んだもの + std::uint32_t end_ms_ = 0; +}; + +// esp_timer 基準の現在時刻 [ms] (32 ビットで折り返す。差は int32 で取ること)。 +std::uint32_t now_ms(); + +} // namespace stackchan::app diff --git a/main/demo_loop.cpp b/main/demo_loop.cpp index 2a1ad96..2f870ba 100644 --- a/main/demo_loop.cpp +++ b/main/demo_loop.cpp @@ -73,6 +73,18 @@ constexpr const char* kTag = "stackchan"; static app::Speech speech; speech.configure(jtts_config_json); + // say() blocks this task while it synthesises (audio already plays chunk by + // chunk), so the mouth is animated from Speech's own timer meanwhile. + speech.set_mouth_sink([g_state](const app::Speech::Mouth& m) { + g_state->face.mouth_open.store(m.open, std::memory_order_relaxed); + g_state->face.mouth_form.store(m.form, std::memory_order_relaxed); + }); + // Balloon text follows the chunks: each chunk's part of the phrase goes up + // when that chunk starts sounding (same timer, so it also works while say() + // is blocked synthesising the later chunks). + speech.set_subtitle_sink([g_state](const std::string& text, std::uint32_t hold_ms) { + g_state->set_balloon_text(text, hold_ms); + }); // LT timekeeper — ticked every loop iteration; speaks through the same // Speech instance (so the avatar's mouth moves) and publishes state for @@ -351,6 +363,7 @@ constexpr const char* kTag = "stackchan"; // bus) and don't touch mouth_open. if (audio_streaming) { if (speech.is_speaking()) speech.stop(); + g_state->face.mouth_form.store(-1.0f, std::memory_order_relaxed); vTaskDelay(pdMS_TO_TICKS(100)); continue; } @@ -366,6 +379,7 @@ constexpr const char* kTag = "stackchan"; // While the conversation is thinking / speaking it owns the avatar — // stand down completely. if (!allow_idle_demo) { + g_state->face.mouth_form.store(-1.0f, std::memory_order_relaxed); vTaskDelay(pdMS_TO_TICKS(100)); continue; } @@ -377,8 +391,11 @@ constexpr const char* kTag = "stackchan"; // lip-sync task (main/mic_lip_sync_task.cpp), if active, owns // `mouth_open` without us overwriting it with 0 every tick. if (jtts_idle_enabled) { - // Mouth opens with the current speech envelope; closed while silent. - g_state->face.mouth_open.store(speech.current_mouth_open(), std::memory_order_relaxed); + // Mouth follows the vowel being spoken (or the speech envelope + // when the engine has no vowel timeline); closed while silent. + const auto mouth = speech.current_mouth(); + g_state->face.mouth_open.store(mouth.open, std::memory_order_relaxed); + g_state->face.mouth_form.store(mouth.form, std::memory_order_relaxed); // The "Wi-Fi: 切断中" balloon and the babble suppression below only // make sense when the assistant actually needs the network — i.e. @@ -399,27 +416,29 @@ constexpr const char* kTag = "stackchan"; next_speech_ms = now_ms + 1500; } - // Kick off a new babble + balloon once the previous balloon is done - // (callback resets balloon_in_flight) AND audio is idle AND the - // random dwell time has elapsed. Suppressed while Wi-Fi is down so - // the disconnected balloon stays visible. + // Kick off a new babble once the previous balloon is gone AND audio + // is idle AND the random dwell time has elapsed. Suppressed while + // Wi-Fi is down so the disconnected balloon stays visible. if (!wifi_warning_active && now_ms >= next_speech_ms && !speech.is_speaking() && - !balloon_in_flight.load(std::memory_order_acquire)) { + !balloon_in_flight.load(std::memory_order_acquire) && + !g_state->balloon_visible()) { // Speak a phrase and show ITS display text in the balloon — - // babble() returns the display (発話内容) of the same phrase - // it synthesises (発声内容), so screen and voice always match. - const std::string display = speech.babble(esp_random()); - if (!display.empty()) { - balloon_in_flight.store(true, std::memory_order_release); - g_state->set_balloon_text(display, /*hold_ms=*/0, [] { - balloon_in_flight.store(false, std::memory_order_release); - }); - } + // babble() speaks the reading (発声内容) of the same phrase + // whose display text (発話内容) the subtitle sink above shows + // chunk by chunk, in step with the sound (or all at once if + // nothing could be synthesised), so screen and voice always match. + (void)speech.babble(esp_random()); next_speech_ms = now_ms + rand_range_ms(kSpeechMinMs, kSpeechMaxMs); } + } else { + // Someone else (mic lip-sync) owns the mouth: follow mouth_open. + g_state->face.mouth_form.store(-1.0f, std::memory_order_relaxed); } + } else { + // Conversation is (idly) active and owns the mouth. + g_state->face.mouth_form.store(-1.0f, std::memory_order_relaxed); } // Nadenade: poll the top sensor and look for a directional stroke diff --git a/main/render_task.cpp b/main/render_task.cpp index 10a9e8d..afd462d 100644 --- a/main/render_task.cpp +++ b/main/render_task.cpp @@ -138,6 +138,7 @@ void render_task_entry(void* arg) int last_expression = -1; std::uint32_t last_balloon_version = 0; + std::uint32_t balloon_applied_version = 0; // version of the balloon the avatar is showing std::uint32_t last_face_config_version = 0; std::uint32_t last_face_bytecode_version = 0; std::string balloon_scratch; @@ -218,6 +219,7 @@ void render_task_entry(void* arg) last_expression = expr; } avatar.set_mouth_open(args.state->face.mouth_open.load(std::memory_order_relaxed)); + avatar.set_mouth_form(args.state->face.mouth_form.load(std::memory_order_relaxed)); avatar.set_gaze(args.state->face.gaze_target_h.load(std::memory_order_relaxed), args.state->face.gaze_target_v.load(std::memory_order_relaxed)); @@ -225,14 +227,15 @@ void render_task_entry(void* arg) if (balloon_version != last_balloon_version) { if (args.state->balloon_visible()) { std::uint32_t hold_ms = 0; - args.state->snapshot_balloon(balloon_scratch, hold_ms); + args.state->snapshot_balloon(balloon_scratch, hold_ms, balloon_applied_version); avatar.set_balloon_text(balloon_scratch, hold_ms); balloon_pending = true; + last_balloon_version = balloon_applied_version; } else { avatar.clear_balloon(); balloon_pending = false; + last_balloon_version = balloon_version; } - last_balloon_version = balloon_version; } // avatar.tick() opens the frame (begin_frame) and draws the face; @@ -263,7 +266,10 @@ void render_task_entry(void* arg) if (balloon_pending && avatar.is_balloon_done()) { balloon_pending = false; - args.state->notify_balloon_complete(); + // Completion of *this* balloon only: if the next one was set in the + // meantime (chunk subtitles), the notify is ignored and it gets applied + // on the next loop instead of being wiped. + args.state->notify_balloon_complete(balloon_applied_version); } // Use vTaskDelay (not vTaskDelayUntil) so the IDLE task on this core diff --git a/main/settings_sinks.cpp b/main/settings_sinks.cpp index 03e9e97..2b29523 100644 --- a/main/settings_sinks.cpp +++ b/main/settings_sinks.cpp @@ -24,6 +24,7 @@ #include "avatar/expression.hpp" #include "config_service/config_service.hpp" #include "config_service/config_store.hpp" +#include "chunk_player.hpp" #include "config_service/settings_registry.hpp" #include "speech.hpp" #include "utf8.hpp" @@ -180,9 +181,55 @@ std::uint16_t read_speaker_volume_pct() return g_state->speaker.volume_pct.load(std::memory_order_relaxed); } +// Body of the say worker: synthesise `kana_utf8` chunk by chunk and play it. +// Kept out of the task lambda so every local (PCM buffers, the player) is +// destroyed before the task deletes itself — vTaskDelete never returns, so +// anything still alive in the task function would leak. +void say_worker_body(std::unique_ptr kana_text) +{ + std::u32string kana = stackchan::app::decode_utf8(*kana_text); + if (kana.empty()) { + ESP_LOGW(kTag, "say: empty / invalid utf8"); + return; + } + // Use the user's jtts settings (voice / pitch / mora / + // formant / vibrato) cached at boot. Falls back to the + // default-options preset when no JSON has been saved yet + // (g_say_opts_ready stays false until app_main sets it). + stackchan::jtts::Options opt = g_say_opts_ready + ? g_say_opts + : stackchan::app::resolve_speech_options("", stackchan::app::Speech::kSampleRate); + + // Synthesise chunk by chunk and play while the next chunk is being made + // (sound starts after the first chunk, not the whole text). Wait for the + // speaker to be free first so we never talk over another utterance. + // Channel 2: Speech uses 1, conversation 0. + while (M5.Speaker.isPlaying()) vTaskDelay(pdMS_TO_TICKS(20)); + stackchan::app::ChunkPlayer player{2}; + bool any = false; + const auto r = stackchan::jtts::synthesize_stream( + kana, + [&](stackchan::jtts::SynthChunk&& chunk) { + if (chunk.pcm.empty()) return true; + // sanoTTS は 22.05 kHz 固定なので、チャンクが持つ出力レートで鳴らす + // (BLE / HTTP の jtts-say もこれで sanoTTS を通る)。 + player.enqueue(std::move(chunk.pcm), chunk.sample_rate); + any = true; + return true; + }, + opt); + if (!r) { + ESP_LOGW(kTag, "say synth fail: %s%s", stackchan::jtts::to_string(r.error()), + any ? " (cut short)" : ""); + } + if (!any) return; + while (!player.finished() || M5.Speaker.isPlaying()) vTaskDelay(pdMS_TO_TICKS(20)); + stackchan::wifi_config::mcp_events::publish_say_done(); +} + // Spawn a PSRAM-stack worker that synthesises `kana_utf8` via jtts and -// pushes it through M5.Speaker.playRaw. Shared by /mcp/say (external MCP -// gate) and the settings-page test-speak buttons (BLE chr + /api/jtts-say, +// pushes it through M5.Speaker (chunk by chunk). Shared by /mcp/say (external +// MCP gate) and the settings-page test-speak buttons (BLE chr + /api/jtts-say, // HTTP-auth gate). Returns immediately; the heap-owned string is freed // either by the worker or on task-create failure here. void start_say_worker(std::string_view kana_utf8) @@ -192,46 +239,11 @@ void start_say_worker(std::string_view kana_utf8) // /mcp/say wiring (steady-state internal-RAM largest is ~10 KiB after // conversation_task TLS, so an internal-RAM 12 KiB stack alloc would // silently fail). The worker only touches PSRAM-friendly surfaces - // (jtts buffers, PCM vector, M5.Speaker.playRaw enqueue). + // (jtts buffers, PCM vectors, M5.Speaker.playRaw enqueue). constexpr UBaseType_t kCaps = stackchan::kNoFlashTaskStackCaps; const BaseType_t rc = xTaskCreatePinnedToCoreWithCaps( +[](void* arg) { - // Body in an immediately-invoked lambda: vTaskDeleteWithCaps() - // never returns, so locals (PCM buffer, text) must be destroyed - // before it — otherwise each /api/say leaks the whole buffer. - [&] { - std::unique_ptr kana_text{static_cast(arg)}; - std::u32string kana = stackchan::app::decode_utf8(*kana_text); - if (kana.empty()) { - ESP_LOGW(kTag, "say: empty / invalid utf8"); - return; - } - // Use the user's jtts settings (voice / pitch / mora / - // formant / vibrato) cached at boot. Falls back to the - // default-options preset when no JSON has been saved yet - // (g_say_opts_ready stays false until app_main sets it). - stackchan::jtts::Options opt = g_say_opts_ready - ? g_say_opts - : stackchan::app::resolve_speech_options("", stackchan::app::Speech::kSampleRate); - // synthesize_ex は実際の出力レートを返す (sanoTTS は 22.05 kHz、他は - // opt.sample_rate_hz)。BLE / HTTP の jtts-say もこれで sanoTTS を通る。 - std::uint32_t rate = opt.sample_rate_hz; - std::vector pcm; - if (auto r = stackchan::jtts::synthesize_ex(kana, pcm, opt); !r) { - ESP_LOGW(kTag, "say synth fail: %s", - stackchan::jtts::to_string(r.error())); - return; - } else { - rate = *r; - } - if (pcm.empty()) { - return; - } - while (M5.Speaker.isPlaying()) vTaskDelay(pdMS_TO_TICKS(20)); - M5.Speaker.playRaw(pcm.data(), pcm.size(), rate, /*stereo=*/false); - while (M5.Speaker.isPlaying()) vTaskDelay(pdMS_TO_TICKS(20)); - stackchan::wifi_config::mcp_events::publish_say_done(); - }(); + say_worker_body(std::unique_ptr{static_cast(arg)}); vTaskDeleteWithCaps(nullptr); }, // Pin to CPU 0 — CPU 1 hosts speaker/mic/render/servo and a diff --git a/main/shared_state.hpp b/main/shared_state.hpp index 161ae06..283d316 100644 --- a/main/shared_state.hpp +++ b/main/shared_state.hpp @@ -76,6 +76,11 @@ class SharedState { // --- Avatar face (render_task reads every frame) ----------------------- struct Face { std::atomic mouth_open{0.0f}; + // Mouth shape, 0 = wide .. 1 = narrow (avatar DSL `mouth_form`). Only + // the jtts vowel lip-sync (demo_loop) sets it; every other mouth_open + // producer (mic, conversation, audio stream) leaves it at -1 = "not + // set", which makes the face follow mouth_open like a level meter. + std::atomic mouth_form{-1.0f}; std::atomic expression{static_cast(stackchan::avatar::Expression::Neutral)}; // External gaze target (Avatar::set_gaze inputs). Updated by the // touch-driven gaze-follow path in demo_loop; read by render_task @@ -322,10 +327,11 @@ class SharedState { // --- Balloon (mutex + completion callback; render_task consumes) ------- // Show `text` in the balloon. - // - hold_ms: minimum on-screen time (0 = use avatar defaults — short - // text holds a few seconds, long text plays one marquee pass). + // - hold_ms: on-screen time (0 = use avatar defaults — fitting text holds + // a few seconds, overflowing text scrolls once; with a value the scroll + // is timed to reach the end of the text within it). // - on_complete: invoked once when the balloon finishes (after hold or - // after a marquee pass). Fired from the render task; the + // after the scroll). Fired from the render task; the // implementation must be cheap and thread-safe. void set_balloon_text(std::string_view text, std::uint32_t hold_ms = 0, @@ -350,16 +356,22 @@ class SharedState { balloon_visible_.store(false, std::memory_order_release); } - // Called by the render task when the avatar finishes displaying the - // current balloon. Hides the balloon and invokes the completion callback - // (if any) outside the lock. - void notify_balloon_complete() + // Called by the render task when the avatar finishes displaying a balloon. + // `version` is the balloon_version() of the balloon that finished (the one + // snapshot_balloon() returned when the render task applied it). If the + // balloon has been replaced or cleared since — e.g. the next chunk's + // subtitle was set while the previous one was timing out — the completion is + // stale and is ignored, so it can never hide the newer balloon. Otherwise + // hides the balloon and invokes the completion callback (if any) outside the + // lock. + void notify_balloon_complete(std::uint32_t version) { BalloonCompletionCallback cb; { std::lock_guard lock{balloon_mutex_}; - if (!balloon_visible_.load(std::memory_order_relaxed)) { - return; // already cleared + if (!balloon_visible_.load(std::memory_order_relaxed) || + balloon_version_.load(std::memory_order_relaxed) != version) { + return; // already cleared, or replaced by a newer balloon } balloon_text_.clear(); balloon_hold_ms_ = 0; @@ -384,12 +396,15 @@ class SharedState { return balloon_visible_.load(std::memory_order_acquire); } - // Copies the current text + hold time into the supplied outputs. - void snapshot_balloon(std::string& text_out, std::uint32_t& hold_ms_out) const + // Copies the current text + hold time + version into the supplied outputs + // (one consistent snapshot: all three are read under the lock). Pass the + // version back to notify_balloon_complete() when this balloon finishes. + void snapshot_balloon(std::string& text_out, std::uint32_t& hold_ms_out, std::uint32_t& version_out) const { std::lock_guard lock{balloon_mutex_}; text_out = balloon_text_; hold_ms_out = balloon_hold_ms_; + version_out = balloon_version_.load(std::memory_order_relaxed); } // --- Versioned slots (VersionedValue facade — see the template above) -- diff --git a/main/speech.cpp b/main/speech.cpp index f902e10..945ed4c 100644 --- a/main/speech.cpp +++ b/main/speech.cpp @@ -4,12 +4,15 @@ #include #include "speech.hpp" #include "utf8.hpp" +#include #include #include #include #include #include +#include +#include #include #include @@ -102,6 +105,30 @@ void apply_engine(jtts::Options& opt, const cJSON* item) } } +// Mouth pose per vowel, in the avatar's terms: `open` scales the rectangle's +// height, `form` its width (0 = wide, 1 = narrow). None = lips closed. +// あ: tall, a little narrower than at rest い: wide and flat +// う: narrow, small opening え: wide, half open +// お: narrow-ish, fairly tall (rounded) +constexpr Speech::Mouth kClosedMouth{0.0f, 0.0f}; + +constexpr Speech::Mouth mouth_for_vowel(jtts::Vowel v) noexcept +{ + switch (v) { + case jtts::Vowel::A: return {1.00f, 0.30f}; + case jtts::Vowel::I: return {0.25f, 0.00f}; + case jtts::Vowel::U: return {0.30f, 1.00f}; + case jtts::Vowel::E: return {0.55f, 0.15f}; + case jtts::Vowel::O: return {0.75f, 0.85f}; + case jtts::Vowel::None: break; + } + return kClosedMouth; +} + +// Time to glide from the previous vowel's pose to the next one. Short enough +// to keep fast speech crisp, long enough that the lips don't snap. +constexpr float kMouthBlendMs = 60.0f; + void build_envelope_from_pcm(const std::vector& pcm, std::vector& envelope, std::uint32_t sample_rate, std::uint32_t step_ms) @@ -177,6 +204,8 @@ void Speech::configure(const std::string& json) for (const auto& p : kDefaultPhrases) { phrases_.push_back({std::string(p.display), std::u32string(p.reading)}); } + phrase_order_ = PhraseOrder::Random; + next_phrase_ = 0; initialised_ = true; if (json.empty()) { @@ -190,6 +219,14 @@ void Speech::configure(const std::string& json) apply_options_json(opts_, root); + // phrase_order: "random" (default) or "sequential" (top to bottom, looping). + // Unknown / missing values keep the default. + const cJSON* order = cJSON_GetObjectItemCaseSensitive(root, "phrase_order"); + if (cJSON_IsString(order) && order->valuestring != nullptr && + std::strcmp(order->valuestring, "sequential") == 0) { + phrase_order_ = PhraseOrder::Sequential; + } + // phrases: array whose elements are either // - a string "こんにちわ" (display == reading), or // - an object {"text":"こんにちは","reading":"こんにちわ"} @@ -220,9 +257,10 @@ void Speech::configure(const std::string& json) if (!parsed.empty()) phrases_ = std::move(parsed); } cJSON_Delete(root); - ESP_LOGI(kTag, "jtts config: voice=%s f0=%.0f mora=%.0fms phrases=%zu", + ESP_LOGI(kTag, "jtts config: voice=%s f0=%.0f mora=%.0fms phrases=%zu order=%s", opts_.voice == jtts::Voice::Female ? "female" : "male", - opts_.f0_hz, opts_.mora_ms, phrases_.size()); + opts_.f0_hz, opts_.mora_ms, phrases_.size(), + phrase_order_ == PhraseOrder::Sequential ? "sequential" : "random"); } std::string Speech::babble(std::uint32_t seed) @@ -233,59 +271,157 @@ std::string Speech::babble(std::uint32_t seed) if (phrases_.empty()) { return {}; } - const Phrase& phrase = phrases_[seed % phrases_.size()]; - // Couldn't pronounce → still return the display text so the caller shows - // the matching balloon (no audio / mouth movement in that case). - (void)say(phrase.reading); + std::size_t index; + if (phrase_order_ == PhraseOrder::Sequential) { + index = next_phrase_ % phrases_.size(); + next_phrase_ = index + 1; // stays < size + 1, so it never overflows + } else { + index = seed % phrases_.size(); + } + const Phrase& phrase = phrases_[index]; + // Couldn't start (busy / task create failed) → still show the text once, so + // the caller's balloon is never lost. Otherwise the subtitle sink is driven + // by the synthesis task, chunk by chunk (and shows the whole text if nothing + // could be synthesised). The display text is returned either way. + if (!say_impl(phrase.reading, &phrase.display) && subtitle_sink_) { + subtitle_sink_(phrase.display, 0); + } return phrase.display; } struct Speech::SynthJob { Speech* self; std::u32string reading; + std::string display; // 吹き出しに出す表示テキスト (空 = 吹き出しは呼び出し側に任せる) jtts::Options opt; std::uint32_t gen; }; void Speech::synth_task(void* arg) { - // vTaskDeleteWithCaps() never returns, so every local (the old PCM buffer - // handed back by swap(), the envelope, the job) must be destroyed BEFORE - // it is called — otherwise ~60 KB leak per utterance. Hence the body - // lives in an immediately-invoked lambda and the delete happens after it. + // vTaskDeleteWithCaps() never returns, so every local (the job, PCM buffers, + // the subtitle mapper, ...) must be destroyed BEFORE it is called — + // otherwise ~60 KB leak per utterance. Hence the body lives in an + // immediately-invoked lambda and the delete happens after it. [&] { - std::unique_ptr job{static_cast(arg)}; - Speech* self = job->self; - std::vector pcm; - auto r = jtts::synthesize_ex(job->reading, pcm, job->opt); - if (!r || pcm.empty() || self->gen_.load(std::memory_order_acquire) != job->gen) { - // 合成失敗、無音、または stop() で取り消された。 + std::unique_ptr job{static_cast(arg)}; + Speech* self = job->self; + self->run_utterance(*job); self->synthesizing_.store(false, std::memory_order_release); - return; - } - const std::uint32_t rate = *r; - std::vector envelope; - build_envelope_from_pcm(pcm, envelope, rate, kEnvelopeStepMs); - { - std::lock_guard lock(self->buf_mutex_); - self->pcm_.swap(pcm); - self->envelope_.swap(envelope); - self->play_rate_ = rate; - self->duration_ms_.store( - static_cast(static_cast(self->pcm_.size()) * 1000.0f / - static_cast(rate)), - std::memory_order_relaxed); - self->start_ms_.store(static_cast(esp_timer_get_time() / 1000), - std::memory_order_release); - M5.Speaker.playRaw(self->pcm_.data(), self->pcm_.size(), rate, /*stereo=*/false, - /*repeat=*/1, /*channel=*/-1, /*stop_current_sound=*/true); - } - self->synthesizing_.store(false, std::memory_order_release); }(); vTaskDeleteWithCaps(nullptr); } +void Speech::run_utterance(const SynthJob& job) +{ + // Balloon text follows the chunks (only when there is text and a sink). + std::optional subtitles; + if (!job.display.empty() && subtitle_sink_) { + subtitles.emplace(job.display, job.reading); + } + + // stop() bumps gen_: everything after that is discarded. + const auto cancelled = [&] { return gen_.load(std::memory_order_acquire) != job.gen; }; + + bool first = true; + bool any = false; + const auto on_chunk = [&](jtts::SynthChunk&& chunk) -> bool { + const std::size_t samples = chunk.pcm.size(); + if (samples == 0) { + return true; + } + if (cancelled()) { + return false; + } + // sanoTTS outputs 22.05 kHz; the chunk carries the actual rate. + const std::uint32_t rate = chunk.sample_rate != 0 ? chunk.sample_rate : job.opt.sample_rate_hz; + + // This chunk's part of the balloon text (chunks arrive in order). + std::string subtitle; + if (subtitles) { + subtitle = subtitles->next(chunk.text); + } + // Envelope (only for engines without a vowel timeline) must be taken + // before the PCM is handed to the player. + std::vector env; + if (chunk.visemes.empty()) { + build_envelope_from_pcm(chunk.pcm, env, rate, kEnvelopeStepMs); + } + const auto dur_ms = static_cast(static_cast(samples) * 1000u / rate); + + if (first) { + // A previous utterance may still be sounding: cut it now that the new + // one has audio, and free its buffers. + player_.begin(); + } + + // Queue the chunk (blocks while the speaker's 2 slots are full) and learn + // when it will actually start sounding. + const auto start_opt = player_.enqueue(std::move(chunk.pcm), rate, cancelled); + if (!start_opt) { + return false; // stop() came in + } + const std::uint32_t start = *start_opt; + + { + std::lock_guard lock(buf_mutex_); + if (first) { + visemes_.clear(); + envelope_.clear(); + subs_.clear(); + next_sub_ = 0; + duration_ms_.store(dur_ms, std::memory_order_relaxed); + // 0 means "idle", so never publish a start of exactly 0. + start_ms_.store(start != 0 ? start : 1, std::memory_order_release); + } + const std::uint32_t base = start_ms_.load(std::memory_order_relaxed); + const std::uint32_t offset = first ? 0 : start - base; + for (const auto& e : chunk.visemes) { + visemes_.push_back({e.start_ms + offset, e.vowel}); + } + if (!env.empty()) { + if (envelope_.size() < offset / kEnvelopeStepMs) { + envelope_.resize(offset / kEnvelopeStepMs, 0.0f); // silence up to the chunk + } + envelope_.insert(envelope_.end(), env.begin(), env.end()); + } + if (!subtitle.empty()) { + subs_.push_back({offset, dur_ms, std::move(subtitle)}); + } + duration_ms_.store(offset + dur_ms, std::memory_order_relaxed); + } + if (first) { + // The first balloon text goes up right now, before the timer takes over. + pump_subtitles(); + start_mouth_timer(); + } + first = false; + any = true; + return true; + }; + + const auto r = jtts::synthesize_stream(job.reading, on_chunk, job.opt); + if (!r && r.error() != jtts::Error::Cancelled) { + ESP_LOGW(kTag, "say: synthesis %s%s", jtts::to_string(r.error()), + any ? " (utterance cut short)" : ""); + } + if (cancelled()) { + // stop() already stopped the speaker; give it a moment to finish reading the + // last block, then free what we queued. + vTaskDelay(pdMS_TO_TICKS(30)); + player_.release(); + } else if (!any && subtitle_sink_ && !job.display.empty()) { + // Nothing could be synthesised: still show the text once. + subtitle_sink_(job.display, 0); + } +} + bool Speech::say(std::u32string_view reading) +{ + return say_impl(reading, nullptr); +} + +bool Speech::say_impl(std::u32string_view reading, const std::string* display) { if (!initialised_) { configure(""); // first-call lazy init with defaults @@ -296,7 +432,8 @@ bool Speech::say(std::u32string_view reading) } jtts::Options opt = opts_; opt.sample_rate_hz = kSampleRate; // 他エンジンの既定レート。sanoTTS は 22.05 kHz を返す - auto* job = new SynthJob{this, std::u32string{reading}, opt, gen_.load(std::memory_order_acquire)}; + auto* job = new SynthJob{this, std::u32string{reading}, display != nullptr ? *display : std::string{}, + opt, gen_.load(std::memory_order_acquire)}; // スタックは PSRAM (flash への書き込みはしない)。CPU 0 — CPU 1 は描画 / サーボ / スピーカー。 const BaseType_t rc = xTaskCreatePinnedToCoreWithCaps(&synth_task, "speech_synth", 16 * 1024, job, tskIDLE_PRIORITY + 2, nullptr, 0, @@ -312,13 +449,18 @@ bool Speech::say(std::u32string_view reading) void Speech::stop() { - // 進行中の合成があれば結果を捨てさせる (タスク自体は合成完了まで走る)。 - gen_.fetch_add(1, std::memory_order_acq_rel); - if (M5.Speaker.isPlaying()) { - M5.Speaker.stop(); - } + stop_mouth_timer(); + // 進行中の合成があれば結果を捨てさせ (gen_)、鳴っている音を止める (タスク自体は今の + // チャンクの合成が終わるまで走る)。gen_ の更新と停止は ChunkPlayer のミューテックスの + // 下で行うので、stop() の後に遅れて鳴り出すチャンクは無い。 + player_.stop([this] { gen_.fetch_add(1, std::memory_order_acq_rel); }); + std::lock_guard lock(buf_mutex_); start_ms_.store(0, std::memory_order_release); duration_ms_.store(0, std::memory_order_release); + visemes_.clear(); + envelope_.clear(); + subs_.clear(); + next_sub_ = 0; } bool Speech::is_speaking() const @@ -334,23 +476,127 @@ bool Speech::is_speaking() const return (now - start) < duration_ms_.load(std::memory_order_relaxed); } -float Speech::current_mouth_open() const +void Speech::set_mouth_sink(MouthSink sink) +{ + mouth_sink_ = std::move(sink); +} + +void Speech::set_subtitle_sink(SubtitleSink sink) +{ + subtitle_sink_ = std::move(sink); +} + +// Show the newest balloon text whose chunk has started sounding. If several are +// due (the timer was late) only the latest is shown — the earlier ones are stale. +void Speech::pump_subtitles() +{ + if (!subtitle_sink_) { + return; + } + Subtitle due; + bool have = false; + { + std::lock_guard lock(buf_mutex_); + const std::uint32_t start = start_ms_.load(std::memory_order_relaxed); + if (start == 0) { + return; + } + const std::uint32_t elapsed = static_cast(esp_timer_get_time() / 1000) - start; + while (next_sub_ < subs_.size() && elapsed >= subs_[next_sub_].at_ms) { + due = subs_[next_sub_]; + have = true; + ++next_sub_; + } + } + if (have) { + subtitle_sink_(due.text, due.hold_ms); + } +} + +void Speech::start_mouth_timer() +{ + if (!mouth_sink_ && !subtitle_sink_) { + return; + } + if (mouth_timer_ == nullptr) { + esp_timer_create_args_t args{}; + args.callback = [](void* self) { static_cast(self)->on_mouth_timer(); }; + args.arg = this; + args.dispatch_method = ESP_TIMER_TASK; + args.name = "speech_mouth"; + args.skip_unhandled_events = true; + if (esp_timer_create(&args, &mouth_timer_) != ESP_OK) { + mouth_timer_ = nullptr; + ESP_LOGW(kTag, "mouth timer create failed — mouth/balloon will not follow the chunks"); + return; + } + } + // Already running (ESP_ERR_INVALID_STATE) is fine. + (void)esp_timer_start_periodic(mouth_timer_, kMouthTimerUs); +} + +void Speech::stop_mouth_timer() +{ + if (mouth_timer_ != nullptr) { + (void)esp_timer_stop(mouth_timer_); // not running (ESP_ERR_INVALID_STATE) is fine + } +} + +void Speech::on_mouth_timer() +{ + if (mouth_sink_) { + mouth_sink_(current_mouth()); + } + pump_subtitles(); + // Nothing left to animate: close the mouth once and go quiet. is_speaking() is + // also true while chunks are still being synthesised, so a slow chunk leaves + // a gap, not an end. + if (!is_speaking()) { + (void)esp_timer_stop(mouth_timer_); + if (mouth_sink_) { + mouth_sink_(Mouth{}); + } + } +} + +Speech::Mouth Speech::current_mouth() const { - std::lock_guard lock(buf_mutex_); const std::uint32_t start = start_ms_.load(std::memory_order_acquire); - if (start == 0 || envelope_.empty()) { - return 0.0f; + if (start == 0) { + return {}; } const std::uint32_t now = static_cast(esp_timer_get_time() / 1000); const std::uint32_t elapsed = now - start; + + std::lock_guard lock(buf_mutex_); if (elapsed >= duration_ms_.load(std::memory_order_relaxed)) { - return 0.0f; + return {}; } + + if (!visemes_.empty()) { + // Vowel lip-sync: find the event in effect and glide toward its pose + // from the previous event's pose. + const auto next = std::upper_bound( + visemes_.begin(), visemes_.end(), elapsed, + [](std::uint32_t t, const jtts::VisemeEvent& e) { return t < e.start_ms; }); + if (next == visemes_.begin()) { + return {kClosedMouth.open, kClosedMouth.form}; + } + const auto cur = std::prev(next); + const Mouth to = mouth_for_vowel(cur->vowel); + const Mouth from = cur == visemes_.begin() ? kClosedMouth : mouth_for_vowel(std::prev(cur)->vowel); + float t = static_cast(elapsed - cur->start_ms) / kMouthBlendMs; + t = t > 1.0f ? 1.0f : t; + t = t * t * (3.0f - 2.0f * t); // smoothstep + return {from.open + (to.open - from.open) * t, from.form + (to.form - from.form) * t}; + } + + // No vowel timeline (unit-concatenation): mouth follows the loudness. const std::size_t idx = elapsed / kEnvelopeStepMs; if (idx >= envelope_.size()) { - return 0.0f; + return {}; } - return envelope_[idx]; + return {envelope_[idx], -1.0f}; } } // namespace stackchan::app diff --git a/main/speech.hpp b/main/speech.hpp index a2d0df6..3c0e985 100644 --- a/main/speech.hpp +++ b/main/speech.hpp @@ -5,12 +5,16 @@ #include #include +#include #include #include #include +#include #include +#include "chunk_player.hpp" + namespace stackchan::app { // Parse the user's jtts config JSON (the same blob Speech::configure @@ -23,9 +27,10 @@ jtts::Options resolve_speech_options(const std::string& json, std::uint32_t sample_rate); // Synthesises a short "babble" speech-like utterance and plays it through -// M5.Speaker. While the clip is playing, `current_mouth_open()` returns the -// instantaneous envelope (peak amplitude) of the audio, which the render -// task uses to drive the avatar's mouth. +// M5.Speaker. While the clip is playing, `current_mouth()` returns the mouth +// pose the render task should show: the vowel (あ/い/う/え/お) being spoken +// when the engine can report one (jtts formant / HMM engines), otherwise the +// instantaneous envelope (peak amplitude) of the audio. class Speech { public: // Sample rate of synthesised audio. 16 kHz int16. @@ -40,35 +45,97 @@ class Speech { // ignored entirely. Call once at startup, before the first babble. void configure(const std::string& json); - // Start a fresh utterance (non-blocking — M5.Speaker queues it). - // `seed` selects which phrase to speak (seed % phrase count). Returns the - // *display* text (発話内容) of the chosen phrase so the caller can show a - // matching balloon — synthesis uses that phrase's separate *reading* - // (発声内容, kana). Returns an empty string only when there are no phrases. + // Which phrase babble() picks next. Random (default) uses the caller's + // seed; Sequential walks the phrase list top to bottom and wraps around. + enum class PhraseOrder : std::uint8_t { Random, Sequential }; + + // Start a fresh utterance (non-blocking — synthesis runs in its own task and + // M5.Speaker queues the audio). Random order: `seed` selects the phrase + // (seed % phrase count). Sequential order: `seed` is ignored, the next line + // in the list is used (restarting from the first line after configure()). + // Returns the *display* text (発話内容) of the chosen phrase — synthesis uses + // that phrase's separate *reading* (発声内容, kana). Returns an empty string + // only when there are no phrases. + // + // The balloon is driven through set_subtitle_sink(): each chunk's part of the + // display text is shown as that chunk starts sounding (if nothing could be + // synthesised the whole text is shown once, so the caller need not show it). std::string babble(std::uint32_t seed); // Speak an arbitrary kana string (発声内容; jtts has no kanji dictionary). - // Same synthesis + envelope path as babble() so the avatar's mouth moves. - // Cuts off any in-flight utterance. Returns false when synthesis failed - // or the reading contained nothing speakable. Caller-side balloon text is - // the caller's business (it usually differs from the reading). + // Same synthesis + lip-sync path as babble() so the avatar's mouth moves. + // Non-blocking: synthesis runs in its own task (sanoTTS takes 1-2 s per + // utterance; blocking the caller — demo_loop — would stall M5.update() and + // drop touches). Returns false when the reading is empty, the previous + // synthesis is still running, or the task could not be created. Caller-side + // balloon text is the caller's business (it usually differs from the reading). + // A new utterance cuts off one that is still playing once its first chunk is ready. + // + // Long text is synthesised chunk by chunk and *played while the next chunk + // is being synthesised*: sound starts after the first chunk (≈ 1 s) instead + // of after the whole utterance. bool say(std::u32string_view reading); // Cancel any in-flight babble so we can hand the speaker / I2S bus to - // someone else (e.g. mic loopback). After this is_speaking() returns false. + // someone else (e.g. mic loopback). Discards a synthesis in progress (its + // task ends after the chunk it is on) and stops the audio. Safe to call + // from any task. is_speaking() stays true until that task has exited. void stop(); - // 0..1 envelope at "now". Returns 0 if nothing is playing. - float current_mouth_open() const; + // Mouth pose at "now". `open` is 0..1 (0 = closed); `form` is 0 = wide .. + // 1 = narrow, or -1 when the pose comes from the audio envelope and the + // face should follow `open` alone. Both are 0 / -1 when nothing is playing. + struct Mouth { + float open = 0.0f; + float form = -1.0f; + }; + Mouth current_mouth() const; + + // Called every 20 ms with the current mouth while an utterance is being + // synthesised / played, from the esp_timer task: keep it tiny (atomic + // stores). Independent of how often the caller polls current_mouth(). Set + // before the first say(). + using MouthSink = std::function; + void set_mouth_sink(MouthSink sink); + + // Balloon text synchronised with the chunks: babble() shows each chunk's part + // of the display text (発話内容) at the moment that chunk starts sounding, using + // the punctuation-based mapping in jtts/subtitle.hpp (falls back to showing the + // whole text with the first chunk when the punctuation of reading and display + // do not line up). `hold_ms` is the chunk's audio length. Called on the + // esp_timer task and on the synthesis task (first chunk / failure): keep it + // cheap. Set before the first babble(). + using SubtitleSink = std::function; + void set_subtitle_sink(SubtitleSink sink); bool is_speaking() const; private: + // Speaker channel for jtts speech (conversation playback uses 0). + static constexpr int kSpeakerChannel = 1; + // Mouth update period while speaking. + static constexpr std::uint32_t kMouthTimerUs = 20'000; + // Pre-computed envelope (peak amplitude per kEnvelopeStepMs window), - // normalised to 0..1. Indexed by elapsed window count. + // normalised to 0..1, indexed by elapsed window count. Only filled for + // chunks the engine gave no vowel timeline for (unit-concatenation). std::vector envelope_; - // PCM kept alive while M5.Speaker plays it asynchronously. - std::vector pcm_; + // Vowel timeline of the current utterance in ms since start_ms_ (jtts + // formant / HMM engines). Chunk timelines are appended at the chunk's + // scheduled playback start, so a synthesis underrun (gap) keeps the mouth + // closed and later chunks stay in sync with the sound. + std::vector visemes_; + // Chunk PCM queue for M5.Speaker (keeps buffers alive while they play). + ChunkPlayer player_{kSpeakerChannel}; + + // Subtitle timeline of the current utterance: `at_ms` is relative to start_ms_. + struct Subtitle { + std::uint32_t at_ms = 0; + std::uint32_t hold_ms = 0; + std::string text; + }; + std::vector subs_; // guarded by buf_mutex_ + std::size_t next_sub_ = 0; // guarded by buf_mutex_: first entry not yet shown // A single babble phrase. `display` (発話内容) is the UTF-8 text shown in // the balloon — free-form, may contain kanji/punctuation. `reading` @@ -85,22 +152,36 @@ class Speech { // preset, ~8 short Japanese phrases). jtts::Options opts_; std::vector phrases_; + PhraseOrder phrase_order_{PhraseOrder::Random}; + std::size_t next_phrase_{0}; // Sequential: index of the phrase babble() speaks next bool initialised_{false}; + // Playback start of the first chunk / total scheduled length so far (grows + // as chunks are queued). 0 start = idle. std::atomic start_ms_{0}; std::atomic duration_ms_{0}; - // 直近の合成出力のレート。sanoTTS は 22.05 kHz、他は kSampleRate。 - std::uint32_t play_rate_{kSampleRate}; // 合成は別タスクで行う (sanoTTS は 1 発話 1〜2 秒かかり、呼び出し元 = demo_loop を // ブロックすると M5.update() が止まってタッチを取りこぼす)。synthesizing_ の間は // is_speaking() が true。stop() は gen_ を進めて進行中の合成結果を捨てる。 std::atomic synthesizing_{false}; std::atomic gen_{0}; - // pcm_ / envelope_ / play_rate_ の差し替えと current_mouth_open() の読み出しを直列化。 + // envelope_ / visemes_ / subs_ / start_ms_ / duration_ms_ の更新 (合成タスク) と + // 読み出し (口のタイマー / demo_loop) を直列化。 mutable std::mutex buf_mutex_; struct SynthJob; static void synth_task(void* arg); + // 合成タスクの本体: チャンクごとに合成 → 再生キューへ → 口形 / 吹き出しの時刻を記録。 + void run_utterance(const SynthJob& job); + bool say_impl(std::u32string_view reading, const std::string* display); + + MouthSink mouth_sink_; + SubtitleSink subtitle_sink_; + esp_timer_handle_t mouth_timer_ = nullptr; + void pump_subtitles(); + void start_mouth_timer(); + void stop_mouth_timer(); + void on_mouth_timer(); }; } // namespace stackchan::app diff --git a/tools/avatar_dsl/README.md b/tools/avatar_dsl/README.md index e8e2aad..6304d84 100644 --- a/tools/avatar_dsl/README.md +++ b/tools/avatar_dsl/README.md @@ -69,6 +69,7 @@ while cond do ... end -- 無限ループ防止策は無し、 -- read-only コンテキスト変数 (ホストから注入): -- canvas_w canvas_h canvas_scale now_ms -- breath eye_open gaze_h gaze_v mouth_open -- 0..1 / -1..1 +-- mouth_form -- 0 (wide) .. 1 (narrow)。未設定時は mouth_open と同値 -- expr -- enum 0..5 -- primary background secondary balloon_fg balloon_bg -- RGB565 -- eye_radius eye_off_x eye_off_y diff --git a/tools/avatar_dsl/opcodes.js b/tools/avatar_dsl/opcodes.js index 58421c5..20f7eef 100644 --- a/tools/avatar_dsl/opcodes.js +++ b/tools/avatar_dsl/opcodes.js @@ -72,6 +72,7 @@ export const Var = Object.freeze({ cheek_radius: 0x1C, cheek_off_x: 0x1D, cheek_off_y: 0x1E, + mouth_form: 0x1F, // Accessory slots (FaceTuning::accessories bitmask). accessory_N is 0/1. accessories: 0x20, accessory_0: 0x21, diff --git a/tools/settings.html b/tools/settings.html index 2022f40..271e45f 100644 --- a/tools/settings.html +++ b/tools/settings.html @@ -815,6 +815,11 @@

jtts ボイス

(漢字は発声時に無視されます)。| を省くと表示と読みが同じになります。 例: こんにちは | こんにちわ

+ +

@@ -1785,6 +1790,8 @@

ファームウェア更新 (OTA)

const phrases = StackchanSettings.jttsPhrasesFromText( document.getElementById('jtts-phrases').value); if (phrases.length > 0) obj.phrases = phrases; + const order = document.getElementById('jtts-phrase-order').value; + if (order === 'sequential') obj.phrase_order = order; return obj; } @@ -1797,6 +1804,7 @@

ファームウェア更新 (OTA)

document.getElementById(id).value = ''; } document.getElementById('jtts-phrases').value = ''; + document.getElementById('jtts-phrase-order').value = 'random'; if (!json) return; let obj; try { obj = JSON.parse(json); } @@ -1819,6 +1827,9 @@

ファームウェア更新 (OTA)

document.getElementById('jtts-phrases').value = StackchanSettings.jttsPhrasesToText(obj.phrases); } + if (obj.phrase_order === 'sequential') { + document.getElementById('jtts-phrase-order').value = 'sequential'; + } } // Serialise form state for change-detection. JSON.stringify on a fresh object diff --git a/wasm/avatar_wasm.cpp b/wasm/avatar_wasm.cpp index d5d4fa8..da295dc 100644 --- a/wasm/avatar_wasm.cpp +++ b/wasm/avatar_wasm.cpp @@ -248,6 +248,12 @@ EMSCRIPTEN_KEEPALIVE void avatar_set_mouth(float ratio) g_ctx.mouth_open_ratio = ratio < 0.0f ? 0.0f : (ratio > 1.0f ? 1.0f : ratio); } +// Mouth shape, 0 = wide .. 1 = narrow; negative = follow mouth_open. +EMSCRIPTEN_KEEPALIVE void avatar_set_mouth_form(float ratio) +{ + g_ctx.mouth_form_ratio = ratio < 0.0f ? -1.0f : (ratio > 1.0f ? 1.0f : ratio); +} + EMSCRIPTEN_KEEPALIVE void avatar_set_manual_gaze(int on, float h, float v) { g_manual_gaze = on != 0; diff --git a/wasm/build.sh b/wasm/build.sh index bac0dc3..c1403d0 100755 --- a/wasm/build.sh +++ b/wasm/build.sh @@ -76,7 +76,7 @@ node "$ROOT/tools/avatar_dsl/inject.mjs" \ "omega=$ROOT/assets/omega_mouth.avdsl" \ "aokko=$ROOT/assets/aokko_face.avdsl" -EXPORTS='_avatar_init,_avatar_set_size,_avatar_width,_avatar_height,_avatar_framebuffer,_avatar_set_expression,_avatar_set_mouth,_avatar_set_manual_gaze,_avatar_set_saccade,_avatar_set_blink,_avatar_set_breath,_avatar_set_colors,_avatar_set_eyebrows_visible,_avatar_set_eye_params,_avatar_set_eyebrow_params,_avatar_set_mouth_params,_avatar_set_cheeks_visible,_avatar_set_cheek_params,_avatar_set_accessories,_avatar_set_direct,_avatar_tick,_avatar_load_bytecode,_avatar_reset_bytecode,_malloc,_free' +EXPORTS='_avatar_init,_avatar_set_size,_avatar_width,_avatar_height,_avatar_framebuffer,_avatar_set_expression,_avatar_set_mouth,_avatar_set_mouth_form,_avatar_set_manual_gaze,_avatar_set_saccade,_avatar_set_blink,_avatar_set_breath,_avatar_set_colors,_avatar_set_eyebrows_visible,_avatar_set_eye_params,_avatar_set_eyebrow_params,_avatar_set_mouth_params,_avatar_set_cheeks_visible,_avatar_set_cheek_params,_avatar_set_accessories,_avatar_set_direct,_avatar_tick,_avatar_load_bytecode,_avatar_reset_bytecode,_malloc,_free' # Shared compile inputs/flags for both outputs (same C++, same exports). Kept # on single lines so the values interpolate cleanly into the `bash -c` script