From f815017153255461dd8f55aa69b9d4e73ea97caf Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 14 Mar 2026 17:30:11 +0000 Subject: [PATCH 1/4] feat: integrate speech testing into main pictogram flow MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Wire the speech practice pipeline (listen → record → results) into the OutputScreen so users can practice pronunciation of generated sentences directly from the pictogram navigator. The "Play" button now launches a 3-phase flow: paced TTS playback, hold-to-record with Azure evaluation, and karaoke-style pronunciation feedback with per-word scores. - Add resolveArasaacImages utility for dynamic sentence word pictograms - Make PronunciationFeedback handle missing image URLs gracefully - Add GitHub Actions CI workflow (lint + test + build) https://claude.ai/code/session_01L7SVMJvoKVtzxK2CCJHirQ --- .github/workflows/ci.yml | 30 +++ app/pictogram/page.tsx | 296 +++++++++++++++++++++++---- components/PronunciationFeedback.tsx | 24 ++- lib/resolveArasaacImages.ts | 51 +++++ 4 files changed, 353 insertions(+), 48 deletions(-) create mode 100644 .github/workflows/ci.yml create mode 100644 lib/resolveArasaacImages.ts diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..15ab664 --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,30 @@ +name: CI + +on: + push: + branches: [main] + pull_request: + branches: [main] + +jobs: + ci: + runs-on: ubuntu-latest + + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-node@v4 + with: + node-version: 20 + cache: npm + + - run: npm ci + + - name: Lint + run: npm run lint + + - name: Test + run: npm test + + - name: Build + run: npm run build diff --git a/app/pictogram/page.tsx b/app/pictogram/page.tsx index 13219c9..835f49b 100644 --- a/app/pictogram/page.tsx +++ b/app/pictogram/page.tsx @@ -5,6 +5,13 @@ import DesktopLayout from "@/components/DesktopLayout"; import { pictogramTree, PictogramNode } from "@/lib/pictogramTree"; import { useArasaacImage } from "@/lib/useArasaacImage"; import { useSpeech } from "@/hooks/useSpeech"; +import { useSpeechPacing } from "@/hooks/useSpeechPacing"; +import AudioRecorder from "@/components/AudioRecorder"; +import PronunciationFeedback from "@/components/PronunciationFeedback"; +import { + resolveArasaacImages, + WordPictogram, +} from "@/lib/resolveArasaacImages"; function IconChevronSmall() { return ( @@ -189,8 +196,34 @@ function TrailChips({ trail }: { trail: PictogramNode[] }) { ); } +// ─── Types for speech evaluation ───────────────────────────────────────────── + +interface EvaluationWord { + word: string; + accuracy_score: number; + is_omitted: boolean; +} + +interface EvaluationResult { + target: string; + overall_accuracy: number; + words: EvaluationWord[]; +} + +function IconMic() { + return ( + + + + + + ); +} + // ─── Output screen ──────────────────────────────────────────────────────────── +type OutputPhase = "output" | "listen" | "record" | "results"; + function OutputScreen({ trail, onBack, @@ -200,10 +233,17 @@ function OutputScreen({ }) { const [sentence, setSentence] = useState(null); const [loading, setLoading] = useState(true); + const [phase, setPhase] = useState("output"); + const [evalResult, setEvalResult] = useState(null); + const [evalError, setEvalError] = useState(null); + const [wordPictograms, setWordPictograms] = useState([]); const fetchSentence = useCallback(async () => { setLoading(true); setSentence(null); + setPhase("output"); + setEvalResult(null); + setEvalError(null); try { const res = await fetch("/api/generate", { method: "POST", @@ -216,13 +256,15 @@ function OutputScreen({ }), }); const data = await res.json(); - if (data.sentence) { - setSentence(data.sentence); - } else { - setSentence("I need help."); - } + const text = data.sentence || "I need help."; + setSentence(text); + // Pre-resolve pictogram images for the sentence words + const words = text.trim().split(/\s+/); + resolveArasaacImages(words).then(setWordPictograms); } catch { - setSentence("I need help."); + const fallback = "I need help."; + setSentence(fallback); + resolveArasaacImages(fallback.replace(".", "").split(/\s+/)).then(setWordPictograms); } finally { setLoading(false); } @@ -234,66 +276,242 @@ function OutputScreen({ }, [fetchSentence]); const { speak } = useSpeech(); + const { speakWithPacing } = useSpeechPacing(); - // Auto-speak when sentence arrives + // Auto-speak when sentence arrives (only in output phase) useEffect(() => { - if (sentence) { + if (sentence && phase === "output") { speak(sentence, { rate: 0.85, pitch: 1 }); } - }, [sentence, speak]); + }, [sentence, speak, phase]); + + const handleStartPractice = useCallback(() => { + if (!sentence) return; + setEvalResult(null); + setEvalError(null); + setPhase("listen"); + speakWithPacing(sentence); + }, [sentence, speakWithPacing]); + + const handleEvalResult = useCallback((result: EvaluationResult) => { + setEvalResult(result); + setPhase("results"); + }, []); + + const handleEvalError = useCallback((err: string) => { + setEvalError(err); + }, []); + + const handleTryAgain = useCallback(() => { + setEvalResult(null); + setEvalError(null); + setPhase("listen"); + if (sentence) speakWithPacing(sentence); + }, [sentence, speakWithPacing]); + + // ── Phase: output (sentence display) ─────────────────────────────────── + + if (phase === "output") { + return ( +
+
+ {}} /> +
- return ( -
-
- {}} /> +
+ {loading ? ( +
+
+
+
+ ) : ( +
+ {sentence} +
+ )} + +
+ + +
+
+ +
+ + +
+ ); + } + + // ── Phase: listen (paced TTS playback) ───────────────────────────────── + + if (phase === "listen") { + return ( +
+
+ {}} /> +
-
- {loading ? ( -
-
-
+
+
+ Listen carefully
- ) : ( -
+
{sentence}
- )} -
+ + +
+ +
+
+
+ ); + } + + // ── Phase: record (microphone input) ─────────────────────────────────── + + if (phase === "record") { + return ( +
+
+ {}} /> +
+ +
+
+ Hold the mic and say: +
+
+ “{sentence}” +
+ + + + {evalError && ( +
+ {evalError} +
+ )} + +
+ ); + } + + // ── Phase: results (pronunciation feedback) ──────────────────────────── + + return ( +
+
+ {}} /> +
+ +
+ {evalResult && ( + <> +
+ Overall: {evalResult.overall_accuracy}% +
+ + + +
+ {evalResult.words.map((w, i) => ( +
+ {w.word} + = 70 ? "text-green-600" : "text-red-500" + }`} + > + {w.is_omitted ? "Omitted" : `${w.accuracy_score}%`} + +
+ ))} +
+ + )} +
diff --git a/components/PronunciationFeedback.tsx b/components/PronunciationFeedback.tsx index 7e79c98..93c7c81 100644 --- a/components/PronunciationFeedback.tsx +++ b/components/PronunciationFeedback.tsx @@ -20,7 +20,7 @@ import React, { useState, useEffect, useRef, useCallback } from "react"; interface Pictogram { id: string; label: string; - imageUrl: string; + imageUrl: string | null; } interface WordScore { @@ -34,7 +34,7 @@ type WordStatus = "pending" | "active" | "correct" | "incorrect"; interface WordState { id: string; label: string; - imageUrl: string; + imageUrl: string | null; status: WordStatus; score: number; isOmitted: boolean; @@ -280,13 +280,19 @@ export default function PronunciationFeedback({ }} aria-label={entry.label} > - {/* Pictogram image */} - {/* eslint-disable-next-line @next/next/no-img-element */} - + {/* Pictogram image or word fallback */} + {entry.imageUrl ? ( + /* eslint-disable-next-line @next/next/no-img-element */ + + ) : ( + + {entry.label} + + )} {/* Score ring (circular progress) */} {(entry.status === "correct" || entry.status === "incorrect") && ( diff --git a/lib/resolveArasaacImages.ts b/lib/resolveArasaacImages.ts new file mode 100644 index 0000000..91100ee --- /dev/null +++ b/lib/resolveArasaacImages.ts @@ -0,0 +1,51 @@ +/** + * Resolves ARASAAC pictogram image URLs for a list of words. + * Uses the same API endpoint as useArasaacImage hook but as a plain async function + * suitable for non-hook contexts. + */ + +export interface WordPictogram { + id: string; + label: string; + imageUrl: string | null; +} + +const imageCache = new Map(); + +async function fetchImageUrl(keyword: string): Promise { + if (imageCache.has(keyword)) return imageCache.get(keyword)!; + + try { + const res = await fetch( + `https://api.arasaac.org/v1/pictograms/en/search/${encodeURIComponent(keyword)}` + ); + const data = await res.json(); + if (Array.isArray(data) && data.length > 0) { + const id = data[0]._id; + const url = `https://static.arasaac.org/pictograms/${id}/${id}_2500.png`; + imageCache.set(keyword, url); + return url; + } + } catch { + // ignore fetch errors + } + + imageCache.set(keyword, null); + return null; +} + +export async function resolveArasaacImages( + words: string[] +): Promise { + const results = await Promise.all( + words.map(async (word, i) => { + const imageUrl = await fetchImageUrl(word.toLowerCase()); + return { + id: `word-${i}-${word}`, + label: word, + imageUrl, + }; + }) + ); + return results; +} From 5e1447382c5592794891bc9f4c3e7f6610cb958f Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 14 Mar 2026 17:30:42 +0000 Subject: [PATCH 2/4] chore: update package-lock.json after npm install https://claude.ai/code/session_01L7SVMJvoKVtzxK2CCJHirQ --- package-lock.json | 13 ------------- 1 file changed, 13 deletions(-) diff --git a/package-lock.json b/package-lock.json index a166674..2c3e77a 100644 --- a/package-lock.json +++ b/package-lock.json @@ -109,7 +109,6 @@ "integrity": "sha512-CGOfOJqWjg2qW/Mb6zNsDm+u5vFQ8DxXfbM09z69p5Z6+mE1ikP2jUXw+j42Pf1XTYED2Rni5f95npYeuwMDQA==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "@babel/code-frame": "^7.29.0", "@babel/generator": "^7.29.0", @@ -1918,7 +1917,6 @@ "integrity": "sha512-8kzdPJ3FsNsVIurqBs7oodNnCEVbni9yUEkaHbgptDACOPW04jimGagZ51E6+lXUwJjgnBw+hyko/lkFWCldqw==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "undici-types": "~6.21.0" } @@ -1929,7 +1927,6 @@ "integrity": "sha512-ilcTH/UniCkMdtexkoCN0bI7pMcJDvmQFPvuPvmEaYA/NSfFTAgdUSLAoVjaRJm7+6PvcM+q1zYOwS4wTYMF9w==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "csstype": "^3.2.2" } @@ -1995,7 +1992,6 @@ "integrity": "sha512-XZzOmihLIr8AD1b9hL9ccNMzEMWt/dE2u7NyTY9jJG6YNiNthaD5XtUHVF2uCXZ15ng+z2hT3MVuxnUYhq6k1g==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "@typescript-eslint/scope-manager": "8.57.0", "@typescript-eslint/types": "8.57.0", @@ -2670,7 +2666,6 @@ "integrity": "sha512-UVJyE9MttOsBQIDKw1skb9nAwQuR5wuGD3+82K6JgJlm/Y+KI92oNsMNGZCYdDsVtRHSak0pcV5Dno5+4jh9sw==", "dev": true, "license": "MIT", - "peer": true, "bin": { "acorn": "bin/acorn" }, @@ -3047,7 +3042,6 @@ } ], "license": "MIT", - "peer": true, "dependencies": { "baseline-browser-mapping": "^2.9.0", "caniuse-lite": "^1.0.30001759", @@ -3644,7 +3638,6 @@ "integrity": "sha512-XoMjdBOwe/esVgEvLmNsD3IRHkm7fbKIUGvrleloJXUZgDHig2IPWNniv+GwjyJXzuNqVjlr5+4yVUZjycJwfQ==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "@eslint-community/eslint-utils": "^4.8.0", "@eslint-community/regexpp": "^4.12.1", @@ -3830,7 +3823,6 @@ "integrity": "sha512-whOE1HFo/qJDyX4SnXzP4N6zOWn79WhnCUY/iDR0mPfQZO8wcYE4JClzI2oZrhBnnMUCBCHZhO6VQyoBU95mZA==", "dev": true, "license": "MIT", - "peer": true, "dependencies": { "@rtsao/scc": "^1.1.0", "array-includes": "^3.1.9", @@ -6151,7 +6143,6 @@ "resolved": "https://registry.npmjs.org/react/-/react-19.2.3.tgz", "integrity": "sha512-Ku/hhYbVjOQnXDZFv2+RibmLFGwFdeeKHFcOTlrt7xplBnya5OGn/hIRDsqDiSUcfORsDC7MPxwork8jBwsIWA==", "license": "MIT", - "peer": true, "engines": { "node": ">=0.10.0" } @@ -6161,7 +6152,6 @@ "resolved": "https://registry.npmjs.org/react-dom/-/react-dom-19.2.3.tgz", "integrity": "sha512-yELu4WmLPw5Mr/lmeEpox5rw3RETacE++JgHqQzd2dg+YbJuat3jH4ingc+WPZhxaoFzdv9y33G+F7Nl5O0GBg==", "license": "MIT", - "peer": true, "dependencies": { "scheduler": "^0.27.0" }, @@ -6922,7 +6912,6 @@ "integrity": "sha512-5gTmgEY/sqK6gFXLIsQNH19lWb4ebPDLA4SdLP7dsWkIXHWlG66oPuVvXSGFPppYZz8ZDZq0dYYrbHfBCVUb1Q==", "dev": true, "license": "MIT", - "peer": true, "engines": { "node": ">=12" }, @@ -7095,7 +7084,6 @@ "integrity": "sha512-jl1vZzPDinLr9eUt3J/t7V6FgNEw9QjvBPdysz9KfQDD41fQrC2Y4vKQdiaUpFT4bXlb1RHhLpp8wtm6M5TgSw==", "dev": true, "license": "Apache-2.0", - "peer": true, "bin": { "tsc": "bin/tsc", "tsserver": "bin/tsserver" @@ -7870,7 +7858,6 @@ "integrity": "sha512-rftlrkhHZOcjDwkGlnUtZZkvaPHCsDATp4pGpuOOMDaTdDDXF91wuVDJoWoPsKX/3YPQ5fHuF3STjcYyKr+Qhg==", "dev": true, "license": "MIT", - "peer": true, "funding": { "url": "https://github.com/sponsors/colinhacks" } From a9c0349e35ca0ad20c103b0a6e052502fc4e6f91 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 14 Mar 2026 17:30:11 +0000 Subject: [PATCH 3/4] feat: integrate speech testing into main pictogram flow MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Wire the speech practice pipeline (listen → record → results) into the OutputScreen so users can practice pronunciation of generated sentences directly from the pictogram navigator. The "Play" button now launches a 3-phase flow: paced TTS playback, hold-to-record with Azure evaluation, and karaoke-style pronunciation feedback with per-word scores. - Add resolveArasaacImages utility for dynamic sentence word pictograms - Make PronunciationFeedback handle missing image URLs gracefully - Add GitHub Actions CI workflow (lint + test + build) https://claude.ai/code/session_01L7SVMJvoKVtzxK2CCJHirQ --- .github/workflows/ci.yml | 30 +++ app/pictogram/page.tsx | 296 +++++++++++++++++++++++---- components/PronunciationFeedback.tsx | 24 ++- lib/resolveArasaacImages.ts | 51 +++++ 4 files changed, 353 insertions(+), 48 deletions(-) create mode 100644 .github/workflows/ci.yml create mode 100644 lib/resolveArasaacImages.ts diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml new file mode 100644 index 0000000..15ab664 --- /dev/null +++ b/.github/workflows/ci.yml @@ -0,0 +1,30 @@ +name: CI + +on: + push: + branches: [main] + pull_request: + branches: [main] + +jobs: + ci: + runs-on: ubuntu-latest + + steps: + - uses: actions/checkout@v4 + + - uses: actions/setup-node@v4 + with: + node-version: 20 + cache: npm + + - run: npm ci + + - name: Lint + run: npm run lint + + - name: Test + run: npm test + + - name: Build + run: npm run build diff --git a/app/pictogram/page.tsx b/app/pictogram/page.tsx index 309fb83..f096267 100644 --- a/app/pictogram/page.tsx +++ b/app/pictogram/page.tsx @@ -5,6 +5,13 @@ import DesktopLayout from "@/components/DesktopLayout"; import { pictogramTree, PictogramNode } from "@/lib/pictogramTree"; import { useArasaacImage } from "@/lib/useArasaacImage"; import { useSpeech } from "@/hooks/useSpeech"; +import { useSpeechPacing } from "@/hooks/useSpeechPacing"; +import AudioRecorder from "@/components/AudioRecorder"; +import PronunciationFeedback from "@/components/PronunciationFeedback"; +import { + resolveArasaacImages, + WordPictogram, +} from "@/lib/resolveArasaacImages"; function IconChevronSmall() { return ( @@ -189,8 +196,34 @@ function TrailChips({ trail }: { trail: PictogramNode[] }) { ); } +// ─── Types for speech evaluation ───────────────────────────────────────────── + +interface EvaluationWord { + word: string; + accuracy_score: number; + is_omitted: boolean; +} + +interface EvaluationResult { + target: string; + overall_accuracy: number; + words: EvaluationWord[]; +} + +function IconMic() { + return ( + + + + + + ); +} + // ─── Output screen ──────────────────────────────────────────────────────────── +type OutputPhase = "output" | "listen" | "record" | "results"; + function OutputScreen({ trail, onBack, @@ -200,10 +233,17 @@ function OutputScreen({ }) { const [sentence, setSentence] = useState(null); const [loading, setLoading] = useState(true); + const [phase, setPhase] = useState("output"); + const [evalResult, setEvalResult] = useState(null); + const [evalError, setEvalError] = useState(null); + const [wordPictograms, setWordPictograms] = useState([]); const fetchSentence = useCallback(async () => { setLoading(true); setSentence(null); + setPhase("output"); + setEvalResult(null); + setEvalError(null); try { const res = await fetch("/api/generate", { method: "POST", @@ -216,13 +256,15 @@ function OutputScreen({ }), }); const data = await res.json(); - if (data.sentence) { - setSentence(data.sentence); - } else { - setSentence("I need help."); - } + const text = data.sentence || "I need help."; + setSentence(text); + // Pre-resolve pictogram images for the sentence words + const words = text.trim().split(/\s+/); + resolveArasaacImages(words).then(setWordPictograms); } catch { - setSentence("I need help."); + const fallback = "I need help."; + setSentence(fallback); + resolveArasaacImages(fallback.replace(".", "").split(/\s+/)).then(setWordPictograms); } finally { setLoading(false); } @@ -238,66 +280,242 @@ function OutputScreen({ }, [fetchSentence]); const { speak } = useSpeech(); + const { speakWithPacing } = useSpeechPacing(); - // Auto-speak when sentence arrives + // Auto-speak when sentence arrives (only in output phase) useEffect(() => { - if (sentence) { + if (sentence && phase === "output") { speak(sentence, { rate: 0.85, pitch: 1 }); } - }, [sentence, speak]); + }, [sentence, speak, phase]); + + const handleStartPractice = useCallback(() => { + if (!sentence) return; + setEvalResult(null); + setEvalError(null); + setPhase("listen"); + speakWithPacing(sentence); + }, [sentence, speakWithPacing]); + + const handleEvalResult = useCallback((result: EvaluationResult) => { + setEvalResult(result); + setPhase("results"); + }, []); + + const handleEvalError = useCallback((err: string) => { + setEvalError(err); + }, []); + + const handleTryAgain = useCallback(() => { + setEvalResult(null); + setEvalError(null); + setPhase("listen"); + if (sentence) speakWithPacing(sentence); + }, [sentence, speakWithPacing]); + + // ── Phase: output (sentence display) ─────────────────────────────────── + + if (phase === "output") { + return ( +
+
+ {}} /> +
- return ( -
-
- {}} /> +
+ {loading ? ( +
+
+
+
+ ) : ( +
+ {sentence} +
+ )} + +
+ + +
+
+ +
+ + +
+ ); + } + + // ── Phase: listen (paced TTS playback) ───────────────────────────────── + + if (phase === "listen") { + return ( +
+
+ {}} /> +
-
- {loading ? ( -
-
-
+
+
+ Listen carefully
- ) : ( -
+
{sentence}
- )} -
+ + +
+ +
+
+
+ ); + } + + // ── Phase: record (microphone input) ─────────────────────────────────── + + if (phase === "record") { + return ( +
+
+ {}} /> +
+ +
+
+ Hold the mic and say: +
+
+ “{sentence}” +
+ + + + {evalError && ( +
+ {evalError} +
+ )} + +
+ ); + } + + // ── Phase: results (pronunciation feedback) ──────────────────────────── + + return ( +
+
+ {}} /> +
+ +
+ {evalResult && ( + <> +
+ Overall: {evalResult.overall_accuracy}% +
+ + + +
+ {evalResult.words.map((w, i) => ( +
+ {w.word} + = 70 ? "text-green-600" : "text-red-500" + }`} + > + {w.is_omitted ? "Omitted" : `${w.accuracy_score}%`} + +
+ ))} +
+ + )} +
diff --git a/components/PronunciationFeedback.tsx b/components/PronunciationFeedback.tsx index 7e79c98..93c7c81 100644 --- a/components/PronunciationFeedback.tsx +++ b/components/PronunciationFeedback.tsx @@ -20,7 +20,7 @@ import React, { useState, useEffect, useRef, useCallback } from "react"; interface Pictogram { id: string; label: string; - imageUrl: string; + imageUrl: string | null; } interface WordScore { @@ -34,7 +34,7 @@ type WordStatus = "pending" | "active" | "correct" | "incorrect"; interface WordState { id: string; label: string; - imageUrl: string; + imageUrl: string | null; status: WordStatus; score: number; isOmitted: boolean; @@ -280,13 +280,19 @@ export default function PronunciationFeedback({ }} aria-label={entry.label} > - {/* Pictogram image */} - {/* eslint-disable-next-line @next/next/no-img-element */} - + {/* Pictogram image or word fallback */} + {entry.imageUrl ? ( + /* eslint-disable-next-line @next/next/no-img-element */ + + ) : ( + + {entry.label} + + )} {/* Score ring (circular progress) */} {(entry.status === "correct" || entry.status === "incorrect") && ( diff --git a/lib/resolveArasaacImages.ts b/lib/resolveArasaacImages.ts new file mode 100644 index 0000000..91100ee --- /dev/null +++ b/lib/resolveArasaacImages.ts @@ -0,0 +1,51 @@ +/** + * Resolves ARASAAC pictogram image URLs for a list of words. + * Uses the same API endpoint as useArasaacImage hook but as a plain async function + * suitable for non-hook contexts. + */ + +export interface WordPictogram { + id: string; + label: string; + imageUrl: string | null; +} + +const imageCache = new Map(); + +async function fetchImageUrl(keyword: string): Promise { + if (imageCache.has(keyword)) return imageCache.get(keyword)!; + + try { + const res = await fetch( + `https://api.arasaac.org/v1/pictograms/en/search/${encodeURIComponent(keyword)}` + ); + const data = await res.json(); + if (Array.isArray(data) && data.length > 0) { + const id = data[0]._id; + const url = `https://static.arasaac.org/pictograms/${id}/${id}_2500.png`; + imageCache.set(keyword, url); + return url; + } + } catch { + // ignore fetch errors + } + + imageCache.set(keyword, null); + return null; +} + +export async function resolveArasaacImages( + words: string[] +): Promise { + const results = await Promise.all( + words.map(async (word, i) => { + const imageUrl = await fetchImageUrl(word.toLowerCase()); + return { + id: `word-${i}-${word}`, + label: word, + imageUrl, + }; + }) + ); + return results; +} From 23a6888d3906f45b24eb667864f1edaeb3c18727 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 14 Mar 2026 18:01:43 +0000 Subject: [PATCH 4/4] feat: update speech practice to word-by-word teaching flow MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Rework the speech practice integration to match the new word-by-word approach: each word is spoken 3× at progressively slower rates, then the learner pronounces just that word, and gets pass/fail feedback before moving to the next. Includes progress dots, ARASAAC pictogram images per word, Azure missing key fallback, and a "great job" completion screen. https://claude.ai/code/session_01L7SVMJvoKVtzxK2CCJHirQ --- app/pictogram/page.tsx | 570 ++++++++++++++++++++++++----------------- 1 file changed, 339 insertions(+), 231 deletions(-) diff --git a/app/pictogram/page.tsx b/app/pictogram/page.tsx index f096267..48078ae 100644 --- a/app/pictogram/page.tsx +++ b/app/pictogram/page.tsx @@ -5,9 +5,7 @@ import DesktopLayout from "@/components/DesktopLayout"; import { pictogramTree, PictogramNode } from "@/lib/pictogramTree"; import { useArasaacImage } from "@/lib/useArasaacImage"; import { useSpeech } from "@/hooks/useSpeech"; -import { useSpeechPacing } from "@/hooks/useSpeechPacing"; import AudioRecorder from "@/components/AudioRecorder"; -import PronunciationFeedback from "@/components/PronunciationFeedback"; import { resolveArasaacImages, WordPictogram, @@ -196,7 +194,7 @@ function TrailChips({ trail }: { trail: PictogramNode[] }) { ); } -// ─── Types for speech evaluation ───────────────────────────────────────────── +// ─── Speech evaluation types ────────────────────────────────────────────────── interface EvaluationWord { word: string; @@ -210,20 +208,308 @@ interface EvaluationResult { words: EvaluationWord[]; } -function IconMic() { +// ─── Speech helpers ─────────────────────────────────────────────────────────── + +// Each repetition gets slower and more emphatic +const SPEECH_PASSES = [ + { rate: 0.4, pitch: 1.2 }, + { rate: 0.2, pitch: 1.1 }, + { rate: 0.1, pitch: 1.0 }, +]; + +function speakAt(text: string, rate: number, pitch: number): Promise { + return new Promise((resolve) => { + if (typeof window === "undefined" || !window.speechSynthesis) { resolve(); return; } + window.speechSynthesis.cancel(); + const utt = new SpeechSynthesisUtterance(text); + utt.rate = rate; utt.pitch = pitch; utt.volume = 1; utt.lang = "en-US"; + utt.onend = () => resolve(); utt.onerror = () => resolve(); + window.speechSynthesis.speak(utt); + }); +} + +function msDelay(ms: number): Promise { + return new Promise((resolve) => setTimeout(resolve, ms)); +} + +// ─── Word-by-word speech practice ──────────────────────────────────────────── + +type WordPhase = "teaching" | "pronounce" | "result"; + +function SpeechPractice({ + words, + onBack, +}: { + words: WordPictogram[]; + onBack: () => void; +}) { + const [wordIndex, setWordIndex] = useState(0); + const [phase, setWordPhase] = useState("teaching"); + const [wordResult, setWordResult] = useState(null); + const [allDone, setAllDone] = useState(false); + const [error, setError] = useState(null); + const [isSpeaking, setIsSpeaking] = useState(false); + const [speakPass, setSpeakPass] = useState(0); + const cancelledRef = useRef(false); + + const currentWord = words[wordIndex]; + // Single-letter "I" gets a period so TTS doesn't spell it out + const ttsText = currentWord.label === "I" ? "I." : currentWord.label; + + const playWordSlowly = useCallback(async () => { + if (isSpeaking) return; + cancelledRef.current = false; + setIsSpeaking(true); + for (let i = 0; i < SPEECH_PASSES.length; i++) { + if (cancelledRef.current) break; + setSpeakPass(i); + const { rate, pitch } = SPEECH_PASSES[i]; + await speakAt(ttsText, rate, pitch); + if (i < SPEECH_PASSES.length - 1 && !cancelledRef.current) await msDelay(1800); + } + setIsSpeaking(false); + }, [ttsText, isSpeaking]); + + // Auto-play when entering teaching phase + useEffect(() => { + if (phase !== "teaching") return; + let isCancelled = false; + cancelledRef.current = false; + const run = async () => { + setIsSpeaking(true); + await msDelay(400); + for (let i = 0; i < SPEECH_PASSES.length; i++) { + if (isCancelled || cancelledRef.current) break; + setSpeakPass(i); + const { rate, pitch } = SPEECH_PASSES[i]; + await speakAt(ttsText, rate, pitch); + if (i < SPEECH_PASSES.length - 1 && !isCancelled && !cancelledRef.current) { + await msDelay(1800); + } + } + if (!isCancelled) setIsSpeaking(false); + }; + run(); + return () => { + isCancelled = true; + cancelledRef.current = true; + window.speechSynthesis?.cancel(); + }; + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [wordIndex, phase]); + + const handleResult = useCallback((result: EvaluationResult) => { + setWordResult(result.words[0] ?? null); + setError(null); + setWordPhase("result"); + }, []); + + const handleNext = useCallback(() => { + cancelledRef.current = true; + window.speechSynthesis?.cancel(); + if (wordIndex < words.length - 1) { + setWordIndex((i) => i + 1); + setWordResult(null); + setError(null); + setWordPhase("teaching"); + } else { + setAllDone(true); + } + }, [wordIndex, words.length]); + + const handleRetry = useCallback(() => { + setWordResult(null); + setError(null); + setWordPhase("teaching"); + }, []); + + // ── All done ───────────────────────────────────────────────────────────── + + if (allDone) { + return ( +
+
🎉
+
Great job!
+
You said all the words!
+ +
+ ); + } + + // ── Main practice screen ────────────────────────────────────────────────── + return ( - - - - - +
+ {/* Progress dots */} +
+ {words.map((_, i) => ( +
+ ))} +
+ +
+ {/* Pictogram */} +
+ {currentWord.imageUrl ? ( + // eslint-disable-next-line @next/next/no-img-element + {currentWord.label} + ) : ( + {currentWord.label} + )} +
+ + {/* Word */} +
+ {currentWord.label} +
+ + {/* Teaching phase */} + {phase === "teaching" && ( +
+
+
+ 🔊 + {speakPass === 0 ? "Listen…" : speakPass === 1 ? "Slower…" : "Very slow…"} +
+
+ {SPEECH_PASSES.map((_, i) => ( +
+ ))} +
+
+ + + + +
+ )} + + {/* Pronounce phase */} + {phase === "pronounce" && ( +
+
+ Say: “{currentWord.label}” +
+ + setError(e)} + /> + + +
+ )} + + {/* Result phase */} + {phase === "result" && wordResult && ( +
+ {wordResult.accuracy_score >= 70 ? ( + <> +
✅
+
Well done!
+
{wordResult.accuracy_score}% correct
+ + + ) : ( + <> +
🔄
+
Try again!
+
+ {wordResult.is_omitted ? "Not heard" : `${wordResult.accuracy_score}% correct`} +
+ + + + )} +
+ )} + + {/* Error display */} + {error && ( +
+ {error.includes("AZURE_SPEECH_KEY") ? ( + <> +
Speech check not set up yet
+
Azure credentials are missing. You can still practice!
+ + + ) : ( +
{error}
+ )} +
+ )} +
+ +
+ +
+
); } // ─── Output screen ──────────────────────────────────────────────────────────── -type OutputPhase = "output" | "listen" | "record" | "results"; - function OutputScreen({ trail, onBack, @@ -233,38 +519,32 @@ function OutputScreen({ }) { const [sentence, setSentence] = useState(null); const [loading, setLoading] = useState(true); - const [phase, setPhase] = useState("output"); - const [evalResult, setEvalResult] = useState(null); - const [evalError, setEvalError] = useState(null); + const [showPractice, setShowPractice] = useState(false); const [wordPictograms, setWordPictograms] = useState([]); const fetchSentence = useCallback(async () => { setLoading(true); setSentence(null); - setPhase("output"); - setEvalResult(null); - setEvalError(null); + setShowPractice(false); try { const res = await fetch("/api/generate", { method: "POST", headers: { "Content-Type": "application/json" }, body: JSON.stringify({ trail: trail.map((n) => n.label), - context: trail - .map((n) => n.llmContext) - .filter(Boolean), + context: trail.map((n) => n.llmContext).filter(Boolean), }), }); const data = await res.json(); const text = data.sentence || "I need help."; setSentence(text); - // Pre-resolve pictogram images for the sentence words + // Pre-resolve pictogram images for each sentence word const words = text.trim().split(/\s+/); resolveArasaacImages(words).then(setWordPictograms); } catch { const fallback = "I need help."; setSentence(fallback); - resolveArasaacImages(fallback.replace(".", "").split(/\s+/)).then(setWordPictograms); + resolveArasaacImages(["I", "need", "help"]).then(setWordPictograms); } finally { setLoading(false); } @@ -272,7 +552,6 @@ function OutputScreen({ const fetchedRef = useRef(false); - // Fetch on mount useEffect(() => { if (fetchedRef.current) return; fetchedRef.current = true; @@ -280,242 +559,71 @@ function OutputScreen({ }, [fetchSentence]); const { speak } = useSpeech(); - const { speakWithPacing } = useSpeechPacing(); - // Auto-speak when sentence arrives (only in output phase) + // Auto-speak when sentence arrives useEffect(() => { - if (sentence && phase === "output") { + if (sentence && !showPractice) { speak(sentence, { rate: 0.85, pitch: 1 }); } - }, [sentence, speak, phase]); - - const handleStartPractice = useCallback(() => { - if (!sentence) return; - setEvalResult(null); - setEvalError(null); - setPhase("listen"); - speakWithPacing(sentence); - }, [sentence, speakWithPacing]); - - const handleEvalResult = useCallback((result: EvaluationResult) => { - setEvalResult(result); - setPhase("results"); - }, []); + }, [sentence, speak, showPractice]); - const handleEvalError = useCallback((err: string) => { - setEvalError(err); - }, []); - - const handleTryAgain = useCallback(() => { - setEvalResult(null); - setEvalError(null); - setPhase("listen"); - if (sentence) speakWithPacing(sentence); - }, [sentence, speakWithPacing]); - - // ── Phase: output (sentence display) ─────────────────────────────────── - - if (phase === "output") { - return ( -
-
- {}} /> -
- -
- {loading ? ( -
-
-
-
- ) : ( -
- {sentence} -
- )} - -
- - -
-
- -
- - -
-
- ); + if (showPractice && wordPictograms.length > 0) { + return setShowPractice(false)} />; } - // ── Phase: listen (paced TTS playback) ───────────────────────────────── - - if (phase === "listen") { - return ( -
-
- {}} /> -
+ return ( +
+
+ {}} /> +
-
-
- Listen carefully +
+ {loading ? ( +
+
+
-
+ ) : ( +
{sentence}
+ )} +
- -
- -
- -
-
- ); - } - - // ── Phase: record (microphone input) ─────────────────────────────────── - - if (phase === "record") { - return ( -
-
- {}} /> -
- -
-
- Hold the mic and say: -
-
- “{sentence}” -
- - - - {evalError && ( -
- {evalError} -
- )} - -
- ); - } - - // ── Phase: results (pronunciation feedback) ──────────────────────────── - - return ( -
-
- {}} /> -
- -
- {evalResult && ( - <> -
- Overall: {evalResult.overall_accuracy}% -
- - - -
- {evalResult.words.map((w, i) => ( -
- {w.word} - = 70 ? "text-green-600" : "text-red-500" - }`} - > - {w.is_omitted ? "Omitted" : `${w.accuracy_score}%`} - -
- ))} -
- - )} -
- +