diff --git a/cmd/server/main.go b/cmd/server/main.go index 9c677a6..efdd170 100644 --- a/cmd/server/main.go +++ b/cmd/server/main.go @@ -200,7 +200,11 @@ func main() { // back to the browser's Web Speech API on its own. if ttsHandler, ok := tts.New(cfg); ok { pr.Mount("/tts", ttsHandler.Routes()) - log.Printf("read-aloud enabled (TTS endpoint=%s)", cfg.TTSEndpoint) + // Name the languages, not just the English endpoint: which + // voices a deployment actually reached is the thing worth + // seeing at boot, and a missing sidecar is silent otherwise + // (a 404 the client answers by quietly using Web Speech). + log.Printf("read-aloud enabled (voices: %s)", strings.Join(ttsHandler.Languages(), ", ")) } }) }) diff --git a/deploy/README.md b/deploy/README.md index 1c4c442..d4e7bab 100644 --- a/deploy/README.md +++ b/deploy/README.md @@ -579,6 +579,35 @@ Petal's env then carries `TTS_ENDPOINT=http://127.0.0.1:5005`, maps language → instance from config, so another language is another instance plus an env pair, no code change. +**Adding a language (Phase 21 made this literal).** Petal discovers its Piper +instances from the environment: English is the unsuffixed +`TTS_ENDPOINT`/`TTS_VOICE_EN`, and every other language is a +`TTS_ENDPOINT_`/`TTS_VOICE_` pair. `` is the *base* tag — +`PT`, not `PT_PT`, because an environment variable name cannot hold a hyphen and +only one Portuguese model is loaded regardless. Both halves must be set: an +endpoint with no voice is dropped, so a half-finished language reads to the +browser as "no voice here, use Web Speech" instead of erroring on every tap. The +startup line names what it actually resolved: + +``` +read-aloud enabled (voices: en=en_US-amy-medium, pt=pt_PT-tugão-medium, zh=zh_CN-huayan-medium) +``` + +**Portuguese: `pt_PT-tugão-medium` is the only European voice Piper ships.** The +other five `pt_*` models in the catalogue are all Brazilian, so the voice has to +be named explicitly for the same reason the Hunspell dictionary did (Phase 21): +the obvious default is the wrong country. Check what exists before assuming: + +```bash +docker exec petal-piper-en python -c "import urllib.request,json; \ +d=json.load(urllib.request.urlopen('https://huggingface.co/rhasspy/piper-voices/resolve/main/voices.json')); \ +print([k for k in d if k.startswith('pt')])" +``` + +**Slow replay.** `POST /api/tts` takes `slow: true`, which raises Piper's +`length_scale` to about 4/3 (≈0.75× pace). It is a separate cache entry, not a +playback-rate trick, so the slow clip is synthesized once and then instant. + **Piper version note:** piper-tts moved synthesis from `POST /` to `POST /synthesize` in 1.6.0, with an identical request body. `TTS_PATH` selects which — it defaults to `/`, and both the VPS compose and millenia's `start.sh` diff --git a/deploy/petal.env.example b/deploy/petal.env.example index e48223d..e6f3948 100644 --- a/deploy/petal.env.example +++ b/deploy/petal.env.example @@ -49,8 +49,17 @@ LLM_TIMEOUT=90s # --- Read-aloud (Piper sidecars) --------------------------------------------- # Endpoints are wired in docker-compose.yml; these pick the voice each sidecar # loads. Changing one means recreating that container so it downloads the model. +# +# A language is routable only when both halves are set — a TTS_ENDPOINT_XX with +# no TTS_VOICE_XX reads as "no voice for this language" and the browser's own +# synthesizer takes over, rather than as an instance that errors on every +# request. Adding fr or es is a compose service plus a pair of lines here. +# +# pt_PT-tugão-medium is the only European Portuguese voice Piper ships; every +# other pt model in the catalogue is Brazilian. TTS_VOICE_EN=en_US-amy-medium TTS_VOICE_ZH=zh_CN-huayan-medium +TTS_VOICE_PT=pt_PT-tugão-medium TTS_AUDIO_FORMAT=mp3 TTS_TIMEOUT=15s diff --git a/docker-compose.yml b/docker-compose.yml index 23eaf6e..1063346 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -46,6 +46,11 @@ services: # separate containers; the handler maps language → instance from config. TTS_ENDPOINT: http://piper-en:5000 TTS_ENDPOINT_ZH: http://piper-zh:5000 + # A language is discovered from the TTS_ENDPOINT_/TTS_VOICE_ + # pair, so fr and es cost a service and two lines rather than a code + # change. is the base tag — an env var name can't hold pt-PT's + # hyphen, and there is one Portuguese voice loaded either way. + TTS_ENDPOINT_PT: http://piper-pt:5000 # The sidecars run piper-tts 1.6.0, which serves synthesis on # /synthesize; millenia's older server keeps the default "/". TTS_PATH: /synthesize @@ -75,6 +80,7 @@ services: depends_on: - piper-en - piper-zh + - piper-pt labels: traefik.enable: "true" traefik.docker.network: traefik @@ -123,6 +129,24 @@ services: networks: - internal + # European Portuguese, for the pt-PT pair. pt_PT-tugão-medium is the *only* + # European voice in Piper's catalogue — the other five Portuguese models are + # all pt_BR — so the default anyone reaches for is the Brazilian one, exactly + # as it was with the Hunspell dictionary in Phase 21. Named here rather than + # left to the image default for that reason. + piper-pt: + build: + context: deploy/piper + image: petal-piper:local + container_name: petal-piper-pt + restart: unless-stopped + environment: + PIPER_VOICE: ${TTS_VOICE_PT:-pt_PT-tugão-medium} + volumes: + - piper-voices:/voices + networks: + - internal + networks: # Created and owned by the host's Traefik stack. traefik: diff --git a/internal/config/config.go b/internal/config/config.go index 70f1d81..fbc448d 100644 --- a/internal/config/config.go +++ b/internal/config/config.go @@ -2,6 +2,7 @@ package config import ( "os" + "strings" "time" ) @@ -29,10 +30,16 @@ type Config struct { // TTS (read-aloud). Off unless TTSEndpoint is set — when empty, the /api/tts // route isn't mounted and the frontend falls back to the browser's Web Speech // API. Endpoint points at a local Piper HTTP server. - TTSEndpoint string // Piper instance serving the English voice - TTSEndpointZH string // Piper instance serving the Chinese voice; empty = zh falls back to Web Speech - TTSVoiceEN string // Piper voice id for English (e.g. en_US-amy-medium) - TTSVoiceZH string // Piper voice id for Chinese (e.g. zh_CN-huayan-medium) + TTSEndpoint string // Piper instance serving the English voice; also the on/off switch + // TTSVoices is every language Petal can read aloud, keyed by base language + // tag ("en", "zh", "pt", …). Each Piper server loads exactly one model, so + // a language *is* an instance — and the instances are discovered from the + // environment rather than named in this struct: one + // TTS_ENDPOINT_/TTS_VOICE_ pair per language, so the fr and es + // pairs cost a compose service and two lines of .env rather than a code + // change. English keeps the unsuffixed TTS_ENDPOINT/TTS_VOICE_EN it has + // always had. + TTSVoices map[string]TTSVoice // TTSPath is the path Piper serves synthesis on. Piper moved it from "/" to // "/synthesize" in 1.6.0 with an unchanged request body, so this is a // version knob, not a feature: millenia's older server keeps the default, @@ -56,6 +63,12 @@ type Config struct { AllowedSubs string } +// TTSVoice is one Piper instance and the single voice it has loaded. +type TTSVoice struct { + Endpoint string + Voice string +} + // AuthEnabled reports whether real logins are configured. When false, Petal // resolves every request to the local user. func (c *Config) AuthEnabled() bool { @@ -77,14 +90,12 @@ func Load() *Config { LLMChatModel: env("LLM_CHAT_MODEL", ""), LLMTimeout: envDuration("LLM_TIMEOUT", 30*time.Second), - TTSEndpoint: env("TTS_ENDPOINT", ""), - TTSEndpointZH: env("TTS_ENDPOINT_ZH", ""), - TTSVoiceEN: env("TTS_VOICE_EN", "en_US-amy-medium"), - TTSVoiceZH: env("TTS_VOICE_ZH", "zh_CN-huayan-medium"), - TTSPath: env("TTS_PATH", "/"), - TTSCacheDir: env("TTS_CACHE_DIR", "./data/tts"), - TTSTimeout: envDuration("TTS_TIMEOUT", 15*time.Second), - TTSFormat: env("TTS_AUDIO_FORMAT", "mp3"), + TTSEndpoint: env("TTS_ENDPOINT", ""), + TTSVoices: ttsVoices(os.Environ()), + TTSPath: env("TTS_PATH", "/"), + TTSCacheDir: env("TTS_CACHE_DIR", "./data/tts"), + TTSTimeout: envDuration("TTS_TIMEOUT", 15*time.Second), + TTSFormat: env("TTS_AUDIO_FORMAT", "mp3"), AuthentikURL: env("AUTHENTIK_URL", ""), AuthentikClientID: env("AUTHENTIK_CLIENT_ID", ""), @@ -93,6 +104,64 @@ func Load() *Config { } } +// ttsVoices reads the Piper instances out of an environment slice (as returned +// by os.Environ) into a map keyed by base language tag. +// +// English is the unsuffixed pair, TTS_ENDPOINT + TTS_VOICE_EN, because that is +// what every deployment already sets and read-aloud has always been English +// first. Every other language is a TTS_ENDPOINT_/TTS_VOICE_ pair, +// discovered rather than enumerated — TTS_ENDPOINT_ZH is what millenia and the +// VPS already use, and TTS_ENDPOINT_PT is all the Portuguese pair needs. +// +// is the *base* tag: an environment variable name cannot hold the hyphen +// in "pt-PT", and the handler routes on the base tag anyway (a request for +// pt-PT, pt-BR or bare pt reaches the same instance, because there is only one +// Portuguese voice loaded). A pair is ignored unless both halves are set: half +// a configuration should read as "no voice for this language" and fall back to +// the browser, not as an instance that answers every request with an error. +func ttsVoices(environ []string) map[string]TTSVoice { + vals := make(map[string]string, len(environ)) + for _, kv := range environ { + if k, v, ok := strings.Cut(kv, "="); ok { + vals[k] = v + } + } + + voices := map[string]TTSVoice{} + add := func(lang, endpoint, voice string) { + endpoint = strings.TrimRight(strings.TrimSpace(endpoint), "/") + voice = strings.TrimSpace(voice) + if endpoint == "" || voice == "" { + return + } + voices[lang] = TTSVoice{Endpoint: endpoint, Voice: voice} + } + + // The two languages that shipped before this was a map keep their voice + // defaults, so an existing deployment that names only the endpoints (as + // millenia's unit does) sounds exactly as it did. + voiceOr := func(key, fallback string) string { + if v := strings.TrimSpace(vals[key]); v != "" { + return v + } + return fallback + } + + add("en", vals["TTS_ENDPOINT"], voiceOr("TTS_VOICE_EN", "en_US-amy-medium")) + for k, endpoint := range vals { + suffix, ok := strings.CutPrefix(k, "TTS_ENDPOINT_") + if !ok || suffix == "" { + continue + } + voice := vals["TTS_VOICE_"+suffix] + if suffix == "ZH" { + voice = voiceOr("TTS_VOICE_ZH", "zh_CN-huayan-medium") + } + add(strings.ToLower(suffix), endpoint, voice) + } + return voices +} + func env(key, fallback string) string { if v := os.Getenv(key); v != "" { return v diff --git a/internal/config/config_test.go b/internal/config/config_test.go new file mode 100644 index 0000000..9372d96 --- /dev/null +++ b/internal/config/config_test.go @@ -0,0 +1,83 @@ +package config + +import "testing" + +// The Piper instances are discovered from the environment rather than named in +// code, so that a new pair costs a compose service and two .env lines. These +// assert the discovery rule, including the two shapes that already exist in the +// wild (millenia's systemd unit and the VPS compose file). +func TestTTSVoicesDiscovery(t *testing.T) { + voices := ttsVoices([]string{ + "TTS_ENDPOINT=http://piper-en:5000", + "TTS_VOICE_EN=en_US-amy-medium", + "TTS_ENDPOINT_ZH=http://piper-zh:5000/", + "TTS_VOICE_ZH=zh_CN-huayan-medium", + "TTS_ENDPOINT_PT=http://piper-pt:5000", + "TTS_VOICE_PT=pt_PT-tugão-medium", + // Noise that must not become a language. + "TTS_PATH=/synthesize", + "PATH=/usr/bin", + }) + + want := map[string]TTSVoice{ + "en": {"http://piper-en:5000", "en_US-amy-medium"}, + // The trailing slash is trimmed here so the synthesis path concatenates + // cleanly rather than producing a double slash at every call site. + "zh": {"http://piper-zh:5000", "zh_CN-huayan-medium"}, + "pt": {"http://piper-pt:5000", "pt_PT-tugão-medium"}, + } + if len(voices) != len(want) { + t.Fatalf("discovered %v, want %v", voices, want) + } + for lang, w := range want { + if voices[lang] != w { + t.Errorf("%s = %+v, want %+v", lang, voices[lang], w) + } + } +} + +// Half a configuration is not a language. An endpoint with no voice (or the +// reverse) must read as "no voice for this language" — a 404 the client answers +// by falling back to Web Speech — rather than as an instance that exists and +// errors on every request. +func TestTTSVoicesIgnoresHalfConfiguredLanguages(t *testing.T) { + voices := ttsVoices([]string{ + "TTS_ENDPOINT=http://piper-en:5000", + "TTS_VOICE_EN=en_US-amy-medium", + "TTS_ENDPOINT_FR=http://piper-fr:5000", // no TTS_VOICE_FR + "TTS_VOICE_ES=es_ES-davefx-medium", // no TTS_ENDPOINT_ES + }) + if _, ok := voices["fr"]; ok { + t.Errorf("fr routed with no voice configured") + } + if _, ok := voices["es"]; ok { + t.Errorf("es routed with no endpoint configured") + } + if len(voices) != 1 { + t.Errorf("discovered %v, want English only", voices) + } +} + +// A deployment that predates the map names only the endpoints and relies on the +// voice defaults; it must sound exactly as it did. +func TestTTSVoicesKeepsTheOriginalDefaults(t *testing.T) { + voices := ttsVoices([]string{ + "TTS_ENDPOINT=http://127.0.0.1:5005", + "TTS_ENDPOINT_ZH=http://127.0.0.1:5006", + }) + if got := voices["en"].Voice; got != "en_US-amy-medium" { + t.Errorf("en voice = %q, want the default", got) + } + if got := voices["zh"].Voice; got != "zh_CN-huayan-medium" { + t.Errorf("zh voice = %q, want the default", got) + } +} + +// Read-aloud is off when no English instance is configured; nothing else may +// switch it on. (tts.New gates on TTSEndpoint, so a stray TTS_ENDPOINT_PT with +// no English sibling must not produce a routable map that outlives that gate.) +func TestTTSVoicesEmptyWithoutEndpoints(t *testing.T) { + if voices := ttsVoices([]string{"TTS_VOICE_EN=en_US-amy-medium"}); len(voices) != 0 { + t.Errorf("discovered %v, want none", voices) + } +} diff --git a/internal/tts/handler.go b/internal/tts/handler.go index bd1b806..26f7060 100644 --- a/internal/tts/handler.go +++ b/internal/tts/handler.go @@ -19,6 +19,7 @@ import ( "os" "os/exec" "path/filepath" + "sort" "strings" "time" "unicode/utf8" @@ -36,6 +37,12 @@ const maxTextBytes = 4000 // with the words, mirroring the old utterance.rate = 0.95. Higher = slower. const lengthScale = 1.1 +// slowLengthScale is the "say it slower" replay (SUGGESTIONS §5e): roughly 0.75× +// the normal pace, which is the speed listening drills have used for decades. +// Piper stretches durations rather than resampling, so the voice keeps its pitch +// instead of turning into a slowed tape. +const slowLengthScale = lengthScale / 0.75 + // audioFormat describes one output encoding: the cache-file extension, the // response Content-Type, and the ffmpeg args that turn Piper's WAV (on stdin) // into this format (on stdout). A nil ffmpegArgs means "serve the WAV as-is". @@ -105,16 +112,14 @@ func New(cfg *config.Config) (*Handler, bool) { format = formats["mp3"] } - // Map by base language so en-US, en-GB, etc. all resolve to the English - // instance (the client sends BCP-47 tags like the old Web Speech path did). - // A language is only routable when both its endpoint and voice are set; - // otherwise the client falls back to Web Speech for that language. + // Keyed by base language so en-US, en-GB — and pt-PT, pt-BR, bare pt — + // resolve to the one instance that has that language's model loaded (the + // client sends BCP-47 tags, as the old Web Speech path did). Config has + // already dropped any language configured by halves, so an unroutable + // language reaches the client as a 404 and falls back to Web Speech. routes := map[string]route{} - if cfg.TTSVoiceEN != "" { - routes["en"] = route{strings.TrimRight(cfg.TTSEndpoint, "/"), cfg.TTSVoiceEN} - } - if cfg.TTSEndpointZH != "" && cfg.TTSVoiceZH != "" { - routes["zh"] = route{strings.TrimRight(cfg.TTSEndpointZH, "/"), cfg.TTSVoiceZH} + for lang, v := range cfg.TTSVoices { + routes[lang] = route{endpoint: v.Endpoint, voice: v.Voice} } if err := os.MkdirAll(cfg.TTSCacheDir, 0o755); err != nil { @@ -132,6 +137,19 @@ func New(cfg *config.Config) (*Handler, bool) { }, true } +// Languages lists the base language tags this handler can synthesize, sorted, +// each with the voice serving it — for the startup line, so a deployment says +// which sidecars it actually reached rather than which ones it was configured +// to want. +func (h *Handler) Languages() []string { + out := make([]string, 0, len(h.routes)) + for lang, rt := range h.routes { + out = append(out, lang+"="+rt.voice) + } + sort.Strings(out) + return out +} + // Routes mounts the synthesis endpoint. Mount under "/tts" so the full path is // POST /api/tts. func (h *Handler) Routes() chi.Router { @@ -140,11 +158,13 @@ func (h *Handler) Routes() chi.Router { return r } -// synthRequest is the body the editor posts: a passage and the BCP-47 language -// tag it's written in (e.g. "en-US", "zh-CN"). +// synthRequest is the body the editor posts: a passage, the BCP-47 language tag +// it's written in (e.g. "en-US", "zh-CN", "pt-PT"), and whether to say it slowly +// — the replay a learner reaches for when the sentence went past too fast. type synthRequest struct { Text string `json:"text"` Lang string `json:"lang"` + Slow bool `json:"slow"` } // synth resolves a voice for the requested language, returns cached audio when @@ -180,9 +200,17 @@ func (h *Handler) synth(w http.ResponseWriter, r *http.Request) { return } - // Content-addressed: identical (voice, text) → identical clip. The format - // extension keeps encodings from colliding in the same dir. - sum := sha256.Sum256([]byte(rt.voice + "\n" + text)) + scale := lengthScale + if req.Slow { + scale = slowLengthScale + } + + // Content-addressed: identical (voice, pace, text) → identical clip. The pace + // belongs in the key — without it the slow replay of a word already heard at + // normal speed would be served from cache at normal speed, which is the one + // request where the difference is the whole point. The format extension keeps + // encodings from colliding in the same dir. + sum := sha256.Sum256([]byte(fmt.Sprintf("%s\n%.3f\n%s", rt.voice, scale, text))) name := hex.EncodeToString(sum[:])[:32] + h.format.ext path := filepath.Join(h.cacheDir, name) @@ -191,7 +219,7 @@ func (h *Handler) synth(w http.ResponseWriter, r *http.Request) { return } - audio, err := h.synthesize(r.Context(), rt, text) + audio, err := h.synthesize(r.Context(), rt, text, scale) if err != nil { http.Error(w, "synthesis failed", http.StatusBadGateway) fmt.Fprintf(os.Stderr, "tts: synthesize: %v\n", err) @@ -222,11 +250,11 @@ func (h *Handler) serve(w http.ResponseWriter, r *http.Request, path string) { // synthesize POSTs to the route's Piper instance, then transcodes the returned // WAV when the configured format calls for it. -func (h *Handler) synthesize(ctx context.Context, rt route, text string) ([]byte, error) { +func (h *Handler) synthesize(ctx context.Context, rt route, text string, scale float64) ([]byte, error) { body, _ := json.Marshal(map[string]any{ "text": text, "voice": rt.voice, - "length_scale": lengthScale, + "length_scale": scale, }) httpReq, err := http.NewRequestWithContext(ctx, http.MethodPost, rt.endpoint+h.synthURI, bytes.NewReader(body)) if err != nil { diff --git a/internal/tts/handler_test.go b/internal/tts/handler_test.go index 5019b9b..b34ec21 100644 --- a/internal/tts/handler_test.go +++ b/internal/tts/handler_test.go @@ -22,6 +22,7 @@ func newStubPiper(t *testing.T, body []byte) (*httptest.Server, *int32, *synthEc _ = json.NewDecoder(r.Body).Decode(&req) last.voice, _ = req["voice"].(string) last.text, _ = req["text"].(string) + last.scale, _ = req["length_scale"].(float64) w.Header().Set("Content-Type", "audio/wav") _, _ = w.Write(body) })) @@ -29,7 +30,10 @@ func newStubPiper(t *testing.T, body []byte) (*httptest.Server, *int32, *synthEc return srv, &calls, last } -type synthEcho struct{ voice, text string } +type synthEcho struct { + voice, text string + scale float64 +} // newHandler builds a wav-format handler (no ffmpeg) pointed at a stub server. func newHandler(t *testing.T, endpoint string) *Handler { @@ -88,7 +92,12 @@ func TestSynthPathNormalisation(t *testing.T) { func post(t *testing.T, h *Handler, text, lang string) *httptest.ResponseRecorder { t.Helper() - b, _ := json.Marshal(synthRequest{Text: text, Lang: lang}) + return postReq(t, h, synthRequest{Text: text, Lang: lang}) +} + +func postReq(t *testing.T, h *Handler, body synthRequest) *httptest.ResponseRecorder { + t.Helper() + b, _ := json.Marshal(body) req := httptest.NewRequest(http.MethodPost, "/", bytes.NewReader(b)) rr := httptest.NewRecorder() h.synth(rr, req) @@ -192,8 +201,87 @@ func TestTextIsCapped(t *testing.T) { } } +// The slow replay is the whole of SUGGESTIONS §5e: same text, same voice, more +// time per phoneme. +func TestSlowRequestStretchesTheVoice(t *testing.T) { + srv, _, last := newStubPiper(t, []byte("RIFF....fake-wav")) + h := newHandler(t, srv.URL) + + if rr := postReq(t, h, synthRequest{Text: "reception", Lang: "en-US"}); rr.Code != http.StatusOK { + t.Fatalf("status = %d, want 200", rr.Code) + } + if last.scale != lengthScale { + t.Fatalf("normal length_scale = %v, want %v", last.scale, lengthScale) + } + + if rr := postReq(t, h, synthRequest{Text: "reception", Lang: "en-US", Slow: true}); rr.Code != http.StatusOK { + t.Fatalf("slow status = %d, want 200", rr.Code) + } + if last.scale != slowLengthScale { + t.Fatalf("slow length_scale = %v, want %v", last.scale, slowLengthScale) + } + if slowLengthScale <= lengthScale { + t.Fatalf("slowLengthScale %v is not slower than %v", slowLengthScale, lengthScale) + } +} + +// The pace has to be part of the cache key. Without it, asking for the slow +// replay of a word already heard at normal speed serves the normal clip — the +// one request where hearing the difference is the entire point. +func TestSlowClipIsNotServedFromTheNormalCache(t *testing.T) { + srv, calls, last := newStubPiper(t, []byte("RIFF....fake-wav")) + h := newHandler(t, srv.URL) + + postReq(t, h, synthRequest{Text: "reception", Lang: "en-US"}) + postReq(t, h, synthRequest{Text: "reception", Lang: "en-US", Slow: true}) + if *calls != 2 { + t.Fatalf("piper calls = %d, want 2 (the slow clip is a different clip)", *calls) + } + if last.scale != slowLengthScale { + t.Fatalf("second call length_scale = %v, want the slow one", last.scale) + } + + // …and each pace still caches on its own. + postReq(t, h, synthRequest{Text: "reception", Lang: "en-US", Slow: true}) + postReq(t, h, synthRequest{Text: "reception", Lang: "en-US"}) + if *calls != 2 { + t.Fatalf("piper calls = %d, want 2 (both paces now cached)", *calls) + } +} + +// A Portuguese request must reach the Portuguese instance on the base tag alone: +// env var names cannot hold the hyphen in pt-PT, so config keys the map on "pt" +// and the handler has to meet it there. pt-BR resolves to the same instance +// because there is only one Portuguese voice loaded — and it is the European one. +func TestPortugueseRoutesOnTheBaseTag(t *testing.T) { + enSrv, enCalls, _ := newStubPiper(t, []byte("EN-wav")) + ptSrv, ptCalls, ptLast := newStubPiper(t, []byte("PT-wav")) + h := &Handler{ + routes: map[string]route{ + "en": {strings.TrimRight(enSrv.URL, "/"), "en_US-amy-medium"}, + "pt": {strings.TrimRight(ptSrv.URL, "/"), "pt_PT-tugão-medium"}, + }, + cacheDir: t.TempDir(), + format: formats["wav"], + client: http.DefaultClient, + } + + // Distinct text per tag, so a cache hit can't stand in for a route. + for i, tag := range []string{"pt-PT", "pt", "pt-BR"} { + if rr := post(t, h, strings.Repeat("receção ", i+1), tag); rr.Code != http.StatusOK { + t.Fatalf("%s status = %d, want 200", tag, rr.Code) + } + } + if *ptCalls != 3 || *enCalls != 0 { + t.Fatalf("calls en=%d pt=%d, want en=0 pt=3", *enCalls, *ptCalls) + } + if ptLast.voice != "pt_PT-tugão-medium" { + t.Fatalf("pt voice = %q, want the European Portuguese voice", ptLast.voice) + } +} + func TestBaseLang(t *testing.T) { - cases := map[string]string{"en-US": "en", "EN_gb": "en", "zh-CN": "zh", "en": "en", "": ""} + cases := map[string]string{"en-US": "en", "EN_gb": "en", "zh-CN": "zh", "pt-PT": "pt", "en": "en", "": ""} for in, want := range cases { if got := baseLang(in); got != want { t.Errorf("baseLang(%q) = %q, want %q", in, got, want) diff --git a/web/src/audio/speech.test.ts b/web/src/audio/speech.test.ts new file mode 100644 index 0000000..4e8db13 --- /dev/null +++ b/web/src/audio/speech.test.ts @@ -0,0 +1,76 @@ +import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest' +import { nativeLang, speak, stopSpeech } from './speech' +import { resetPackForTests, setPackLang } from '../i18n' + +// Read-aloud has two jobs beyond "make a sound": ask for the right pace, and ask +// in the right language. Both are decided at the call site and travel in the +// request body, so this checks the body — the part a component author can get +// wrong without anything failing loudly. + +let bodies: Array> + +beforeEach(() => { + bodies = [] + vi.stubGlobal( + 'fetch', + vi.fn((_url: string, init: RequestInit) => { + bodies.push(JSON.parse(String(init.body))) + // Never resolves to audio: the fallback path needs no window.Audio here, + // and rejecting would run the Web Speech branch instead of the server one. + return new Promise(() => {}) + }), + ) +}) + +afterEach(() => { + stopSpeech() + vi.unstubAllGlobals() + resetPackForTests() +}) + +describe('speak', () => { + it('asks for the normal pace by default', () => { + speak('reception') + expect(bodies).toHaveLength(1) + expect(bodies[0]).toMatchObject({ text: 'reception', lang: 'en-US', slow: false }) + }) + + it('asks for the slow replay when the slow control is used', () => { + speak('reception', undefined, true) + expect(bodies[0]).toMatchObject({ text: 'reception', slow: true }) + }) + + it('still detects Chinese by script, so a zh selection is never read in English', () => { + speak('你好世界') + expect(bodies[0]).toMatchObject({ lang: 'zh-CN' }) + }) + + it('sends nothing for empty text', () => { + speak(' ') + expect(bodies).toHaveLength(0) + }) +}) + +describe('nativeLang', () => { + // The voice for her own language comes from the pack, not from the letters. + // "comum" is spelled the same in both halves of the pt pair, so a detector + // would have to guess; the component that knows it is rendering her language + // says so instead. + it('follows the pair language', () => { + setPackLang('zh') + expect(nativeLang()).toBe('zh-CN') + setPackLang('pt-PT') + expect(nativeLang()).toBe('pt-PT') + }) + + it('names a European Portuguese voice, never a Brazilian one', () => { + setPackLang('pt-PT') + expect(nativeLang()).not.toBe('pt-BR') + }) + + it('is what a Latin-pair lookup speaks the other reading in', () => { + setPackLang('pt-PT') + speak('comum', nativeLang()) + expect(bodies[0]).toMatchObject({ text: 'comum', lang: 'pt-PT' }) + }) +}) diff --git a/web/src/audio/speech.ts b/web/src/audio/speech.ts index 79a63b8..eefc694 100644 --- a/web/src/audio/speech.ts +++ b/web/src/audio/speech.ts @@ -6,6 +6,8 @@ // (TTS disabled) or unreachable, we fall back to the browser's Web Speech API so // the buttons still do something. No model or network is strictly required. +import { pack } from '../i18n' + // speechSupported reports whether read-aloud can do anything at all. Audio // playback is universal, so as long as we can construct an Audio element OR the // Web Speech API exists, the buttons should show. The server path is tried at @@ -51,8 +53,10 @@ function pickVoice(lang: string): SpeechSynthesisVoice | undefined { } // speakWebSpeech is the fallback: the browser's built-in synthesizer. A touch -// slower than default so learners can follow along. -function speakWebSpeech(text: string, lang: string): void { +// slower than default so learners can follow along, and slower still when the +// slow replay was asked for — the fallback should degrade in voice quality, not +// in what the button does. +function speakWebSpeech(text: string, lang: string, slow: boolean): void { if (!webSpeechSupported()) return const synth = window.speechSynthesis synth.cancel() @@ -60,7 +64,7 @@ function speakWebSpeech(text: string, lang: string): void { utterance.lang = lang const voice = pickVoice(lang) if (voice) utterance.voice = voice - utterance.rate = 0.95 + utterance.rate = slow ? 0.7 : 0.95 synth.speak(utterance) } @@ -74,13 +78,26 @@ export function detectLang(text: string): string { return CJK.test(text) ? 'zh-CN' : 'en-US' } +// nativeLang is the locale of the writer's own language — the voice for the +// *other* reading of a word that exists in both halves of a Latin pair. +// +// It is asked for explicitly rather than detected, and that is the point. A +// script boundary can be detected (the CJK test above); "comum" cannot. So the +// component that knows it is rendering her language says so, and everything +// rendering English lets the default stand. No guess, therefore no wrong guess +// about her writing — the same rule the both-directions gloss follows. +export function nativeLang(): string { + return pack().locale +} + // speak reads `text` aloud, cancelling anything already in flight so rapid taps // don't queue up. `lang` defaults to a guess from the text (Chinese vs English) // so callers can just pass the selection; pass an explicit locale to override. -// It tries the server's neural voice first and silently falls back to the browser -// voice if that's unavailable (route off, network error, or a 404 for a language -// with no configured voice). -export function speak(text: string, lang = detectLang(text)): void { +// `slow` asks for the stretched replay (SUGGESTIONS §5e) — the second tap on a +// sentence that went by too fast. It tries the server's neural voice first and +// silently falls back to the browser voice if that's unavailable (route off, +// network error, or a 404 for a language with no configured voice). +export function speak(text: string, lang = detectLang(text), slow = false): void { if (!text.trim()) return stopSpeech() const seq = ++requestSeq @@ -88,7 +105,7 @@ export function speak(text: string, lang = detectLang(text)): void { fetch('/api/tts', { method: 'POST', headers: { 'Content-Type': 'application/json' }, - body: JSON.stringify({ text, lang }), + body: JSON.stringify({ text, lang, slow }), }) .then((res) => { if (!res.ok) throw new Error(`tts ${res.status}`) @@ -115,6 +132,6 @@ export function speak(text: string, lang = detectLang(text)): void { // Server TTS unavailable for this request — use the browser voice instead, // unless a newer tap has already superseded this one. if (seq !== requestSeq) return - speakWebSpeech(text, lang) + speakWebSpeech(text, lang, slow) }) } diff --git a/web/src/components/Editor/EditorCore.tsx b/web/src/components/Editor/EditorCore.tsx index c747527..3e6fef5 100644 --- a/web/src/components/Editor/EditorCore.tsx +++ b/web/src/components/Editor/EditorCore.tsx @@ -1109,6 +1109,7 @@ export function EditorCore({ style={{ top: selection.top, left: selection.left, transform: 'translateY(calc(-100% - 8px))' }} onRewrite={handleRewrite} onSpeak={speechSupported() ? () => speak(selection.text) : null} + onSpeakSlow={speechSupported() ? () => speak(selection.text, undefined, true) : null} /> )} {rewrite && ( diff --git a/web/src/components/Editor/SelectionBubble.tsx b/web/src/components/Editor/SelectionBubble.tsx index 516a0e5..1fa65ab 100644 --- a/web/src/components/Editor/SelectionBubble.tsx +++ b/web/src/components/Editor/SelectionBubble.tsx @@ -30,11 +30,15 @@ interface Props { // Read the selected text aloud (null when speech isn't available — the button // is then hidden). onSpeak: (() => void) | null + // The same passage, said slowly. A whole sentence replayed at three-quarter + // speed is the case SUGGESTIONS §5e is actually about — a word she can look + // up, but a sentence only goes past once. + onSpeakSlow: (() => void) | null } const CJK = "'Nunito','PingFang SC','Microsoft YaHei','Noto Sans CJK SC',sans-serif" -export function SelectionBubble({ style, onRewrite, onSpeak }: Props) { +export function SelectionBubble({ style, onRewrite, onSpeak, onSpeakSlow }: Props) { const pk = usePack() const [natural, ...tones] = REWRITE_STYLES @@ -85,6 +89,20 @@ export function SelectionBubble({ style, onRewrite, onSpeak }: Props) { )} + {onSpeakSlow && ( + + )} + {tones.map((t) => ( diff --git a/web/src/components/Editor/WordCard.tsx b/web/src/components/Editor/WordCard.tsx index 7a0ef50..a23d3b9 100644 --- a/web/src/components/Editor/WordCard.tsx +++ b/web/src/components/Editor/WordCard.tsx @@ -1,5 +1,5 @@ import type { WordInfo } from '../../api/client' -import { speak, speechSupported } from '../../audio/speech' +import { nativeLang, speak, speechSupported } from '../../audio/speech' import { usePack } from '../../i18n' import { wordBand } from './wordband' @@ -77,16 +77,32 @@ export function WordCard({ word, info, loading, saved, onToggleSave, style, onRe )} {speechSupported() && ( - + <> + + {/* The same word, stretched out. A learner replaying a word at + three-quarter speed is one of the oldest listening aids there + is, and Piper does it by lengthening durations rather than + slowing the tape, so it stays a voice rather than a groan. */} + + )} @@ -197,9 +213,27 @@ export function WordCard({ word, info, loading, saved, onToggleSave, style, onRe className="mt-3 rounded-xl px-2.5 py-2" style={{ background: 'var(--color-surface-alt)' }} > -

- {t.editor.alsoIn} -

+
+

+ {t.editor.alsoIn} +

+ {/* Her language, in her language's voice. The pack names the locale + (nativeLang) rather than anything guessing from the letters: + "comum" is spelled the same either way, and an English voice + reading it is the mistake this whole block exists to avoid. */} + {speechSupported() && ( + + )} +

{reverse.gloss || word} {reverse.phonetic && ( diff --git a/web/src/components/Garden/GardenPanel.tsx b/web/src/components/Garden/GardenPanel.tsx index 32ab711..05a53a1 100644 --- a/web/src/components/Garden/GardenPanel.tsx +++ b/web/src/components/Garden/GardenPanel.tsx @@ -445,15 +445,29 @@ function ReviewSession({

{card.word} {speechSupported() && ( - + <> + + {/* A word she has just failed to recall is exactly the word + worth hearing stretched out. */} + + )}
{card.phonetic && ( diff --git a/web/src/i18n/i18n.test.ts b/web/src/i18n/i18n.test.ts index 68ec55b..5b4cb39 100644 --- a/web/src/i18n/i18n.test.ts +++ b/web/src/i18n/i18n.test.ts @@ -118,6 +118,16 @@ describe('the zh pack', () => { expect(empties).toEqual([]) }) + // The voice read-aloud speaks this pair in. A pack that names a locale no + // Piper voice exists for degrades to Web Speech, which is survivable; a pack + // that names the *wrong region* does not announce itself at all — it just + // reads her language back to her in the accent the pair exists to avoid. + it.each(PACKS)('names a speakable locale for its own language ($code)', (p) => { + expect(p.locale, `${p.code} has no locale`).toMatch(/^[a-z]{2}(-[A-Za-z]{2,4})?$/) + expect(p.locale.split('-')[0]).toBe(p.code.split('-')[0]) + if (p.code === 'pt-PT') expect(p.locale).toBe('pt-PT') // never pt-BR + }) + it.each(PACKS)('labels every companion, tone and style ($code)', async (p) => { const { COMPANIONS } = await import('../components/Companion/companions') for (const c of COMPANIONS) { diff --git a/web/src/i18n/packs/pt-PT.ts b/web/src/i18n/packs/pt-PT.ts index 1bf6473..705c18d 100644 --- a/web/src/i18n/packs/pt-PT.ts +++ b/web/src/i18n/packs/pt-PT.ts @@ -31,6 +31,7 @@ import type { Pack } from '../types' export const ptPT: Pack = { code: 'pt-PT', nativeName: 'Português', + locale: 'pt-PT', app: { duplicateTitle: (title) => `${title} (cópia)`, @@ -184,6 +185,8 @@ export const ptPT: Pack = { inGarden: 'Já está no jardim · In your garden (tap to remove)', saveToGarden: 'Guardar no jardim · Save to garden', readAloud: 'Ler em voz alta · Read aloud', + readSlowly: 'Ler devagar · Read slowly', + readAloudNative: 'Ler em português · Read in Portuguese', lookingUp: 'A procurar… · Looking up…', definition: 'Definição · Definition', synonyms: 'Sinónimos · Synonyms', @@ -242,6 +245,7 @@ export const ptPT: Pack = { due: 'a rever · due', seen: (reps, intervalDays) => `${reps}× revista · seen ${reps}× · intervalo ${intervalDays}d`, readAloud: '🔊 Ler', + readSlowly: '🐢 Devagar', source: '📄 Origem · Source', remove: '🗑 Remover', growing: (n) => `🐱💤 ${n} flor${n === 1 ? '' : 'es'} no jardim · ${n} blossom${n > 1 ? 's' : ''} growing`, diff --git a/web/src/i18n/packs/zh.ts b/web/src/i18n/packs/zh.ts index add84b6..2178dde 100644 --- a/web/src/i18n/packs/zh.ts +++ b/web/src/i18n/packs/zh.ts @@ -13,6 +13,7 @@ import type { Pack } from '../types' export const zh: Pack = { code: 'zh', nativeName: '中文', + locale: 'zh-CN', app: { duplicateTitle: (title) => `${title} (副本)`, @@ -169,6 +170,8 @@ export const zh: Pack = { inGarden: '已在词汇花园 · In your garden (tap to remove)', saveToGarden: '加入词汇花园 · Save to garden', readAloud: '朗读 · Read aloud', + readSlowly: '慢速朗读 · Read slowly', + readAloudNative: '用中文朗读 · Read in Chinese', lookingUp: '查找中… · Looking up…', definition: '释义 · Definition', synonyms: '近义词 · Synonyms', @@ -228,6 +231,7 @@ export const zh: Pack = { due: '待复习 · due', seen: (reps, intervalDays) => `复习 ${reps} 次 · seen ${reps}× · 间隔 ${intervalDays}d`, readAloud: '🔊 朗读', + readSlowly: '🐢 慢速', source: '📄 出处 · Source', remove: '🗑 移除', growing: (n) => `🐱💤 ${n} 朵花在花园里 · ${n} blossom${n > 1 ? 's' : ''} growing`, diff --git a/web/src/i18n/types.ts b/web/src/i18n/types.ts index 0905044..f0980b7 100644 --- a/web/src/i18n/types.ts +++ b/web/src/i18n/types.ts @@ -31,6 +31,13 @@ export interface Pack { // for anywhere Petal has to say which pair this is. code: PairLang nativeName: string + // The BCP-47 locale to *speak* this language in — what read-aloud sends to + // Piper (and to the browser's Web Speech fallback). It is not derivable from + // `code`: zh is a pair language but zh-CN is a voice, and a pack is the only + // place that knows which regional voice its pair should be read in. pt-PT is + // spelled out for the same reason the prompts spell it out — the default + // Portuguese voice anyone reaches for is Brazilian. + locale: string app: { // A duplicated document's title. A function, not a suffix: where the marker @@ -129,6 +136,13 @@ export interface Pack { inGarden: string saveToGarden: string readAloud: string + // The same passage, said slowly (SUGGESTIONS §5e). Only ever offered for + // English: it is the language she is learning to hear. + readSlowly: string + // Read the *other* reading aloud — the one in her own language, in her own + // language's voice. Sits on the `alsoIn` block, so a pack whose pair has no + // collisions never sees it rendered. + readAloudNative: string lookingUp: string definition: string synonyms: string @@ -170,6 +184,9 @@ export interface Pack { due: string seen: (reps: number, intervalDays: number) => string readAloud: string + // Short label for the slow replay on a flashcard, where a word she is + // trying to recall is exactly the word worth hearing stretched out. + readSlowly: string source: string remove: string growing: (n: number) => string