From 91c072ab8257f8771fe203477677fb20db5355bc Mon Sep 17 00:00:00 2001
From: luke-speechify <289678208+luke-speechify@users.noreply.github.com>
Date: Fri, 21 Aug 2026 12:52:46 +0100
Subject: [PATCH] feat: add multilingual-tts demo
Read text in any of 30+ languages with the Speechify TTS API, one native
voice per locale. One-page Next.js playground: pick a language, hear it
spoken by a voice native to that locale via POST /v1/audio/speech with the
`language` parameter on simba-3.0. API key stays server-side; Turnstile
fails open locally.
Six languages wired up (en, de, es, fr, it, pt-BR), each with a native
catalogue voice; 30+ more available. Registered as a hostable service in
pnpm-workspace.yaml and vercel.json. Includes a Playwright e2e that hits the
live API when SPEECHIFY_API_KEY is set.
---
demos/multilingual-tts/.env.example | 1 +
demos/multilingual-tts/.gitignore | 6 +
demos/multilingual-tts/README.md | 53 +++++++
demos/multilingual-tts/app/api/speak/route.ts | 73 +++++++++
demos/multilingual-tts/app/globals.css | 114 ++++++++++++++
demos/multilingual-tts/app/layout.tsx | 21 +++
demos/multilingual-tts/app/lib/turnstile.ts | 34 ++++
demos/multilingual-tts/app/page.tsx | 147 ++++++++++++++++++
demos/multilingual-tts/demo.json | 6 +
.../e2e/multilingual-tts.spec.ts | 35 +++++
demos/multilingual-tts/next-env.d.ts | 6 +
demos/multilingual-tts/next.config.ts | 11 ++
demos/multilingual-tts/package.json | 27 ++++
demos/multilingual-tts/playwright.config.ts | 20 +++
demos/multilingual-tts/tsconfig.json | 27 ++++
pnpm-lock.yaml | 72 ++++++++-
pnpm-workspace.yaml | 1 +
vercel.json | 8 +
18 files changed, 658 insertions(+), 4 deletions(-)
create mode 100644 demos/multilingual-tts/.env.example
create mode 100644 demos/multilingual-tts/.gitignore
create mode 100644 demos/multilingual-tts/README.md
create mode 100644 demos/multilingual-tts/app/api/speak/route.ts
create mode 100644 demos/multilingual-tts/app/globals.css
create mode 100644 demos/multilingual-tts/app/layout.tsx
create mode 100644 demos/multilingual-tts/app/lib/turnstile.ts
create mode 100644 demos/multilingual-tts/app/page.tsx
create mode 100644 demos/multilingual-tts/demo.json
create mode 100644 demos/multilingual-tts/e2e/multilingual-tts.spec.ts
create mode 100644 demos/multilingual-tts/next-env.d.ts
create mode 100644 demos/multilingual-tts/next.config.ts
create mode 100644 demos/multilingual-tts/package.json
create mode 100644 demos/multilingual-tts/playwright.config.ts
create mode 100644 demos/multilingual-tts/tsconfig.json
diff --git a/demos/multilingual-tts/.env.example b/demos/multilingual-tts/.env.example
new file mode 100644
index 0000000..534cec4
--- /dev/null
+++ b/demos/multilingual-tts/.env.example
@@ -0,0 +1 @@
+SPEECHIFY_API_KEY=your_api_key_here
diff --git a/demos/multilingual-tts/.gitignore b/demos/multilingual-tts/.gitignore
new file mode 100644
index 0000000..ad644c6
--- /dev/null
+++ b/demos/multilingual-tts/.gitignore
@@ -0,0 +1,6 @@
+node_modules
+.next
+.env
+.env.local
+test-results
+playwright-report
diff --git a/demos/multilingual-tts/README.md b/demos/multilingual-tts/README.md
new file mode 100644
index 0000000..785e4db
--- /dev/null
+++ b/demos/multilingual-tts/README.md
@@ -0,0 +1,53 @@
+# multilingual-tts
+
+Read text in any of 30+ languages with the Speechify TTS API, one native voice per language. Pick a language, hear it spoken by a voice native to that locale.
+
+Pairs with the speechify.ai post **"Text-to-speech in 30+ languages with the Speechify API."**
+
+## What it does
+
+- A one-page Next.js playground with a language picker and a text box. Choose a language, press **Speak**, and the text is synthesized in a voice native to that locale.
+- `POST /api/speak` (Node runtime) calls Speechify's `POST /v1/audio/speech` with `model: "simba-3.0"`, `audio_format: "mp3"`, and the `language` parameter, then returns the audio as base64.
+- The whole feature is that one `language` field. `simba-3.0` is the multilingual, streaming-native model.
+
+## The six wired-up languages
+
+Each uses a voice native to the locale (from the public catalogue). `simba-3.0` officially supports these; the catalogue carries 30+ more locales.
+
+| Language | Locale | Voice |
+| --- | --- | --- |
+| English | `en-US` | `alfonso` |
+| German | `de-DE` | `amalia` |
+| Spanish | `es-MX` | `aitana` |
+| French | `fr-FR` | `adeline` |
+| Italian | `it-IT` | `alessia` |
+| Portuguese (Brazil) | `pt-BR` | `adriana` |
+
+To add a language: pick a voice for the locale at [platform.speechify.ai](https://platform.speechify.ai), then add a row to `LANGUAGES` in `app/page.tsx` and to `ALLOWED_LANGUAGES` in `app/api/speak/route.ts`.
+
+## Notes
+
+- `simba-3.2` and `simba-english` are English-only. Non-English synthesis uses `simba-3.0` (or `simba-multilingual` on an API version pinned before `2026-09-21`, which sunsets `2026-11-21`).
+- The Speechify API key stays server-side. The browser only ever calls this app's own `/api/speak` route.
+- Abuse protection is Cloudflare Turnstile, which fails open when `TURNSTILE_SECRET_KEY` is unset (local dev).
+
+## Run it
+
+```bash
+cp .env.example .env # add your SPEECHIFY_API_KEY
+pnpm install
+pnpm dev # http://localhost:8781/multilingual-tts
+```
+
+End-to-end test against the live API (needs `SPEECHIFY_API_KEY`):
+
+```bash
+pnpm e2e
+```
+
+## Environment
+
+| Variable | Required | Purpose |
+| --- | --- | --- |
+| `SPEECHIFY_API_KEY` | yes | Server-side Speechify API key |
+| `TURNSTILE_SECRET_KEY` | no | Enables the Turnstile abuse gate; fails open when unset |
diff --git a/demos/multilingual-tts/app/api/speak/route.ts b/demos/multilingual-tts/app/api/speak/route.ts
new file mode 100644
index 0000000..04e6a2e
--- /dev/null
+++ b/demos/multilingual-tts/app/api/speak/route.ts
@@ -0,0 +1,73 @@
+import { NextResponse } from "next/server";
+import { verifyTurnstile } from "../../lib/turnstile";
+
+export const runtime = "nodejs";
+
+// The whole feature is one field: `language`. We call the REST speech endpoint
+// directly (rather than the Node SDK) so the `language` parameter passes
+// through verbatim and the demo stays dependency-light. `simba-3.0` is the
+// multilingual, streaming-native model; each language uses a voice native to
+// that locale (see app/page.tsx). Omit `language` and the model infers it from
+// the voice's own locale, but we send it to show the parameter doing the work.
+const SPEECH_URL = "https://api.speechify.ai/v1/audio/speech";
+
+// Locales this demo wires up, kept in sync with the picker in app/page.tsx.
+const ALLOWED_LANGUAGES = new Set([
+ "en-US",
+ "de-DE",
+ "es-MX",
+ "fr-FR",
+ "it-IT",
+ "pt-BR",
+]);
+
+export async function POST(req: Request) {
+ if (!(await verifyTurnstile(req))) {
+ return NextResponse.json({ error: "Forbidden" }, { status: 403 });
+ }
+
+ const { text, voiceId, language } = (await req.json().catch(() => ({}))) as {
+ text?: unknown;
+ voiceId?: unknown;
+ language?: unknown;
+ };
+
+ if (typeof text !== "string" || !text.trim()) {
+ return NextResponse.json({ error: "text is required" }, { status: 400 });
+ }
+ if (typeof voiceId !== "string" || !voiceId) {
+ return NextResponse.json({ error: "voiceId is required" }, { status: 400 });
+ }
+ if (typeof language !== "string" || !ALLOWED_LANGUAGES.has(language)) {
+ return NextResponse.json(
+ { error: "unsupported language" },
+ { status: 400 },
+ );
+ }
+
+ const upstream = await fetch(SPEECH_URL, {
+ method: "POST",
+ headers: {
+ Authorization: `Bearer ${process.env.SPEECHIFY_API_KEY}`,
+ "content-type": "application/json",
+ },
+ body: JSON.stringify({
+ input: text.slice(0, 2000),
+ voice_id: voiceId,
+ model: "simba-3.0",
+ audio_format: "mp3",
+ language,
+ }),
+ });
+
+ if (!upstream.ok) {
+ const detail = await upstream.text().catch(() => "");
+ return NextResponse.json(
+ { error: detail || "Speechify request failed" },
+ { status: upstream.status || 502 },
+ );
+ }
+
+ const data = (await upstream.json()) as { audio_data?: string };
+ return NextResponse.json({ audio: data.audio_data });
+}
diff --git a/demos/multilingual-tts/app/globals.css b/demos/multilingual-tts/app/globals.css
new file mode 100644
index 0000000..a14bba2
--- /dev/null
+++ b/demos/multilingual-tts/app/globals.css
@@ -0,0 +1,114 @@
+/* Speechify brand base — mirrors demos.speechify.ai/site (speechify.ai/brand).
+ * ABC Diatype is licensed and NOT committed; it is loaded cross-origin from
+ * speechify.ai/fonts (served with Access-Control-Allow-Origin: *). Monochrome
+ * palette, thin display type, pill ink buttons, sentence-case voice. */
+
+@font-face { font-family: "ABCDiatype"; src: url("https://speechify.ai/fonts/ABCDiatype-Thin.woff2") format("woff2"); font-weight: 100; font-style: normal; font-display: swap; }
+@font-face { font-family: "ABCDiatype"; src: url("https://speechify.ai/fonts/ABCDiatype-Light.woff2") format("woff2"); font-weight: 300; font-style: normal; font-display: swap; }
+@font-face { font-family: "ABCDiatype"; src: url("https://speechify.ai/fonts/ABCDiatype-Regular.woff2") format("woff2"); font-weight: 400; font-style: normal; font-display: swap; }
+@font-face { font-family: "ABCDiatype"; src: url("https://speechify.ai/fonts/ABCDiatype-Medium.woff2") format("woff2"); font-weight: 500; font-style: normal; font-display: swap; }
+@font-face { font-family: "ABCDiatype"; src: url("https://speechify.ai/fonts/ABCDiatype-Bold.woff2") format("woff2"); font-weight: 700; font-style: normal; font-display: swap; }
+
+:root {
+ --font-sans: "ABCDiatype", ui-sans-serif, system-ui, -apple-system, sans-serif;
+ --font-mono: ui-monospace, SFMono-Regular, Menlo, Monaco, "Cascadia Code", monospace;
+
+ --surface-page: #ffffff;
+ --surface-card: #ffffff;
+ --surface-subtle: #f5f5f5;
+ --surface-raised: #fafafa;
+ --text-primary: #0a0a0a;
+ --text-secondary: #525252;
+ --text-tertiary: #666666;
+ --border-subtle: #e5e5e5;
+ --border-strong: #d1d1d4;
+ --action: #0a0a0a;
+ --action-hover: #2a2a2e;
+ --action-foreground: #fafafa;
+ --focus-ring: rgba(10, 10, 10, 0.22);
+ --success: #00c270;
+ --danger: #b42318;
+ --radius-md: 8px;
+ --radius-lg: 12px;
+ --radius-pill: 9999px;
+}
+
+@media (prefers-color-scheme: dark) {
+ :root {
+ --surface-page: #0a0a0a;
+ --surface-card: #101010;
+ --surface-subtle: #161616;
+ --surface-raised: #141414;
+ --text-primary: #fafafa;
+ --text-secondary: #b3b3b3;
+ --text-tertiary: #8a8a8a;
+ --border-subtle: #262626;
+ --border-strong: #3a3a3a;
+ --action: #fafafa;
+ --action-hover: #e5e5e5;
+ --action-foreground: #0a0a0a;
+ --focus-ring: rgba(250, 250, 250, 0.28);
+ }
+}
+
+* { box-sizing: border-box; }
+
+body {
+ margin: 0;
+ padding: 4rem 1.5rem;
+ background: var(--surface-page);
+ color: var(--text-primary);
+ font-family: var(--font-sans);
+ font-weight: 400;
+ line-height: 1.55;
+ -webkit-font-smoothing: antialiased;
+ text-rendering: optimizeLegibility;
+}
+
+main { max-width: 44rem; margin: 0 auto; display: flex; flex-direction: column; gap: 1.75rem; }
+
+h1 { font-weight: 100; font-size: clamp(2.25rem, 6vw, 3.5rem); line-height: 1.02; letter-spacing: -0.03em; margin: 0 0 0.5rem; }
+h2 { font-weight: 300; letter-spacing: -0.01em; margin: 0 0 0.5rem; font-size: 1.125rem; }
+
+.eyebrow { font-family: var(--font-mono); font-size: 13px; font-weight: 500; letter-spacing: 0.02em; color: var(--text-tertiary); margin: 0 0 0.75rem; }
+.lead { max-width: 62ch; color: var(--text-secondary); font-size: 1.0625rem; line-height: 1.55; margin: 0; }
+
+a { color: var(--text-primary); text-underline-offset: 2px; }
+
+code, kbd, samp { font-family: var(--font-mono); }
+code { background: var(--surface-subtle); border: 1px solid var(--border-subtle); padding: 0.08em 0.38em; border-radius: 5px; font-size: 0.88em; }
+
+.step { display: flex; flex-direction: column; gap: 0.5rem; }
+
+select, textarea {
+ width: 100%;
+ font-family: var(--font-sans);
+ font-size: 1rem;
+ color: var(--text-primary);
+ background: var(--surface-raised);
+ border: 1px solid var(--border-strong);
+ border-radius: var(--radius-md);
+ padding: 0.75rem 0.875rem;
+}
+textarea { resize: vertical; line-height: 1.5; }
+select:focus, textarea:focus { outline: none; border-color: var(--action); box-shadow: 0 0 0 3px var(--focus-ring); }
+
+.play { display: flex; align-items: center; gap: 1rem; flex-wrap: wrap; }
+
+.btn {
+ font-family: var(--font-sans);
+ font-size: 1rem;
+ font-weight: 500;
+ border: none;
+ border-radius: var(--radius-pill);
+ padding: 0.75rem 1.75rem;
+ cursor: pointer;
+}
+.btn-primary { background: var(--action); color: var(--action-foreground); }
+.btn-primary:hover:not(:disabled) { background: var(--action-hover); }
+.btn:disabled { opacity: 0.55; cursor: default; }
+
+.err { color: var(--danger); font-size: 0.9375rem; margin: 0; }
+
+footer { border-top: 1px solid var(--border-subtle); padding-top: 1.25rem; color: var(--text-secondary); font-size: 0.9375rem; }
+footer p { margin: 0; }
diff --git a/demos/multilingual-tts/app/layout.tsx b/demos/multilingual-tts/app/layout.tsx
new file mode 100644
index 0000000..3f48410
--- /dev/null
+++ b/demos/multilingual-tts/app/layout.tsx
@@ -0,0 +1,21 @@
+import type { Metadata } from "next";
+import type { ReactNode } from "react";
+import Script from "next/script";
+import "./globals.css";
+
+export const metadata: Metadata = {
+ title: "Text-to-speech in 30+ languages with Speechify",
+ description:
+ "Pick a language and hear it in a native voice. One Speechify TTS request with a language parameter; the API key stays server-side.",
+};
+
+export default function RootLayout({ children }: { children: ReactNode }) {
+ return (
+
+
+
+ {children}
+
+
+ );
+}
diff --git a/demos/multilingual-tts/app/lib/turnstile.ts b/demos/multilingual-tts/app/lib/turnstile.ts
new file mode 100644
index 0000000..5661dbb
--- /dev/null
+++ b/demos/multilingual-tts/app/lib/turnstile.ts
@@ -0,0 +1,34 @@
+// Verifies a Turnstile token against Cloudflare siteverify. Returns true iff
+// the caller is allowed to proceed.
+//
+// Fail-open contract: when TURNSTILE_SECRET_KEY isn't set (local dev, fork
+// deploys, anywhere the operator hasn't configured Turnstile) OR when the
+// siteverify request itself errors, returns true. The alternative is breaking
+// the demo whenever Turnstile isn't configured, a worse experience than
+// leaving the abuse gate briefly open. Real prod hardening would flip this to
+// fail-closed; this is a reference demo.
+const SITEVERIFY_URL =
+ "https://challenges.cloudflare.com/turnstile/v0/siteverify";
+
+export async function verifyTurnstile(req: Request): Promise {
+ const secret = process.env.TURNSTILE_SECRET_KEY;
+ if (!secret) return true;
+
+ const token = req.headers.get("x-turnstile-token");
+ if (!token) return false;
+
+ const form = new URLSearchParams();
+ form.set("secret", secret);
+ form.set("response", token);
+ const remoteip = req.headers.get("x-forwarded-for")?.split(",")[0]?.trim();
+ if (remoteip) form.set("remoteip", remoteip);
+
+ try {
+ const cf = await fetch(SITEVERIFY_URL, { method: "POST", body: form });
+ if (!cf.ok) return true;
+ const result = (await cf.json()) as { success?: boolean };
+ return Boolean(result?.success);
+ } catch {
+ return true;
+ }
+}
diff --git a/demos/multilingual-tts/app/page.tsx b/demos/multilingual-tts/app/page.tsx
new file mode 100644
index 0000000..6e70525
--- /dev/null
+++ b/demos/multilingual-tts/app/page.tsx
@@ -0,0 +1,147 @@
+"use client";
+
+import { useEffect, useRef, useState } from "react";
+
+// Six languages wired up, each with a voice native to that locale. simba-3.0
+// officially supports these; the catalogue carries 30+ more. Browse them at
+// platform.speechify.ai. To add a language: find a voice for the locale and
+// add a row here (and to ALLOWED_LANGUAGES in app/api/speak/route.ts).
+const LANGUAGES = [
+ { code: "en-US", label: "English", voiceId: "alfonso", sample: "Hello. This is Speechify reading in English." },
+ { code: "de-DE", label: "German", voiceId: "amalia", sample: "Hallo. Das ist Speechify auf Deutsch." },
+ { code: "es-MX", label: "Spanish", voiceId: "aitana", sample: "Hola. Esto es Speechify en español." },
+ { code: "fr-FR", label: "French", voiceId: "adeline", sample: "Bonjour. Voici Speechify en français." },
+ { code: "it-IT", label: "Italian", voiceId: "alessia", sample: "Ciao. Questo è Speechify in italiano." },
+ { code: "pt-BR", label: "Portuguese (Brazil)", voiceId: "adriana", sample: "Olá. Este é o Speechify em português." },
+] as const;
+
+// This app is mounted under a basePath (see next.config.ts), so the route
+// handler lives at `${BASE_PATH}/api/speak`. Client fetches are not basePath-aware.
+const BASE_PATH = "/multilingual-tts";
+
+type State = "idle" | "loading" | "playing" | "error";
+
+type TurnstileHandle = {
+ getToken: (opts?: { timeout?: number }) => Promise;
+ reset: () => void;
+};
+
+declare global {
+ interface Window {
+ SpeechifyTurnstile?: {
+ render: (target: string | HTMLElement, options?: unknown) => Promise;
+ };
+ }
+}
+
+export default function Home() {
+ const [code, setCode] = useState<(typeof LANGUAGES)[number]["code"]>("fr-FR");
+ const lang = LANGUAGES.find((l) => l.code === code)!;
+ const [text, setText] = useState(lang.sample);
+ const [state, setState] = useState("idle");
+ const [error, setError] = useState("");
+ const audioRef = useRef(null);
+ const turnstileRef = useRef(null);
+
+ useEffect(() => {
+ let cancelled = false;
+ (async () => {
+ while (!window.SpeechifyTurnstile && !cancelled) {
+ await new Promise((r) => setTimeout(r, 30));
+ }
+ if (cancelled) return;
+ turnstileRef.current = await window.SpeechifyTurnstile!.render("#turnstile-container");
+ })();
+ return () => {
+ cancelled = true;
+ };
+ }, []);
+
+ function pickLanguage(next: (typeof LANGUAGES)[number]["code"]) {
+ const prev = LANGUAGES.find((l) => l.code === code)!;
+ const nextLang = LANGUAGES.find((l) => l.code === next)!;
+ setCode(next);
+ // Swap the sample only if the user hasn't typed their own text.
+ if (text === prev.sample || text.trim() === "") setText(nextLang.sample);
+ }
+
+ async function speak() {
+ if (!text.trim() || state === "loading") return;
+ setState("loading");
+ setError("");
+ try {
+ const headers: Record = { "content-type": "application/json" };
+ const token = turnstileRef.current ? await turnstileRef.current.getToken() : null;
+ turnstileRef.current?.reset();
+ if (token) headers["x-turnstile-token"] = token;
+
+ const res = await fetch(`${BASE_PATH}/api/speak`, {
+ method: "POST",
+ headers,
+ body: JSON.stringify({ text, voiceId: lang.voiceId, language: lang.code }),
+ });
+ if (!res.ok) throw new Error(await res.text());
+
+ const { audio } = (await res.json()) as { audio: string };
+ const el = audioRef.current!;
+ el.src = `data:audio/mpeg;base64,${audio}`;
+ el.onended = () => setState("idle");
+ await el.play();
+ setState("playing");
+ } catch (e) {
+ setError(e instanceof Error ? e.message : "Something went wrong");
+ setState("error");
+ }
+ }
+
+ return (
+
+
+
SpeechifyAI · Multilingual TTS
+
Text-to-speech in 30+ languages.
+
+ One request, one language parameter. Pick a language below and hear it in a
+ voice native to that locale, synthesized by simba-3.0. The API key stays on
+ the server.
+