diff --git a/.gitignore b/.gitignore index b1c6b9c..279303f 100644 --- a/.gitignore +++ b/.gitignore @@ -13,6 +13,9 @@ __pycache__/ .env.local !.env.example +# Voice-cloning consent challenge cache (single-use, per-run) +.consent-challenge.json + # Generated audio / artifacts *.mp3 *.wav diff --git a/agents/python-recipes.md b/agents/python-recipes.md index 866ab2a..e31922e 100644 --- a/agents/python-recipes.md +++ b/agents/python-recipes.md @@ -21,7 +21,7 @@ name = "audio-python-quickstart" version = "0.1.0" requires-python = ">=3.10" dependencies = [ - "speechify-api>=3.0.1", + "speechify-api>=4.0.0", "python-dotenv>=1.0.0", ] ``` diff --git a/pnpm-lock.yaml b/pnpm-lock.yaml index 320214f..76608fb 100644 --- a/pnpm-lock.yaml +++ b/pnpm-lock.yaml @@ -7,8 +7,8 @@ settings: catalogs: default: '@speechify/api': - specifier: 3.0.1 - version: 3.0.1 + specifier: 4.0.1 + version: 4.0.1 '@types/node': specifier: ^22.10.2 version: 22.20.0 @@ -114,7 +114,7 @@ importers: dependencies: '@speechify/api': specifier: 'catalog:' - version: 3.0.1 + version: 4.0.1 dotenv: specifier: 'catalog:' version: 16.6.1 @@ -133,7 +133,7 @@ importers: dependencies: '@speechify/api': specifier: 'catalog:' - version: 3.0.1 + version: 4.0.1 dotenv: specifier: 'catalog:' version: 16.6.1 @@ -152,7 +152,7 @@ importers: dependencies: '@speechify/api': specifier: 'catalog:' - version: 3.0.1 + version: 4.0.1 dotenv: specifier: 'catalog:' version: 16.6.1 @@ -171,7 +171,7 @@ importers: dependencies: '@speechify/api': specifier: 'catalog:' - version: 3.0.1 + version: 4.0.1 dotenv: specifier: 'catalog:' version: 16.6.1 @@ -190,7 +190,7 @@ importers: dependencies: '@speechify/api': specifier: 'catalog:' - version: 3.0.1 + version: 4.0.1 dotenv: specifier: 'catalog:' version: 16.6.1 @@ -363,8 +363,8 @@ packages: cpu: [x64] os: [win32] - '@speechify/api@3.0.1': - resolution: {integrity: sha512-bEqbyvwGW0A/cgq+ZsQJVP3Hm4/0Bm1sniVtPLBh7vDNG2f/YY8y8sweWojoUnhIeIAL6bHBde43+VZLA2MYpA==} + '@speechify/api@4.0.1': + resolution: {integrity: sha512-OsfZbx4o47kIZ29OmntUj/9J2LYrlntRHfAwZ8uRLCw2gsu5zgQf46yasDgGe9fNgC5XI1+JfJwCweYVs5w+4w==} engines: {node: '>=18.0.0'} '@types/node@22.20.0': @@ -482,7 +482,7 @@ snapshots: '@esbuild/win32-x64@0.28.1': optional: true - '@speechify/api@3.0.1': {} + '@speechify/api@4.0.1': {} '@types/node@22.20.0': dependencies: diff --git a/pnpm-workspace.yaml b/pnpm-workspace.yaml index fc5bfcb..449afdc 100644 --- a/pnpm-workspace.yaml +++ b/pnpm-workspace.yaml @@ -9,7 +9,7 @@ packages: # Single source of truth for shared dependency versions. # Recipes reference these with "catalog:" instead of pinning their own. catalog: - "@speechify/api": 3.0.1 + "@speechify/api": 4.0.1 tsx: ^4.19.2 typescript: ^5.7.2 "@types/node": ^22.10.2 diff --git a/recipes/audio/python/sdk/quickstart/pyproject.toml b/recipes/audio/python/sdk/quickstart/pyproject.toml index 95f77ba..095d413 100644 --- a/recipes/audio/python/sdk/quickstart/pyproject.toml +++ b/recipes/audio/python/sdk/quickstart/pyproject.toml @@ -4,6 +4,6 @@ version = "0.1.0" description = "Synthesize speech to an MP3 file with the Speechify TTS API." requires-python = ">=3.10" dependencies = [ - "speechify-api>=3.0.1", + "speechify-api>=4.0.0", "python-dotenv>=1.0.0", ] diff --git a/recipes/audio/python/sdk/quickstart/uv.lock b/recipes/audio/python/sdk/quickstart/uv.lock index da30115..d8c59d6 100644 --- a/recipes/audio/python/sdk/quickstart/uv.lock +++ b/recipes/audio/python/sdk/quickstart/uv.lock @@ -37,7 +37,7 @@ dependencies = [ [package.metadata] requires-dist = [ { name = "python-dotenv", specifier = ">=1.0.0" }, - { name = "speechify-api", specifier = ">=3.0.1" }, + { name = "speechify-api", specifier = ">=4.0.0" }, ] [[package]] @@ -249,7 +249,7 @@ wheels = [ [[package]] name = "speechify-api" -version = "3.0.1" +version = "4.0.0" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "httpx" }, @@ -257,9 +257,9 @@ dependencies = [ { name = "pydantic-core" }, { name = "typing-extensions" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/6d/67/961acfbf3355c73b85c9a85e7d2ca4c38a9f14261c845ba9c804940969de/speechify_api-3.0.1.tar.gz", hash = "sha256:e73a084fd6845cb2f92bff2e4dfa6d58052ff97e8112b38dc67ebbf47610aa1a", size = 48230, upload-time = "2026-07-10T19:41:34.679Z" } +sdist = { url = "https://files.pythonhosted.org/packages/90/3c/4cef771d24e89577b5819b12b351434deb3b5bfd647c21c57c5ece6b2935/speechify_api-4.0.0.tar.gz", hash = "sha256:461ca9f3b6ec3f95a604f299c59152c46bda594a441570a80a4f04d88657f11e", size = 66631, upload-time = "2026-08-18T18:39:42.634Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/b3/7d/52daedb2aa7e1df49f39a03b0d87ea25cd281588102f4231cffbcbf32038/speechify_api-3.0.1-py3-none-any.whl", hash = "sha256:7e09d09bee792a348a5dbfe0e07a806152ba2797c70b399b6e5544b320808cf8", size = 75381, upload-time = "2026-07-10T19:41:33.749Z" }, + { url = "https://files.pythonhosted.org/packages/fd/70/b783459bd5729da204dfd077301a7a85399a5ccd0e3bb53c655ef421e600/speechify_api-4.0.0-py3-none-any.whl", hash = "sha256:0d811377a4a2a20eeca3c3f443b7853a379df1a8d9fa859f264f8d70b8fd36f0", size = 103832, upload-time = "2026-08-18T18:39:41.424Z" }, ] [[package]] diff --git a/recipes/audio/python/sdk/speech-marks/pyproject.toml b/recipes/audio/python/sdk/speech-marks/pyproject.toml index a208a9f..64f007b 100644 --- a/recipes/audio/python/sdk/speech-marks/pyproject.toml +++ b/recipes/audio/python/sdk/speech-marks/pyproject.toml @@ -4,6 +4,6 @@ version = "0.1.0" description = "Generate WebVTT captions from word-level speech marks with the Speechify TTS API." requires-python = ">=3.10" dependencies = [ - "speechify-api>=3.0.1", + "speechify-api>=4.0.0", "python-dotenv>=1.0.0", ] diff --git a/recipes/audio/python/sdk/speech-marks/uv.lock b/recipes/audio/python/sdk/speech-marks/uv.lock index 9e4a77b..53c0c7a 100644 --- a/recipes/audio/python/sdk/speech-marks/uv.lock +++ b/recipes/audio/python/sdk/speech-marks/uv.lock @@ -37,7 +37,7 @@ dependencies = [ [package.metadata] requires-dist = [ { name = "python-dotenv", specifier = ">=1.0.0" }, - { name = "speechify-api", specifier = ">=3.0.1" }, + { name = "speechify-api", specifier = ">=4.0.0" }, ] [[package]] @@ -249,7 +249,7 @@ wheels = [ [[package]] name = "speechify-api" -version = "3.0.1" +version = "4.0.0" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "httpx" }, @@ -257,9 +257,9 @@ dependencies = [ { name = "pydantic-core" }, { name = "typing-extensions" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/6d/67/961acfbf3355c73b85c9a85e7d2ca4c38a9f14261c845ba9c804940969de/speechify_api-3.0.1.tar.gz", hash = "sha256:e73a084fd6845cb2f92bff2e4dfa6d58052ff97e8112b38dc67ebbf47610aa1a", size = 48230, upload-time = "2026-07-10T19:41:34.679Z" } +sdist = { url = "https://files.pythonhosted.org/packages/90/3c/4cef771d24e89577b5819b12b351434deb3b5bfd647c21c57c5ece6b2935/speechify_api-4.0.0.tar.gz", hash = "sha256:461ca9f3b6ec3f95a604f299c59152c46bda594a441570a80a4f04d88657f11e", size = 66631, upload-time = "2026-08-18T18:39:42.634Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/b3/7d/52daedb2aa7e1df49f39a03b0d87ea25cd281588102f4231cffbcbf32038/speechify_api-3.0.1-py3-none-any.whl", hash = "sha256:7e09d09bee792a348a5dbfe0e07a806152ba2797c70b399b6e5544b320808cf8", size = 75381, upload-time = "2026-07-10T19:41:33.749Z" }, + { url = "https://files.pythonhosted.org/packages/fd/70/b783459bd5729da204dfd077301a7a85399a5ccd0e3bb53c655ef421e600/speechify_api-4.0.0-py3-none-any.whl", hash = "sha256:0d811377a4a2a20eeca3c3f443b7853a379df1a8d9fa859f264f8d70b8fd36f0", size = 103832, upload-time = "2026-08-18T18:39:41.424Z" }, ] [[package]] diff --git a/recipes/audio/python/sdk/ssml-emotion/pyproject.toml b/recipes/audio/python/sdk/ssml-emotion/pyproject.toml index 068fdcb..d7317d4 100644 --- a/recipes/audio/python/sdk/ssml-emotion/pyproject.toml +++ b/recipes/audio/python/sdk/ssml-emotion/pyproject.toml @@ -4,6 +4,6 @@ version = "0.1.0" description = "Control emotion, pitch, rate, pauses & emphasis via SSML with the Speechify TTS API." requires-python = ">=3.10" dependencies = [ - "speechify-api>=3.0.1", + "speechify-api>=4.0.0", "python-dotenv>=1.0.0", ] diff --git a/recipes/audio/python/sdk/ssml-emotion/uv.lock b/recipes/audio/python/sdk/ssml-emotion/uv.lock index c2a7deb..800cb10 100644 --- a/recipes/audio/python/sdk/ssml-emotion/uv.lock +++ b/recipes/audio/python/sdk/ssml-emotion/uv.lock @@ -37,7 +37,7 @@ dependencies = [ [package.metadata] requires-dist = [ { name = "python-dotenv", specifier = ">=1.0.0" }, - { name = "speechify-api", specifier = ">=3.0.1" }, + { name = "speechify-api", specifier = ">=4.0.0" }, ] [[package]] @@ -249,7 +249,7 @@ wheels = [ [[package]] name = "speechify-api" -version = "3.0.1" +version = "4.0.0" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "httpx" }, @@ -257,9 +257,9 @@ dependencies = [ { name = "pydantic-core" }, { name = "typing-extensions" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/6d/67/961acfbf3355c73b85c9a85e7d2ca4c38a9f14261c845ba9c804940969de/speechify_api-3.0.1.tar.gz", hash = "sha256:e73a084fd6845cb2f92bff2e4dfa6d58052ff97e8112b38dc67ebbf47610aa1a", size = 48230, upload-time = "2026-07-10T19:41:34.679Z" } +sdist = { url = "https://files.pythonhosted.org/packages/90/3c/4cef771d24e89577b5819b12b351434deb3b5bfd647c21c57c5ece6b2935/speechify_api-4.0.0.tar.gz", hash = "sha256:461ca9f3b6ec3f95a604f299c59152c46bda594a441570a80a4f04d88657f11e", size = 66631, upload-time = "2026-08-18T18:39:42.634Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/b3/7d/52daedb2aa7e1df49f39a03b0d87ea25cd281588102f4231cffbcbf32038/speechify_api-3.0.1-py3-none-any.whl", hash = "sha256:7e09d09bee792a348a5dbfe0e07a806152ba2797c70b399b6e5544b320808cf8", size = 75381, upload-time = "2026-07-10T19:41:33.749Z" }, + { url = "https://files.pythonhosted.org/packages/fd/70/b783459bd5729da204dfd077301a7a85399a5ccd0e3bb53c655ef421e600/speechify_api-4.0.0-py3-none-any.whl", hash = "sha256:0d811377a4a2a20eeca3c3f443b7853a379df1a8d9fa859f264f8d70b8fd36f0", size = 103832, upload-time = "2026-08-18T18:39:41.424Z" }, ] [[package]] diff --git a/recipes/audio/python/sdk/streaming/pyproject.toml b/recipes/audio/python/sdk/streaming/pyproject.toml index e88111c..b63b026 100644 --- a/recipes/audio/python/sdk/streaming/pyproject.toml +++ b/recipes/audio/python/sdk/streaming/pyproject.toml @@ -4,6 +4,6 @@ version = "0.1.0" description = "Stream synthesized audio to a file as it is generated with the Speechify TTS API." requires-python = ">=3.10" dependencies = [ - "speechify-api>=3.0.1", + "speechify-api>=4.0.0", "python-dotenv>=1.0.0", ] diff --git a/recipes/audio/python/sdk/streaming/uv.lock b/recipes/audio/python/sdk/streaming/uv.lock index 392f096..06dafc8 100644 --- a/recipes/audio/python/sdk/streaming/uv.lock +++ b/recipes/audio/python/sdk/streaming/uv.lock @@ -37,7 +37,7 @@ dependencies = [ [package.metadata] requires-dist = [ { name = "python-dotenv", specifier = ">=1.0.0" }, - { name = "speechify-api", specifier = ">=3.0.1" }, + { name = "speechify-api", specifier = ">=4.0.0" }, ] [[package]] @@ -249,7 +249,7 @@ wheels = [ [[package]] name = "speechify-api" -version = "3.0.1" +version = "4.0.0" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "httpx" }, @@ -257,9 +257,9 @@ dependencies = [ { name = "pydantic-core" }, { name = "typing-extensions" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/6d/67/961acfbf3355c73b85c9a85e7d2ca4c38a9f14261c845ba9c804940969de/speechify_api-3.0.1.tar.gz", hash = "sha256:e73a084fd6845cb2f92bff2e4dfa6d58052ff97e8112b38dc67ebbf47610aa1a", size = 48230, upload-time = "2026-07-10T19:41:34.679Z" } +sdist = { url = "https://files.pythonhosted.org/packages/90/3c/4cef771d24e89577b5819b12b351434deb3b5bfd647c21c57c5ece6b2935/speechify_api-4.0.0.tar.gz", hash = "sha256:461ca9f3b6ec3f95a604f299c59152c46bda594a441570a80a4f04d88657f11e", size = 66631, upload-time = "2026-08-18T18:39:42.634Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/b3/7d/52daedb2aa7e1df49f39a03b0d87ea25cd281588102f4231cffbcbf32038/speechify_api-3.0.1-py3-none-any.whl", hash = "sha256:7e09d09bee792a348a5dbfe0e07a806152ba2797c70b399b6e5544b320808cf8", size = 75381, upload-time = "2026-07-10T19:41:33.749Z" }, + { url = "https://files.pythonhosted.org/packages/fd/70/b783459bd5729da204dfd077301a7a85399a5ccd0e3bb53c655ef421e600/speechify_api-4.0.0-py3-none-any.whl", hash = "sha256:0d811377a4a2a20eeca3c3f443b7853a379df1a8d9fa859f264f8d70b8fd36f0", size = 103832, upload-time = "2026-08-18T18:39:41.424Z" }, ] [[package]] diff --git a/recipes/audio/python/sdk/voice-cloning/.env.example b/recipes/audio/python/sdk/voice-cloning/.env.example index 4cb51e4..cdea4b1 100644 --- a/recipes/audio/python/sdk/voice-cloning/.env.example +++ b/recipes/audio/python/sdk/voice-cloning/.env.example @@ -1 +1,11 @@ +# Get a key at https://platform.speechify.ai/api-keys +# Voice cloning must be enabled on your plan. SPEECHIFY_API_KEY= + +# Full name of the person consenting to have their voice cloned. +CONSENT_FULL_NAME=Jane Doe + +# Optional — override where the recipe reads the audio from. +# Defaults to sample.wav and consent.wav in the recipe folder. +# SAMPLE_PATH=./sample.wav +# CONSENT_RECORDING_PATH=./consent.wav diff --git a/recipes/audio/python/sdk/voice-cloning/README.md b/recipes/audio/python/sdk/voice-cloning/README.md index 1bde1e8..79476f6 100644 --- a/recipes/audio/python/sdk/voice-cloning/README.md +++ b/recipes/audio/python/sdk/voice-cloning/README.md @@ -1,7 +1,7 @@ # Text-to-Speech: voice cloning (Python) Clone a voice from an audio sample, synthesize speech with the clone, then delete it — the -full create → use → delete lifecycle. +full create → use → delete lifecycle, with **verified consent**. ## Prerequisites @@ -10,6 +10,9 @@ full create → use → delete lifecycle. pointing to [Speechify pricing](https://speechify.ai/pricing) (the API returns `402 voice_cloning_not_included`). - Python 3.10+ and [uv](https://docs.astral.sh/uv/) +- **Two recordings of the same consenting person** (see [Consent](#consent)): + - a voice **sample** to clone — 10–30s of clean speech + - a **consent recording** — that person reading the challenge phrase this recipe prints ## Setup @@ -20,29 +23,41 @@ uv sync ## Run +Because consent is verified against a phrase the API generates, this is a two-step run: + ```bash -uv run main.py +uv run main.py # 1st run: prints the phrase to read, then exits +# record the speaker reading that phrase → consent.wav, and their voice → sample.wav +uv run main.py # 2nd run: clones, synthesizes, deletes ``` -Produces an `output.mp3` spoken in the cloned voice, then removes the cloned voice. +Put `sample.wav` and `consent.wav` in the recipe folder, or point `SAMPLE_PATH` / +`CONSENT_RECORDING_PATH` at them. Produces an `output.mp3` in the cloned voice, then +removes the cloned voice. ## What it does -- `client.voices.create(...)` — clones a voice from `fixtures/spacewalk.wav`. Required - fields: `name`, `gender`, `sample` (a readable binary file of 10–30s of clean speech), - and `consent`. +- `client.voices.consent_challenges.create(full_name=...)` — starts a consent challenge and + returns a `phrase` the speaker must read aloud (cached in `.consent-challenge.json` + between runs; single-use). +- `client.voices.create(...)` — clones the voice. Required: `name`, `gender`, `sample` (a + readable binary file of 10–30s of clean speech), `consent_challenge_id`, and + `consent_recording` (the speaker reading the phrase). - `client.audio.speech(...)` with `voice_id` set to the new voice's id. -- `client.voices.delete(id)` — cleans up so personal voices don't accumulate. +- `client.voices.delete(voice_id)` — cleans up so personal voices don't accumulate. ## Consent -`consent` is a **required** JSON string attesting you have the speaker's permission to -clone their voice (`{"fullName": "...", "email": "..."}`). Only clone voices you are -authorized to. The bundled `fixtures/spacewalk.wav` is ~26s of NASA ISS spacewalk audio -(U.S. government work, public domain); replace it with your own consented sample for real -use. +Cloning requires **verified consent**. You create a consent challenge, the speaker records +themselves reading the returned `phrase`, and that recording is sent as `consent_recording` +and retained as the consent record. The `consent_recording` **must be the same person** as +the voice `sample` — so there is no shippable sample that comes with valid consent, and this +recipe is deliberately bring-your-own-audio. Only clone voices you are authorized to. +> The old `consent` JSON field (`fullName` + `email`) is removed in SDK 4.x — see the +> [migration guide](https://docs.speechify.ai/build/guides/deprecations/migrating-voice-cloning-consent). +> > Note: `voices.list()` is eventually consistent — a just-deleted voice may still appear in > the list briefly. The delete itself is immediate (a subsequent delete returns 404). > -> Voice cloning reference: https://docs.speechify.ai/tts/guides/voice-cloning +> Voice cloning reference: https://docs.speechify.ai/build/guides/voice-cloning/overview diff --git a/recipes/audio/python/sdk/voice-cloning/fixtures/spacewalk.wav b/recipes/audio/python/sdk/voice-cloning/fixtures/spacewalk.wav deleted file mode 100644 index 498be74..0000000 Binary files a/recipes/audio/python/sdk/voice-cloning/fixtures/spacewalk.wav and /dev/null differ diff --git a/recipes/audio/python/sdk/voice-cloning/main.py b/recipes/audio/python/sdk/voice-cloning/main.py index e185fd0..3e58cb9 100644 --- a/recipes/audio/python/sdk/voice-cloning/main.py +++ b/recipes/audio/python/sdk/voice-cloning/main.py @@ -1,13 +1,35 @@ import base64 import json import os +from datetime import datetime, timezone from dotenv import load_dotenv from speechify import Speechify from speechify.core.api_error import ApiError -# Bundled sample: ~26s of NASA ISS spacewalk audio (public domain). -SAMPLE_PATH = os.path.join(os.path.dirname(__file__), "fixtures", "spacewalk.wav") +# Cloning requires VERIFIED consent: the speaker records themselves reading a +# phrase Speechify returns, and that recording is kept as the consent record. +# It must be the SAME person as the voice sample, so this recipe is +# bring-your-own-audio — there is no sample that ships with valid consent. +HERE = os.path.dirname(__file__) +CONSENT_FULL_NAME = os.environ.get("CONSENT_FULL_NAME", "Jane Doe") +# The voice to clone (10-30s of clean speech) and the consent recording (the +# same person reading the challenge phrase, 5-30s). Same speaker in both. +SAMPLE_PATH = os.environ.get("SAMPLE_PATH", os.path.join(HERE, "sample.wav")) +CONSENT_RECORDING_PATH = os.environ.get("CONSENT_RECORDING_PATH", os.path.join(HERE, "consent.wav")) +# The challenge is single-use and its phrase is dynamic, so we cache it between +# runs: run once to get the phrase, record it, run again to submit. +CHALLENGE_CACHE = os.path.join(HERE, ".consent-challenge.json") + + +def load_challenge(): + if not os.path.exists(CHALLENGE_CACHE): + return None + with open(CHALLENGE_CACHE) as f: + c = json.load(f) + if datetime.fromisoformat(c["expires_at"]) <= datetime.now(timezone.utc): + return None # expired -> make a fresh one + return c def main() -> None: @@ -19,16 +41,44 @@ def main() -> None: client = Speechify(token=token) - # 1. Clone a voice from an audio sample (10-30s of clean speech works well). - # `consent` is REQUIRED: a JSON string attesting you have the speaker's - # permission to clone their voice. Use the real consenting person's details. + # 1. Get (or reuse) a consent challenge. Its `phrase` is what the speaker + # must read aloud; `id` ties the recording to this consent on the create. + challenge = load_challenge() + if challenge is None: + created = client.voices.consent_challenges.create(full_name=CONSENT_FULL_NAME) + challenge = { + "id": created.id, + "phrase": created.phrase, + "expires_at": created.expires_at.isoformat(), + } + with open(CHALLENGE_CACHE, "w") as f: + json.dump(challenge, f, indent=2) + + # 2. Make sure we have both recordings before spending the (single-use) challenge. + missing = [] + if not os.path.exists(SAMPLE_PATH): + missing.append(f" sample: {SAMPLE_PATH} ({CONSENT_FULL_NAME}'s voice, 10-30s of clean speech)") + if not os.path.exists(CONSENT_RECORDING_PATH): + missing.append(f" consent: {CONSENT_RECORDING_PATH} (the SAME person reading the phrase below)") + if missing: + raise SystemExit( + f"\nConsent required. Have {CONSENT_FULL_NAME} record themselves reading this phrase, " + "exactly as written:\n\n" + f' "{challenge["phrase"]}"\n\n' + "Then provide these files and re-run (paths override via SAMPLE_PATH / CONSENT_RECORDING_PATH):\n" + + "\n".join(missing) + + f"\n\nChallenge expires {challenge['expires_at']}.\n" + ) + + # 3. Clone the voice: the sample + the consent recording + the challenge id. try: - with open(SAMPLE_PATH, "rb") as sample: + with open(SAMPLE_PATH, "rb") as sample, open(CONSENT_RECORDING_PATH, "rb") as consent: voice = client.voices.create( name="cookbook-cloned-voice", gender="male", sample=sample, - consent=json.dumps({"fullName": "Jane Doe", "email": "jane@example.com"}), + consent_challenge_id=challenge["id"], + consent_recording=consent, ) except ApiError as err: # Voice cloning is gated by plan. A 402 means it isn't included in yours. @@ -38,10 +88,11 @@ def main() -> None: "Upgrade to a plan that includes voice cloning: https://speechify.ai/pricing\n" ) raise + os.remove(CHALLENGE_CACHE) # challenge is spent print(f"Cloned voice created: {voice.id} ({voice.display_name}, type={voice.type})") try: - # 2. Synthesize speech using the cloned voice — pass its id as voice_id. + # 4. Synthesize speech using the cloned voice — pass its id as voice_id. speech = client.audio.speech( input="Hello from a voice cloned with the Speechify API.", voice_id=voice.id, @@ -52,7 +103,7 @@ def main() -> None: f.write(base64.b64decode(speech.audio_data)) print("Wrote output.mp3") finally: - # 3. Clean up so cloned voices don't accumulate on your account. + # 5. Clean up so cloned voices don't accumulate on your account. # Remove this to keep the voice and reuse it later via voice.id. client.voices.delete(voice.id) print(f"Deleted cloned voice {voice.id}") diff --git a/recipes/audio/python/sdk/voice-cloning/pyproject.toml b/recipes/audio/python/sdk/voice-cloning/pyproject.toml index 30f972f..f6c3a5c 100644 --- a/recipes/audio/python/sdk/voice-cloning/pyproject.toml +++ b/recipes/audio/python/sdk/voice-cloning/pyproject.toml @@ -4,6 +4,6 @@ version = "0.1.0" description = "Clone a voice from an audio sample, synthesize with it, then delete it (Speechify TTS API)." requires-python = ">=3.10" dependencies = [ - "speechify-api>=3.0.1", + "speechify-api>=4.0.0", "python-dotenv>=1.0.0", ] diff --git a/recipes/audio/python/sdk/voice-cloning/uv.lock b/recipes/audio/python/sdk/voice-cloning/uv.lock index 458db63..4a853c2 100644 --- a/recipes/audio/python/sdk/voice-cloning/uv.lock +++ b/recipes/audio/python/sdk/voice-cloning/uv.lock @@ -37,7 +37,7 @@ dependencies = [ [package.metadata] requires-dist = [ { name = "python-dotenv", specifier = ">=1.0.0" }, - { name = "speechify-api", specifier = ">=3.0.1" }, + { name = "speechify-api", specifier = ">=4.0.0" }, ] [[package]] @@ -249,7 +249,7 @@ wheels = [ [[package]] name = "speechify-api" -version = "3.0.1" +version = "4.0.0" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "httpx" }, @@ -257,9 +257,9 @@ dependencies = [ { name = "pydantic-core" }, { name = "typing-extensions" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/6d/67/961acfbf3355c73b85c9a85e7d2ca4c38a9f14261c845ba9c804940969de/speechify_api-3.0.1.tar.gz", hash = "sha256:e73a084fd6845cb2f92bff2e4dfa6d58052ff97e8112b38dc67ebbf47610aa1a", size = 48230, upload-time = "2026-07-10T19:41:34.679Z" } +sdist = { url = "https://files.pythonhosted.org/packages/90/3c/4cef771d24e89577b5819b12b351434deb3b5bfd647c21c57c5ece6b2935/speechify_api-4.0.0.tar.gz", hash = "sha256:461ca9f3b6ec3f95a604f299c59152c46bda594a441570a80a4f04d88657f11e", size = 66631, upload-time = "2026-08-18T18:39:42.634Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/b3/7d/52daedb2aa7e1df49f39a03b0d87ea25cd281588102f4231cffbcbf32038/speechify_api-3.0.1-py3-none-any.whl", hash = "sha256:7e09d09bee792a348a5dbfe0e07a806152ba2797c70b399b6e5544b320808cf8", size = 75381, upload-time = "2026-07-10T19:41:33.749Z" }, + { url = "https://files.pythonhosted.org/packages/fd/70/b783459bd5729da204dfd077301a7a85399a5ccd0e3bb53c655ef421e600/speechify_api-4.0.0-py3-none-any.whl", hash = "sha256:0d811377a4a2a20eeca3c3f443b7853a379df1a8d9fa859f264f8d70b8fd36f0", size = 103832, upload-time = "2026-08-18T18:39:41.424Z" }, ] [[package]] diff --git a/recipes/audio/typescript/sdk/streaming/src/index.ts b/recipes/audio/typescript/sdk/streaming/src/index.ts index de15b6b..6d8f436 100644 --- a/recipes/audio/typescript/sdk/streaming/src/index.ts +++ b/recipes/audio/typescript/sdk/streaming/src/index.ts @@ -16,11 +16,13 @@ async function main() { // long inputs. const response = await client.audio.stream({ Accept: "audio/mpeg", - input: - "Streaming lets you start playing audio before the whole clip is ready. " + - "This sentence is being synthesized and written to disk chunk by chunk.", - voice_id: "geffen_32", - model: "simba-3.2", + body: { + input: + "Streaming lets you start playing audio before the whole clip is ready. " + + "This sentence is being synthesized and written to disk chunk by chunk.", + voice_id: "geffen_32", + model: "simba-3.2", + }, }); const outFile = "output.mp3"; diff --git a/recipes/audio/typescript/sdk/voice-cloning/.env.example b/recipes/audio/typescript/sdk/voice-cloning/.env.example index 6f5312e..cdea4b1 100644 --- a/recipes/audio/typescript/sdk/voice-cloning/.env.example +++ b/recipes/audio/typescript/sdk/voice-cloning/.env.example @@ -1,3 +1,11 @@ # Get a key at https://platform.speechify.ai/api-keys # Voice cloning must be enabled on your plan. SPEECHIFY_API_KEY= + +# Full name of the person consenting to have their voice cloned. +CONSENT_FULL_NAME=Jane Doe + +# Optional — override where the recipe reads the audio from. +# Defaults to sample.wav and consent.wav in the recipe folder. +# SAMPLE_PATH=./sample.wav +# CONSENT_RECORDING_PATH=./consent.wav diff --git a/recipes/audio/typescript/sdk/voice-cloning/README.md b/recipes/audio/typescript/sdk/voice-cloning/README.md index f2988ea..0c0a479 100644 --- a/recipes/audio/typescript/sdk/voice-cloning/README.md +++ b/recipes/audio/typescript/sdk/voice-cloning/README.md @@ -1,7 +1,7 @@ # Text-to-Speech: voice cloning (TypeScript) Clone a voice from an audio sample, synthesize speech with the clone, then delete it — the -full create → use → delete lifecycle. +full create → use → delete lifecycle, with **verified consent**. ## Prerequisites @@ -10,6 +10,9 @@ full create → use → delete lifecycle. pointing to [Speechify pricing](https://speechify.ai/pricing) (the API returns `402 voice_cloning_not_included`). - Node 20+ +- **Two recordings of the same consenting person** (see [Consent](#consent)): + - a voice **sample** to clone — 10–30s of clean speech + - a **consent recording** — that person reading the challenge phrase this recipe prints ## Setup @@ -20,29 +23,41 @@ pnpm install ## Run +Because consent is verified against a phrase the API generates, this is a two-step run: + ```bash -pnpm start +pnpm start # 1st run: prints the phrase to read, then exits +# record the speaker reading that phrase → consent.wav, and their voice → sample.wav +pnpm start # 2nd run: clones, synthesizes, deletes ``` -Produces an `output.mp3` spoken in the cloned voice, then removes the cloned voice. +Put `sample.wav` and `consent.wav` in the recipe folder, or point `SAMPLE_PATH` / +`CONSENT_RECORDING_PATH` at them. Produces an `output.mp3` in the cloned voice, then +removes the cloned voice. ## What it does -- `client.voices.create(...)` — clones a voice from `fixtures/spacewalk.wav`. Required - fields: `name`, `gender`, `sample` (a readable stream of 10–30s of clean speech), and - `consent`. +- `client.voices.consentChallenges.create({ full_name })` — starts a consent challenge and + returns a `phrase` the speaker must read aloud (cached in `.consent-challenge.json` + between runs; single-use). +- `client.voices.create(...)` — clones the voice. Required: `name`, `gender`, `sample` (a + readable stream of 10–30s of clean speech), `consent_challenge_id`, and + `consent_recording` (the speaker reading the phrase). - `client.audio.speech(...)` with `voice_id` set to the new voice's id. -- `client.voices.delete({ id })` — cleans up so personal voices don't accumulate. +- `client.voices.delete({ voice_id })` — cleans up so personal voices don't accumulate. ## Consent -`consent` is a **required** JSON string attesting you have the speaker's permission to -clone their voice (`{"fullName": "...", "email": "..."}`). Only clone voices you are -authorized to. The bundled `fixtures/spacewalk.wav` is ~26s of NASA ISS spacewalk audio -(U.S. government work, public domain); replace it with your own consented sample for real -use. +Cloning requires **verified consent**. You create a consent challenge, the speaker records +themselves reading the returned `phrase`, and that recording is sent as `consent_recording` +and retained as the consent record. The `consent_recording` **must be the same person** as +the voice `sample` — so there is no shippable sample that comes with valid consent, and this +recipe is deliberately bring-your-own-audio. Only clone voices you are authorized to. +> The old `consent` JSON field (`fullName` + `email`) is removed in SDK 4.x — see the +> [migration guide](https://docs.speechify.ai/build/guides/deprecations/migrating-voice-cloning-consent). +> > Note: `voices.list()` is eventually consistent — a just-deleted voice may still appear in > the list briefly. The delete itself is immediate (a subsequent delete returns 404). > -> Voice cloning reference: https://docs.speechify.ai/tts/guides/voice-cloning +> Voice cloning reference: https://docs.speechify.ai/build/guides/voice-cloning/overview diff --git a/recipes/audio/typescript/sdk/voice-cloning/fixtures/spacewalk.wav b/recipes/audio/typescript/sdk/voice-cloning/fixtures/spacewalk.wav deleted file mode 100644 index 498be74..0000000 Binary files a/recipes/audio/typescript/sdk/voice-cloning/fixtures/spacewalk.wav and /dev/null differ diff --git a/recipes/audio/typescript/sdk/voice-cloning/src/index.ts b/recipes/audio/typescript/sdk/voice-cloning/src/index.ts index 2014a6f..d4d98a8 100644 --- a/recipes/audio/typescript/sdk/voice-cloning/src/index.ts +++ b/recipes/audio/typescript/sdk/voice-cloning/src/index.ts @@ -9,20 +9,67 @@ if (!process.env.SPEECHIFY_API_KEY) { const client = new SpeechifyClient({ token: process.env.SPEECHIFY_API_KEY }); -// Bundled sample: ~26s of NASA ISS spacewalk audio (public domain). -const samplePath = path.resolve(import.meta.dirname, "../fixtures/spacewalk.wav"); +// Cloning requires VERIFIED consent: the speaker records themselves reading a +// phrase Speechify returns, and that recording is kept as the consent record. +// The recording must be the SAME person as the voice sample. So this recipe is +// bring-your-own-audio — there is no sample that ships with valid consent. +const CONSENT_FULL_NAME = process.env.CONSENT_FULL_NAME ?? "Jane Doe"; +const dir = import.meta.dirname; +// The voice to clone (10–30s of clean speech) and the consent recording (the +// same person reading the challenge phrase, 5–30s). Same speaker in both. +const samplePath = path.resolve(process.env.SAMPLE_PATH ?? path.join(dir, "../sample.wav")); +const consentPath = path.resolve(process.env.CONSENT_RECORDING_PATH ?? path.join(dir, "../consent.wav")); +// The challenge is single-use and its phrase is dynamic, so we cache it between +// runs: run once to get the phrase, record it, run again to submit. +const challengeCache = path.join(dir, "../.consent-challenge.json"); + +interface CachedChallenge { + id: string; + phrase: string; + expires_at: string; +} + +function loadChallenge(): CachedChallenge | null { + if (!fs.existsSync(challengeCache)) return null; + const c = JSON.parse(fs.readFileSync(challengeCache, "utf8")) as CachedChallenge; + if (new Date(c.expires_at).getTime() <= Date.now()) return null; // expired → make a fresh one + return c; +} async function main() { - // 1. Clone a voice from an audio sample (10–30s of clean speech works well). - // `consent` is REQUIRED: a JSON string attesting you have the speaker's - // permission to clone their voice. Use the real consenting person's details. + // 1. Get (or reuse) a consent challenge. Its `phrase` is what the speaker + // must read aloud; `id` ties the recording to this consent on the create. + let challenge = loadChallenge(); + if (!challenge) { + challenge = await client.voices.consentChallenges.create({ full_name: CONSENT_FULL_NAME }); + fs.writeFileSync(challengeCache, JSON.stringify(challenge, null, 2)); + } + + // 2. Make sure we have both recordings before spending the (single-use) challenge. + const missing = [ + fs.existsSync(samplePath) ? null : ` sample: ${samplePath} (${CONSENT_FULL_NAME}'s voice, 10–30s of clean speech)`, + fs.existsSync(consentPath) ? null : ` consent: ${consentPath} (the SAME person reading the phrase below)`, + ].filter(Boolean); + if (missing.length > 0) { + console.log( + `\nConsent required. Have ${CONSENT_FULL_NAME} record themselves reading this phrase, exactly as written:\n\n` + + ` "${challenge.phrase}"\n\n` + + `Then provide these files and re-run (paths override via SAMPLE_PATH / CONSENT_RECORDING_PATH):\n` + + missing.join("\n") + + `\n\nChallenge expires ${challenge.expires_at}.\n`, + ); + process.exit(1); + } + + // 3. Clone the voice: the sample + the consent recording + the challenge id. let voice; try { voice = await client.voices.create({ name: "cookbook-cloned-voice", gender: "male", sample: fs.createReadStream(samplePath), - consent: JSON.stringify({ fullName: "Jane Doe", email: "jane@example.com" }), + consent_challenge_id: challenge.id, + consent_recording: fs.createReadStream(consentPath), }); } catch (err) { // Voice cloning is gated by plan. A 402 means it isn't included in yours. @@ -35,10 +82,11 @@ async function main() { } throw err; } + fs.rmSync(challengeCache, { force: true }); // challenge is spent console.log(`Cloned voice created: ${voice.id} (${voice.display_name}, type=${voice.type})`); try { - // 2. Synthesize speech using the cloned voice — pass its id as voice_id. + // 4. Synthesize speech using the cloned voice — pass its id as voice_id. const speech = await client.audio.speech({ input: "Hello from a voice cloned with the Speechify API.", voice_id: voice.id, @@ -48,7 +96,7 @@ async function main() { fs.writeFileSync("output.mp3", Buffer.from(speech.audio_data, "base64")); console.log("Wrote output.mp3"); } finally { - // 3. Clean up so cloned voices don't accumulate on your account. + // 5. Clean up so cloned voices don't accumulate on your account. // Remove this to keep the voice and reuse it later via voice.id. await client.voices.delete({ voice_id: voice.id }); console.log(`Deleted cloned voice ${voice.id}`); diff --git a/templates/python/recipe/pyproject.toml b/templates/python/recipe/pyproject.toml index aa430c4..cdba0c6 100644 --- a/templates/python/recipe/pyproject.toml +++ b/templates/python/recipe/pyproject.toml @@ -3,6 +3,6 @@ name = "change-me-python-recipe" version = "0.1.0" requires-python = ">=3.10" dependencies = [ - "speechify-api>=3.0.1", + "speechify-api>=4.0.0", "python-dotenv>=1.0.0", ]