{
  "reviewed": "2026-10-08",
  "scope": "Documented capabilities and editorial workflow fits, not benchmark results",
  "tools": [
    {
      "id": "elevenlabs",
      "name": "ElevenLabs",
      "kind": "Creator studio + API",
      "status": "active",
      "uses": [
        "narration",
        "dialogue",
        "languages",
        "live"
      ],
      "summary": "A broad creator workflow for narration, voices, dialogue and dubbing. Choose the model family as carefully as the voice.",
      "features": [
        {
          "name": "Models",
          "value": "v4 / v4 Turbo; v3; Multilingual v2; Flash v2.5"
        },
        {
          "name": "Direction",
          "value": "Inline audio tags on supported v3/v4 models"
        },
        {
          "name": "Languages",
          "value": "v4: 90+; coverage differs for other families"
        },
        {
          "name": "Voice identity",
          "value": "Library, design and cloning; access depends on feature and plan"
        },
        {
          "name": "Dialogue / dubbing",
          "value": "Multi-speaker generation; separate dubbing product"
        },
        {
          "name": "Access",
          "value": "Web editor and API; commercial terms depend on plan"
        }
      ],
      "pros": [
        "A practical starting point for creators who prefer a web interface.",
        "Several model families let you separate performance from fast delivery.",
        "Dubbing is available alongside speech generation."
      ],
      "cons": [
        "Audio tags are model-specific; a voice can behave differently after a model switch.",
        "Dubbing v2 website output is automatic; transcript editing/regeneration via API requires Enterprise.",
        "Credits, allowances and temporary promotions complicate price comparisons."
      ],
      "workflow": [
        "Pick v4 for an expressive short narration and audition two library voices.",
        "Use one performance instruction, generate a short take, then listen for skipped words.",
        "Lock the voice/model, regenerate only the changed section and assemble in your editor."
      ],
      "example": "[reassuring] You are right on time. [pause] We can take this one step at a time.",
      "price": "Listed v4 API rate: $0.08/1K characters before the temporary $0.022 promotion ending October 12, 2026. Our planner uses the regular rate, not the promotion. Check plan allowances and rights.",
      "sources": [
        "eleven",
        "tags",
        "eleven-price",
        "dubbing"
      ],
      "demo": "https://elevenlabs.io/text-to-speech"
    },
    {
      "id": "cartesia",
      "name": "Cartesia Sonic",
      "kind": "Streaming speech API",
      "status": "active",
      "uses": [
        "live",
        "languages",
        "narration"
      ],
      "summary": "An API-oriented option for conversational speech and repeatable deployment. Sonic 3.6 offers a dated snapshot as well as an automatically updated alias.",
      "features": [
        {
          "name": "Model",
          "value": "Sonic 3.6; dated snapshot 2026-08-27"
        },
        {
          "name": "Languages",
          "value": "44 documented languages"
        },
        {
          "name": "Direction",
          "value": "Model-specific emotion, speed and volume controls"
        },
        {
          "name": "Voice identity",
          "value": "Voice library and consent-based cloning"
        },
        {
          "name": "Delivery",
          "value": "Streaming API; playground for auditioning"
        },
        {
          "name": "Deployment",
          "value": "Hosted; enterprise deployment options require a separate agreement"
        }
      ],
      "pros": [
        "Streaming suits responses that begin before the whole passage is ready.",
        "Dated versions make your own regression checks reproducible.",
        "Locale and normalization controls help structured speech."
      ],
      "cons": [
        "It is an API workflow rather than a complete audiobook or video editor.",
        "Provider latency claims are not end-to-end conversation measurements.",
        "Commercial use, credits and concurrency depend on your plan."
      ],
      "workflow": [
        "Audition a voice with your actual product vocabulary.",
        "Pin a dated model snapshot and test your chosen locale.",
        "Stream short turns and test interruptions in the application, not just the playground."
      ],
      "example": "A short support response: “I found your booking. Your reference is A, one, zero, four.” Test the same line with the intended locale.",
      "price": "Credit-based self-serve plans; inspect the current character conversion, concurrency and commercial terms. Do not compare a credit count directly with another vendor’s credits.",
      "sources": [
        "cartesia",
        "cartesia-controls",
        "cartesia-price",
        "cartesia-demo"
      ],
      "demo": "https://play.cartesia.ai/"
    },
    {
      "id": "deepgram",
      "name": "Deepgram Aura",
      "kind": "Speech infrastructure API",
      "status": "active",
      "uses": [
        "live",
        "narration"
      ],
      "summary": "Useful to evaluate when spoken reminders, identifiers and support conversations matter more than theatrical performance. Speech recognition and agents are separate services.",
      "features": [
        {
          "name": "Model",
          "value": "Aura-2 voice catalog"
        },
        {
          "name": "Direction",
          "value": "Voice selection and text preparation; inspect supported controls"
        },
        {
          "name": "Voice identity",
          "value": "Named voice IDs from the documented catalog"
        },
        {
          "name": "Delivery",
          "value": "REST and streaming speech endpoints"
        },
        {
          "name": "Related products",
          "value": "Separate transcription and voice-agent APIs"
        },
        {
          "name": "Language check",
          "value": "Match the exact voice ID to its supported language"
        }
      ],
      "pros": [
        "The published examples make numbers and tricky vocabulary tangible.",
        "Speech and transcription can live within one provider stack.",
        "Streaming can reduce waiting for the entire output."
      ],
      "cons": [
        "A launch-page preference study is conducted by the provider, not by us.",
        "Voice synthesis alone does not listen, reason or handle interruption.",
        "Catalog language support is not interchangeable with transcription language support."
      ],
      "workflow": [
        "Choose an Aura-2 voice appropriate to the listener and language.",
        "Test dates, amounts, names and your most frequent identifiers.",
        "Integrate short streamed responses, then measure the entire request-to-playback path."
      ],
      "example": "An appointment reminder with “March fourth”, “three oh five in the afternoon” and a letter-by-letter reference code.",
      "price": "Character-based synthesis; transcription and full voice-agent usage have separate rates. Check the current pay-as-you-go tier and language/model.",
      "sources": [
        "deepgram",
        "deepgram-demo",
        "deepgram-price"
      ],
      "demo": "https://playground.deepgram.com/"
    },
    {
      "id": "openai",
      "name": "OpenAI speech",
      "kind": "Instruction-driven speech API",
      "status": "active",
      "uses": [
        "narration",
        "languages",
        "live"
      ],
      "summary": "A focused speech endpoint for apps and scripted content. Our original studio examples use one pinned model, with the script and instructions published.",
      "features": [
        {
          "name": "Our example model",
          "value": "gpt-4o-mini-tts-2025-12-15"
        },
        {
          "name": "Direction",
          "value": "Natural-language instructions; speed parameter"
        },
        {
          "name": "Request limit",
          "value": "4,096 input characters per speech request"
        },
        {
          "name": "Formats",
          "value": "MP3, Opus, AAC, FLAC, WAV and PCM"
        },
        {
          "name": "Voice identity",
          "value": "Built-in voices; custom voices are eligibility-gated"
        },
        {
          "name": "Live conversation",
          "value": "The speech endpoint differs from a Realtime conversation API"
        }
      ],
      "pros": [
        "Separate text and delivery instructions make controlled lessons easy to reproduce.",
        "Output formats suit both editing and application playback.",
        "A dated model makes the provenance of these examples explicit."
      ],
      "cons": [
        "This endpoint is not a complete creator timeline or dubbing editor.",
        "Instructions guide a performance; they are not a guarantee of exact timing or stress.",
        "Actual billing is token-based; the per-minute price is an estimate."
      ],
      "workflow": [
        "Write a short, plain-language script and choose a built-in voice.",
        "Add a concise tone/pacing instruction and generate two takes.",
        "Listen with the transcript, correct pronunciation and export the chosen take."
      ],
      "example": "Text: “You are right on time.” Instructions: Warm and reassuring; gentle pace; speak to one worried person.",
      "price": "The pricing page estimates gpt-4o-mini-tts at $0.015 per generated minute. Actual charges use text and audio tokens; retries create additional output.",
      "sources": [
        "openai",
        "openai-api",
        "openai-price"
      ],
      "demo": "https://www.openai.fm/"
    },
    {
      "id": "gemini",
      "name": "Google Gemini TTS",
      "kind": "Directed speech + dialogue API",
      "status": "active",
      "uses": [
        "dialogue",
        "narration",
        "languages"
      ],
      "summary": "A route for style-directed narration and dialogue. Google’s current guide documents both speaker configuration and per-turn direction.",
      "features": [
        {
          "name": "Direction",
          "value": "Style instructions and speech metadata in the current interface"
        },
        {
          "name": "Dialogue",
          "value": "Up to two prebuilt speakers in a single multi-speaker request"
        },
        {
          "name": "Custom dialogue",
          "value": "Custom voices require separate turns in the documented workflow"
        },
        {
          "name": "Output",
          "value": "Audio output; inspect WAV/PCM handling for assembly"
        },
        {
          "name": "Languages",
          "value": "Check the supported language list for the chosen model"
        },
        {
          "name": "Model / interface",
          "value": "Pin the exact model and SDK/API version from the current guide"
        }
      ],
      "pros": [
        "Useful to explore a scripted exchange between two speakers.",
        "Turn-level direction fits dialogue rather than a single narration style.",
        "The documentation includes single- and multi-speaker workflows."
      ],
      "cons": [
        "Two-speaker generation is not an unlimited cast in one request.",
        "Custom-voice turns need separate synthesis and assembly.",
        "Different Gemini/Cloud interfaces expose different controls and billing."
      ],
      "workflow": [
        "Create a two-person exchange with short, clearly labeled turns.",
        "Assign prebuilt voices and a restrained style for each turn.",
        "Check voice separation and silence, then assemble and caption the output."
      ],
      "example": "HOST: “What changes when we slow down?” GUEST: “The listener gets a moment to think.” Keep the two styles calm and distinct.",
      "price": "Model-specific API pricing; inspect text-input and audio-output units for your selected interface. A Gemini API allowance is not a Google Cloud TTS allowance.",
      "sources": [
        "gemini",
        "google-price"
      ],
      "demo": "https://aistudio.google.com/"
    },
    {
      "id": "chirp",
      "name": "Google Chirp 3 HD",
      "kind": "Cloud speech synthesis",
      "status": "active",
      "uses": [
        "languages",
        "narration",
        "live"
      ],
      "summary": "A cloud voice catalog with documented text and pronunciation controls. The examples below show what a real pause or pronunciation override changes.",
      "features": [
        {
          "name": "Model",
          "value": "Chirp 3: HD, Google Cloud Text-to-Speech"
        },
        {
          "name": "Direction",
          "value": "Documented SSML; voice controls include preview features"
        },
        {
          "name": "Pacing",
          "value": "Documented speaking-rate control, 0.25–2.0"
        },
        {
          "name": "Pronunciation",
          "value": "IPA/X-SAMPA overrides in supported locales"
        },
        {
          "name": "Pauses",
          "value": "Pause markup uses the markup field, not plain text"
        },
        {
          "name": "Access",
          "value": "Cloud project, billing and synthesis API"
        }
      ],
      "pros": [
        "Explicit control can be useful for names, codes and instructional pacing.",
        "The voice catalog lets you inspect locale-specific options.",
        "Published examples include the same phrase with and without a pause."
      ],
      "cons": [
        "Preview controls have language and launch-stage limitations.",
        "Pause duration can vary with context and some tags may be ignored.",
        "Cloud setup takes more effort than an all-in-one creator editor."
      ],
      "workflow": [
        "Choose the exact locale and voice ID.",
        "Test a phrase with a difficult name and a deliberate pause.",
        "Use supported pronunciation controls and validate the actual audio before scaling."
      ],
      "example": "Compare “Let me take a look, yes, I see it.” with the same text using [pause long] in the documented markup field.",
      "price": "Published Chirp 3 HD rate: $30 per million characters after the listed free allowance. Planner excludes allowances, taxes and other Cloud services.",
      "sources": [
        "chirp",
        "google-price"
      ],
      "demo": "https://cloud.google.com/text-to-speech"
    },
    {
      "id": "azure",
      "name": "Microsoft Azure Speech",
      "kind": "Cloud speech + SSML",
      "status": "active",
      "uses": [
        "languages",
        "narration",
        "live"
      ],
      "summary": "A route for teams that need structured control over pronunciation and pacing within an existing Microsoft speech workflow.",
      "features": [
        {
          "name": "Direction",
          "value": "SSML for pronunciation, rate, pitch, pauses and more"
        },
        {
          "name": "Styles",
          "value": "Speaking styles and roles depend on the selected voice"
        },
        {
          "name": "Access",
          "value": "Speech SDK, REST and voice-specific configuration"
        },
        {
          "name": "Output",
          "value": "Select an audio format appropriate to the playback channel"
        },
        {
          "name": "Voice / language",
          "value": "Verify the exact voice’s locale and supported features"
        },
        {
          "name": "Workflow",
          "value": "Cloud resource and credentials; editor or app integration"
        }
      ],
      "pros": [
        "SSML makes the intended structure explicit.",
        "SDK and REST routes support integration into existing systems.",
        "Voice-specific styles can support instructional speech."
      ],
      "cons": [
        "Not every voice supports every SSML element or style.",
        "Markup support does not guarantee the pronunciation you intended.",
        "Resource region, voice availability and the selected tier need checking."
      ],
      "workflow": [
        "Choose a supported neural voice and inspect its style list.",
        "Add one pronunciation override and one pause in SSML.",
        "Synthesize a short test, review by ear and export at the required audio format."
      ],
      "example": "An SSML learning example: <speak version=\"1.0\" xml:lang=\"en-US\"><voice name=\"en-US-JennyNeural\">Take a breath.<break time=\"500ms\"/>Then continue.</voice></speak>",
      "price": "Cloud billing depends on the voice tier and region. Check the Speech pricing page for your resource; synthesis, custom voice work and other services differ.",
      "sources": [
        "azure",
        "azure-how"
      ],
      "demo": "https://speech.microsoft.com/portal/voicegallery"
    },
    {
      "id": "kokoro",
      "name": "Kokoro-82M",
      "kind": "Open-weight local model",
      "status": "active",
      "uses": [
        "local",
        "narration"
      ],
      "summary": "A small open-weight speech model for people who want to operate their own runtime. The model card is the authoritative project entry point.",
      "features": [
        {
          "name": "Size",
          "value": "82 million parameters"
        },
        {
          "name": "Weights license",
          "value": "Apache 2.0"
        },
        {
          "name": "Model card release",
          "value": "v1.0: January 27, 2025"
        },
        {
          "name": "Voice workflow",
          "value": "Select supplied voices and language pipeline"
        },
        {
          "name": "Deployment",
          "value": "Local runtime or a third-party host"
        },
        {
          "name": "Control",
          "value": "Text/phoneme preparation and pipeline settings; no promised arbitrary acting prompts"
        }
      ],
      "pros": [
        "Your deployment can run without sending each script to a hosted speech API.",
        "Open weights allow experimentation with the synthesis pipeline.",
        "A useful alternative when hosting control matters more than a creator suite."
      ],
      "cons": [
        "Local operation still costs compute, setup and maintenance.",
        "A third-party hosted endpoint has its own pricing and data handling.",
        "Open weights do not provide an unlimited cloning or dialogue editor."
      ],
      "workflow": [
        "Start from the author’s model card and installation instructions.",
        "Choose a language pipeline and supplied voice; synthesize a short script.",
        "Inspect pronunciation and runtime speed on your own hardware."
      ],
      "example": "Try one 30-word tutorial introduction, then change the supplied voice while keeping text and runtime settings fixed.",
      "price": "No hosted vendor generation fee when run locally; hardware and operation still cost money. Third-party API prices are separate. License applies to the published weights.",
      "sources": [
        "kokoro",
        "kokoro-samples"
      ],
      "demo": "https://huggingface.co/spaces/hexgrad/Kokoro-TTS"
    },
    {
      "id": "hume",
      "name": "Hume Octave",
      "kind": "Retiring service / historical reference",
      "status": "retiring",
      "uses": [],
      "summary": "An instructive example of voice design and acting controls, but not a recommended dependency for a new project: Hume has announced a voice API sunset.",
      "features": [
        {
          "name": "Status",
          "value": "TTS and EVI access ends November 13, 2026, 12:01 a.m. EST"
        },
        {
          "name": "Data",
          "value": "The notice says account data is deleted after November 13"
        },
        {
          "name": "Acting direction",
          "value": "Description field: Octave 1; Octave 2 support listed as coming soon"
        },
        {
          "name": "Other controls",
          "value": "Speed and trailing silence supported across the documented models"
        },
        {
          "name": "Learning value",
          "value": "Distinguishes voice identity from delivery instructions"
        },
        {
          "name": "New-project fit",
          "value": "Excluded from our recommended starting points"
        }
      ],
      "pros": [
        "The archived concepts help explain identity, intention and timing.",
        "Version-specific docs expose why a control should not be generalized across models."
      ],
      "cons": [
        "Announced access deadline makes a new production integration unsuitable.",
        "Export needed material and confirm the notice with Hume before the closure.",
        "Old articles and demos can outlive the product they describe."
      ],
      "workflow": [
        "Read the retirement notice before taking any action.",
        "If you are an existing customer, export needed audio and review your migration requirements.",
        "Audition alternatives with your own scripts, not a vendor leaderboard alone."
      ],
      "example": "A performance brief can specify a calm voice, a measured pace and silence after an utterance; the supported fields vary by Octave version.",
      "price": "Historical reference only. Do not budget a new production project around an API with an announced closure.",
      "sources": [
        "hume",
        "hume-controls"
      ],
      "demo": "https://dev.hume.ai/voice/docs/text-to-speech-tts/overview"
    }
  ],
  "rates": [
    {
      "id": "eleven-v4",
      "name": "ElevenLabs v4 · regular listed API rate",
      "rate": 80,
      "unit": "million-characters",
      "source": "eleven-price",
      "note": "$0.08/1K characters. Excludes the temporary October 12 promotion and plan allowances."
    },
    {
      "id": "chirp",
      "name": "Google Chirp 3 HD · listed character rate",
      "rate": 30,
      "unit": "million-characters",
      "source": "google-price",
      "note": "$30/million characters after free usage. No allowance is deducted here."
    },
    {
      "id": "openai",
      "name": "OpenAI mini TTS · estimated output-minute rate",
      "rate": 0.015,
      "unit": "estimated-minute",
      "source": "openai-price",
      "note": "Published $0.015/minute estimate. Actual billing uses text and audio tokens; duration here is modeled from your words-per-minute assumption."
    }
  ]
}
