{"items":[{"provider":"elevenlabs","providerDisplayName":"ElevenLabs","providerDisplayDescription":"AI audio platform — lifelike text-to-speech and multi-speaker dialogue in 70+ languages, speech-to-text transcription (90+ languages, diarization, timestamps), sound-effect and music generation, voice changing, background-noise removal, and forced alignment — with a free catalog of selectable voices.","endpoint":"/v1/audio-isolation","displayName":"Voice Isolator","displayDescription":"Remove background noise, music and ambient sounds from a recording, returning clean studio-quality speech (MP3, base64).","price":{"type":"PER_UNIT_MATRIX","amount":{"value":0.12,"currency":"USD"},"selectors":[{"label":"Rate","key":"rate","in":"body"}],"variants":[{"when":{"rate":"standard"},"price":{"type":"PER_UNIT","amount":{"value":0.12,"currency":"USD"},"per":60,"unit":"second"},"label":"Per minute of input audio"}]},"tags":["verified"],"categories":["audio","elevenlabs"]},{"provider":"elevenlabs","providerDisplayName":"ElevenLabs","providerDisplayDescription":"AI audio platform — lifelike text-to-speech and multi-speaker dialogue in 70+ languages, speech-to-text transcription (90+ languages, diarization, timestamps), sound-effect and music generation, voice changing, background-noise removal, and forced alignment — with a free catalog of selectable voices.","endpoint":"/v1/speech-to-speech","displayName":"Voice Changer","displayDescription":"Transform a recording into a different ElevenLabs voice while preserving the original emotion, timing and delivery (MP3, base64 output).","price":{"type":"PER_UNIT_MATRIX","amount":{"value":0.12,"currency":"USD"},"selectors":[{"label":"Model","key":"model_id","in":"body"}],"variants":[{"when":{"model_id":"eleven_multilingual_sts_v2"},"price":{"type":"PER_UNIT","amount":{"value":0.12,"currency":"USD"},"per":60,"unit":"second"},"label":"Multilingual STS v2 — per minute of input audio"},{"when":{"model_id":"eleven_english_sts_v2"},"price":{"type":"PER_UNIT","amount":{"value":0.12,"currency":"USD"},"per":60,"unit":"second"},"label":"English STS v2 — per minute of input audio"}]},"tags":["verified"],"categories":["elevenlabs","speech"]},{"provider":"elevenlabs","providerDisplayName":"ElevenLabs","providerDisplayDescription":"AI audio platform — lifelike text-to-speech and multi-speaker dialogue in 70+ languages, speech-to-text transcription (90+ languages, diarization, timestamps), sound-effect and music generation, voice changing, background-noise removal, and forced alignment — with a free catalog of selectable voices.","endpoint":"/v1/forced-alignment","displayName":"Forced Alignment","displayDescription":"Align a known transcript to its audio recording, returning precise word-level timestamps (subtitles, audiobook timings, karaoke).","price":{"type":"PER_UNIT_MATRIX","amount":{"value":0.22,"currency":"USD"},"selectors":[{"label":"Rate","key":"rate","in":"body"}],"variants":[{"when":{"rate":"standard"},"price":{"type":"PER_UNIT","amount":{"value":0.22,"currency":"USD"},"per":3600,"unit":"second"},"label":"Per hour of input audio"}]},"tags":["verified"],"categories":["elevenlabs","speech"]},{"provider":"elevenlabs","providerDisplayName":"ElevenLabs","providerDisplayDescription":"AI audio platform — lifelike text-to-speech and multi-speaker dialogue in 70+ languages, speech-to-text transcription (90+ languages, diarization, timestamps), sound-effect and music generation, voice changing, background-noise removal, and forced alignment — with a free catalog of selectable voices.","endpoint":"/voices","displayName":"List Voices","displayDescription":"List the ElevenLabs voices available to this account (voice_id, name, category, labels, preview_url).","price":{"type":"PER_CALL","amount":{"value":0,"currency":"USD"}},"tags":["verified"],"categories":["elevenlabs","speech"]},{"provider":"elevenlabs","providerDisplayName":"ElevenLabs","providerDisplayDescription":"AI audio platform — lifelike text-to-speech and multi-speaker dialogue in 70+ languages, speech-to-text transcription (90+ languages, diarization, timestamps), sound-effect and music generation, voice changing, background-noise removal, and forced alignment — with a free catalog of selectable voices.","endpoint":"/v1/speech-to-text","displayName":"Speech to Text","displayDescription":"Transcribe an audio or video file from a URL with ElevenLabs Scribe — 90+ languages, word timestamps, speaker diarization, audio-event tagging, optional entity detection and keyterm biasing.","price":{"type":"PER_UNIT_MATRIX","amount":{"value":0.22,"currency":"USD"},"selectors":[{"label":"Entity detection","key":"entity_detection","in":"body"},{"label":"Keyterm prompting","key":"keyterms_enabled","in":"body"}],"variants":[{"when":{"entity_detection":"false","keyterms_enabled":"false"},"price":{"type":"PER_UNIT","amount":{"value":0.22,"currency":"USD"},"per":3600,"unit":"second"},"label":"Base — per hour of audio"},{"when":{"entity_detection":"false","keyterms_enabled":"true"},"price":{"type":"PER_UNIT","amount":{"value":0.27,"currency":"USD"},"per":3600,"unit":"second"},"label":"With keyterm prompting — per hour of audio"},{"when":{"entity_detection":"true","keyterms_enabled":"false"},"price":{"type":"PER_UNIT","amount":{"value":0.29,"currency":"USD"},"per":3600,"unit":"second"},"label":"With entity detection — per hour of audio"},{"when":{"entity_detection":"true","keyterms_enabled":"true"},"price":{"type":"PER_UNIT","amount":{"value":0.34,"currency":"USD"},"per":3600,"unit":"second"},"label":"With entity detection + keyterms — per hour of audio"}]},"tags":["verified"],"categories":["elevenlabs","speech"]},{"provider":"elevenlabs","providerDisplayName":"ElevenLabs","providerDisplayDescription":"AI audio platform — lifelike text-to-speech and multi-speaker dialogue in 70+ languages, speech-to-text transcription (90+ languages, diarization, timestamps), sound-effect and music generation, voice changing, background-noise removal, and forced alignment — with a free catalog of selectable voices.","endpoint":"/v1/music","displayName":"Music Generation","displayDescription":"Generate studio-grade music (MP3) in any style from a natural language prompt with Eleven Music — vocals or instrumental, commercial-use cleared.","price":{"type":"PER_UNIT_MATRIX","amount":{"value":0.15,"currency":"USD"},"selectors":[{"label":"Model","key":"model_id","in":"body"}],"variants":[{"when":{"model_id":"music_v1"},"price":{"type":"PER_UNIT","amount":{"value":0.15,"currency":"USD"},"per":60,"unit":"second"},"label":"Music v1 — per minute of generated audio"},{"when":{"model_id":"music_v2"},"price":{"type":"PER_UNIT","amount":{"value":0.15,"currency":"USD"},"per":60,"unit":"second"},"label":"Music v2 — per minute of generated audio"}]},"tags":["verified"],"categories":["elevenlabs","music-generation"]},{"provider":"elevenlabs","providerDisplayName":"ElevenLabs","providerDisplayDescription":"AI audio platform — lifelike text-to-speech and multi-speaker dialogue in 70+ languages, speech-to-text transcription (90+ languages, diarization, timestamps), sound-effect and music generation, voice changing, background-noise removal, and forced alignment — with a free catalog of selectable voices.","endpoint":"/v1/text-to-dialogue","displayName":"Text to Dialogue","displayDescription":"Generate immersive multi-speaker dialogue audio (MP3) from text + voice_id pairs with ElevenLabs eleven_v3.","price":{"type":"PER_UNIT_MATRIX","amount":{"value":0.1,"currency":"USD"},"selectors":[{"label":"Model","key":"model_id","in":"body"}],"variants":[{"when":{"model_id":"eleven_v3"},"price":{"type":"PER_UNIT","amount":{"value":0.1,"currency":"USD"},"per":1000,"unit":"character"},"label":"Eleven v3 — per 1,000 characters"}]},"tags":["verified"],"categories":["elevenlabs","speech"]},{"provider":"elevenlabs","providerDisplayName":"ElevenLabs","providerDisplayDescription":"AI audio platform — lifelike text-to-speech and multi-speaker dialogue in 70+ languages, speech-to-text transcription (90+ languages, diarization, timestamps), sound-effect and music generation, voice changing, background-noise removal, and forced alignment — with a free catalog of selectable voices.","endpoint":"/text-to-speech","displayName":"Text to Speech","displayDescription":"Convert text to lifelike speech audio (MP3) with an ElevenLabs voice.","price":{"type":"PER_UNIT_MATRIX","amount":{"value":0.05,"currency":"USD"},"selectors":[{"label":"Model","key":"model_id","in":"body"}],"variants":[{"when":{"model_id":"eleven_multilingual_v2"},"price":{"type":"PER_UNIT","amount":{"value":0.1,"currency":"USD"},"per":1000,"unit":"character"},"label":"Multilingual v2 — per 1,000 characters"},{"when":{"model_id":"eleven_flash_v2_5"},"price":{"type":"PER_UNIT","amount":{"value":0.05,"currency":"USD"},"per":1000,"unit":"character"},"label":"Flash v2.5 — per 1,000 characters"},{"when":{"model_id":"eleven_v3"},"price":{"type":"PER_UNIT","amount":{"value":0.1,"currency":"USD"},"per":1000,"unit":"character"},"label":"Eleven v3 — per 1,000 characters"}]},"tags":["verified"],"categories":["elevenlabs","speech"]},{"provider":"elevenlabs","providerDisplayName":"ElevenLabs","providerDisplayDescription":"AI audio platform — lifelike text-to-speech and multi-speaker dialogue in 70+ languages, speech-to-text transcription (90+ languages, diarization, timestamps), sound-effect and music generation, voice changing, background-noise removal, and forced alignment — with a free catalog of selectable voices.","endpoint":"/v1/sound-generation","displayName":"Sound Effects","displayDescription":"Generate a high-quality sound effect (MP3) from a text description — cinematic hits, Foley, ambience, loops, musical elements.","price":{"type":"PER_UNIT_MATRIX","amount":{"value":0.12,"currency":"USD"},"selectors":[{"label":"Model","key":"model_id","in":"body"}],"variants":[{"when":{"model_id":"eleven_text_to_sound_v2"},"price":{"type":"PER_UNIT","amount":{"value":0.12,"currency":"USD"},"per":60,"unit":"second"},"label":"Text to Sound v2 — per minute of generated audio"}]},"tags":["verified"],"categories":["audio","elevenlabs"]}],"total":9}