{"rewrite":{"id":"r_9b28859ddc0465e3b7a8365f","clusterId":"c_c0714fada39f75c88c9d465a","slug":"meta-releases-muse-voice-transcribe-its-first-real-time-speech-recognition-model","model":"deepseek-v4-flash","headline":"Meta Releases Muse Voice Transcribe, Its First Real-Time Speech Recognition Model","summary":"Meta Superintelligence Labs released Muse Voice Transcribe, its first real-time speech recognition model. It handles streaming recognition, identifies 20 or more speakers, and detects the end of speech. It ranks first on Artificial Analysis with a final transcription error rate of 3.1 percent. Pricing runs $0.18 per hour or $3 per 1000 minutes. Trained on more than 70 languages, it is available from September 2 via Meta Model API, Meta AI for Mac, and Muse Code.","whyItMatters":"Meta's first real-time speech recognition model tops Artificial Analysis rankings and is priced below rival transcription models, giving the Muse Spark family a direct entry into the streaming voice segment.","webCardHtml":"\u003cp\u003eMuse Voice Transcribe is an autoregressive multimodal model in the Muse Spark family, Meta\u0026#39;s line of reasoning models. Its speaker identification score on Artificial Analysis also shows a lower misrecognition rate than other models, and it natively supports code-switching, where bilingual speakers alternate languages within or across sentences.\u003c/p\u003e\u003cp\u003eThe model handles audio longer than one hour and conversations with more than 20 participants without any post-recording speaker labeling or level adjustment. Twenty-five languages are extensively validated, including Japanese, English, Mandarin Chinese, Korean, and Hindi.\u003c/p\u003e","blueskyPost":"Meta released Muse Voice Transcribe, its first real-time speech recognition model. It handles 20+ speakers, detects speech end, and ranks first on Artificial Analysis. Priced at $0.18/hour. Live now via Meta Model API, Meta AI for Mac, and Muse Code.","twitterPost":"Meta's Muse Voice Transcribe is out: real-time speech recognition, 20+ speaker identification, end-of-utterance detection. #1 on Artificial Analysis, 3.1% final error rate, $0.18/hour. Live now via Meta Model API, Meta AI for Mac, Muse Code.","threadsPost":null,"newsletterBlurb":"Meta Superintelligence Labs released Muse Voice Transcribe, its first real-time speech recognition model and a member of the Muse Spark family. It ranks first on Artificial Analysis with a 3.1 percent final transcription error rate, supports 20 or more speakers, and costs $0.18 per hour. The model is live from September 2 via Meta Model API, Meta AI for Mac, and Muse Code.","attributionJson":"[{\"source\":\"GIGAZINE\",\"url\":\"https://gigazine.net/news/20260902-meta-muse-voice-transcribe/\",\"title\":\"Metaがリアルタイム文字起こしAI「Muse Voice Transcribe」をリリース\"}]","lintFlagsJson":null,"lintHits":0,"costUsd":0,"inputTokens":5201,"outputTokens":4281,"status":"published","repairAttempts":0,"nextRepairAt":null,"factsAttemptedAt":1788358818,"createdAt":"2026-09-02T14:15:06.000Z","publishedAt":"2026-09-02T14:17:24.000Z","updatedAt":"2026-09-02T14:17:24.000Z"},"cluster":{"id":"c_c0714fada39f75c88c9d465a","canonicalTitle":"Metaがリアルタイム文字起こしAI「Muse Voice Transcribe」をリリース","representativeArticleId":"a_19bad1da4be88420410990cf","sourceCount":1,"writtenSourceCount":1,"writeAttempts":0,"isSolo":true,"entitiesJson":"{\"anime_titles\":[],\"manga_titles\":[],\"work_titles\":[],\"studios\":[],\"people\":[],\"type\":\"news\",\"domain\":\"other\",\"is_roundup\":false}","contentType":"news","status":"published","firstSeenAt":"2026-09-02T08:30:00.000Z","lastSeenAt":"2026-09-02T08:30:00.000Z","updatedAt":"2026-09-02T14:17:24.000Z"},"attribution":[{"source":"GIGAZINE","url":"https://gigazine.net/news/20260902-meta-muse-voice-transcribe/","title":"Metaがリアルタイム文字起こしAI「Muse Voice Transcribe」をリリース"}],"entities":{"anime_titles":[],"manga_titles":[],"work_titles":[],"studios":[],"people":[],"type":"news","domain":"other","is_roundup":false},"keyFacts":null}
