{"omc_version":"0.1","id":"espnet-owsm-ctc-v4-1b","name":"Owsm CTC v4 1B","provider":"ESPnet","released":"2025-01-16","licence":"Creative Commons Attribution 4.0","url":"https://huggingface.co/espnet/owsm_ctc_v4_1B","description":"Owsm CTC v4 is a one-billion-parameter speech-to-text model from the ESPnet project that turns audio into written words across 75 languages. It is extremely fast on benchmark hardware and accurate on clean recordings, though its error rate rises sharply on accented speech and meetings.","modalities":["audio"],"context_window":null,"capabilities":{"transcription":2},"x_llmap_capability_sources":{"transcription":{"suite":"asr-wer","suite_name":"Open ASR WER","rank":57,"of":74,"score":6.662857143,"higher_is_better":false,"variant":null}},"relationships":[{"type":"successor-of","id":"espnet-owsm-ctc-v3-1-1b"},{"type":"same-family","id":"espnet-owsm-ctc-v3-1-1b"}],"card_meta":{"created":"2026-08-01T22:45:46.036Z","created_by":"LLMap pipeline","last_updated":"2026-08-03T20:17:31.340Z","source":"https://llmap.ai/models/espnet-owsm-ctc-v4-1b","omc_spec":"https://github.com/openmodelcard/spec"}}