{"omc_version":"0.1","id":"espnet-owsm-ctc-v3-2-ft-1b","name":"Owsm CTC v3.2 ft 1B","provider":"ESPnet","released":"2024-09-24","licence":"Creative Commons Attribution 4.0","url":"https://huggingface.co/espnet/owsm_ctc_v3.2_ft_1B","description":"Owsm CTC v3.2 ft 1B is a one-billion-parameter speech-to-text model from ESPnet that turns audio into written text. It is extremely fast and permissively licensed, though its accuracy drops sharply on difficult audio such as meetings and accented speech.","modalities":["audio"],"context_window":null,"capabilities":{"transcription":1.5},"x_llmap_capability_sources":{"transcription":{"suite":"asr-wer","suite_name":"Open ASR WER","rank":63,"of":74,"score":7.25,"higher_is_better":false,"variant":null}},"relationships":[],"card_meta":{"created":"2026-08-01T22:45:45.898Z","created_by":"LLMap pipeline","last_updated":"2026-08-03T20:17:31.137Z","source":"https://llmap.ai/models/espnet-owsm-ctc-v3-2-ft-1b","omc_spec":"https://github.com/openmodelcard/spec"}}