{"omc_version":"0.1","id":"espnet-owsm-ctc-v4-1b","name":"Owsm CTC v4 1B","provider":"ESPnet","released":"2025-01-16","licence":"Creative Commons Attribution 4.0","url":"https://huggingface.co/espnet/owsm_ctc_v4_1B","description":"Owsm CTC v4 is a one-billion-parameter speech-to-text model released in early 2025. It turns audio into written words extremely fast on benchmark hardware, though its accuracy falls sharply on anything less than clean, read-aloud recordings.","modalities":["audio"],"context_window":null,"capabilities":{"transcription":2},"x_llmap_capability_sources":{"transcription":{"suite":"asr-wer","suite_name":"Open ASR WER","rank":57,"of":74,"score":6.662857143,"higher_is_better":false,"variant":null}},"relationships":[{"type":"successor-of","id":"espnet-owsm-ctc-v3-1-1b"},{"type":"same-family","id":"espnet-owsm-ctc-v3-1-1b"}],"card_meta":{"created":"2026-08-01T22:45:46.036Z","created_by":"LLMap pipeline","last_updated":"2026-08-04T21:30:31.337Z","source":"https://llmap.ai/models/espnet-owsm-ctc-v4-1b","omc_spec":"https://github.com/openmodelcard/spec"}}