{"omc_version":"0.1","id":"microsoft-phi-4-multimodal-instruct","name":"Phi 4 Multimodal Instruct","provider":"Microsoft","released":"2025-02-24","licence":"MIT License","url":"https://huggingface.co/microsoft/Phi-4-multimodal-instruct","description":"Phi 4 Multimodal Instruct is a six-billion-parameter speech-to-text model from Microsoft with a permissive MIT licence. It turns audio into written text with excellent accuracy on clean recordings and very fast batch processing, though it struggles with meetings and accented speech.","modalities":["audio"],"context_window":null,"capabilities":{"transcription":3.5},"x_llmap_capability_sources":{"transcription":{"suite":"asr-wer","suite_name":"Open ASR WER","rank":29,"of":74,"score":5.424285714,"higher_is_better":false,"variant":null}},"relationships":[],"card_meta":{"created":"2026-08-01T22:45:48.116Z","created_by":"LLMap pipeline","last_updated":"2026-08-03T20:17:34.477Z","source":"https://llmap.ai/models/microsoft-phi-4-multimodal-instruct","omc_spec":"https://github.com/openmodelcard/spec"}}