{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/transcribe","entry":"transcribe","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":8,"n_papers_ran":1,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":8,"n_samples_ran":1,"n_samples_fingerprinted":0,"n_places":8,"n_places_pointer_only":2,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":1,"unverified":7},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2604.13073","paper":"/paper/arxiv-2604-13073","title":"OmniTrace: A Unified Framework for Generation-Time Attribution in Omni-Modal LLMs","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"openai/whisper","path":"whisper/transcribe.py","file_url":"https://github.com/openai/whisper/blob/HEAD/whisper/transcribe.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"700202ff899b0c2d","mcp_get_code":{"code_sha256":"700202ff899b0c2d"}},{"arxiv_id":"2601.18415","paper":"/paper/arxiv-2601-18415","title":"Pisets: A Robust Speech Recognition System for Lectures and Interviews","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"bond005/pisets","path":"asr/asr.py","file_url":"https://github.com/bond005/pisets/blob/HEAD/asr/asr.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8e9edbe2537edcae","mcp_get_code":{"code_sha256":"8e9edbe2537edcae"}},{"arxiv_id":"2601.03973","paper":"/paper/arxiv-2601-03973","title":"Muse: Towards Reproducible Long-Form Song Generation with Fine-Grained Style Control","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"yuhui1038/Muse","path":"eval_pipeline/fill_missing.py","file_url":"https://github.com/yuhui1038/Muse/blob/HEAD/eval_pipeline/fill_missing.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"58e1684834942c02","mcp_get_code":{"code_sha256":"58e1684834942c02"}},{"arxiv_id":"2409.07556","paper":"/paper/ssr-speech-towards-stable-safe-and-robust","title":"SSR-Speech: Towards Stable, Safe and Robust Zero-shot Text-based Speech Editing and Synthesis","date":"2024-09-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"WangHelin1997/SSR-Speech","path":"inference_v2.py","file_url":"https://github.com/WangHelin1997/SSR-Speech/blob/HEAD/inference_v2.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"dc3571b4c872f039","mcp_get_code":{"code_sha256":"dc3571b4c872f039"}},{"arxiv_id":"2406.10082","paper":"/paper/whisper-flamingo-integrating-visual-features","title":"Whisper-Flamingo: Integrating Visual Features into Whisper for Audio-Visual Speech Recognition and Translation","date":"2024-06-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"roudimit/whisper-flamingo","path":"whisper/transcribe.py","file_url":"https://github.com/roudimit/whisper-flamingo/blob/HEAD/whisper/transcribe.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"7bebc0cb7f2ab5b0","mcp_get_code":{"code_sha256":"7bebc0cb7f2ab5b0"}},{"arxiv_id":"2309.15701","paper":"/paper/hyporadise-an-open-baseline-for-generative-1","title":"HyPoradise: An Open Baseline for Generative Speech Recognition with Large Language Models","date":"2023-09-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hypotheses-paradise/hypo2trans","path":"generate_data/whisper/whisper/transcribe.py","file_url":"https://github.com/hypotheses-paradise/hypo2trans/blob/HEAD/generate_data/whisper/whisper/transcribe.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"51e0aca11651c4d4","mcp_get_code":{"code_sha256":"51e0aca11651c4d4"}},{"arxiv_id":"2212.04356","paper":"/paper/robust-speech-recognition-via-large-scale-1","title":"Robust Speech Recognition via Large-Scale Weak Supervision","date":"2022-12-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"briansidp/whisperbiasing","path":"whisper/transcribe.py","file_url":"https://github.com/briansidp/whisperbiasing/blob/HEAD/whisper/transcribe.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"738c6e74fd16381d","mcp_get_code":{"code_sha256":"738c6e74fd16381d"}},{"arxiv_id":"2210.13352","paper":"/paper/esb-a-benchmark-for-multi-domain-end-to-end","title":"ESB: A Benchmark For Multi-Domain End-to-End Speech Recognition","date":"2022-10-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"huggingface/open_asr_leaderboard","path":"nemo_asr/run_eval_salm.py","file_url":"https://github.com/huggingface/open_asr_leaderboard/blob/HEAD/nemo_asr/run_eval_salm.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"465981d7412a5bb3","mcp_get_code":{"code_sha256":"465981d7412a5bb3"}}]}