{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/lhotse-a-speech-data-representation-library","title":"Lhotse: a speech data representation library for the modern deep learning ecosystem","arxiv_id":"2110.12561","date":"2021-10-25","proceeding":null,"authors":["Piotr Żelasko","Daniel Povey","Jan \"Yenda\" Trmal","Sanjeev Khudanpur"],"abstract":"Speech data is notoriously difficult to work with due to a variety of codecs, lengths of recordings, and meta-data formats. We present Lhotse, a speech data representation library that draws upon lessons learned from Kaldi speech recognition toolkit and brings its concepts into the modern deep learning ecosystem. Lhotse provides a common JSON description format with corresponding Python classes and data preparation recipes for over 30 popular speech corpora. Various datasets can be easily combined together and re-purposed for different tasks. The library handles multi-channel recordings, long recordings, local and cloud storage, lazy and on-the-fly operations amongst other features. We introduce Cut and CutSet concepts, which simplify common data wrangling tasks for audio and help incorporate acoustic context of speech utterances. Finally, we show how Lhotse leverages PyTorch data API abstractions and adopts them to handle speech data for deep learning.","url_abs":"https://arxiv.org/abs/2110.12561v1","url_pdf":"https://arxiv.org/pdf/2110.12561v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"links_only","authors_date_abstract":"arXiv metadata, CC0 1.0 (https://info.arxiv.org/help/license), from the Kaggle arXiv metadata snapshot of 2026-09-12"},"code_links":[{"paper_slug":"lhotse-a-speech-data-representation-library","repo_url":"https://github.com/lhotse-speech/lhotse","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"pytorch","reach":{"status":"ok","spdx":"Apache-2.0"}}],"tasks":[],"methods":[],"datasets_introduced":[],"methods_introduced":[],"results":[],"syntology":{"atlas_url":"https://app.syntology.ai/?focus=2110.12561","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2110.12561"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-24T18:15:14+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"deterministic:regex_extraction","url":"https://github.com/lhotse-speech/lhotse","reach":{"status":"ok","spdx":"Apache-2.0"}}],"summary":{"unverified":6},"by_repo_kind":{"official":{"samples":6,"ran":0,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"65f2858d8f27cb3b","entry":"attach_graph_origin","repo":"lhotse-speech/lhotse","repo_kind":"official","path":"lhotse/lazy.py","file_url":"https://github.com/lhotse-speech/lhotse/blob/HEAD/lhotse/lazy.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"65f2858d8f27cb3b"}},{"code_sha256_prefix":"23e1a7fb7a027001","entry":"dynamic_lru_cache","repo":"lhotse-speech/lhotse","repo_kind":"official","path":"lhotse/caching.py","file_url":"https://github.com/lhotse-speech/lhotse/blob/HEAD/lhotse/caching.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"23e1a7fb7a027001"}},{"code_sha256_prefix":"dec717886f21abcb","entry":"floor_duration_to_milliseconds","repo":"lhotse-speech/lhotse","repo_kind":"official","path":"lhotse/kaldi.py","file_url":"https://github.com/lhotse-speech/lhotse/blob/HEAD/lhotse/kaldi.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"dec717886f21abcb"}},{"code_sha256_prefix":"013578fd2d443769","entry":"index_pack_collection_key","repo":"lhotse-speech/lhotse","repo_kind":"official","path":"lhotse/index_pack.py","file_url":"https://github.com/lhotse-speech/lhotse/blob/HEAD/lhotse/index_pack.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"013578fd2d443769"}},{"code_sha256_prefix":"692503911cd198f5","entry":"normalize_graph_token","repo":"lhotse-speech/lhotse","repo_kind":"official","path":"lhotse/lazy.py","file_url":"https://github.com/lhotse-speech/lhotse/blob/HEAD/lhotse/lazy.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"692503911cd198f5"}},{"code_sha256_prefix":"a7b7db3accd82f35","entry":"resolve_iterator_source","repo":"lhotse-speech/lhotse","repo_kind":"official","path":"lhotse/lazy.py","file_url":"https://github.com/lhotse-speech/lhotse/blob/HEAD/lhotse/lazy.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"mcp_get_code":{"code_sha256":"a7b7db3accd82f35"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}