{"url":"/dataset/pytorrent","name":"PyTorrent","full_name":"PyTorrent","description_markdown":"PyTorrent  contains 218,814 Python package libraries from PyPI and Anaconda environment. This is because earlier studies have shown that much of the code is redundant and Python packages from these environments are better in quality and are well-documented. PyTorrent enables users (such as data scientists, students, etc.) to build off the shelf machine learning models directly without spending months of effort on large infrastructure.","description_withheld":null,"homepage":"https://github.com/fla-sil/PyTorrent","introduced_date":"2021-10-04","introduced_date_note":null,"introduced_by":{"paper":"/paper/pytorrent-a-python-library-corpus-for-large","title":"PyTorrent: A Python Library Corpus for Large-scale Language Models","first_author":null,"url":null},"license":null,"modalities":[{"name":"Texts","url":"/datasets/modality/texts"}],"tasks":[{"name":"Code Generation","url":"/task/code-generation","datasets_with_task":"/datasets/task/code-generation"},{"name":"Code Search","url":"/task/code-search","datasets_with_task":"/datasets/task/code-search"},{"name":"Code Documentation Generation","url":"/task/code-documentation-generation","datasets_with_task":"/datasets/task/code-documentation-generation"},{"name":"Code Completion","url":"/task/code-completion","datasets_with_task":"/datasets/task/code-completion"},{"name":"Annotated Code Search","url":"/task/annotated-code-search","datasets_with_task":"/datasets/task/annotated-code-search"},{"name":"Code Translation","url":"/task/code-translation","datasets_with_task":"/datasets/task/code-translation"},{"name":"Code Comment Generation","url":"/task/code-comment-generation","datasets_with_task":"/datasets/task/code-comment-generation"},{"name":"Text-to-Code Generation","url":"/task/text-to-code-generation","datasets_with_task":"/datasets/task/text-to-code-generation"},{"name":"Code Repair","url":"/task/code-repair","datasets_with_task":"/datasets/task/code-repair"},{"name":"Code Summarization","url":"/task/code-summarization-1","datasets_with_task":"/datasets/task/code-summarization-1"},{"name":"Contextual Embedding for Source Code","url":"/task/contextual-embedding-for-source-code","datasets_with_task":"/datasets/task/contextual-embedding-for-source-code"},{"name":"CodeSearchNet - Java","url":"/task/codesearchnet-java","datasets_with_task":"/datasets/task/codesearchnet-java"},{"name":"Code Classification","url":"/task/code-classification","datasets_with_task":"/datasets/task/code-classification"}],"languages":[{"name":"English","url":"/datasets/language/english"}],"variants":["PyTorrent"],"data_loaders":[],"num_papers_in_archive":5,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28"},"benchmarks":[],"papers_with_a_benchmark_row":[],"syntology_totals":{"read_at":"2026-09-24T18:15:14+00:00","papers_with_samples":0,"samples_harvested":0,"samples_ran":0,"samples_unverified":0,"pointer_only_for_licence":0,"papers_with_no_sample_that_ran":0,"note":"the per-paper counts above, summed; not a rate"},"papers_note":"The archive never published its papers-using-dataset list; these are papers with a leaderboard row on this dataset's benchmarks."}