{"url":"/task/code-search","name":"Code Search","slug":"code-search","description_markdown":"The goal of **Code Search** is to retrieve code fragments from a large code corpus that most closely match a developer’s intent, which is expressed in natural language.\r\n\r\n\r\n<span class=\"description-source\">Source: [When Deep Learning Met Code Search ](https://arxiv.org/abs/1905.03813)</span>","categories":[{"name":"Computer Code","url":"/area/computer-code"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":125,"papers_with_code":60,"benchmarks":7,"benchmark_tables_in_archive":7,"benchmark_tables_shown":7,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":14,"subtasks":1,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/code-search-on-codesearchnet","slug":"code-search-on-codesearchnet","dataset":"CodeSearchNet","dataset_url":"/dataset/codesearchnet","rows_in_archive":6,"metrics":["Overall","Go","Ruby","Python","Java","JS","PHP"],"first_row_in_archive_order":{"model":"cpt-code M","paper_title":"Text and Code Embeddings by Contrastive Pre-Training","paper_url":"/paper/text-and-code-embeddings-by-contrastive-pre","paper_date":"2022-01-24","arxiv_id":"2201.10005","code_links":[{"title":"openmatch/coco-dr","url":"https://github.com/openmatch/coco-dr"}],"syntology":null}},{"leaderboard":"/sota/code-search-on-codesc","slug":"code-search-on-codesc","dataset":"CoDesc","dataset_url":"/dataset/codesc","rows_in_archive":3,"metrics":["Test MRR"],"first_row_in_archive_order":{"model":"Self-attention","paper_title":"CoDesc: A Large Code-Description Parallel Dataset","paper_url":"/paper/codesc-a-large-code-description-parallel","paper_date":"2021-05-29","arxiv_id":"2105.14220","code_links":[{"title":"csebuetnlp/CoDesc","url":"https://github.com/csebuetnlp/CoDesc"}],"syntology":{"n":8,"n_ran":1,"n_unverified":7,"n_pointer_only":0}}},{"leaderboard":"/sota/code-search-on-codexglue-advtest","slug":"code-search-on-codexglue-advtest","dataset":"CodeXGLUE - AdvTest","dataset_url":"/dataset/codexglue","rows_in_archive":3,"metrics":["MRR"],"first_row_in_archive_order":{"model":"CodeT5+ 770M","paper_title":"CodeT5+: Open Code Large Language Models for Code Understanding and Generation","paper_url":"/paper/codet5-open-code-large-language-models-for","paper_date":"2023-05-13","arxiv_id":"2305.07922","code_links":[{"title":"salesforce/codet5","url":"https://github.com/salesforce/codet5"},{"title":"leiluk1/codesearcher","url":"https://github.com/leiluk1/codesearcher"}],"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":0}}},{"leaderboard":"/sota/code-search-on","slug":"code-search-on","dataset":null,"dataset_url":null,"rows_in_archive":1,"metrics":["nDCG@10"],"first_row_in_archive_order":{"model":"Voyage-code-002","paper_title":"CoIR: A Comprehensive Benchmark for Code Information Retrieval Models","paper_url":"/paper/coir-a-comprehensive-benchmark-for-code","paper_date":"2024-07-03","arxiv_id":"2407.02883","code_links":[{"title":"coir-team/coir","url":"https://github.com/coir-team/coir"}],"syntology":{"n":5,"n_ran":2,"n_unverified":3,"n_pointer_only":0}}},{"leaderboard":"/sota/code-search-on-codesearchnet-ruby","slug":"code-search-on-codesearchnet-ruby","dataset":"CodeSearchNet - Ruby","dataset_url":"/dataset/codesearchnet","rows_in_archive":1,"metrics":["MRR"],"first_row_in_archive_order":{"model":"Uni-SBT","paper_title":"Multimodal Representation for Neural Code Search","paper_url":"/paper/multimodal-representation-for-neural-code","paper_date":"2021-07-02","arxiv_id":"2107.00992","code_links":[{"title":"jianguda/mrncs","url":"https://github.com/jianguda/mrncs"}],"syntology":null}},{"leaderboard":"/sota/code-search-on-codexglue-webquerytest","slug":"code-search-on-codexglue-webquerytest","dataset":"CodeXGLUE - WebQueryTest","dataset_url":"/dataset/codexglue","rows_in_archive":1,"metrics":["Accuracy","F1"],"first_row_in_archive_order":{"model":"CodeBERT","paper_title":"CodeXGLUE: A Machine Learning Benchmark Dataset for Code Understanding and Generation","paper_url":"/paper/codexglue-a-machine-learning-benchmark","paper_date":"2021-02-09","arxiv_id":"2102.04664","code_links":[{"title":"microsoft/CodeXGLUE","url":"https://github.com/microsoft/CodeXGLUE"},{"title":"facebookresearch/CodeGen","url":"https://github.com/facebookresearch/CodeGen"},{"title":"sberbank-ai/fusion_brain_aij2021","url":"https://github.com/sberbank-ai/fusion_brain_aij2021"},{"title":"kilimanj4r0/code-summarization-beyond-function-level","url":"https://github.com/kilimanj4r0/code-summarization-beyond-function-level"},{"title":"yueyuel/programgen-lms-reliability","url":"https://github.com/yueyuel/programgen-lms-reliability"},{"title":"Avmb/semantic_neq_game","url":"https://github.com/Avmb/semantic_neq_game"},{"title":"deeplearnxmu/unigencoder","url":"https://github.com/deeplearnxmu/unigencoder"}],"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}}},{"leaderboard":"/sota/code-search-on-coir","slug":"code-search-on-coir","dataset":"CoIR","dataset_url":"/dataset/coir","rows_in_archive":1,"metrics":["nDCG@10"],"first_row_in_archive_order":{"model":"Voyage-code-002","paper_title":"CoIR: A Comprehensive Benchmark for Code Information Retrieval Models","paper_url":"/paper/coir-a-comprehensive-benchmark-for-code","paper_date":"2024-07-03","arxiv_id":"2407.02883","code_links":[{"title":"coir-team/coir","url":"https://github.com/coir-team/coir"}],"syntology":{"n":5,"n_ran":2,"n_unverified":3,"n_pointer_only":0}}}],"datasets":[{"url":"/dataset/codesearchnet","name":"CodeSearchNet","full_name":"","num_papers_in_archive":308},{"url":"/dataset/codexglue","name":"CodeXGLUE","full_name":"","num_papers_in_archive":205},{"url":"/dataset/conala","name":"CoNaLa","full_name":"CMU CoNaLa, the Code/Natural Language Challenge","num_papers_in_archive":77},{"url":"/dataset/xlcost","name":"XLCoST","full_name":"Cross-Lingual Code Snippet","num_papers_in_archive":22},{"url":"/dataset/staqc","name":"StaQC","full_name":"","num_papers_in_archive":19},{"url":"/dataset/coir","name":"CoIR","full_name":"Code Information Retrieval Benchmark","num_papers_in_archive":18},{"url":"/dataset/pytorrent","name":"PyTorrent","full_name":"PyTorrent","num_papers_in_archive":5},{"url":"/dataset/codesc","name":"CoDesc","full_name":"","num_papers_in_archive":4},{"url":"/dataset/cosqa-1","name":"CoSQA+","full_name":"CoSQA_Plus","num_papers_in_archive":1},{"url":"/dataset/isadetect-dataset","name":"ISAdetect dataset","full_name":"ISAdetect binary file and object code dataset","num_papers_in_archive":1},{"url":"/dataset/res-q","name":"RES-Q","full_name":"RES-Q: Evaluating Code-Editing Large Language Model Systems at the Repository Scale","num_papers_in_archive":1},{"url":"/dataset/search4code","name":"Search4Code","full_name":null,"num_papers_in_archive":1},{"url":"/dataset/washed-contract","name":"washed_contract","full_name":"","num_papers_in_archive":1},{"url":"/dataset/codescan","name":"CodeSCAN","full_name":"ScreenCast ANalysis for Video Programming Tutorials","num_papers_in_archive":0}],"subtasks":[{"url":"/task/annotated-code-search","name":"Annotated Code Search"}],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":60,"tagged_in_all":125,"items":[{"url":"/paper/codesearchnet-challenge-evaluating-the-state","title":"CodeSearchNet Challenge: Evaluating the State of Semantic Code Search","date":"2019-09-20","arxiv_id":"1909.09436","repositories_listed":14,"syntology":{"n":17,"n_ran":8,"n_unverified":9,"n_pointer_only":0}},{"url":"/paper/codexglue-a-machine-learning-benchmark","title":"CodeXGLUE: A Machine Learning Benchmark Dataset for Code Understanding and Generation","date":"2021-02-09","arxiv_id":"2102.04664","repositories_listed":7,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/autocoderover-autonomous-program-improvement","title":"AutoCodeRover: Autonomous Program Improvement","date":"2024-04-08","arxiv_id":"2404.05427","repositories_listed":5,"syntology":{"n":4,"n_ran":0,"n_unverified":4,"n_pointer_only":1}},{"url":"/paper/melt-mining-effective-lightweight","title":"MELT: Mining Effective Lightweight Transformations from Pull Requests","date":"2023-08-28","arxiv_id":"2308.14687","repositories_listed":2,"syntology":null},{"url":"/paper/structure-aware-language-model-pretraining","title":"Structure-Aware Language Model Pretraining Improves Dense Retrieval on Structured Data","date":"2023-05-31","arxiv_id":"2305.19912","repositories_listed":2,"syntology":{"n":8,"n_ran":5,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/codet5-open-code-large-language-models-for","title":"CodeT5+: Open Code Large Language Models for Code Understanding and Generation","date":"2023-05-13","arxiv_id":"2305.07922","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/unixcoder-unified-cross-modal-pre-training","title":"UniXcoder: Unified Cross-Modal Pre-training for Code Representation","date":"2022-03-08","arxiv_id":"2203.03850","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_unverified":2,"n_pointer_only":0}},{"url":"/paper/memorization-and-generalization-in-neural","title":"Memorization and Generalization in Neural Code Intelligence Models","date":"2021-06-16","arxiv_id":"2106.08704","repositories_listed":2,"syntology":null},{"url":"/paper/dobf-a-deobfuscation-pre-training-objective","title":"DOBF: A Deobfuscation Pre-Training Objective for Programming Languages","date":"2021-02-15","arxiv_id":"2102.07492","repositories_listed":2,"syntology":null},{"url":"/paper/concra-a-convolutional-neural-network-code","title":"CoNCRA: A Convolutional Neural Network Code Retrieval Approach","date":"2020-09-03","arxiv_id":"2009.01959","repositories_listed":2,"syntology":null},{"url":"/paper/when-deep-learning-met-code-search","title":"When Deep Learning Met Code Search","date":"2019-05-09","arxiv_id":"1905.03813","repositories_listed":2,"syntology":null},{"url":"/paper/zero-shot-cross-domain-code-search-without","title":"Zero-Shot Cross-Domain Code Search without Fine-Tuning","date":"2025-04-10","arxiv_id":"2504.07740","repositories_listed":1,"syntology":null},{"url":"/paper/repository-level-code-search-with-neural","title":"Repository-level Code Search with Neural Retrieval Methods","date":"2025-02-10","arxiv_id":"2502.07067","repositories_listed":1,"syntology":null},{"url":"/paper/isotropy-matters-soft-zca-whitening-of","title":"Isotropy Matters: Soft-ZCA Whitening of Embeddings for Semantic Code Search","date":"2024-11-26","arxiv_id":"2411.17538","repositories_listed":1,"syntology":null},{"url":"/paper/codesam-source-code-representation-learning","title":"CodeSAM: Source Code Representation Learning by Infusing Self-Attention with Multi-Code-View Graphs","date":"2024-11-21","arxiv_id":"2411.14611","repositories_listed":1,"syntology":null},{"url":"/paper/vic-virtual-compiler-is-all-you-need-for","title":"ViC: Virtual Compiler Is All You Need For Assembly Code Search","date":"2024-08-10","arxiv_id":"2408.06385","repositories_listed":1,"syntology":null},{"url":"/paper/coir-a-comprehensive-benchmark-for-code","title":"CoIR: A Comprehensive Benchmark for Code Information Retrieval Models","date":"2024-07-03","arxiv_id":"2407.02883","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/cosqa-enhancing-code-search-dataset-with","title":"CoSQA+: Pioneering the Multi-Choice Code Search Benchmark with Test-Driven Agents","date":"2024-06-17","arxiv_id":"2406.11589","repositories_listed":1,"syntology":null},{"url":"/paper/repoqa-evaluating-long-context-code","title":"RepoQA: Evaluating Long Context Code Understanding","date":"2024-06-10","arxiv_id":"2406.06025","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/advanced-detection-of-source-code-clones-via","title":"Advanced Detection of Source Code Clones via an Ensemble of Unsupervised Similarity Measures","date":"2024-05-03","arxiv_id":"2405.02095","repositories_listed":1,"syntology":null},{"url":"/paper/procqa-a-large-scale-community-based","title":"ProCQA: A Large-scale Community-based Programming Question Answering Dataset for Code Search","date":"2024-03-25","arxiv_id":"2403.16702","repositories_listed":1,"syntology":null},{"url":"/paper/source-code-clone-detection-using","title":"Source Code Clone Detection Using Unsupervised Similarity Measures","date":"2024-01-18","arxiv_id":"2401.09885","repositories_listed":1,"syntology":null},{"url":"/paper/rewriting-the-code-a-simple-method-for-large","title":"Rewriting the Code: A Simple Method for Large Language Model Augmented Code Search","date":"2024-01-09","arxiv_id":"2401.04514","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_unverified":4,"n_pointer_only":9}},{"url":"/paper/gencodesearchnet-a-benchmark-test-suite-for","title":"GenCodeSearchNet: A Benchmark Test Suite for Evaluating Generalization in Programming Language Understanding","date":"2023-11-16","arxiv_id":"2311.09707","repositories_listed":1,"syntology":null},{"url":"/paper/transformcode-a-contrastive-learning","title":"TransformCode: A Contrastive Learning Framework for Code Embedding via Subtree Transformation","date":"2023-11-10","arxiv_id":"2311.08157","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-negative-pairs-in-code-search","title":"Rethinking Negative Pairs in Code Search","date":"2023-10-12","arxiv_id":"2310.08069","repositories_listed":1,"syntology":null},{"url":"/paper/language-models-are-universal-embedders","title":"Language Models are Universal Embedders","date":"2023-10-12","arxiv_id":"2310.08232","repositories_listed":1,"syntology":null},{"url":"/paper/constructing-multilingual-code-search-dataset","title":"Constructing Multilingual Code Search Dataset Using Neural Machine Translation","date":"2023-06-27","arxiv_id":"2306.15604","repositories_listed":1,"syntology":null},{"url":"/paper/backdooring-neural-code-search","title":"Backdooring Neural Code Search","date":"2023-05-27","arxiv_id":"2305.17506","repositories_listed":1,"syntology":{"n":14,"n_ran":2,"n_unverified":12,"n_pointer_only":0}},{"url":"/paper/the-vault-a-comprehensive-multilingual","title":"The Vault: A Comprehensive Multilingual Dataset for Advancing Code Understanding and Generation","date":"2023-05-09","arxiv_id":"2305.06156","repositories_listed":1,"syntology":null}],"syntology_records":10,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}