{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/jaccard-similarity","entry":"jaccard_similarity","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":19,"n_papers_ran":12,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":17,"n_samples_ran":11,"n_samples_fingerprinted":7,"n_places":19,"n_places_pointer_only":2,"by_status":{"ran_honours":2,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":9,"unverified":6},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2608.16643","paper":"/paper/arxiv-2608-16643","title":"Toward Better Assessment of LLMs' Performance in Clinical Error Detection","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"healthylaife/paired-clinical-eval","path":"build_pairs.py","file_url":"https://github.com/healthylaife/paired-clinical-eval/blob/HEAD/build_pairs.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f31a7f3584994961","mcp_get_code":{"code_sha256":"f31a7f3584994961"}},{"arxiv_id":"2604.22760","paper":"/paper/arxiv-2604-22760","title":"Quantifying Divergence in Inter-LLM Communication Through API Retrieval and Ranking","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"aeris-lab/llmrank","path":"aaai_lamas_study.ipynb.py","file_url":"https://github.com/aeris-lab/llmrank/blob/HEAD/aaai_lamas_study.ipynb.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9ffccdf789b0e482","mcp_get_code":{"code_sha256":"9ffccdf789b0e482"}},{"arxiv_id":"2603.12478","paper":"/paper/arxiv-2603-12478","title":"Less Data, Faster Convergence: Goal-Driven Data Optimization for Multimodal Instruction Tuning","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"rujiewu/GDO","path":"gdo/extract_six_metrics.py","file_url":"https://github.com/rujiewu/GDO/blob/HEAD/gdo/extract_six_metrics.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cd6a7e9e2afeead9","mcp_get_code":{"code_sha256":"cd6a7e9e2afeead9"}},{"arxiv_id":"2506.20803","paper":"/paper/the-ideation-execution-gap-execution-outcomes","title":"The Ideation-Execution Gap: Execution Outcomes of LLM-Generated versus Human Research Ideas","date":"2025-06-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"NoviScl/AI-Researcher","path":"ai_researcher/src/analyze_experiment_plans_semantic_similarity.py","file_url":"https://github.com/NoviScl/AI-Researcher/blob/HEAD/ai_researcher/src/analyze_experiment_plans_semantic_similarity.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"aa1065b49e195e70","mcp_get_code":{"code_sha256":"aa1065b49e195e70"}},{"arxiv_id":"2409.04109","paper":"/paper/can-llms-generate-novel-research-ideas-a","title":"Can LLMs Generate Novel Research Ideas? A Large-Scale Human Study with 100+ NLP Researchers","date":"2024-09-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"simplaj/AI-Researcher-Spark","path":"ai_researcher/src/analyze_experiment_plans_semantic_similarity.py","file_url":"https://github.com/simplaj/AI-Researcher-Spark/blob/HEAD/ai_researcher/src/analyze_experiment_plans_semantic_similarity.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"aa1065b49e195e70","mcp_get_code":{"code_sha256":"aa1065b49e195e70"}},{"arxiv_id":"2405.19740","paper":"/paper/perteval-unveiling-real-knowledge-capacity-of","title":"PertEval: Unveiling Real Knowledge Capacity of LLMs with Knowledge-Invariant Perturbations","date":"2024-05-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aigc-apps/PertEval","path":"transition_analysis.py","file_url":"https://github.com/aigc-apps/PertEval/blob/HEAD/transition_analysis.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"9c137effe97e6ce1","mcp_get_code":{"code_sha256":"9c137effe97e6ce1"}},{"arxiv_id":"2405.15523","paper":"/paper/mosaic-memory-fuzzy-duplication-in-copyright","title":"The Mosaic Memory of Large Language Models","date":"2024-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"computationalprivacy/mosaic_memory","path":"src/gen_canaries.py","file_url":"https://github.com/computationalprivacy/mosaic_memory/blob/HEAD/src/gen_canaries.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"60ab76511145062d","mcp_get_code":{"code_sha256":"60ab76511145062d"}},{"arxiv_id":"2403.09732","paper":"/paper/pet-sql-a-prompt-enhanced-two-stage-text-to","title":"PET-SQL: A Prompt-Enhanced Two-Round Refinement of Text-to-SQL with Cross-consistency","date":"2024-03-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhshlii/petsql","path":"src/sources/sql_gen/sql_gen_utils.py","file_url":"https://github.com/zhshlii/petsql/blob/HEAD/src/sources/sql_gen/sql_gen_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a1579e74824be11f","mcp_get_code":{"code_sha256":"a1579e74824be11f"}},{"arxiv_id":"2402.14700","paper":"/paper/unveiling-linguistic-regions-in-large","title":"Unveiling Linguistic Regions in Large Language Models","date":"2024-02-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zzhang0179/Unveiling-Linguistic-Regions-in-LLMs","path":"region_selection/extract_accumulated_core_linguistic_region.py","file_url":"https://github.com/zzhang0179/Unveiling-Linguistic-Regions-in-LLMs/blob/HEAD/region_selection/extract_accumulated_core_linguistic_region.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"055c2b5ab3275ac2","mcp_get_code":{"code_sha256":"055c2b5ab3275ac2"}},{"arxiv_id":"2402.09363","paper":"/paper/copyright-traps-for-large-language-models","title":"Copyright Traps for Large Language Models","date":"2024-02-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"computationalprivacy/copyright-traps","path":"src/utils.py","file_url":"https://github.com/computationalprivacy/copyright-traps/blob/HEAD/src/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4eef3bc43ceb9a6e","mcp_get_code":{"code_sha256":"4eef3bc43ceb9a6e"}},{"arxiv_id":"2402.06221","paper":"/paper/resumeflow-an-llm-facilitated-pipeline-for","title":"ResumeFlow: An LLM-facilitated Pipeline for Personalized Resume Generation and Refinement","date":"2024-02-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Ztrimus/job-llm","path":"zlm/utils/metrics.py","file_url":"https://github.com/Ztrimus/job-llm/blob/HEAD/zlm/utils/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"78dddf7fab0bf963","mcp_get_code":{"code_sha256":"78dddf7fab0bf963"}},{"arxiv_id":"2401.16744","paper":"/paper/sharp-explaining-rankings-with-shapley-values","title":"ShaRP: A Novel Feature Importance Framework for Ranking","date":"2024-01-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dataresponsibly/sharp","path":"sharp/metrics/_base.py","file_url":"https://github.com/dataresponsibly/sharp/blob/HEAD/sharp/metrics/_base.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9363f8db4167743d","mcp_get_code":{"code_sha256":"9363f8db4167743d"}},{"arxiv_id":"2310.13132","paper":"/paper/ask-me-in-english-instead-cross-lingual","title":"Better to Ask in English: Cross-Lingual Evaluation of Large Language Models for Healthcare Queries","date":"2023-10-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"claws-lab/XLingEval","path":"consistency/consistency_answer_evaluation.py","file_url":"https://github.com/claws-lab/XLingEval/blob/HEAD/consistency/consistency_answer_evaluation.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"9a52e4e9f92b3c06","mcp_get_code":{"code_sha256":"9a52e4e9f92b3c06"}},{"arxiv_id":"2310.11248","paper":"/paper/crosscodeeval-a-diverse-and-multilingual","title":"CrossCodeEval: A Diverse and Multilingual Benchmark for Cross-File Code Completion","date":"2023-10-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"amazon-science/cceval","path":"prompt_builder/rerank_utils.py","file_url":"https://github.com/amazon-science/cceval/blob/HEAD/prompt_builder/rerank_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ae5ed04d466c8051","mcp_get_code":{"code_sha256":"ae5ed04d466c8051"}},{"arxiv_id":"2305.14292","paper":"/paper/wikichat-a-few-shot-llm-based-chatbot","title":"WikiChat: Stopping the Hallucination of Large Language Model Chatbots by Few-Shot Grounding on Wikipedia","date":"2023-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"stanford-oval/wikichat","path":"pipelines/dialogue_state.py","file_url":"https://github.com/stanford-oval/wikichat/blob/HEAD/pipelines/dialogue_state.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"34a2e796a199d00d","mcp_get_code":{"code_sha256":"34a2e796a199d00d"}},{"arxiv_id":"2204.09888","paper":"/paper/fairness-in-graph-mining-a-survey","title":"Fairness in Graph Mining: A Survey","date":"2022-04-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yushundong/pygdebias","path":"pygdebias/debiasing/GUIDE.py","file_url":"https://github.com/yushundong/pygdebias/blob/HEAD/pygdebias/debiasing/GUIDE.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-2-Clause","inline_ok":true,"code_sha256_prefix":"917ebaecb4caaf74","mcp_get_code":{"code_sha256":"917ebaecb4caaf74"}},{"arxiv_id":"2111.13654","paper":"/paper/do-language-models-have-beliefs-methods-for","title":"Do Language Models Have Beliefs? Methods for Detecting, Updating, and Visualizing Model Beliefs","date":"2021-11-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"peterbhase/slag-belief-updating","path":"metrics.py","file_url":"https://github.com/peterbhase/slag-belief-updating/blob/HEAD/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f4cecdbe8b090444","mcp_get_code":{"code_sha256":"f4cecdbe8b090444"}},{"arxiv_id":"1906.10263","paper":"/paper/dlime-a-deterministic-local-interpretable","title":"DLIME: A Deterministic Local Interpretable Model-Agnostic Explanations Approach for Computer-Aided Diagnosis Systems","date":"2019-06-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rehmanzafar/dlime_experiments","path":"experiments_bc_nn.py","file_url":"https://github.com/rehmanzafar/dlime_experiments/blob/HEAD/experiments_bc_nn.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9a52e4e9f92b3c06","mcp_get_code":{"code_sha256":"9a52e4e9f92b3c06"}},{"arxiv_id":"2022.findings-naacl.156","paper":null,"title":"arXiv:2022.findings-naacl.156","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"MJ-Jang/beyond-distributional","path":"src/metrics/eval_metrics.py","file_url":"https://github.com/MJ-Jang/beyond-distributional/blob/HEAD/src/metrics/eval_metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-2-Clause","inline_ok":true,"code_sha256_prefix":"c3fb804a84a0ac5b","mcp_get_code":{"code_sha256":"c3fb804a84a0ac5b"}}]}