{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/extract-ans","entry":"extract_ans","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":9,"n_papers_ran":7,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":17,"n_samples_ran":14,"n_samples_fingerprinted":8,"n_places":19,"n_places_pointer_only":2,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":1,"ran_fixture":0,"ran":13,"unverified":3},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2502.14815","paper":"/paper/optimizing-model-selection-for-compound-ai","title":"Optimizing Model Selection for Compound AI Systems","date":"2025-02-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LLMSELECTOR/LLMSELECTOR","path":"llmselector/llmselector/compoundai/module/debate.py","file_url":"https://github.com/LLMSELECTOR/LLMSELECTOR/blob/HEAD/llmselector/llmselector/compoundai/module/debate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ae7782a357cf3a90","mcp_get_code":{"code_sha256":"ae7782a357cf3a90"}},{"arxiv_id":"2502.14815","paper":"/paper/optimizing-model-selection-for-compound-ai","title":"Optimizing Model Selection for Compound AI Systems","date":"2025-02-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LLMSELECTOR/LLMSELECTOR","path":"llmselector/llmselector/compoundai/module/majorityvote.py","file_url":"https://github.com/LLMSELECTOR/LLMSELECTOR/blob/HEAD/llmselector/llmselector/compoundai/module/majorityvote.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4ec0ffdf43c1c1c5","mcp_get_code":{"code_sha256":"4ec0ffdf43c1c1c5"}},{"arxiv_id":"2501.06252","paper":"/paper/text-transformer-2-self-adaptive-llms","title":"Transformer-Squared: Self-adaptive LLMs","date":"2025-01-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"SakanaAI/self-adaptive-llms","path":"tasks/cls.py","file_url":"https://github.com/SakanaAI/self-adaptive-llms/blob/HEAD/tasks/cls.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"86a430f986c7a911","mcp_get_code":{"code_sha256":"86a430f986c7a911"}},{"arxiv_id":"2410.04199","paper":"/paper/longgenbench-long-context-generation","title":"LongGenBench: Long-context Generation Benchmark","date":"2024-10-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dominic789654/longgenbench","path":"longgenbench_GSM8K_openai.py","file_url":"https://github.com/dominic789654/longgenbench/blob/HEAD/longgenbench_GSM8K_openai.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cf6841a6599707b0","mcp_get_code":{"code_sha256":"cf6841a6599707b0"}},{"arxiv_id":"2407.12709","paper":"/paper/mome-mixture-of-multimodal-experts-for","title":"MoME: Mixture of Multimodal Experts for Generalist Multimodal Large Language Models","date":"2024-07-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jiutian-vl/mome","path":"demo_mome.py","file_url":"https://github.com/jiutian-vl/mome/blob/HEAD/demo_mome.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"439482b5435bb1d9","mcp_get_code":{"code_sha256":"439482b5435bb1d9"}},{"arxiv_id":"2406.02919","paper":"/paper/multifaceteval-multifaceted-evaluation-to","title":"MultifacetEval: Multifaceted Evaluation to Probe LLMs in Mastering Medical Knowledge","date":"2024-06-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thumlp/multifaceteval","path":"MAQ_answer_analysis_medqa.py","file_url":"https://github.com/thumlp/multifaceteval/blob/HEAD/MAQ_answer_analysis_medqa.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a621bd45961d965f","mcp_get_code":{"code_sha256":"a621bd45961d965f"}},{"arxiv_id":"2406.02919","paper":"/paper/multifaceteval-multifaceted-evaluation-to","title":"MultifacetEval: Multifaceted Evaluation to Probe LLMs in Mastering Medical Knowledge","date":"2024-06-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thumlp/multifaceteval","path":"MAQ_answer_analysis_medqa_cotsc.py","file_url":"https://github.com/thumlp/multifaceteval/blob/HEAD/MAQ_answer_analysis_medqa_cotsc.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"24796f0e8267d08f","mcp_get_code":{"code_sha256":"24796f0e8267d08f"}},{"arxiv_id":"2406.02919","paper":"/paper/multifaceteval-multifaceted-evaluation-to","title":"MultifacetEval: Multifaceted Evaluation to Probe LLMs in Mastering Medical Knowledge","date":"2024-06-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thumlp/multifaceteval","path":"MCQ_answer_analysis_medqa.py","file_url":"https://github.com/thumlp/multifaceteval/blob/HEAD/MCQ_answer_analysis_medqa.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5fe4ca43949d2549","mcp_get_code":{"code_sha256":"5fe4ca43949d2549"}},{"arxiv_id":"2406.02919","paper":"/paper/multifaceteval-multifaceted-evaluation-to","title":"MultifacetEval: Multifaceted Evaluation to Probe LLMs in Mastering Medical Knowledge","date":"2024-06-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thumlp/multifaceteval","path":"MCQ_answer_analysis_medqa_cotsc.py","file_url":"https://github.com/thumlp/multifaceteval/blob/HEAD/MCQ_answer_analysis_medqa_cotsc.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"172aa259fd1693c0","mcp_get_code":{"code_sha256":"172aa259fd1693c0"}},{"arxiv_id":"2406.02919","paper":"/paper/multifaceteval-multifaceted-evaluation-to","title":"MultifacetEval: Multifaceted Evaluation to Probe LLMs in Mastering Medical Knowledge","date":"2024-06-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thumlp/multifaceteval","path":"RQ_answer_analysis_medqa.py","file_url":"https://github.com/thumlp/multifaceteval/blob/HEAD/RQ_answer_analysis_medqa.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"be9589b46024780d","mcp_get_code":{"code_sha256":"be9589b46024780d"}},{"arxiv_id":"2406.02919","paper":"/paper/multifaceteval-multifaceted-evaluation-to","title":"MultifacetEval: Multifaceted Evaluation to Probe LLMs in Mastering Medical Knowledge","date":"2024-06-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thumlp/multifaceteval","path":"RQ_answer_analysis_medqa_cotsc.py","file_url":"https://github.com/thumlp/multifaceteval/blob/HEAD/RQ_answer_analysis_medqa_cotsc.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"66dd4af07fab516b","mcp_get_code":{"code_sha256":"66dd4af07fab516b"}},{"arxiv_id":"2406.02919","paper":"/paper/multifaceteval-multifaceted-evaluation-to","title":"MultifacetEval: Multifaceted Evaluation to Probe LLMs in Mastering Medical Knowledge","date":"2024-06-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thumlp/multifaceteval","path":"TFQ_answer_analysis_medqa.py","file_url":"https://github.com/thumlp/multifaceteval/blob/HEAD/TFQ_answer_analysis_medqa.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e508780be7f14ee4","mcp_get_code":{"code_sha256":"e508780be7f14ee4"}},{"arxiv_id":"2406.02919","paper":"/paper/multifaceteval-multifaceted-evaluation-to","title":"MultifacetEval: Multifaceted Evaluation to Probe LLMs in Mastering Medical Knowledge","date":"2024-06-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thumlp/multifaceteval","path":"TFQ_answer_analysis_medqa_cotsc.py","file_url":"https://github.com/thumlp/multifaceteval/blob/HEAD/TFQ_answer_analysis_medqa_cotsc.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"685e445c8af90359","mcp_get_code":{"code_sha256":"685e445c8af90359"}},{"arxiv_id":"2404.09127","paper":"/paper/confidence-calibration-and-rationalization","title":"Confidence Calibration and Rationalization for LLMs via Multi-Agent Deliberation","date":"2024-04-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"minnesotanlp/collaborative-calibration","path":"collaborative-calibration/simulation_utils.py","file_url":"https://github.com/minnesotanlp/collaborative-calibration/blob/HEAD/collaborative-calibration/simulation_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"90d7825b0ed1e03b","mcp_get_code":{"code_sha256":"90d7825b0ed1e03b"}},{"arxiv_id":"2403.05527","paper":"/paper/gear-an-efficient-kv-cache-compression","title":"GEAR: An Efficient KV Cache Compression Recipe for Near-Lossless Generative Inference of LLM","date":"2024-03-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"haokang-timmy/gear","path":"GenerationBench/GenerationTest/evaluation_bbh_cot.py","file_url":"https://github.com/haokang-timmy/gear/blob/HEAD/GenerationBench/GenerationTest/evaluation_bbh_cot.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3bbfc4445eea20a9","mcp_get_code":{"code_sha256":"3bbfc4445eea20a9"}},{"arxiv_id":"2310.05736","paper":"/paper/llmlingua-compressing-prompts-for-accelerated","title":"LLMLingua: Compressing Prompts for Accelerated Inference of Large Language Models","date":"2023-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"FranxYao/chain-of-thought-hub","path":"BBH/run_bbh_claude_instant_v1.0.py","file_url":"https://github.com/FranxYao/chain-of-thought-hub/blob/HEAD/BBH/run_bbh_claude_instant_v1.0.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9ad0c8eaff610fe8","mcp_get_code":{"code_sha256":"9ad0c8eaff610fe8"}},{"arxiv_id":"2310.05736","paper":"/paper/llmlingua-compressing-prompts-for-accelerated","title":"LLMLingua: Compressing Prompts for Accelerated Inference of Large Language Models","date":"2023-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"FranxYao/chain-of-thought-hub","path":"BBH/run_bbh_claude_v1.3.py","file_url":"https://github.com/FranxYao/chain-of-thought-hub/blob/HEAD/BBH/run_bbh_claude_v1.3.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"295b8fa768a029bc","mcp_get_code":{"code_sha256":"295b8fa768a029bc"}},{"arxiv_id":"2305.17306","paper":"/paper/chain-of-thought-hub-a-continuous-effort-to","title":"Chain-of-Thought Hub: A Continuous Effort to Measure Large Language Models' Reasoning Performance","date":"2023-05-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"franxyao/chain-of-thought-hub","path":"BBH/run_bbh_claude_instant_v1.0.py","file_url":"https://github.com/franxyao/chain-of-thought-hub/blob/HEAD/BBH/run_bbh_claude_instant_v1.0.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9ad0c8eaff610fe8","mcp_get_code":{"code_sha256":"9ad0c8eaff610fe8"}},{"arxiv_id":"2305.17306","paper":"/paper/chain-of-thought-hub-a-continuous-effort-to","title":"Chain-of-Thought Hub: A Continuous Effort to Measure Large Language Models' Reasoning Performance","date":"2023-05-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"franxyao/chain-of-thought-hub","path":"BBH/run_bbh_claude_v1.3.py","file_url":"https://github.com/franxyao/chain-of-thought-hub/blob/HEAD/BBH/run_bbh_claude_v1.3.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"295b8fa768a029bc","mcp_get_code":{"code_sha256":"295b8fa768a029bc"}}]}