{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/parse-output","entry":"parse_output","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":21,"n_papers_ran":11,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":23,"n_samples_ran":12,"n_samples_fingerprinted":7,"n_places":24,"n_places_pointer_only":13,"by_status":{"ran_honours":1,"ran_violates":0,"ran_draft_wrong":4,"ran_fixture":0,"ran":7,"unverified":11},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.01788","paper":"/paper/arxiv-2609-01788","title":"VakyArth: Evaluating Pragmatic Competence in LLMs across Indic Languages","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"Usneek1/VakyArth","path":"src/tasks/mcq.py","file_url":"https://github.com/Usneek1/VakyArth/blob/HEAD/src/tasks/mcq.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"60a978213610f448","mcp_get_code":{"code_sha256":"60a978213610f448"}},{"arxiv_id":"2609.01788","paper":"/paper/arxiv-2609-01788","title":"VakyArth: Evaluating Pragmatic Competence in LLMs across Indic Languages","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"Usneek1/VakyArth","path":"src/tasks/nli.py","file_url":"https://github.com/Usneek1/VakyArth/blob/HEAD/src/tasks/nli.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"0345df54e251f515","mcp_get_code":{"code_sha256":"0345df54e251f515"}},{"arxiv_id":"2608.17070","paper":"/paper/arxiv-2608-17070","title":"Certified but Private: Scalable Zero-Knowledge Proofs for Neural Network Guarantees","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"youweizhong/PANDA","path":"evaluation/run_panda.py","file_url":"https://github.com/youweizhong/PANDA/blob/HEAD/evaluation/run_panda.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1f4bb09ed2b3b4bc","mcp_get_code":{"code_sha256":"1f4bb09ed2b3b4bc"}},{"arxiv_id":"2604.15676","paper":"/paper/arxiv-2604-15676","title":"EvoRAG: Making Knowledge Graph-based RAG Automatically Evolve through Feedback-driven Backpropagation","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"iDC-NEU/EvoRAG","path":"chat/json_parse.py","file_url":"https://github.com/iDC-NEU/EvoRAG/blob/HEAD/chat/json_parse.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0d38071c84b6e6d5","mcp_get_code":{"code_sha256":"0d38071c84b6e6d5"}},{"arxiv_id":"2504.07954","paper":"/paper/perception-r1-pioneering-perception-policy","title":"Perception-R1: Pioneering Perception Policy with Reinforcement Learning","date":"2025-04-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"linkangheng/pr1","path":"eval/evaluate_counting.py","file_url":"https://github.com/linkangheng/pr1/blob/HEAD/eval/evaluate_counting.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"779b00437fd21789","mcp_get_code":{"code_sha256":"779b00437fd21789"}},{"arxiv_id":"2504.04953","paper":"/paper/m-prometheus-a-suite-of-open-multilingual-llm","title":"M-Prometheus: A Suite of Open Multilingual LLM Judges","date":"2025-04-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"prometheus-eval/prometheus-eval","path":"eval/run_evaluate.py","file_url":"https://github.com/prometheus-eval/prometheus-eval/blob/HEAD/eval/run_evaluate.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a1d5ae63fb237d0d","mcp_get_code":{"code_sha256":"a1d5ae63fb237d0d"}},{"arxiv_id":"2411.07130","paper":"/paper/retrieval-or-global-context-understanding-on","title":"Retrieval or Global Context Understanding? On Many-Shot In-Context Learning for Long-Context Evaluation","date":"2024-11-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"launchnlp/ManyICLBench","path":"utils.py","file_url":"https://github.com/launchnlp/ManyICLBench/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ac940764d0fa9610","mcp_get_code":{"code_sha256":"ac940764d0fa9610"}},{"arxiv_id":"2410.16198","paper":"/paper/improve-vision-language-model-chain-of","title":"Improve Vision Language Model Chain-of-thought Reasoning","date":"2024-10-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"riflezhang/llava-hound-dpo","path":"llava_hound_dpo/eval/gpt4v_eval.py","file_url":"https://github.com/riflezhang/llava-hound-dpo/blob/HEAD/llava_hound_dpo/eval/gpt4v_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"fba50892dd813a72","mcp_get_code":{"code_sha256":"fba50892dd813a72"}},{"arxiv_id":"2410.13394","paper":"/paper/cross-lingual-auto-evaluation-for-assessing","title":"Cross-Lingual Auto Evaluation for Assessing Multilingual LLMs","date":"2024-10-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ai4bharat/cia","path":"CIA/parser.py","file_url":"https://github.com/ai4bharat/cia/blob/HEAD/CIA/parser.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"49cf8b805cce2f5b","mcp_get_code":{"code_sha256":"49cf8b805cce2f5b"}},{"arxiv_id":"2407.14309","paper":"/paper/how-to-engage-your-readers-generating-guiding","title":"How to Engage Your Readers? Generating Guiding Questions to Promote Active Reading","date":"2024-07-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"eth-lre/engage-your-readers","path":"src/common_utils.py","file_url":"https://github.com/eth-lre/engage-your-readers/blob/HEAD/src/common_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"17f5a8d5d5701f52","mcp_get_code":{"code_sha256":"17f5a8d5d5701f52"}},{"arxiv_id":"2407.10834","paper":"/paper/metallm-a-high-performant-and-cost-efficient","title":"MetaLLM: A High-performant and Cost-efficient Dynamic Framework for Wrapping LLMs","date":"2024-07-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mail-research/metallm-wrapper","path":"benchmark.py","file_url":"https://github.com/mail-research/metallm-wrapper/blob/HEAD/benchmark.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"8003ae884ff0ba3a","mcp_get_code":{"code_sha256":"8003ae884ff0ba3a"}},{"arxiv_id":"2402.10051","paper":"/paper/swissnyf-tool-grounded-llm-agents-for-black","title":"SwissNYF: Tool Grounded LLM Agents for Black Box Setting","date":"2024-02-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"iclr-dummy-user/swissnyf","path":"swissnyf/pipeline/top_gun.py","file_url":"https://github.com/iclr-dummy-user/swissnyf/blob/HEAD/swissnyf/pipeline/top_gun.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1462953ea9a7df14","mcp_get_code":{"code_sha256":"1462953ea9a7df14"}},{"arxiv_id":"2310.05657","paper":"/paper/a-closer-look-into-automatic-evaluation-using","title":"A Closer Look into Automatic Evaluation Using Large Language Models","date":"2023-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"d223302/a-closer-look-to-llm-evaluation","path":"all_eval.py","file_url":"https://github.com/d223302/a-closer-look-to-llm-evaluation/blob/HEAD/all_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0dd7128d9b597b16","mcp_get_code":{"code_sha256":"0dd7128d9b597b16"}},{"arxiv_id":"2310.05657","paper":"/paper/a-closer-look-into-automatic-evaluation-using","title":"A Closer Look into Automatic Evaluation Using Large Language Models","date":"2023-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"d223302/a-closer-look-to-llm-evaluation","path":"meta_eval_summeval.py","file_url":"https://github.com/d223302/a-closer-look-to-llm-evaluation/blob/HEAD/meta_eval_summeval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d153a96e419db1c6","mcp_get_code":{"code_sha256":"d153a96e419db1c6"}},{"arxiv_id":"2310.05657","paper":"/paper/a-closer-look-into-automatic-evaluation-using","title":"A Closer Look into Automatic Evaluation Using Large Language Models","date":"2023-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"d223302/a-closer-look-to-llm-evaluation","path":"significance.py","file_url":"https://github.com/d223302/a-closer-look-to-llm-evaluation/blob/HEAD/significance.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e746ffd256b84ce5","mcp_get_code":{"code_sha256":"e746ffd256b84ce5"}},{"arxiv_id":"2308.12066","paper":"/paper/pre-gated-moe-an-algorithm-system-co-design","title":"Pre-gated MoE: An Algorithm-System Co-Design for Fast and Scalable Mixture-of-Expert Inference","date":"2023-08-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ranggihwang/pregated_moe","path":"scripts/eval_all.py","file_url":"https://github.com/ranggihwang/pregated_moe/blob/HEAD/scripts/eval_all.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1577dbcb7e57025b","mcp_get_code":{"code_sha256":"1577dbcb7e57025b"}},{"arxiv_id":"2307.02477","paper":"/paper/reasoning-or-reciting-exploring-the","title":"Reasoning or Reciting? Exploring the Capabilities and Limitations of Language Models Through Counterfactual Tasks","date":"2023-07-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhaofengwu/counterfactual-evaluation","path":"arithmetic/eval_ccc.py","file_url":"https://github.com/zhaofengwu/counterfactual-evaluation/blob/HEAD/arithmetic/eval_ccc.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5352f13bd6bc44b3","mcp_get_code":{"code_sha256":"5352f13bd6bc44b3"}},{"arxiv_id":"2304.13861","paper":"/paper/is-a-prompt-and-a-few-samples-all-you-need","title":"The Parrot Dilemma: Human-Labeled vs. LLM-augmented Data in Classification Tasks","date":"2023-04-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AGMoller/worker_vs_gpt","path":"src/worker_vs_gpt/utils.py","file_url":"https://github.com/AGMoller/worker_vs_gpt/blob/HEAD/src/worker_vs_gpt/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"576cc3f3886cae51","mcp_get_code":{"code_sha256":"576cc3f3886cae51"}},{"arxiv_id":"2303.16634","paper":"/paper/gpteval-nlg-evaluation-using-gpt-4-with","title":"G-Eval: NLG Evaluation using GPT-4 with Better Human Alignment","date":"2023-03-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nlpyang/geval","path":"meta_eval_summeval.py","file_url":"https://github.com/nlpyang/geval/blob/HEAD/meta_eval_summeval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"94622d9c122d3bb0","mcp_get_code":{"code_sha256":"94622d9c122d3bb0"}},{"arxiv_id":"2211.10797","paper":"/paper/an-empirical-study-on-contrastive-search-and","title":"An Empirical Study On Contrastive Search And Contrastive Decoding For Open-ended Text Generation","date":"2022-11-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"a115616e0c6ba1e8","mcp_get_code":{"code_sha256":"a115616e0c6ba1e8"}},{"arxiv_id":"2210.14140","paper":"/paper/contrastive-search-is-what-you-need-for","title":"Contrastive Search Is What You Need For Neural Text Generation","date":"2022-10-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yxuansu/contrastive_search_is_what_you_need","path":"code_generation/inference.py","file_url":"https://github.com/yxuansu/contrastive_search_is_what_you_need/blob/HEAD/code_generation/inference.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a115616e0c6ba1e8","mcp_get_code":{"code_sha256":"a115616e0c6ba1e8"}},{"arxiv_id":"1911.01547","paper":"/paper/the-measure-of-intelligence","title":"On the Measure of Intelligence","date":"2019-11-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"artyompal/kaggle-abstract-reasoning","path":"unit_tests.py","file_url":"https://github.com/artyompal/kaggle-abstract-reasoning/blob/HEAD/unit_tests.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ca5e60c977324c1e","mcp_get_code":{"code_sha256":"ca5e60c977324c1e"}},{"arxiv_id":"2025.emnlp-main.796","paper":null,"title":"arXiv:2025.emnlp-main.796","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"yukyunglee/CheckEval","path":"src/aggregation.py","file_url":"https://github.com/yukyunglee/CheckEval/blob/HEAD/src/aggregation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5a19af1ebc05880d","mcp_get_code":{"code_sha256":"5a19af1ebc05880d"}},{"arxiv_id":"2023.findings-emnlp.593","paper":null,"title":"arXiv:2023.findings-emnlp.593","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"csarkar/tagE","path":"src/metrics.py","file_url":"https://github.com/csarkar/tagE/blob/HEAD/src/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"85133ee681701f42","mcp_get_code":{"code_sha256":"85133ee681701f42"}}]}