{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/fmt","entry":"fmt","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":14,"n_papers_ran":10,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":15,"n_samples_ran":11,"n_samples_fingerprinted":6,"n_places":15,"n_places_pointer_only":6,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":3,"ran_fixture":0,"ran":8,"unverified":4},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2608.17070","paper":"/paper/arxiv-2608-17070","title":"Certified but Private: Scalable Zero-Knowledge Proofs for Neural Network Guarantees","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"youweizhong/PANDA","path":"evaluation/reporting/component_report.py","file_url":"https://github.com/youweizhong/PANDA/blob/HEAD/evaluation/reporting/component_report.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f4b95b988ca47e28","mcp_get_code":{"code_sha256":"f4b95b988ca47e28"}},{"arxiv_id":"2606.17838","paper":"/paper/arxiv-2606-17838","title":"Environment-Grounded Automated Prompt Optimization for LLM Game Agents","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"ReanFernandes/rapoa","path":"final_paper_plotting/gen_condition_summary.py","file_url":"https://github.com/ReanFernandes/rapoa/blob/HEAD/final_paper_plotting/gen_condition_summary.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5416d3f7ac720a6d","mcp_get_code":{"code_sha256":"5416d3f7ac720a6d"}},{"arxiv_id":"2605.21404","paper":"/paper/arxiv-2605-21404","title":"What Twelve LLM Agent Benchmark Papers Disclose About Themselves: A Pilot Audit and an Open Scoring Schema","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"mahdinaser/reprobe-audit","path":"score_corpus.py","file_url":"https://github.com/mahdinaser/reprobe-audit/blob/HEAD/score_corpus.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7ce2982f28f2b074","mcp_get_code":{"code_sha256":"7ce2982f28f2b074"}},{"arxiv_id":"2605.12882","paper":"/paper/arxiv-2605-12882","title":"CiteVQA: Benchmarking Evidence Attribution for Trustworthy Document Intelligence","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"opendatalab/CiteVQA","path":"eval/summarize.py","file_url":"https://github.com/opendatalab/CiteVQA/blob/HEAD/eval/summarize.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e172f46060bcf6d9","mcp_get_code":{"code_sha256":"e172f46060bcf6d9"}},{"arxiv_id":"2605.09253","paper":"/paper/arxiv-2605-09253","title":"Cornerstones or Stumbling Blocks? Deciphering the Rock Tokens in On-Policy Distillation","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"YuxuanJiang1/Rock-Token","path":"rock_detection/compare_checkpoints.py","file_url":"https://github.com/YuxuanJiang1/Rock-Token/blob/HEAD/rock_detection/compare_checkpoints.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2a10cc0c29182e7d","mcp_get_code":{"code_sha256":"2a10cc0c29182e7d"}},{"arxiv_id":"2605.07093","paper":"/paper/arxiv-2605-07093","title":"The Translation Tax Is Not a Scalar: A Counterfactual Audit of English-Source Cue Inheritance in Chinese Multilingual Benchmarks","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"chi-mi-rvard/translation-tax-supplement","path":"run_include_lean.py","file_url":"https://github.com/chi-mi-rvard/translation-tax-supplement/blob/HEAD/run_include_lean.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"d8a81f0b768c77a5","mcp_get_code":{"code_sha256":"d8a81f0b768c77a5"}},{"arxiv_id":"2605.00604","paper":"/paper/arxiv-2605-00604","title":"Affinity Is Not Enough: Recovering the Free Energy Principle in Mixture-of-Experts","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"russellwmy/affinity-is-not-enough","path":"prototype/format_tables.py","file_url":"https://github.com/russellwmy/affinity-is-not-enough/blob/HEAD/prototype/format_tables.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"28cb7991e57a1649","mcp_get_code":{"code_sha256":"28cb7991e57a1649"}},{"arxiv_id":"2604.20043","paper":"/paper/arxiv-2604-20043","title":"TriEx: A Game-based Tri-View Framework for Explaining Internal Reasoning in Multi-Agent LLMs","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"Einsam1819/TriEx","path":"experiments/exp2c_intervention/draw.py","file_url":"https://github.com/Einsam1819/TriEx/blob/HEAD/experiments/exp2c_intervention/draw.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a382113ab2825052","mcp_get_code":{"code_sha256":"a382113ab2825052"}},{"arxiv_id":"2601.19448","paper":"/paper/arxiv-2601-19448","title":"From Internal Diagnosis to External Auditing: A VLM-Driven Paradigm for Data-Free Online Backdoor Defense","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"binyxu/PRISM","path":"rebuttal_experiments/reviewer_r3x4/W1_semantic_blend_attack/summarize_results.py","file_url":"https://github.com/binyxu/PRISM/blob/HEAD/rebuttal_experiments/reviewer_r3x4/W1_semantic_blend_attack/summarize_results.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0c07631b012f1cf9","mcp_get_code":{"code_sha256":"0c07631b012f1cf9"}},{"arxiv_id":"2509.19885","paper":"/paper/arxiv-2509-19885","title":"Towards Self-Supervised Foundation Models for Critical Care Time Series","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"Katja-Jagd/YAIB","path":"finetuning_results/pretrained_BAT/summarize_results.py","file_url":"https://github.com/Katja-Jagd/YAIB/blob/HEAD/finetuning_results/pretrained_BAT/summarize_results.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ff33ca62f52a82c6","mcp_get_code":{"code_sha256":"ff33ca62f52a82c6"}},{"arxiv_id":"2406.12208","paper":"/paper/knowledge-fusion-by-evolving-weights-of","title":"Knowledge Fusion By Evolving Weights of Language Models","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"duguodong7/model-evolution","path":"src/model_merge/base.py","file_url":"https://github.com/duguodong7/model-evolution/blob/HEAD/src/model_merge/base.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"89e8e1726d953764","mcp_get_code":{"code_sha256":"89e8e1726d953764"}},{"arxiv_id":"2406.12208","paper":"/paper/knowledge-fusion-by-evolving-weights-of","title":"Knowledge Fusion By Evolving Weights of Language Models","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"duguodong7/model-evolution","path":"src/model_merge/evolver.py","file_url":"https://github.com/duguodong7/model-evolution/blob/HEAD/src/model_merge/evolver.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"9c4701cb9ebe6693","mcp_get_code":{"code_sha256":"9c4701cb9ebe6693"}},{"arxiv_id":"2204.01959","paper":"/paper/data-augmentation-for-intent-classification-1","title":"Data Augmentation for Intent Classification with Off-the-shelf Large Language Models","date":"2022-04-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"elementai/data-augmentation-with-llms","path":"runners/compile_results.py","file_url":"https://github.com/elementai/data-augmentation-with-llms/blob/HEAD/runners/compile_results.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"47b1f7854f607d9f","mcp_get_code":{"code_sha256":"47b1f7854f607d9f"}},{"arxiv_id":"2104.11353","paper":"/paper/optimal-cost-design-for-model-predictive","title":"Optimal Cost Design for Model Predictive Control","date":"2021-04-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"avikj/l4dc-mpc-ocd","path":"experiments/run_mpc_ord.py","file_url":"https://github.com/avikj/l4dc-mpc-ocd/blob/HEAD/experiments/run_mpc_ord.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5b18d781b0676171","mcp_get_code":{"code_sha256":"5b18d781b0676171"}},{"arxiv_id":"1902.03616","paper":"/paper/elki-a-large-open-source-library-for-data","title":"ELKI: A large open-source library for data analysis - ELKI Release 0.7.5 \"Heidelberg\"","date":"2019-02-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"elki-project/elki","path":"elki-core-math/src/test/resources/elki/math/statistics/distribution/distribution-gen-testdata.py","file_url":"https://github.com/elki-project/elki/blob/HEAD/elki-core-math/src/test/resources/elki/math/statistics/distribution/distribution-gen-testdata.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"AGPL-3.0","inline_ok":false,"code_sha256_prefix":"da9abdf1e7a039dc","mcp_get_code":{"code_sha256":"da9abdf1e7a039dc"}}]}