{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/generate-answer","entry":"generate_answer","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":12,"n_papers_ran":3,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":11,"n_samples_ran":2,"n_samples_fingerprinted":0,"n_places":12,"n_places_pointer_only":6,"by_status":{"ran_honours":0,"ran_violates":1,"ran_draft_wrong":0,"ran_fixture":0,"ran":1,"unverified":9},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2607.00605","paper":"/paper/arxiv-2607-00605","title":"Auditing Forgetting in Limited Memory Language Models","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"raeesiarya/LMLMAudit","path":"src/lmlm-audit/run_audit.py","file_url":"https://github.com/raeesiarya/LMLMAudit/blob/HEAD/src/lmlm-audit/run_audit.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1908bdfde38728bb","mcp_get_code":{"code_sha256":"1908bdfde38728bb"}},{"arxiv_id":"2606.18986","paper":"/paper/arxiv-2606-18986","title":"Beyond Tokenization: Direct Timestep Embedding and Contrastive Alignment for Time-Series Question Answering","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"YafengWu/CADE","path":"Time-MQA_Full_FT/merge.py","file_url":"https://github.com/YafengWu/CADE/blob/HEAD/Time-MQA_Full_FT/merge.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"74e845a6129041fd","mcp_get_code":{"code_sha256":"74e845a6129041fd"}},{"arxiv_id":"2601.01552","paper":"/paper/arxiv-2601-01552","title":"HalluZig: Hallucination Detection using Zigzag Persistence","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"TDA-Jyamiti/halluzig","path":"label_generated_gpt4o.py","file_url":"https://github.com/TDA-Jyamiti/halluzig/blob/HEAD/label_generated_gpt4o.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c55f41cc0b59f46f","mcp_get_code":{"code_sha256":"c55f41cc0b59f46f"}},{"arxiv_id":"2502.20082","paper":"/paper/longrope2-near-lossless-llm-context-window","title":"LongRoPE2: Near-Lossless LLM Context Window Scaling","date":"2025-02-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/longrope","path":"evaluation/passkey.py","file_url":"https://github.com/microsoft/longrope/blob/HEAD/evaluation/passkey.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f2a976adb1d0806e","mcp_get_code":{"code_sha256":"f2a976adb1d0806e"}},{"arxiv_id":"2412.19513","paper":"/paper/confidence-v-s-critique-a-decomposition-of","title":"Confidence v.s. Critique: A Decomposition of Self-Correction Capability for LLMs","date":"2024-12-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Zhe-Young/SelfCorrectDecompose","path":"eval_sampling.py","file_url":"https://github.com/Zhe-Young/SelfCorrectDecompose/blob/HEAD/eval_sampling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"dd6378ce27ea29b1","mcp_get_code":{"code_sha256":"dd6378ce27ea29b1"}},{"arxiv_id":"2407.07791","paper":"/paper/flooding-spread-of-manipulated-knowledge-in","title":"Flooding Spread of Manipulated Knowledge in LLM-Based Multi-Agent Communities","date":"2024-07-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Jometeorie/KnowledgeSpread","path":"simulation/baseline_easyedit.py","file_url":"https://github.com/Jometeorie/KnowledgeSpread/blob/HEAD/simulation/baseline_easyedit.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"41f7fbef0bb0cb0f","mcp_get_code":{"code_sha256":"41f7fbef0bb0cb0f"}},{"arxiv_id":"2407.00379","paper":"/paper/grapharena-benchmarking-large-language-models","title":"GraphArena: Benchmarking Large Language Models on Graph Computational Problems","date":"2024-06-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"squareRoot3/GraphArena","path":"benchmark_LLM_local.py","file_url":"https://github.com/squareRoot3/GraphArena/blob/HEAD/benchmark_LLM_local.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"31b2e95470d367f8","mcp_get_code":{"code_sha256":"31b2e95470d367f8"}},{"arxiv_id":"2403.05065","paper":"/paper/can-we-obtain-significant-success-in-rst","title":"Can we obtain significant success in RST discourse parsing by using Large Language Models?","date":"2024-03-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nttcslab-nlp/rstparser_eacl24","path":"src/parse/parse_utils.py","file_url":"https://github.com/nttcslab-nlp/rstparser_eacl24/blob/HEAD/src/parse/parse_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"01eb601638ec987f","mcp_get_code":{"code_sha256":"01eb601638ec987f"}},{"arxiv_id":"2402.15018","paper":"/paper/unintended-impacts-of-llm-alignment-on-global","title":"Unintended Impacts of LLM Alignment on Global Representation","date":"2024-02-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"salt-nlp/unintended-impacts-of-alignment","path":"code/md3game.py","file_url":"https://github.com/salt-nlp/unintended-impacts-of-alignment/blob/HEAD/code/md3game.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ea124a9e59aa3d97","mcp_get_code":{"code_sha256":"ea124a9e59aa3d97"}},{"arxiv_id":"2310.14566","paper":"/paper/hallusionbench-you-see-what-you-think-or-you","title":"HallusionBench: An Advanced Diagnostic Suite for Entangled Language Hallucination and Visual Illusion in Large Vision-Language Models","date":"2023-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tianyi-lab/hallusionbench","path":"evaluation.py","file_url":"https://github.com/tianyi-lab/hallusionbench/blob/HEAD/evaluation.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"c5e2fa32f54d0bba","mcp_get_code":{"code_sha256":"c5e2fa32f54d0bba"}},{"arxiv_id":"2211.05655","paper":"/paper/disentqa-disentangling-parametric-and","title":"DisentQA: Disentangling Parametric and Contextual Knowledge with Counterfactual Question Answering","date":"2022-11-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ellaneeman/disent_qa","path":"query_model.py","file_url":"https://github.com/ellaneeman/disent_qa/blob/HEAD/query_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a0de6f5df583e698","mcp_get_code":{"code_sha256":"a0de6f5df583e698"}},{"arxiv_id":"Guan_HallusionBench_An_Advanced_Diagnostic_Suite_for_Entangled_Language_Hallucination_and_CVPR_2024_paper","paper":null,"title":"arXiv:Guan_HallusionBench_An_Advanced_Diagnostic_Suite_for_Entangled_Language_Hallucination_and_CVPR_2024_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"tianyi-lab/HallusionBench","path":"evaluation.py","file_url":"https://github.com/tianyi-lab/HallusionBench/blob/HEAD/evaluation.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"c5e2fa32f54d0bba","mcp_get_code":{"code_sha256":"c5e2fa32f54d0bba"}}]}