{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/postprocess-text","entry":"postprocess_text","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":21,"n_papers_ran":8,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":12,"n_samples_ran":3,"n_samples_fingerprinted":1,"n_places":23,"n_places_pointer_only":7,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":1,"ran_fixture":0,"ran":2,"unverified":9},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2503.20047","paper":"/paper/med3dvlm-an-efficient-vision-language-model","title":"Med3DVLM: An Efficient Vision-Language Model for 3D Medical Image Analysis","date":"2025-03-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mirthai/med3dvlm","path":"src/eval/eval_caption.py","file_url":"https://github.com/mirthai/med3dvlm/blob/HEAD/src/eval/eval_caption.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8b51adb8a6c014e5","mcp_get_code":{"code_sha256":"8b51adb8a6c014e5"}},{"arxiv_id":"2410.06554","paper":"/paper/the-accuracy-paradox-in-rlhf-when-better","title":"The Accuracy Paradox in RLHF: When Better Reward Models Don't Yield Better Language Models","date":"2024-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"EIT-NLP/AccuracyParadox-RLHF","path":"fgrlhf/evaluators.py","file_url":"https://github.com/EIT-NLP/AccuracyParadox-RLHF/blob/HEAD/fgrlhf/evaluators.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"94a55f838c975848","mcp_get_code":{"code_sha256":"94a55f838c975848"}},{"arxiv_id":"2410.04002","paper":"/paper/take-it-easy-label-adaptive-self","title":"Take It Easy: Label-Adaptive Self-Rationalization for Fact Verification and Explanation Generation","date":"2024-10-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jingyng/label-adaptive-self-rationalization","path":"code/training.py","file_url":"https://github.com/jingyng/label-adaptive-self-rationalization/blob/HEAD/code/training.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8b51adb8a6c014e5","mcp_get_code":{"code_sha256":"8b51adb8a6c014e5"}},{"arxiv_id":"2407.05952","paper":"/paper/h-star-llm-driven-hybrid-sql-text-adaptive","title":"H-STAR: LLM-driven Hybrid SQL-Text Adaptive Reasoning on Tables","date":"2024-06-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nikhilsab/h-star","path":"fetaqa_score.py","file_url":"https://github.com/nikhilsab/h-star/blob/HEAD/fetaqa_score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"aba2984da729d60f","mcp_get_code":{"code_sha256":"aba2984da729d60f"}},{"arxiv_id":"2406.12793","paper":"/paper/chatglm-a-family-of-large-language-models","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thudm/chatglm","path":"composite_demo/conversation.py","file_url":"https://github.com/thudm/chatglm/blob/HEAD/composite_demo/conversation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"554545dbb567c8a2","mcp_get_code":{"code_sha256":"554545dbb567c8a2"}},{"arxiv_id":"2406.02888","paper":"/paper/hydra-model-factorization-framework-for-black","title":"HYDRA: Model Factorization Framework for Black-Box LLM Personalization","date":"2024-06-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"night-chen/HYDRA","path":"metrics/generation_metrics.py","file_url":"https://github.com/night-chen/HYDRA/blob/HEAD/metrics/generation_metrics.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8b51adb8a6c014e5","mcp_get_code":{"code_sha256":"8b51adb8a6c014e5"}},{"arxiv_id":"2404.05970","paper":"/paper/optimization-methods-for-personalizing-large","title":"Optimization Methods for Personalizing Large Language Models through Retrieval Augmentation","date":"2024-04-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lamp-benchmark/lamp","path":"LaMP/metrics/generation_metrics.py","file_url":"https://github.com/lamp-benchmark/lamp/blob/HEAD/LaMP/metrics/generation_metrics.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8b51adb8a6c014e5","mcp_get_code":{"code_sha256":"8b51adb8a6c014e5"}},{"arxiv_id":"2404.05970","paper":"/paper/optimization-methods-for-personalizing-large","title":"Optimization Methods for Personalizing Large Language Models through Retrieval Augmentation","date":"2024-04-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lamp-benchmark/lamp","path":"LaMP/metrics/classification_metrics.py","file_url":"https://github.com/lamp-benchmark/lamp/blob/HEAD/LaMP/metrics/classification_metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1bea9980ec25d1c5","mcp_get_code":{"code_sha256":"1bea9980ec25d1c5"}},{"arxiv_id":"2403.18447","paper":"/paper/can-language-beat-numerical-regression","title":"Can Language Beat Numerical Regression? Language-Based Multimodal Trajectory Prediction","date":"2024-03-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"inhwanbae/LMTrajectory","path":"model/nltoolkit.py","file_url":"https://github.com/inhwanbae/LMTrajectory/blob/HEAD/model/nltoolkit.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"eff19094fe0ec70e","mcp_get_code":{"code_sha256":"eff19094fe0ec70e"}},{"arxiv_id":"2402.16058","paper":"/paper/say-more-with-less-understanding-prompt","title":"Say More with Less: Understanding Prompt Learning Behaviors through Gist Compression","date":"2024-02-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"openmatch/gist-coco","path":"src/inference/generalize/get_generalize_instruction_data.py","file_url":"https://github.com/openmatch/gist-coco/blob/HEAD/src/inference/generalize/get_generalize_instruction_data.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"85f31f5811b9b51a","mcp_get_code":{"code_sha256":"85f31f5811b9b51a"}},{"arxiv_id":"2402.05140","paper":"/paper/tag-llm-repurposing-general-purpose-llms-for","title":"Tag-LLM: Repurposing General-Purpose LLMs for Specialized Domains","date":"2024-02-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sjunhongshen/Tag-LLM","path":"src/metrics.py","file_url":"https://github.com/sjunhongshen/Tag-LLM/blob/HEAD/src/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"85f31f5811b9b51a","mcp_get_code":{"code_sha256":"85f31f5811b9b51a"}},{"arxiv_id":"2402.04315","paper":"/paper/training-language-models-to-generate-text","title":"Training Language Models to Generate Text with Citations via Fine-grained Rewards","date":"2024-02-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hcy123902/atg-w-fg-rw","path":"fgrlhf/evaluators.py","file_url":"https://github.com/hcy123902/atg-w-fg-rw/blob/HEAD/fgrlhf/evaluators.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"94a55f838c975848","mcp_get_code":{"code_sha256":"94a55f838c975848"}},{"arxiv_id":"2312.08914","paper":"/paper/cogagent-a-visual-language-model-for-gui","title":"CogAgent: A Visual Language Model for GUI Agents","date":"2023-12-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thudm/cogvlm","path":"composite_demo/conversation.py","file_url":"https://github.com/thudm/cogvlm/blob/HEAD/composite_demo/conversation.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2ce3e067b29fa052","mcp_get_code":{"code_sha256":"2ce3e067b29fa052"}},{"arxiv_id":"2310.20046","paper":"/paper/which-examples-to-annotate-for-in-context","title":"Which Examples to Annotate for In-Context Learning? Towards Effective and Efficient Selection","date":"2023-10-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"amazon-science/adaptive-in-context-learning","path":"main_adaptive_phases.py","file_url":"https://github.com/amazon-science/adaptive-in-context-learning/blob/HEAD/main_adaptive_phases.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"eff19094fe0ec70e","mcp_get_code":{"code_sha256":"eff19094fe0ec70e"}},{"arxiv_id":"2306.08401","paper":"/paper/livechat-a-large-scale-personalized-dialogue","title":"LiveChat: A Large-Scale Personalized Dialogue Dataset Automatically Constructed from Live Streaming","date":"2023-06-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gaojingsheng/livechat","path":"Tasks/Generation/src/utils.py","file_url":"https://github.com/gaojingsheng/livechat/blob/HEAD/Tasks/Generation/src/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c6948dc7082d01c7","mcp_get_code":{"code_sha256":"c6948dc7082d01c7"}},{"arxiv_id":"2212.09741","paper":"/paper/one-embedder-any-task-instruction-finetuned","title":"One Embedder, Any Task: Instruction-Finetuned Text Embeddings","date":"2022-12-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"HKUNLP/instructor-embedding","path":"evaluation/prompt_retrieval/main_coda_title_gen.py","file_url":"https://github.com/HKUNLP/instructor-embedding/blob/HEAD/evaluation/prompt_retrieval/main_coda_title_gen.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"eff19094fe0ec70e","mcp_get_code":{"code_sha256":"eff19094fe0ec70e"}},{"arxiv_id":"2212.09741","paper":"/paper/one-embedder-any-task-instruction-finetuned","title":"One Embedder, Any Task: Instruction-Finetuned Text Embeddings","date":"2022-12-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"HKUNLP/instructor-embedding","path":"evaluation/prompt_retrieval/main_answerbility.py","file_url":"https://github.com/HKUNLP/instructor-embedding/blob/HEAD/evaluation/prompt_retrieval/main_answerbility.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8d5099dd630ccb92","mcp_get_code":{"code_sha256":"8d5099dd630ccb92"}},{"arxiv_id":"Hong_CogAgent_A_Visual_Language_Model_for_GUI_Agents_CVPR_2024_paper","paper":null,"title":"arXiv:Hong_CogAgent_A_Visual_Language_Model_for_GUI_Agents_CVPR_2024_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"THUDM/CogVLM","path":"composite_demo/conversation.py","file_url":"https://github.com/THUDM/CogVLM/blob/HEAD/composite_demo/conversation.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2ce3e067b29fa052","mcp_get_code":{"code_sha256":"2ce3e067b29fa052"}},{"arxiv_id":"2025.findings-emnlp.228","paper":null,"title":"arXiv:2025.findings-emnlp.228","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"du-nlp-lab/FG-PRM","path":"src/fgrlhf/evaluators.py","file_url":"https://github.com/du-nlp-lab/FG-PRM/blob/HEAD/src/fgrlhf/evaluators.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"94a55f838c975848","mcp_get_code":{"code_sha256":"94a55f838c975848"}},{"arxiv_id":"2024.findings-emnlp.687","paper":null,"title":"arXiv:2024.findings-emnlp.687","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"IAAR-Shanghai/FastMem","path":"src/utils.py","file_url":"https://github.com/IAAR-Shanghai/FastMem/blob/HEAD/src/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"9d087ade51236362","mcp_get_code":{"code_sha256":"9d087ade51236362"}},{"arxiv_id":"2024.findings-emnlp.131","paper":null,"title":"arXiv:2024.findings-emnlp.131","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"siyue-zhang/SynTableQA","path":"metric/squall_tableqa.py","file_url":"https://github.com/siyue-zhang/SynTableQA/blob/HEAD/metric/squall_tableqa.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1bea9980ec25d1c5","mcp_get_code":{"code_sha256":"1bea9980ec25d1c5"}},{"arxiv_id":"2024.findings-acl.8","paper":null,"title":"arXiv:2024.findings-acl.8","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"allenai/chime","path":"chime/src/flanT5/finetune_task3.py","file_url":"https://github.com/allenai/chime/blob/HEAD/chime/src/flanT5/finetune_task3.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"27a7b2dc3a9a34b8","mcp_get_code":{"code_sha256":"27a7b2dc3a9a34b8"}},{"arxiv_id":"2022.findings-emnlp.174","paper":null,"title":"arXiv:2022.findings-emnlp.174","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"subercui/CodeExp","path":"code2text/GPT-J/gpt-j.pre.py","file_url":"https://github.com/subercui/CodeExp/blob/HEAD/code2text/GPT-J/gpt-j.pre.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8b51adb8a6c014e5","mcp_get_code":{"code_sha256":"8b51adb8a6c014e5"}}]}