{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/remove-punctuation","entry":"remove_punctuation","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":14,"n_papers_ran":5,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":14,"n_samples_ran":5,"n_samples_fingerprinted":5,"n_places":14,"n_places_pointer_only":7,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":3,"ran_fixture":0,"ran":2,"unverified":9},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2603.04969","paper":"/paper/arxiv-2603-04969","title":"MPCEval: A Benchmark for Multi-Party Conversation Generation","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"Owen-Yang-18/MPCEval","path":"src/local_content_unsupervised/message.py","file_url":"https://github.com/Owen-Yang-18/MPCEval/blob/HEAD/src/local_content_unsupervised/message.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cad4ae349da9e797","mcp_get_code":{"code_sha256":"cad4ae349da9e797"}},{"arxiv_id":"2601.05707","paper":"/paper/arxiv-2601-05707","title":"Multimodal In-context Learning for ASR of Low-resource Languages","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"ZL-KA/MICL","path":"utils.py","file_url":"https://github.com/ZL-KA/MICL/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e95ee381ed8fb2dd","mcp_get_code":{"code_sha256":"e95ee381ed8fb2dd"}},{"arxiv_id":"2510.14374","paper":"/paper/arxiv-2510-14374","title":"Spatial Preference Rewarding for MLLMs Spatial Understanding","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"hanqiu-hq/SPR","path":"eval/ferret/model_flickr.py","file_url":"https://github.com/hanqiu-hq/SPR/blob/HEAD/eval/ferret/model_flickr.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f65f7ca4ffcd4fa6","mcp_get_code":{"code_sha256":"f65f7ca4ffcd4fa6"}},{"arxiv_id":"2504.05457","paper":null,"title":"arXiv:2504.05457","date":null,"month_inferred_from_arxiv_id":"2025-04","title_source":null,"repo":"vesteinn/vlm-eval","path":"src/vlmeval/calculate_scores/map_predictions.py","file_url":"https://github.com/vesteinn/vlm-eval/blob/HEAD/src/vlmeval/calculate_scores/map_predictions.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6dc5cdc318a06eca","mcp_get_code":{"code_sha256":"6dc5cdc318a06eca"}},{"arxiv_id":"2501.15228","paper":"/paper/improving-retrieval-augmented-generation","title":"Improving Retrieval-Augmented Generation through Multi-Agent Reinforcement Learning","date":"2025-01-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"chenyiqun/MMOA-RAG","path":"LLaMA-Factory/src/llamafactory/train/ppo/trainer_qr_s_g.py","file_url":"https://github.com/chenyiqun/MMOA-RAG/blob/HEAD/LLaMA-Factory/src/llamafactory/train/ppo/trainer_qr_s_g.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a6d95c1b612d9b81","mcp_get_code":{"code_sha256":"a6d95c1b612d9b81"}},{"arxiv_id":"2501.01872","paper":"/paper/turning-logic-against-itself-probing-model","title":"Turning Logic Against Itself : Probing Model Defenses Through Contrastive Questions","date":"2025-01-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"UKPLab/emnlp2025-poate-attack","path":"poate_attack/data_creation/generate_test_data.py","file_url":"https://github.com/UKPLab/emnlp2025-poate-attack/blob/HEAD/poate_attack/data_creation/generate_test_data.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"63a21a195abc5634","mcp_get_code":{"code_sha256":"63a21a195abc5634"}},{"arxiv_id":"2410.21909","paper":"/paper/scenegenagent-precise-industrial-scene","title":"SceneGenAgent: Precise Industrial Scene Generation with Coding Agent","date":"2024-10-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thudm/scenegenagent","path":"sceneinstruct/minhash.py","file_url":"https://github.com/thudm/scenegenagent/blob/HEAD/sceneinstruct/minhash.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6379d172938408fe","mcp_get_code":{"code_sha256":"6379d172938408fe"}},{"arxiv_id":"2408.08693","paper":"/paper/med-pmc-medical-personalized-multi-modal","title":"Med-PMC: Medical Personalized Multi-modal Consultation with a Proactive Ask-First-Observe-Next Paradigm","date":"2024-08-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"liuhc0428/med-pmc","path":"src/metrics/doctor_calaulate_infor.py","file_url":"https://github.com/liuhc0428/med-pmc/blob/HEAD/src/metrics/doctor_calaulate_infor.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"307e937536bd67cc","mcp_get_code":{"code_sha256":"307e937536bd67cc"}},{"arxiv_id":"2311.12022","paper":"/paper/gpqa-a-graduate-level-google-proof-q-a","title":"GPQA: A Graduate-Level Google-Proof Q&A Benchmark","date":"2023-11-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"idavidrein/gpqa","path":"baselines/open_book.py","file_url":"https://github.com/idavidrein/gpqa/blob/HEAD/baselines/open_book.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bc5679c991d7d2e6","mcp_get_code":{"code_sha256":"bc5679c991d7d2e6"}},{"arxiv_id":"2305.18153","paper":"/paper/do-large-language-models-know-what-they-don-t","title":"Do Large Language Models Know What They Don't Know?","date":"2023-05-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yinzhangyue/selfaware","path":"code/eval_model.py","file_url":"https://github.com/yinzhangyue/selfaware/blob/HEAD/code/eval_model.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"756ab533b15630b9","mcp_get_code":{"code_sha256":"756ab533b15630b9"}},{"arxiv_id":"2204.02380","paper":"/paper/clevr-x-a-visual-reasoning-dataset-for","title":"CLEVR-X: A Visual Reasoning Dataset for Natural Language Explanations","date":"2022-04-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"explainableml/clevr-x","path":"question_generation/text_template_handling.py","file_url":"https://github.com/explainableml/clevr-x/blob/HEAD/question_generation/text_template_handling.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"4756f083ebef1e12","mcp_get_code":{"code_sha256":"4756f083ebef1e12"}},{"arxiv_id":"2110.07205","paper":"/paper/speecht5-unified-modal-encoder-decoder-pre","title":"SpeechT5: Unified-Modal Encoder-Decoder Pre-Training for Spoken Language Processing","date":"2021-10-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mbzuai-nlp/artst","path":"scripts/ASR/evaluation.py","file_url":"https://github.com/mbzuai-nlp/artst/blob/HEAD/scripts/ASR/evaluation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"720bf60a08824a68","mcp_get_code":{"code_sha256":"720bf60a08824a68"}},{"arxiv_id":"2004.13922","paper":"/paper/revisiting-pre-trained-models-for-chinese","title":"Revisiting Pre-Trained Models for Chinese Natural Language Processing","date":"2020-04-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ymcui/Chinese-ELECTRA","path":"cmrc2018_drcd_evaluate.py","file_url":"https://github.com/ymcui/Chinese-ELECTRA/blob/HEAD/cmrc2018_drcd_evaluate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"5a2e0cd001c9f26b","mcp_get_code":{"code_sha256":"5a2e0cd001c9f26b"}},{"arxiv_id":"1611.09268","paper":"/paper/ms-marco-a-human-generated-machine-reading","title":"MS MARCO: A Human Generated MAchine Reading COmprehension Dataset","date":"2016-11-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/MSMARCO-Question-Answering","path":"Evaluation/eval_exp.py","file_url":"https://github.com/microsoft/MSMARCO-Question-Answering/blob/HEAD/Evaluation/eval_exp.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0a2e4a7e57b0e25a","mcp_get_code":{"code_sha256":"0a2e4a7e57b0e25a"}}]}