{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/load-items","entry":"load_items","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":10,"n_papers_ran":6,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":10,"n_samples_ran":6,"n_samples_fingerprinted":0,"n_places":11,"n_places_pointer_only":4,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":1,"ran_fixture":0,"ran":5,"unverified":4},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.15530","paper":"/paper/arxiv-2609-15530","title":"Option-Aware Retrieval and Task-Specific VLM Adaptation for Medical VQA","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"Kirscher/MedReason2026","path":"finetune/build_sft_dataset.py","file_url":"https://github.com/Kirscher/MedReason2026/blob/HEAD/finetune/build_sft_dataset.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c8dcd3491ff74701","mcp_get_code":{"code_sha256":"c8dcd3491ff74701"}},{"arxiv_id":"2607.24165","paper":"/paper/arxiv-2607-24165","title":"Do Current Retrievers Cover All the Evidence? A Controlled Study of Conjunctive Cross-Page Retrieval","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"sunggukcha/n_clue","path":"code/evaluate.py","file_url":"https://github.com/sunggukcha/n_clue/blob/HEAD/code/evaluate.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"b035076c22823a53","mcp_get_code":{"code_sha256":"b035076c22823a53"}},{"arxiv_id":"2607.15176","paper":"/paper/arxiv-2607-15176","title":"Benchmarking Multimodal Large Language Models for Scientific Visualization Literacy","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"patdmp/mllm-scivis-lit-benchmark","path":"src/convert_open_source_to_csv.py","file_url":"https://github.com/patdmp/mllm-scivis-lit-benchmark/blob/HEAD/src/convert_open_source_to_csv.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"696497391eafdfef","mcp_get_code":{"code_sha256":"696497391eafdfef"}},{"arxiv_id":"2607.15176","paper":"/paper/arxiv-2607-15176","title":"Benchmarking Multimodal Large Language Models for Scientific Visualization Literacy","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"patdmp/mllm-scivis-lit-benchmark","path":"src/utils.py","file_url":"https://github.com/patdmp/mllm-scivis-lit-benchmark/blob/HEAD/src/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"99b88126210298d8","mcp_get_code":{"code_sha256":"99b88126210298d8"}},{"arxiv_id":"2604.27249","paper":"/paper/arxiv-2604-27249","title":"Instruction Complexity Induces Positional Collapse in Adversarial LLM Evaluation","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"synthiumjp/bcb-sandbagging-pilot","path":"run_study3.py","file_url":"https://github.com/synthiumjp/bcb-sandbagging-pilot/blob/HEAD/run_study3.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e55e0822df93c985","mcp_get_code":{"code_sha256":"e55e0822df93c985"}},{"arxiv_id":"2604.22215","paper":"/paper/arxiv-2604-22215","title":"Verbal Confidence Saturation in 3-9B Open-Weight Instruction-Tuned LLMs: A Pre-Registered Psychometric Validity Screen","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"synthiumjp/koriat","path":"collect_data_v2.py","file_url":"https://github.com/synthiumjp/koriat/blob/HEAD/collect_data_v2.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7a504a494e2c9480","mcp_get_code":{"code_sha256":"7a504a494e2c9480"}},{"arxiv_id":"2505.16994","paper":"/paper/text-r-2-text-ec-towards-large-recommender","title":"$\\text{R}^2\\text{ec}$: Towards Large Recommender Models with Reasoning","date":"2025-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"YRYangang/RRec","path":"preprocess.py","file_url":"https://github.com/YRYangang/RRec/blob/HEAD/preprocess.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5e7e71e747850f1b","mcp_get_code":{"code_sha256":"5e7e71e747850f1b"}},{"arxiv_id":"2501.00353","paper":"/paper/rag-instruct-boosting-llms-with-diverse","title":"RAG-Instruct: Boosting LLMs with Diverse Retrieval-Augmented Instructions","date":"2024-12-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"freedomintelligence/rag-instruct","path":"retrieval_lm/passage_retrieval.py","file_url":"https://github.com/freedomintelligence/rag-instruct/blob/HEAD/retrieval_lm/passage_retrieval.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c3045447a535da01","mcp_get_code":{"code_sha256":"c3045447a535da01"}},{"arxiv_id":"2408.10500","paper":"/paper/sztu-cmu-at-mer2024-improving-emotion-llama","title":"SZTU-CMU at MER2024: Improving Emotion-LLaMA with Conv-Attention for Multimodal Emotion Recognition","date":"2024-08-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zebangcheng/emotion-llama","path":"batch_infer.py","file_url":"https://github.com/zebangcheng/emotion-llama/blob/HEAD/batch_infer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"2f0c6446071c3dd6","mcp_get_code":{"code_sha256":"2f0c6446071c3dd6"}},{"arxiv_id":"2025.findings-acl.307","paper":null,"title":"arXiv:2025.findings-acl.307","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"hyp1231/AmazonReviews2023","path":"product_search_results/eval_search.py","file_url":"https://github.com/hyp1231/AmazonReviews2023/blob/HEAD/product_search_results/eval_search.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5b5901b90ddb592a","mcp_get_code":{"code_sha256":"5b5901b90ddb592a"}},{"arxiv_id":"2025.emnlp-main.192","paper":null,"title":"arXiv:2025.emnlp-main.192","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"FreedomIntelligence/RAG-Instruct","path":"retrieval_lm/passage_retrieval.py","file_url":"https://github.com/FreedomIntelligence/RAG-Instruct/blob/HEAD/retrieval_lm/passage_retrieval.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c3045447a535da01","mcp_get_code":{"code_sha256":"c3045447a535da01"}}]}