{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/process-dataset","entry":"process_dataset","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":20,"n_papers_ran":7,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":21,"n_samples_ran":8,"n_samples_fingerprinted":0,"n_places":21,"n_places_pointer_only":12,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":2,"ran_fixture":0,"ran":6,"unverified":13},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.15400","paper":"/paper/arxiv-2609-15400","title":"BSC-Net: A Small-Branch-Sensitive Structural Continuity Network for Coronary Vessel Segmentation and Quantitative Angiographic Analysis","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"liwx-deeplearning/BSC-Net","path":"postprocessing/vessel_repair.py","file_url":"https://github.com/liwx-deeplearning/BSC-Net/blob/HEAD/postprocessing/vessel_repair.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"55661f629f654c22","mcp_get_code":{"code_sha256":"55661f629f654c22"}},{"arxiv_id":"2608.08459","paper":"/paper/arxiv-2608-08459","title":"Beyond Tables: Doc2DB-Bench for Relationally Faithful Document-to-Database Construction","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"SetonLiang/Doc2DB-Bench","path":"evaluation/clean_llamaextract_result.py","file_url":"https://github.com/SetonLiang/Doc2DB-Bench/blob/HEAD/evaluation/clean_llamaextract_result.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ed2b9770a39a07ce","mcp_get_code":{"code_sha256":"ed2b9770a39a07ce"}},{"arxiv_id":"2607.18232","paper":"/paper/arxiv-2607-18232","title":"It's Not What You Say, It's How You Say It: Evaluating LLM Responses to Expressions of Belief","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"clarakuempel/EoB","path":"score_dataset.py","file_url":"https://github.com/clarakuempel/EoB/blob/HEAD/score_dataset.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9a42667a07ed158c","mcp_get_code":{"code_sha256":"9a42667a07ed158c"}},{"arxiv_id":"2604.14397","paper":"/paper/arxiv-2604-14397","title":"Generating Concept Lexicalizations via Dictionary-Based Cross-Lingual Sense Projection","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"UAlberta-NLP/ExpandNet","path":"expandnet_step3_project.py","file_url":"https://github.com/UAlberta-NLP/ExpandNet/blob/HEAD/expandnet_step3_project.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"db2b44164286fc61","mcp_get_code":{"code_sha256":"db2b44164286fc61"}},{"arxiv_id":"2601.15334","paper":"/paper/arxiv-2601-15334","title":"NO RELIABLE EVIDENCE OF SELF-REPORTED SENTIENCE IN LARGE LANGUAGE MODELS","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"casparwarwick/sentient_machines_public","path":"01_cluster_pipeline/get_continuation.py","file_url":"https://github.com/casparwarwick/sentient_machines_public/blob/HEAD/01_cluster_pipeline/get_continuation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e7a7f254fe7d6701","mcp_get_code":{"code_sha256":"e7a7f254fe7d6701"}},{"arxiv_id":"2505.12632","paper":"/paper/scalable-video-to-dataset-generation-for","title":"Scalable Video-to-Dataset Generation for Cross-Platform Mobile Agents","date":"2025-05-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"runamu/monday","path":"data_processing/extract_scenes.py","file_url":"https://github.com/runamu/monday/blob/HEAD/data_processing/extract_scenes.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c9a868b25b751c41","mcp_get_code":{"code_sha256":"c9a868b25b751c41"}},{"arxiv_id":"2503.23513","paper":"/paper/rare-retrieval-augmented-reasoning-modeling","title":"RARE: Retrieval-Augmented Reasoning Modeling","date":"2025-03-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"open-dataflow/rare","path":"process/process_medqa.py","file_url":"https://github.com/open-dataflow/rare/blob/HEAD/process/process_medqa.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bf7b0dc382946a22","mcp_get_code":{"code_sha256":"bf7b0dc382946a22"}},{"arxiv_id":"2501.13302","paper":"/paper/watching-the-ai-watchdogs-a-fairness-and","title":"Watching the AI Watchdogs: A Fairness and Robustness Analysis of AI Safety Moderation Classifiers","date":"2025-01-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"acharaakshit/FairMod","path":"utils.py","file_url":"https://github.com/acharaakshit/FairMod/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"832c2aa62967a652","mcp_get_code":{"code_sha256":"832c2aa62967a652"}},{"arxiv_id":"2410.23856","paper":"/paper/can-language-models-perform-robust-reasoning","title":"Can Language Models Perform Robust Reasoning in Chain-of-thought Prompting with Noisy Rationales?","date":"2024-10-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dsubuntu/scie","path":"prepare_data_2.py","file_url":"https://github.com/dsubuntu/scie/blob/HEAD/prepare_data_2.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1d31f1da783c7f25","mcp_get_code":{"code_sha256":"1d31f1da783c7f25"}},{"arxiv_id":"2410.22284","paper":"/paper/embedding-based-classifiers-can-detect-prompt","title":"Embedding-based classifiers can detect prompt injection attacks","date":"2024-10-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AhsanAyub/malicious-prompt-detection","path":"binary_classification.py","file_url":"https://github.com/AhsanAyub/malicious-prompt-detection/blob/HEAD/binary_classification.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b3a5bd4d1e163768","mcp_get_code":{"code_sha256":"b3a5bd4d1e163768"}},{"arxiv_id":"2410.03775","paper":"/paper/beyond-correlation-the-impact-of-human","title":"Beyond correlation: The Impact of Human Uncertainty in Measuring the Effectiveness of Automatic Evaluation and LLM-as-a-Judge","date":"2024-10-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"amazon-science/beyondcorrelation","path":"examples/judge_bench_example.py","file_url":"https://github.com/amazon-science/beyondcorrelation/blob/HEAD/examples/judge_bench_example.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"992a1552125a9048","mcp_get_code":{"code_sha256":"992a1552125a9048"}},{"arxiv_id":"2406.18966","paper":"/paper/unigen-a-unified-framework-for-textual","title":"UniGen: A Unified Framework for Textual Dataset Generation Using Large Language Models","date":"2024-06-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"howiehwong/unigen","path":"unigen/utils/challenge.py","file_url":"https://github.com/howiehwong/unigen/blob/HEAD/unigen/utils/challenge.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c81e8bf3d510cc9e","mcp_get_code":{"code_sha256":"c81e8bf3d510cc9e"}},{"arxiv_id":"2405.19681","paper":"/paper/bayesian-online-natural-gradient-bong","title":"Bayesian Online Natural Gradient (BONG)","date":"2024-05-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"petergchang/bong","path":"bong/src/dataloaders.py","file_url":"https://github.com/petergchang/bong/blob/HEAD/bong/src/dataloaders.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"03c0e3a373096d5d","mcp_get_code":{"code_sha256":"03c0e3a373096d5d"}},{"arxiv_id":"2405.05466","paper":"/paper/poser-unmasking-alignment-faking-llms-by","title":"Poser: Unmasking Alignment Faking LLMs by Manipulating Their Internals","date":"2024-05-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sevdeawesome/POSER","path":"src/detection_strategies/can_we_shift_it.py","file_url":"https://github.com/sevdeawesome/POSER/blob/HEAD/src/detection_strategies/can_we_shift_it.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7d71c3773a5cd44f","mcp_get_code":{"code_sha256":"7d71c3773a5cd44f"}},{"arxiv_id":"2402.12997","paper":"/paper/towards-trustworthy-reranking-a-simple-yet","title":"Towards Trustworthy Reranking: A Simple yet Effective Abstention Mechanism","date":"2024-02-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"artefactory/abstention-reranker","path":"abstention_reranker/utils.py","file_url":"https://github.com/artefactory/abstention-reranker/blob/HEAD/abstention_reranker/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f53c7ebd02857bf4","mcp_get_code":{"code_sha256":"f53c7ebd02857bf4"}},{"arxiv_id":"2401.05952","paper":"/paper/llm-as-a-coauthor-the-challenges-of-detecting","title":"LLM-as-a-Coauthor: Can Mixed Human-Written and Machine-Generated Text Be Detected?","date":"2024-01-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dongping-chen/mixset","path":"dataset_loader.py","file_url":"https://github.com/dongping-chen/mixset/blob/HEAD/dataset_loader.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d6120a808c697802","mcp_get_code":{"code_sha256":"d6120a808c697802"}},{"arxiv_id":"2310.17284","paper":"/paper/learning-to-abstract-with-nonparametric","title":"Learning to Abstract with Nonparametric Variational Information Bottleneck","date":"2023-10-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"idiap/nvib_selfattention","path":"data_modules/ArxivDataModule.py","file_url":"https://github.com/idiap/nvib_selfattention/blob/HEAD/data_modules/ArxivDataModule.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e2a2369f2cb401a7","mcp_get_code":{"code_sha256":"e2a2369f2cb401a7"}},{"arxiv_id":"2310.17284","paper":"/paper/learning-to-abstract-with-nonparametric","title":"Learning to Abstract with Nonparametric Variational Information Bottleneck","date":"2023-10-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"idiap/nvib_selfattention","path":"data_modules/SentEvalDataModule.py","file_url":"https://github.com/idiap/nvib_selfattention/blob/HEAD/data_modules/SentEvalDataModule.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2c6d88e3f4e079ca","mcp_get_code":{"code_sha256":"2c6d88e3f4e079ca"}},{"arxiv_id":"2309.00916","paper":"/paper/blsp-bootstrapping-language-speech-pre-1","title":"BLSP: Bootstrapping Language-Speech Pre-training via Behavior Alignment of Continuation Writing","date":"2023-09-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cwang621/blsp","path":"blsp/src/text_instruction_dataset.py","file_url":"https://github.com/cwang621/blsp/blob/HEAD/blsp/src/text_instruction_dataset.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e35ba6d4c2745c08","mcp_get_code":{"code_sha256":"e35ba6d4c2745c08"}},{"arxiv_id":"2308.10278","paper":"/paper/characterchat-learning-towards-conversational","title":"CharacterChat: Learning towards Conversational AI with Personalized Social Support","date":"2023-08-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"morecry/characterchat","path":"model/BERT/util.py","file_url":"https://github.com/morecry/characterchat/blob/HEAD/model/BERT/util.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6681fc262a4fb058","mcp_get_code":{"code_sha256":"6681fc262a4fb058"}},{"arxiv_id":"1508.01211","paper":"/paper/listen-attend-and-spell","title":"Listen, Attend and Spell","date":"2015-08-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"WindQAQ/listen-attend-and-spell","path":"utils/dataset_utils.py","file_url":"https://github.com/WindQAQ/listen-attend-and-spell/blob/HEAD/utils/dataset_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"00ec8a3f07cfb9b9","mcp_get_code":{"code_sha256":"00ec8a3f07cfb9b9"}}]}