{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/extract-label","entry":"extract_label","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":12,"n_papers_ran":5,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":11,"n_samples_ran":5,"n_samples_fingerprinted":4,"n_places":12,"n_places_pointer_only":9,"by_status":{"ran_honours":0,"ran_violates":1,"ran_draft_wrong":2,"ran_fixture":0,"ran":2,"unverified":6},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2606.15307","paper":"/paper/arxiv-2606-15307","title":"Adapting Reinforcement Learning with Chain-of-Thought Supervision for Explainable Detection of Hateful and Propagandistic Memes","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"MohamedBayan/MemeReason","path":"data_prep/build_unlabeled_training_set.py","file_url":"https://github.com/MohamedBayan/MemeReason/blob/HEAD/data_prep/build_unlabeled_training_set.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f85c65c949b629e1","mcp_get_code":{"code_sha256":"f85c65c949b629e1"}},{"arxiv_id":"2603.08286","paper":"/paper/arxiv-2603-08286","title":"LAMUS: A Large-Scale Corpus for Legal Argument Mining from U.S. Caselaw using LLMs","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"LavanyaPobbathi/LAMUS","path":"code/experiment/A_run_4_models_1st.py","file_url":"https://github.com/LavanyaPobbathi/LAMUS/blob/HEAD/code/experiment/A_run_4_models_1st.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"33a56252e8d12114","mcp_get_code":{"code_sha256":"33a56252e8d12114"}},{"arxiv_id":"2603.00696","paper":"/paper/arxiv-2603-00696","title":"DRIV-EX: Counterfactual Explanations for Driving LLMs","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"Amaia-CARDIEL/DRIV_EX","path":"driv_ex/dataset/generate_input_only_highD_data.py","file_url":"https://github.com/Amaia-CARDIEL/DRIV_EX/blob/HEAD/driv_ex/dataset/generate_input_only_highD_data.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"48ce3f13567e8ce8","mcp_get_code":{"code_sha256":"48ce3f13567e8ce8"}},{"arxiv_id":"2510.20369","paper":"/paper/arxiv-2510-20369","title":"Ask a Strong LLM Judge when Your Reward Model is Uncertain","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"zhenghaoxu-gatech/uncertainty-router","path":"ppo/src/eval_pm_router/eval_rewardbench.py","file_url":"https://github.com/zhenghaoxu-gatech/uncertainty-router/blob/HEAD/ppo/src/eval_pm_router/eval_rewardbench.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1acad12079a57e2a","mcp_get_code":{"code_sha256":"1acad12079a57e2a"}},{"arxiv_id":"2410.20651","paper":"/paper/subjective-qa-measuring-subjectivity-in","title":"SubjECTive-QA: Measuring Subjectivity in Earnings Call Transcripts' QA Through Six-Dimensional Feature Analysis","date":"2024-10-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gtfintechlab/SubjECTive-QA","path":"scripts/openai_benchmarking.py","file_url":"https://github.com/gtfintechlab/SubjECTive-QA/blob/HEAD/scripts/openai_benchmarking.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"828a7266ccb957bb","mcp_get_code":{"code_sha256":"828a7266ccb957bb"}},{"arxiv_id":"2410.02884","paper":"/paper/llama-berry-pairwise-optimization-for-o1-like","title":"LLaMA-Berry: Pairwise Optimization for O1-like Olympiad-Level Mathematical Reasoning","date":"2024-10-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"trotsky1997/mathblackbox","path":"gen_mcts_dpo.py","file_url":"https://github.com/trotsky1997/mathblackbox/blob/HEAD/gen_mcts_dpo.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f4e59a05a34d24d9","mcp_get_code":{"code_sha256":"f4e59a05a34d24d9"}},{"arxiv_id":"2407.02855","paper":"/paper/safe-unlearning-a-surprisingly-effective-and","title":"From Theft to Bomb-Making: The Ripple Effect of Unlearning in Defending Against Jailbreak Attacks","date":"2024-07-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thu-coai/safeunlearning","path":"evaluation/score_shieldlm.py","file_url":"https://github.com/thu-coai/safeunlearning/blob/HEAD/evaluation/score_shieldlm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"470eb9f49a78b466","mcp_get_code":{"code_sha256":"470eb9f49a78b466"}},{"arxiv_id":"2406.19146","paper":"/paper/resolving-discrepancies-in-compute-optimal","title":"Resolving Discrepancies in Compute-Optimal Scaling of Language Models","date":"2024-06-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"formll/resolving-scaling-law-discrepencies","path":"paper_tables.py","file_url":"https://github.com/formll/resolving-scaling-law-discrepencies/blob/HEAD/paper_tables.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9370894d3db38de0","mcp_get_code":{"code_sha256":"9370894d3db38de0"}},{"arxiv_id":"2311.18702","paper":"/paper/critiquellm-scaling-llm-as-critic-for","title":"CritiqueLLM: Towards an Informative Critique Generation Model for Evaluation of Large Language Model Generation","date":"2023-11-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thu-coai/critiquellm","path":"evaluation/eval_pairwise.py","file_url":"https://github.com/thu-coai/critiquellm/blob/HEAD/evaluation/eval_pairwise.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0164dbe04ac52b71","mcp_get_code":{"code_sha256":"0164dbe04ac52b71"}},{"arxiv_id":"2305.15005","paper":"/paper/sentiment-analysis-in-the-era-of-large","title":"Sentiment Analysis in the Era of Large Language Models: A Reality Check","date":"2023-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"damo-nlp-sg/llm-sentiment","path":"evaluate.py","file_url":"https://github.com/damo-nlp-sg/llm-sentiment/blob/HEAD/evaluate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"10ef73cfeb03a631","mcp_get_code":{"code_sha256":"10ef73cfeb03a631"}},{"arxiv_id":"2204.09435","paper":"/paper/hephaestus-a-large-scale-multitask-dataset","title":"Hephaestus: A large scale multitask dataset towards InSAR understanding","date":"2022-04-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"orion-ai-lab/hephaestus","path":"utilities/annotation_utils.py","file_url":"https://github.com/orion-ai-lab/hephaestus/blob/HEAD/utilities/annotation_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d64c1408c1ae7f8d","mcp_get_code":{"code_sha256":"d64c1408c1ae7f8d"}},{"arxiv_id":"2024.findings-naacl.246","paper":null,"title":"arXiv:2024.findings-naacl.246","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"DAMO-NLP-SG/LLM-Sentiment","path":"evaluate.py","file_url":"https://github.com/DAMO-NLP-SG/LLM-Sentiment/blob/HEAD/evaluate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"10ef73cfeb03a631","mcp_get_code":{"code_sha256":"10ef73cfeb03a631"}}]}