{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/parse-json-response","entry":"parse_json_response","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":9,"n_papers_ran":7,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":9,"n_samples_ran":7,"n_samples_fingerprinted":4,"n_places":9,"n_places_pointer_only":7,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":1,"ran_fixture":0,"ran":6,"unverified":2},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2608.09209","paper":"/paper/arxiv-2608-09209","title":"UNMASK: Discovering and Causally Verifying Spurious Shortcuts in Text Classifiers","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"chidaksh/spurious_mitigator","path":"src/civil_comments/counterfactual_generation_api.py","file_url":"https://github.com/chidaksh/spurious_mitigator/blob/HEAD/src/civil_comments/counterfactual_generation_api.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2144f4386daf0e96","mcp_get_code":{"code_sha256":"2144f4386daf0e96"}},{"arxiv_id":"2607.10455","paper":"/paper/arxiv-2607-10455","title":"ANCHOR: Automated Alignment Auditing for CLI Agents on Real-World Harm","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"garified/anchor","path":"auditor_model_training/sft_data/sft_data_gen.py","file_url":"https://github.com/garified/anchor/blob/HEAD/auditor_model_training/sft_data/sft_data_gen.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"01de990d571f2414","mcp_get_code":{"code_sha256":"01de990d571f2414"}},{"arxiv_id":"2606.15307","paper":"/paper/arxiv-2606-15307","title":"Adapting Reinforcement Learning with Chain-of-Thought Supervision for Explainable Detection of Hateful and Propagandistic Memes","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"MohamedBayan/MemeReason","path":"annotation/consolidate_annotations.py","file_url":"https://github.com/MohamedBayan/MemeReason/blob/HEAD/annotation/consolidate_annotations.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"98d69637f11a050c","mcp_get_code":{"code_sha256":"98d69637f11a050c"}},{"arxiv_id":"2606.02578","paper":"/paper/arxiv-2606-02578","title":"Mitigating Perceptual Judgment Bias in Multimodal LLM-as-a-Judge via Perceptual Perturbation and Reward Modeling","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"kaist-cvml/perception-judge","path":"prepare-datasets/processing_async.py","file_url":"https://github.com/kaist-cvml/perception-judge/blob/HEAD/prepare-datasets/processing_async.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"3a81d8857ecf1f74","mcp_get_code":{"code_sha256":"3a81d8857ecf1f74"}},{"arxiv_id":"2605.30611","paper":"/paper/arxiv-2605-30611","title":"CRAFTER: A Multi-Agent Harness for Editable Scientific Figure Generation from Diverse Inputs","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"HaozheZhao/Crafter","path":"crafter/editor/raster_to_svg/model_router.py","file_url":"https://github.com/HaozheZhao/Crafter/blob/HEAD/crafter/editor/raster_to_svg/model_router.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"19fe7dc0cf7d4879","mcp_get_code":{"code_sha256":"19fe7dc0cf7d4879"}},{"arxiv_id":"2605.21240","paper":"/paper/arxiv-2605-21240","title":"APEX: Autonomous Policy Exploration for Self-Evolving LLM Agents","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"liushiliushi/APEX1","path":"src/explore_agent/openai_helpers_proxy.py","file_url":"https://github.com/liushiliushi/APEX1/blob/HEAD/src/explore_agent/openai_helpers_proxy.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5d09530257f612ba","mcp_get_code":{"code_sha256":"5d09530257f612ba"}},{"arxiv_id":"2603.21335","paper":"/paper/arxiv-2603-21335","title":"TimeTox: An LLM-Based Pipeline for Automated Extraction of Time Toxicity from Clinical Trial Protocols","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"sakethbuild/TimeTox","path":"agent_comparison.py","file_url":"https://github.com/sakethbuild/TimeTox/blob/HEAD/agent_comparison.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f63dcaa00bc5279a","mcp_get_code":{"code_sha256":"f63dcaa00bc5279a"}},{"arxiv_id":"2603.06183","paper":"/paper/arxiv-2603-06183","title":"CRIMSON: A Clinically-Grounded LLM-Based Metric for Generative Radiology Report Evaluation","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"rajpurkarlab/CRIMSON","path":"CRIMSON/generate_score.py","file_url":"https://github.com/rajpurkarlab/CRIMSON/blob/HEAD/CRIMSON/generate_score.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"78e54daaf9f90526","mcp_get_code":{"code_sha256":"78e54daaf9f90526"}},{"arxiv_id":"2601.01407","paper":"/paper/arxiv-2601-01407","title":"From Emotion Classification to Emotional Reasoning: Enhancing Emotional Intelligence in Large Language Models","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"kernelism/EC2ER","path":"benchmarking/src/utils.py","file_url":"https://github.com/kernelism/EC2ER/blob/HEAD/benchmarking/src/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b0c0b9ce9dd6a70a","mcp_get_code":{"code_sha256":"b0c0b9ce9dd6a70a"}}]}