{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/parse-response","entry":"parse_response","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":39,"n_papers_ran":22,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":43,"n_samples_ran":24,"n_samples_fingerprinted":16,"n_places":44,"n_places_pointer_only":16,"by_status":{"ran_honours":1,"ran_violates":0,"ran_draft_wrong":5,"ran_fixture":0,"ran":18,"unverified":19},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.03432","paper":"/paper/arxiv-2609-03432","title":"Decoupled Analysis-Judging: An Automated Creativity Evaluator Using LLMs in Complex Multi-step Creativity Tasks","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"Jaong/CreaEval","path":"experiment.py","file_url":"https://github.com/Jaong/CreaEval/blob/HEAD/experiment.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"157153a5f87f4a2d","mcp_get_code":{"code_sha256":"157153a5f87f4a2d"}},{"arxiv_id":"2609.00621","paper":"/paper/arxiv-2609-00621","title":"Control-Data Flow Separation: Stable Prompt Optimization in Multi-Agent LLMs","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"yuntian-group/cdsep","path":"cdsep/schema.py","file_url":"https://github.com/yuntian-group/cdsep/blob/HEAD/cdsep/schema.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3e955ba82b75f281","mcp_get_code":{"code_sha256":"3e955ba82b75f281"}},{"arxiv_id":"2607.23519","paper":"/paper/arxiv-2607-23519","title":"Auditing Alignment Controllability in LLMs via Political Axes","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"mbrcic/llm-political-steerability","path":"ingestion/01_collect/collect.py","file_url":"https://github.com/mbrcic/llm-political-steerability/blob/HEAD/ingestion/01_collect/collect.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bcbcfcee1dc27141","mcp_get_code":{"code_sha256":"bcbcfcee1dc27141"}},{"arxiv_id":"2606.31718","paper":"/paper/arxiv-2606-31718","title":"Cross-lingual Relation Extraction with Large Language Models: Zero-Shot, Few-Shot, and Fine-Tuned Evaluation on Romanian","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"DS4AI-UPB/crosslingual-romanian-re","path":"infer_e2e.py","file_url":"https://github.com/DS4AI-UPB/crosslingual-romanian-re/blob/HEAD/infer_e2e.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ffe27e79dc4e4979","mcp_get_code":{"code_sha256":"ffe27e79dc4e4979"}},{"arxiv_id":"2606.28392","paper":"/paper/arxiv-2606-28392","title":"RADIANT-PET: Reasoning-Augmented PET/CT Lesion Segmentation with Large Language Models and Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"jwang-580/RADIANT-PET","path":"eval/infer_api_models.py","file_url":"https://github.com/jwang-580/RADIANT-PET/blob/HEAD/eval/infer_api_models.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"88548e534a655e64","mcp_get_code":{"code_sha256":"88548e534a655e64"}},{"arxiv_id":"2606.28392","paper":"/paper/arxiv-2606-28392","title":"RADIANT-PET: Reasoning-Augmented PET/CT Lesion Segmentation with Large Language Models and Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"jwang-580/RADIANT-PET","path":"eval/infer_gpt_oss.py","file_url":"https://github.com/jwang-580/RADIANT-PET/blob/HEAD/eval/infer_gpt_oss.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6a5a10fbcab9e3a1","mcp_get_code":{"code_sha256":"6a5a10fbcab9e3a1"}},{"arxiv_id":"2606.28392","paper":"/paper/arxiv-2606-28392","title":"RADIANT-PET: Reasoning-Augmented PET/CT Lesion Segmentation with Large Language Models and Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"jwang-580/RADIANT-PET","path":"eval/infer_medgemma.py","file_url":"https://github.com/jwang-580/RADIANT-PET/blob/HEAD/eval/infer_medgemma.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0c270d00c05c496c","mcp_get_code":{"code_sha256":"0c270d00c05c496c"}},{"arxiv_id":"2605.15308","paper":"/paper/arxiv-2605-15308","title":"SMCEVOLVE: Principled Scientific Discovery via Sequential Monte Carlo Evolution","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"kongwanbianjinyu/SMCEvolve","path":"smcevolve/prompts.py","file_url":"https://github.com/kongwanbianjinyu/SMCEvolve/blob/HEAD/smcevolve/prompts.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"47e92f4ea3de4ddf","mcp_get_code":{"code_sha256":"47e92f4ea3de4ddf"}},{"arxiv_id":"2605.13043","paper":"/paper/arxiv-2605-13043","title":"Adaptive Steering and Remasking for Safe Generation in Diffusion Language Models","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"leeyejin1231/DLM_Steering_Remasking","path":"utils/mmlu_eval.py","file_url":"https://github.com/leeyejin1231/DLM_Steering_Remasking/blob/HEAD/utils/mmlu_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bae3b10ba5297700","mcp_get_code":{"code_sha256":"bae3b10ba5297700"}},{"arxiv_id":"2604.22215","paper":"/paper/arxiv-2604-22215","title":"Verbal Confidence Saturation in 3-9B Open-Weight Instruction-Tuned LLMs: A Pre-Registered Psychometric Validity Screen","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"synthiumjp/koriat","path":"collect_data.py","file_url":"https://github.com/synthiumjp/koriat/blob/HEAD/collect_data.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9cf19b493bfa49a0","mcp_get_code":{"code_sha256":"9cf19b493bfa49a0"}},{"arxiv_id":"2604.22215","paper":"/paper/arxiv-2604-22215","title":"Verbal Confidence Saturation in 3-9B Open-Weight Instruction-Tuned LLMs: A Pre-Registered Psychometric Validity Screen","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"synthiumjp/koriat","path":"collect_data_v2.py","file_url":"https://github.com/synthiumjp/koriat/blob/HEAD/collect_data_v2.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"51e6f27b5ffb8628","mcp_get_code":{"code_sha256":"51e6f27b5ffb8628"}},{"arxiv_id":"2604.05564","paper":"/paper/arxiv-2604-05564","title":"THIVLVC: Retrieval Augmented Dependency Parsing for Latin","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"l-pommeret/THIVLVC","path":"src/pipeline/stage1_rag_with_baseline.py","file_url":"https://github.com/l-pommeret/THIVLVC/blob/HEAD/src/pipeline/stage1_rag_with_baseline.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a7e44ce956d21681","mcp_get_code":{"code_sha256":"a7e44ce956d21681"}},{"arxiv_id":"2602.02258","paper":"/paper/arxiv-2602-02258","title":"Alignment-Aware Model Adaptation via Feedback-Guided Optimization","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"facebookresearch/TruthRL","path":"evaluation/evaluate.py","file_url":"https://github.com/facebookresearch/TruthRL/blob/HEAD/evaluation/evaluate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"21fa52735b227537","mcp_get_code":{"code_sha256":"21fa52735b227537"}},{"arxiv_id":"2601.11443","paper":"/paper/arxiv-2601-11443","title":"Predict the Retrieval! Test time adaptation for Retrieval Augmented Generation","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"sunxin000/TTARAG","path":"local_evaluation.py","file_url":"https://github.com/sunxin000/TTARAG/blob/HEAD/local_evaluation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d5f4dbf6439299f0","mcp_get_code":{"code_sha256":"d5f4dbf6439299f0"}},{"arxiv_id":"2601.00223","paper":"/paper/arxiv-2601-00223","title":"JP-TL-Bench: Anchored Pairwise LLM Evaluation for Bidirectional Japanese-English Translation","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"lhl/liquid-ai-hackathon-tokyo","path":"eval/judge-mt.py","file_url":"https://github.com/lhl/liquid-ai-hackathon-tokyo/blob/HEAD/eval/judge-mt.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5b3a1e4d0aa325d5","mcp_get_code":{"code_sha256":"5b3a1e4d0aa325d5"}},{"arxiv_id":"2510.21652","paper":"/paper/arxiv-2510-21652","title":"AstaBench: Rigorous Benchmarking of AI Agents with a Scientific Research Suite","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"allenai/agent-baselines","path":"agent_baselines/solvers/code_agent/llm_agent.py","file_url":"https://github.com/allenai/agent-baselines/blob/HEAD/agent_baselines/solvers/code_agent/llm_agent.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"cdad0f257db7572a","mcp_get_code":{"code_sha256":"cdad0f257db7572a"}},{"arxiv_id":"2507.04127","paper":null,"title":"arXiv:2507.04127","date":null,"month_inferred_from_arxiv_id":"2025-07","title_source":null,"repo":"awslabs/graphrag-toolkit","path":"byokg-rag/src/graphrag_toolkit/byokg_rag/byokg_query_engine.py","file_url":"https://github.com/awslabs/graphrag-toolkit/blob/HEAD/byokg-rag/src/graphrag_toolkit/byokg_rag/byokg_query_engine.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0cac8c1bedf394c9","mcp_get_code":{"code_sha256":"0cac8c1bedf394c9"}},{"arxiv_id":"2504.08600","paper":"/paper/sql-r1-training-natural-language-to-sql","title":"SQL-R1: Training Natural Language to SQL Reasoning Model By Reinforcement Learning","date":null,"month_inferred_from_arxiv_id":"2025-04","title_source":"archive","repo":"DataArcTech/SQL-R1","path":"src/inference.py","file_url":"https://github.com/DataArcTech/SQL-R1/blob/HEAD/src/inference.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4127f7104f276d1b","mcp_get_code":{"code_sha256":"4127f7104f276d1b"}},{"arxiv_id":"2503.21457","paper":"/paper/facebench-a-multi-view-multi-level-facial","title":"FaceBench: A Multi-View Multi-Level Facial Attribute VQA Dataset for Benchmarking Face Perception MLLMs","date":"2025-03-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"CVI-SZU/FaceBench","path":"evaluation/inference.py","file_url":"https://github.com/CVI-SZU/FaceBench/blob/HEAD/evaluation/inference.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a4d72c2b03733d9b","mcp_get_code":{"code_sha256":"a4d72c2b03733d9b"}},{"arxiv_id":"2503.00096","paper":"/paper/2503-00096","title":"BixBench: a Comprehensive Benchmark for LLM-based Agents in Computational Biology","date":"2025-02-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Future-House/BixBench","path":"bixbench/utils.py","file_url":"https://github.com/Future-House/BixBench/blob/HEAD/bixbench/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a6c564d5b5c99437","mcp_get_code":{"code_sha256":"a6c564d5b5c99437"}},{"arxiv_id":"2502.20490","paper":"/paper/egonormia-benchmarking-physical-social-norm","title":"EgoNormia: Benchmarking Physical Social Norm Understanding","date":"2025-02-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"open-social-world/egonormia","path":"src/gen/03_gen_questions.py","file_url":"https://github.com/open-social-world/egonormia/blob/HEAD/src/gen/03_gen_questions.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a78ab5f88a9698d5","mcp_get_code":{"code_sha256":"a78ab5f88a9698d5"}},{"arxiv_id":"2502.18906","paper":"/paper/vem-environment-free-exploration-for-training","title":"VEM: Environment-Free Exploration for Training GUI Agent with Value Environment Model","date":"2025-02-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/gui-agent-rl","path":"utils.py","file_url":"https://github.com/microsoft/gui-agent-rl/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ff74d412d9163765","mcp_get_code":{"code_sha256":"ff74d412d9163765"}},{"arxiv_id":"2412.20299","paper":"/paper/no-preference-left-behind-group","title":"No Preference Left Behind: Group Distributional Preference Optimization","date":"2024-12-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"BigBinnie/GDPO","path":"evaluate_BPC.py","file_url":"https://github.com/BigBinnie/GDPO/blob/HEAD/evaluate_BPC.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"80adbd3522975425","mcp_get_code":{"code_sha256":"80adbd3522975425"}},{"arxiv_id":"2411.13550","paper":"/paper/find-any-part-in-3d","title":"Find Any Part in 3D","date":"2024-11-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ziqi-ma/find3d","path":"dataengine/llm/name_single_part_gemini.py","file_url":"https://github.com/ziqi-ma/find3d/blob/HEAD/dataengine/llm/name_single_part_gemini.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d213559236b7abdd","mcp_get_code":{"code_sha256":"d213559236b7abdd"}},{"arxiv_id":"2411.13550","paper":"/paper/find-any-part-in-3d","title":"Find Any Part in 3D","date":"2024-11-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ziqi-ma/find3d","path":"dataengine/llm/query_orientation.py","file_url":"https://github.com/ziqi-ma/find3d/blob/HEAD/dataengine/llm/query_orientation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1de3152e134b486b","mcp_get_code":{"code_sha256":"1de3152e134b486b"}},{"arxiv_id":"2410.19546","paper":"/paper/bongard-in-wonderland-visual-puzzles-that","title":"Bongard in Wonderland: Visual Puzzles that Still Make AI Go Mad?","date":"2024-10-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ml-research/bongard-in-wonderland","path":"experiments/evaluate/eval_bp_with_solutions.py","file_url":"https://github.com/ml-research/bongard-in-wonderland/blob/HEAD/experiments/evaluate/eval_bp_with_solutions.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"3580d02883c8fc63","mcp_get_code":{"code_sha256":"3580d02883c8fc63"}},{"arxiv_id":"2410.16130","paper":"/paper/can-large-audio-language-models-truly-hear","title":"Can Large Audio-Language Models Truly Hear? Tackling Hallucinations with Multi-Task Assessment and Stepwise Audio Reasoning","date":"2024-10-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kuan2jiu99/audio-hallucination","path":"icassp2025/evaluation.py","file_url":"https://github.com/kuan2jiu99/audio-hallucination/blob/HEAD/icassp2025/evaluation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"698b1eaa3d1639e4","mcp_get_code":{"code_sha256":"698b1eaa3d1639e4"}},{"arxiv_id":"2410.16130","paper":"/paper/can-large-audio-language-models-truly-hear","title":"Can Large Audio-Language Models Truly Hear? Tackling Hallucinations with Multi-Task Assessment and Stepwise Audio Reasoning","date":"2024-10-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kuan2jiu99/audio-hallucination","path":"interspeech2024/evaluation.py","file_url":"https://github.com/kuan2jiu99/audio-hallucination/blob/HEAD/interspeech2024/evaluation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"94016dbb5d97a3d5","mcp_get_code":{"code_sha256":"94016dbb5d97a3d5"}},{"arxiv_id":"2409.07440","paper":"/paper/super-evaluating-agents-on-setting-up-and","title":"SUPER: Evaluating Agents on Setting Up and Executing Tasks from Research Repositories","date":"2024-09-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"allenai/super-benchmark","path":"super/agent/agent.py","file_url":"https://github.com/allenai/super-benchmark/blob/HEAD/super/agent/agent.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7365c17255307e51","mcp_get_code":{"code_sha256":"7365c17255307e51"}},{"arxiv_id":"2408.14469","paper":"/paper/grounded-multi-hop-videoqa-in-long-form","title":"Grounded Multi-Hop VideoQA in Long-Form Egocentric Videos","date":"2024-08-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"qirui-chen/MultiHop-EgoQA","path":"benchmark/metrics/evaluate_answering.py","file_url":"https://github.com/qirui-chen/MultiHop-EgoQA/blob/HEAD/benchmark/metrics/evaluate_answering.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"19864996c48943c7","mcp_get_code":{"code_sha256":"19864996c48943c7"}},{"arxiv_id":"2407.00948","paper":"/paper/the-house-always-wins-a-framework-for","title":"View From Above: A Framework for Evaluating Distribution Shifts in Model Behavior","date":"2024-07-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Bluefin-Tuna/ApartResearch","path":"deception/pyfiles/agent.py","file_url":"https://github.com/Bluefin-Tuna/ApartResearch/blob/HEAD/deception/pyfiles/agent.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7d64ba24b41b4639","mcp_get_code":{"code_sha256":"7d64ba24b41b4639"}},{"arxiv_id":"2406.04744","paper":"/paper/crag-comprehensive-rag-benchmark","title":"CRAG -- Comprehensive RAG Benchmark","date":"2024-06-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"facebookresearch/CRAG","path":"local_evaluation.py","file_url":"https://github.com/facebookresearch/CRAG/blob/HEAD/local_evaluation.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"a7323b55a4495886","mcp_get_code":{"code_sha256":"a7323b55a4495886"}},{"arxiv_id":"2405.20974","paper":"/paper/sayself-teaching-llms-to-express-confidence","title":"SaySelf: Teaching LLMs to Express Confidence with Self-Reflective Rationales","date":"2024-05-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xu1868/SaySelf","path":"utils/utils.py","file_url":"https://github.com/xu1868/SaySelf/blob/HEAD/utils/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"df9fb4e8e291c1f4","mcp_get_code":{"code_sha256":"df9fb4e8e291c1f4"}},{"arxiv_id":"2405.06691","paper":"/paper/fleet-of-agents-coordinated-problem-solving","title":"Fleet of Agents: Coordinated Problem Solving with Large Language Models","date":"2024-05-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"au-clan/FoA","path":"src/agents/crosswords.py","file_url":"https://github.com/au-clan/FoA/blob/HEAD/src/agents/crosswords.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"da3bdfa23d2582f6","mcp_get_code":{"code_sha256":"da3bdfa23d2582f6"}},{"arxiv_id":"2404.10975","paper":"/paper/procedural-dilemma-generation-for-evaluating","title":"Procedural Dilemma Generation for Evaluating Moral Reasoning in Humans and Language Models","date":"2024-04-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cicl-stanford/moral-evals","path":"offtherails/src/evaluate_llm.py","file_url":"https://github.com/cicl-stanford/moral-evals/blob/HEAD/offtherails/src/evaluate_llm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0e5309d0f2137cd8","mcp_get_code":{"code_sha256":"0e5309d0f2137cd8"}},{"arxiv_id":"2404.01261","paper":"/paper/fables-evaluating-faithfulness-and-content","title":"FABLES: Evaluating faithfulness and content selection in book-length summarization","date":"2024-04-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lilakk/booookscore","path":"booookscore/legacy/get_booookscore_v2.py","file_url":"https://github.com/lilakk/booookscore/blob/HEAD/booookscore/legacy/get_booookscore_v2.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"710b33af888f5924","mcp_get_code":{"code_sha256":"710b33af888f5924"}},{"arxiv_id":"2403.02528","paper":"/paper/daco-towards-application-driven-and","title":"DACO: Towards Application-Driven and Comprehensive Data Analysis via Code Generation","date":"2024-03-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shirley-wu/daco","path":"evaluation/eval_helpfulness.py","file_url":"https://github.com/shirley-wu/daco/blob/HEAD/evaluation/eval_helpfulness.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"9bd2a3e45b17af18","mcp_get_code":{"code_sha256":"9bd2a3e45b17af18"}},{"arxiv_id":"2401.11035","paper":"/paper/image-safeguarding-reasoning-with-conditional","title":"Image Safeguarding: Reasoning with Conditional Vision Language Model and Obfuscating Unsafe Content Counterfactually","date":"2024-01-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"SecureAIAutonomyLab/ConditionalVLM","path":"data_scripts/step2_human_verification.py","file_url":"https://github.com/SecureAIAutonomyLab/ConditionalVLM/blob/HEAD/data_scripts/step2_human_verification.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d7840e6c2b484338","mcp_get_code":{"code_sha256":"d7840e6c2b484338"}},{"arxiv_id":"2311.16502","paper":"/paper/mmmu-a-massive-multi-discipline-multimodal","title":"MMMU: A Massive Multi-discipline Multimodal Understanding and Reasoning Benchmark for Expert AGI","date":"2023-11-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"eric-ai-lab/probmed","path":"eval/calculate_score.py","file_url":"https://github.com/eric-ai-lab/probmed/blob/HEAD/eval/calculate_score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"eabb48a101284727","mcp_get_code":{"code_sha256":"eabb48a101284727"}},{"arxiv_id":"2311.04901","paper":"/paper/genome-generative-neuro-symbolic-visual","title":"GENOME: GenerativE Neuro-symbOlic visual reasoning by growing and reusing ModulEs","date":"2023-11-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"umass-foundation-model/genome","path":"engine/gpt.py","file_url":"https://github.com/umass-foundation-model/genome/blob/HEAD/engine/gpt.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c949723a39704cc2","mcp_get_code":{"code_sha256":"c949723a39704cc2"}},{"arxiv_id":"2310.15337","paper":"/paper/moral-foundations-of-large-language-models","title":"Moral Foundations of Large Language Models","date":"2023-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"abdulhaim/moral_foundations_llm","path":"utils/gpt3_utils.py","file_url":"https://github.com/abdulhaim/moral_foundations_llm/blob/HEAD/utils/gpt3_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0f6ba13b6fc5cda6","mcp_get_code":{"code_sha256":"0f6ba13b6fc5cda6"}},{"arxiv_id":"2310.00785","paper":"/paper/booookscore-a-systematic-exploration-of-book","title":"BooookScore: A systematic exploration of book-length summarization in the era of LLMs","date":"2023-10-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lilakk/BooookScore","path":"booookscore/legacy/get_booookscore_v2.py","file_url":"https://github.com/lilakk/BooookScore/blob/HEAD/booookscore/legacy/get_booookscore_v2.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"710b33af888f5924","mcp_get_code":{"code_sha256":"710b33af888f5924"}},{"arxiv_id":"2305.19713","paper":"/paper/red-teaming-language-model-detectors-with","title":"Red Teaming Language Model Detectors with Language Models","date":"2023-05-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shizhouxing/Attack-LM-Detectors","path":"DetectGPT/openai_perturbations.py","file_url":"https://github.com/shizhouxing/Attack-LM-Detectors/blob/HEAD/DetectGPT/openai_perturbations.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0c125fa85591bb34","mcp_get_code":{"code_sha256":"0c125fa85591bb34"}},{"arxiv_id":"2302.08582","paper":"/paper/pretraining-language-models-with-human","title":"Pretraining Language Models with Human Preferences","date":"2023-02-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tomekkorbak/pretraining-with-human-feedback","path":"red_team.py","file_url":"https://github.com/tomekkorbak/pretraining-with-human-feedback/blob/HEAD/red_team.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8d9e4494a1a48703","mcp_get_code":{"code_sha256":"8d9e4494a1a48703"}}]}