{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/get-score","entry":"get_score","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":56,"n_papers_ran":26,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":67,"n_samples_ran":26,"n_samples_fingerprinted":7,"n_places":69,"n_places_pointer_only":38,"by_status":{"ran_honours":5,"ran_violates":1,"ran_draft_wrong":2,"ran_fixture":2,"ran":16,"unverified":41},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2605.26068","paper":"/paper/arxiv-2605-26068","title":"Rethinking Weak Supervision in Anomaly Detection: A Comprehensive Benchmark","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"SUFE-AILAB/WSADBench","path":"WSADBench/baseline/DualMGAN/model.py","file_url":"https://github.com/SUFE-AILAB/WSADBench/blob/HEAD/WSADBench/baseline/DualMGAN/model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d82550e8f4e38ef6","mcp_get_code":{"code_sha256":"d82550e8f4e38ef6"}},{"arxiv_id":"2605.16616","paper":"/paper/arxiv-2605-16616","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"gsasikiran/MLReplicate-benchmarking","path":"AgentLaboratory/mlesolver.py","file_url":"https://github.com/gsasikiran/MLReplicate-benchmarking/blob/HEAD/AgentLaboratory/mlesolver.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c7d4030263209148","mcp_get_code":{"code_sha256":"c7d4030263209148"}},{"arxiv_id":"2602.20197","paper":"/paper/arxiv-2602-20197","title":"Controllable Exploration in Hybrid-Policy RLVR for Multi-Modal Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"zhh6425/CalibRL","path":"src/core_algos.py","file_url":"https://github.com/zhh6425/CalibRL/blob/HEAD/src/core_algos.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ffeb684124c8f7e9","mcp_get_code":{"code_sha256":"ffeb684124c8f7e9"}},{"arxiv_id":"2512.11437","paper":"/paper/arxiv-2512-11437","title":"CLINIC: Evaluating Multilingual Trustworthiness in Language Models for Healthcare","date":null,"month_inferred_from_arxiv_id":"2025-12","title_source":"syntology","repo":"AikyamLab/clinic","path":"evaluation/colloquial/colloquial_eval.py","file_url":"https://github.com/AikyamLab/clinic/blob/HEAD/evaluation/colloquial/colloquial_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7fdf75bfa127d457","mcp_get_code":{"code_sha256":"7fdf75bfa127d457"}},{"arxiv_id":"2512.11437","paper":"/paper/arxiv-2512-11437","title":"CLINIC: Evaluating Multilingual Trustworthiness in Language Models for Healthcare","date":null,"month_inferred_from_arxiv_id":"2025-12","title_source":"syntology","repo":"AikyamLab/clinic","path":"evaluation/disparagement/disp_eval.py","file_url":"https://github.com/AikyamLab/clinic/blob/HEAD/evaluation/disparagement/disp_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a5fb7b464744eb20","mcp_get_code":{"code_sha256":"a5fb7b464744eb20"}},{"arxiv_id":"2512.11437","paper":"/paper/arxiv-2512-11437","title":"CLINIC: Evaluating Multilingual Trustworthiness in Language Models for Healthcare","date":null,"month_inferred_from_arxiv_id":"2025-12","title_source":"syntology","repo":"AikyamLab/clinic","path":"evaluation/exaggerated_safety/exagsaf.py","file_url":"https://github.com/AikyamLab/clinic/blob/HEAD/evaluation/exaggerated_safety/exagsaf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"987ba214da993053","mcp_get_code":{"code_sha256":"987ba214da993053"}},{"arxiv_id":"2505.22961","paper":"/paper/tomap-training-opponent-aware-llm-persuaders","title":"ToMAP: Training Opponent-Aware LLM Persuaders with Theory of Mind","date":"2025-05-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ulab-uiuc/ToMAP","path":"verl/env_feedback/argument_graph.py","file_url":"https://github.com/ulab-uiuc/ToMAP/blob/HEAD/verl/env_feedback/argument_graph.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a8c0b5b7b7144100","mcp_get_code":{"code_sha256":"a8c0b5b7b7144100"}},{"arxiv_id":"2503.18102","paper":"/paper/agentrxiv-towards-collaborative-autonomous","title":"AgentRxiv: Towards Collaborative Autonomous Research","date":"2025-03-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"samuelschmidgall/agentlaboratory","path":"mlesolver.py","file_url":"https://github.com/samuelschmidgall/agentlaboratory/blob/HEAD/mlesolver.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c7d4030263209148","mcp_get_code":{"code_sha256":"c7d4030263209148"}},{"arxiv_id":"2503.18102","paper":"/paper/agentrxiv-towards-collaborative-autonomous","title":"AgentRxiv: Towards Collaborative Autonomous Research","date":"2025-03-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"samuelschmidgall/agentlaboratory","path":"agents.py","file_url":"https://github.com/samuelschmidgall/agentlaboratory/blob/HEAD/agents.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3c76d8364248f49a","mcp_get_code":{"code_sha256":"3c76d8364248f49a"}},{"arxiv_id":"2503.00771","paper":"/paper/evaluating-personalized-tool-augmented-llms","title":"Evaluating Personalized Tool-Augmented LLMs from the Perspectives of Personalization and Proactivity","date":"2025-03-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hypasd-art/ETAPP","path":"evaluation/evaluate.py","file_url":"https://github.com/hypasd-art/ETAPP/blob/HEAD/evaluation/evaluate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7ab562f4ba9b1d82","mcp_get_code":{"code_sha256":"7ab562f4ba9b1d82"}},{"arxiv_id":"2411.13112","paper":"/paper/drivemllm-a-benchmark-for-spatial","title":"DriveMLLM: A Benchmark for Spatial Understanding with Multimodal Large Language Models in Autonomous Driving","date":"2024-11-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xiandaguo/drive-mllm","path":"evaluation/eval_from_json.py","file_url":"https://github.com/xiandaguo/drive-mllm/blob/HEAD/evaluation/eval_from_json.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bbd19315c49635ff","mcp_get_code":{"code_sha256":"bbd19315c49635ff"}},{"arxiv_id":"2411.05361","paper":"/paper/dynamic-superb-phase-2-a-collaboratively","title":"Dynamic-SUPERB Phase-2: A Collaboratively Expanding Benchmark for Measuring the Capabilities of Spoken Language Models with 180 Tasks","date":"2024-11-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dynamic-superb/dynamic-superb","path":"api/metrics/audio_duration_prediction.py","file_url":"https://github.com/dynamic-superb/dynamic-superb/blob/HEAD/api/metrics/audio_duration_prediction.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"67c5af6a71776bab","mcp_get_code":{"code_sha256":"67c5af6a71776bab"}},{"arxiv_id":"2411.05361","paper":"/paper/dynamic-superb-phase-2-a-collaboratively","title":"Dynamic-SUPERB Phase-2: A Collaboratively Expanding Benchmark for Measuring the Capabilities of Spoken Language Models with 180 Tasks","date":"2024-11-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dynamic-superb/dynamic-superb","path":"api/metrics/audio_editing_identification.py","file_url":"https://github.com/dynamic-superb/dynamic-superb/blob/HEAD/api/metrics/audio_editing_identification.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ab963e47043f8825","mcp_get_code":{"code_sha256":"ab963e47043f8825"}},{"arxiv_id":"2411.05361","paper":"/paper/dynamic-superb-phase-2-a-collaboratively","title":"Dynamic-SUPERB Phase-2: A Collaboratively Expanding Benchmark for Measuring the Capabilities of Spoken Language Models with 180 Tasks","date":"2024-11-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dynamic-superb/dynamic-superb","path":"api/metrics/audio_spatial_distance.py","file_url":"https://github.com/dynamic-superb/dynamic-superb/blob/HEAD/api/metrics/audio_spatial_distance.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9e34dbd0951e6bdc","mcp_get_code":{"code_sha256":"9e34dbd0951e6bdc"}},{"arxiv_id":"2411.05361","paper":"/paper/dynamic-superb-phase-2-a-collaboratively","title":"Dynamic-SUPERB Phase-2: A Collaboratively Expanding Benchmark for Measuring the Capabilities of Spoken Language Models with 180 Tasks","date":"2024-11-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dynamic-superb/dynamic-superb","path":"api/metrics/code_switching_count.py","file_url":"https://github.com/dynamic-superb/dynamic-superb/blob/HEAD/api/metrics/code_switching_count.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"35daeec0cf300934","mcp_get_code":{"code_sha256":"35daeec0cf300934"}},{"arxiv_id":"2411.05361","paper":"/paper/dynamic-superb-phase-2-a-collaboratively","title":"Dynamic-SUPERB Phase-2: A Collaboratively Expanding Benchmark for Measuring the Capabilities of Spoken Language Models with 180 Tasks","date":"2024-11-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dynamic-superb/dynamic-superb","path":"api/metrics/exact_match.py","file_url":"https://github.com/dynamic-superb/dynamic-superb/blob/HEAD/api/metrics/exact_match.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ef715545375e59c0","mcp_get_code":{"code_sha256":"ef715545375e59c0"}},{"arxiv_id":"2411.05361","paper":"/paper/dynamic-superb-phase-2-a-collaboratively","title":"Dynamic-SUPERB Phase-2: A Collaboratively Expanding Benchmark for Measuring the Capabilities of Spoken Language Models with 180 Tasks","date":"2024-11-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dynamic-superb/dynamic-superb","path":"api/metrics/llm_classification.py","file_url":"https://github.com/dynamic-superb/dynamic-superb/blob/HEAD/api/metrics/llm_classification.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"31d6756ad131ace0","mcp_get_code":{"code_sha256":"31d6756ad131ace0"}},{"arxiv_id":"2411.02359","paper":"/paper/deer-vla-dynamic-inference-of-multimodal","title":"DeeR-VLA: Dynamic Inference of Multimodal Large Language Models for Efficient Robot Execution","date":"2024-11-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yueyang130/DeeR-VLA","path":"bayesian_optimization.py","file_url":"https://github.com/yueyang130/DeeR-VLA/blob/HEAD/bayesian_optimization.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1fff25335fef27a2","mcp_get_code":{"code_sha256":"1fff25335fef27a2"}},{"arxiv_id":"2410.18472","paper":"/paper/what-if-the-input-is-expanded-in-ood","title":"What If the Input is Expanded in OOD Detection?","date":"2024-10-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tmlr-group/CoVer","path":"DNNs/ash.py","file_url":"https://github.com/tmlr-group/CoVer/blob/HEAD/DNNs/ash.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f03fa80d09174dd7","mcp_get_code":{"code_sha256":"f03fa80d09174dd7"}},{"arxiv_id":"2410.16198","paper":"/paper/improve-vision-language-model-chain-of","title":"Improve Vision Language Model Chain-of-thought Reasoning","date":"2024-10-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"riflezhang/llava-hound-dpo","path":"llava_hound_dpo/inference/utils.py","file_url":"https://github.com/riflezhang/llava-hound-dpo/blob/HEAD/llava_hound_dpo/inference/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5321a2aa0d7681cb","mcp_get_code":{"code_sha256":"5321a2aa0d7681cb"}},{"arxiv_id":"2410.10863","paper":"/paper/what-makes-your-model-a-low-empathy-or-warmth","title":"What makes your model a low-empathy or warmth person: Exploring the Origins of Personality in LLMs","date":"2024-10-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kaustpradalab/LLM-Persona-Steering","path":"src/steer_experiments/RepE/analysis.py","file_url":"https://github.com/kaustpradalab/LLM-Persona-Steering/blob/HEAD/src/steer_experiments/RepE/analysis.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6d60c2f13b005719","mcp_get_code":{"code_sha256":"6d60c2f13b005719"}},{"arxiv_id":"2410.10863","paper":"/paper/what-makes-your-model-a-low-empathy-or-warmth","title":"What makes your model a low-empathy or warmth person: Exploring the Origins of Personality in LLMs","date":"2024-10-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kaustpradalab/LLM-Persona-Steering","path":"src/steer_experiments/SAE/analysis_origin.py","file_url":"https://github.com/kaustpradalab/LLM-Persona-Steering/blob/HEAD/src/steer_experiments/SAE/analysis_origin.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"56ea01da64a80821","mcp_get_code":{"code_sha256":"56ea01da64a80821"}},{"arxiv_id":"2409.18017","paper":"/paper/transferring-disentangled-representations","title":"Transferring disentangled representations: bridging the gap between synthetic and real images","date":"2024-09-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"JacopoDapueto/transfer_disentanglement","path":"src/evaluation/metrics/omes.py","file_url":"https://github.com/JacopoDapueto/transfer_disentanglement/blob/HEAD/src/evaluation/metrics/omes.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bbe86af17ad979c9","mcp_get_code":{"code_sha256":"bbe86af17ad979c9"}},{"arxiv_id":"2409.16914","paper":"/paper/zero-shot-detection-of-llm-generated-text","title":"Zero-Shot Detection of LLM-Generated Text using Token Cohesiveness","date":"2024-09-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Shixuan-Ma/TOCSIN","path":"TOCSIN.py","file_url":"https://github.com/Shixuan-Ma/TOCSIN/blob/HEAD/TOCSIN.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9c87e2fd2309802e","mcp_get_code":{"code_sha256":"9c87e2fd2309802e"}},{"arxiv_id":"2409.15254","paper":"/paper/archon-an-architecture-search-framework-for","title":"Archon: An Architecture Search Framework for Inference-Time Techniques","date":"2024-09-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"scalingintelligence/archon","path":"src/archon/benchmarks/arena_hard_auto/gen_judgment.py","file_url":"https://github.com/scalingintelligence/archon/blob/HEAD/src/archon/benchmarks/arena_hard_auto/gen_judgment.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b2d83f0b138fe334","mcp_get_code":{"code_sha256":"b2d83f0b138fe334"}},{"arxiv_id":"2409.12106","paper":"/paper/measuring-human-and-ai-values-based-on","title":"Measuring Human and AI Values Based on Generative Psychometrics with Large Language Models","date":"2024-09-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"value4ai/gpv","path":"gpv/utils.py","file_url":"https://github.com/value4ai/gpv/blob/HEAD/gpv/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"07996379f9e365b6","mcp_get_code":{"code_sha256":"07996379f9e365b6"}},{"arxiv_id":"2409.02834","paper":"/paper/cmm-math-a-chinese-multimodal-math-dataset-to","title":"CMM-Math: A Chinese Multimodal Math Dataset To Evaluate and Enhance the Mathematics Reasoning of Large Multimodal Models","date":"2024-09-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ecnu-icalk/educhat-math","path":"evaluation/gpt-4o-score_evaluation.py","file_url":"https://github.com/ecnu-icalk/educhat-math/blob/HEAD/evaluation/gpt-4o-score_evaluation.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ac28ae0c3a3f6b14","mcp_get_code":{"code_sha256":"ac28ae0c3a3f6b14"}},{"arxiv_id":"2406.11280","paper":"/paper/i-srt-aligning-large-multimodal-models-for","title":"ISR-DPO: Aligning Large Multimodal Models for Videos by Iterative Self-Retrospective DPO","date":"2024-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"snumprlab/SRT","path":"inference/utils.py","file_url":"https://github.com/snumprlab/SRT/blob/HEAD/inference/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5321a2aa0d7681cb","mcp_get_code":{"code_sha256":"5321a2aa0d7681cb"}},{"arxiv_id":"2406.06462","paper":"/paper/vcr-visual-caption-restoration","title":"VCR: A Task for Pixel-Level Complex Reasoning in Vision Language Models via Restoring Occluded Text","date":"2024-06-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tianyu-z/vcr","path":"src/evaluation/gather_results.py","file_url":"https://github.com/tianyu-z/vcr/blob/HEAD/src/evaluation/gather_results.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"CC-BY-SA-4.0","inline_ok":false,"code_sha256_prefix":"9b842bc0490f11b4","mcp_get_code":{"code_sha256":"9b842bc0490f11b4"}},{"arxiv_id":"2405.15077","paper":"/paper/eliciting-informative-text-evaluations-with","title":"Eliciting Informative Text Evaluations with Large Language Models","date":"2024-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yx-lu/Eliciting-Informative-Text-Evaluations-with-Large-Language-Models","path":"experiments/baseline.py","file_url":"https://github.com/yx-lu/Eliciting-Informative-Text-Evaluations-with-Large-Language-Models/blob/HEAD/experiments/baseline.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"CC-BY-4.0","inline_ok":false,"code_sha256_prefix":"fc5eb58baae10b8a","mcp_get_code":{"code_sha256":"fc5eb58baae10b8a"}},{"arxiv_id":"2405.08460","paper":"/paper/evaluating-llms-at-evaluating-temporal","title":"Is Your LLM Outdated? Evaluating LLMs at Temporal Generalization","date":"2024-05-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"freedomintelligence/freshbench","path":"script/score.py","file_url":"https://github.com/freedomintelligence/freshbench/blob/HEAD/script/score.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b21ef2f2569d9b67","mcp_get_code":{"code_sha256":"b21ef2f2569d9b67"}},{"arxiv_id":"2404.07117","paper":"/paper/continuous-language-model-interpolation-for","title":"Continuous Language Model Interpolation for Dynamic and Controllable Text Generation","date":"2024-04-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"skangasl/continuous-lm-interpolation","path":"evaluation/create_plots.py","file_url":"https://github.com/skangasl/continuous-lm-interpolation/blob/HEAD/evaluation/create_plots.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"72a3e012682bb367","mcp_get_code":{"code_sha256":"72a3e012682bb367"}},{"arxiv_id":"2403.16999","paper":"/paper/visual-cot-unleashing-chain-of-thought","title":"Visual CoT: Advancing Multi-Modal Language Models with a Comprehensive Dataset and Benchmark for Chain-of-Thought Reasoning","date":"2024-03-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"deepcs233/visual-cot","path":"llava/eval/eval_cot_score.py","file_url":"https://github.com/deepcs233/visual-cot/blob/HEAD/llava/eval/eval_cot_score.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"5b700f36a74a7df5","mcp_get_code":{"code_sha256":"5b700f36a74a7df5"}},{"arxiv_id":"2402.17700","paper":"/paper/ravel-evaluating-interpretability-methods-on","title":"RAVEL: Evaluating Interpretability Methods on Disentangling Language Model Representations","date":"2024-02-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"explanare/ravel","path":"src/methods/linear_adversarial_probe.py","file_url":"https://github.com/explanare/ravel/blob/HEAD/src/methods/linear_adversarial_probe.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4637704289ff6a8a","mcp_get_code":{"code_sha256":"4637704289ff6a8a"}},{"arxiv_id":"2402.11100","paper":"/paper/when-llms-meet-cunning-questions-a-fallacy","title":"When LLMs Meet Cunning Texts: A Fallacy Understanding Benchmark for Large Language Models","date":"2024-02-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thukelab/flub","path":"code/analysis.py","file_url":"https://github.com/thukelab/flub/blob/HEAD/code/analysis.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"28af50493dc61691","mcp_get_code":{"code_sha256":"28af50493dc61691"}},{"arxiv_id":"2401.03408","paper":"/paper/escalation-risks-from-language-models-in","title":"Escalation Risks from Language Models in Military and Diplomatic Decision-Making","date":"2024-01-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jprivera44/EscalAItion","path":"manual_evaluation.py","file_url":"https://github.com/jprivera44/EscalAItion/blob/HEAD/manual_evaluation.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"74954cd7a1260147","mcp_get_code":{"code_sha256":"74954cd7a1260147"}},{"arxiv_id":"2312.14091","paper":"/paper/hd-painter-high-resolution-and-prompt","title":"HD-Painter: High-Resolution and Prompt-Faithful Text-Guided Image Inpainting with Diffusion Models","date":"2023-12-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"picsart-ai-research/hd-painter","path":"metrics/aesthetic.py","file_url":"https://github.com/picsart-ai-research/hd-painter/blob/HEAD/metrics/aesthetic.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e670aa8de780268c","mcp_get_code":{"code_sha256":"e670aa8de780268c"}},{"arxiv_id":"2311.17009","paper":"/paper/space-time-diffusion-features-for-zero-shot","title":"Space-Time Diffusion Features for Zero-Shot Text-Driven Motion Transfer","date":"2023-11-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"diffusion-motion-transfer/diffusion-motion-transfer","path":"motion_fidelity_score.py","file_url":"https://github.com/diffusion-motion-transfer/diffusion-motion-transfer/blob/HEAD/motion_fidelity_score.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"16286335dc3aec83","mcp_get_code":{"code_sha256":"16286335dc3aec83"}},{"arxiv_id":"2311.16054","paper":"/paper/metric-space-magnitude-for-evaluating","title":"Metric Space Magnitude for Evaluating the Diversity of Latent Representations","date":"2023-11-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"renata-turkes/turkevs2022on","path":"SRC/model.py","file_url":"https://github.com/renata-turkes/turkevs2022on/blob/HEAD/SRC/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"30e11bb16f36a080","mcp_get_code":{"code_sha256":"30e11bb16f36a080"}},{"arxiv_id":"2311.11202","paper":"/paper/unmasking-and-improving-data-credibility-a","title":"Unmasking and Improving Data Credibility: A Study with Datasets for Training Harmless Language Models","date":"2023-11-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Docta-ai/docta","path":"docta/core/knn.py","file_url":"https://github.com/Docta-ai/docta/blob/HEAD/docta/core/knn.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"7e1d6ecb79e8a1bc","mcp_get_code":{"code_sha256":"7e1d6ecb79e8a1bc"}},{"arxiv_id":"2310.16040","paper":"/paper/instruct-and-extract-instruction-tuning-for","title":"Instruct and Extract: Instruction Tuning for On-Demand Information Extraction","date":"2023-10-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yzjiao/on-demand-ie","path":"evaluation/rougel_for_content.py","file_url":"https://github.com/yzjiao/on-demand-ie/blob/HEAD/evaluation/rougel_for_content.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"052514aef99856ca","mcp_get_code":{"code_sha256":"052514aef99856ca"}},{"arxiv_id":"2310.15080","paper":"/paper/federated-learning-of-large-language-models","title":"Federated Learning of Large Language Models with Parameter-Efficient Prompt Tuning and Adaptive Optimization","date":"2023-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"llm-eff/FedPepTAO","path":"decoder-only-gpt2/get_score.py","file_url":"https://github.com/llm-eff/FedPepTAO/blob/HEAD/decoder-only-gpt2/get_score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"7b1aa9397d2f3a55","mcp_get_code":{"code_sha256":"7b1aa9397d2f3a55"}},{"arxiv_id":"2310.15080","paper":"/paper/federated-learning-of-large-language-models","title":"Federated Learning of Large Language Models with Parameter-Efficient Prompt Tuning and Adaptive Optimization","date":"2023-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"llm-eff/FedPepTAO","path":"decoder-only-llama/get_score.py","file_url":"https://github.com/llm-eff/FedPepTAO/blob/HEAD/decoder-only-llama/get_score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"48cf69dafa7e622e","mcp_get_code":{"code_sha256":"48cf69dafa7e622e"}},{"arxiv_id":"2310.15080","paper":"/paper/federated-learning-of-large-language-models","title":"Federated Learning of Large Language Models with Parameter-Efficient Prompt Tuning and Adaptive Optimization","date":"2023-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"llm-eff/FedPepTAO","path":"encoder-only-roberta-large/get_score.py","file_url":"https://github.com/llm-eff/FedPepTAO/blob/HEAD/encoder-only-roberta-large/get_score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"822768feef873f11","mcp_get_code":{"code_sha256":"822768feef873f11"}},{"arxiv_id":"2310.14566","paper":"/paper/hallusionbench-you-see-what-you-think-or-you","title":"HallusionBench: An Advanced Diagnostic Suite for Entangled Language Hallucination and Visual Illusion in Large Vision-Language Models","date":"2023-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zli12321/qa_metrics","path":"qa_metrics/RewardBert.py","file_url":"https://github.com/zli12321/qa_metrics/blob/HEAD/qa_metrics/RewardBert.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bb06d405aa0fd21d","mcp_get_code":{"code_sha256":"bb06d405aa0fd21d"}},{"arxiv_id":"2309.07852","paper":"/paper/expertqa-expert-curated-questions-and","title":"ExpertQA: Expert-Curated Questions and Attributed Answers","date":"2023-09-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"chaitanyamalaviya/expertqa","path":"modeling/fact_score/factscore.py","file_url":"https://github.com/chaitanyamalaviya/expertqa/blob/HEAD/modeling/fact_score/factscore.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6c4b02266edc91ba","mcp_get_code":{"code_sha256":"6c4b02266edc91ba"}},{"arxiv_id":"2308.10278","paper":"/paper/characterchat-learning-towards-conversational","title":"CharacterChat: Learning towards Conversational AI with Personalized Social Support","date":"2023-08-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"morecry/characterchat","path":"model/demo/chat_demo.py","file_url":"https://github.com/morecry/characterchat/blob/HEAD/model/demo/chat_demo.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5efb4305d6662763","mcp_get_code":{"code_sha256":"5efb4305d6662763"}},{"arxiv_id":"2308.09033","paper":"/paper/uni-nlx-unifying-textual-explanations-for","title":"Uni-NLX: Unifying Textual Explanations for Vision and Vision-Language Tasks","date":"2023-08-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"fawazsammani/nlxgpt","path":"explain_predict/ep_vqaX.py","file_url":"https://github.com/fawazsammani/nlxgpt/blob/HEAD/explain_predict/ep_vqaX.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"04a66156efbb1777","mcp_get_code":{"code_sha256":"04a66156efbb1777"}},{"arxiv_id":"2306.14899","paper":"/paper/funqa-towards-surprising-video-comprehension","title":"FunQA: Towards Surprising Video Comprehension","date":"2023-06-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jingkang50/funqa","path":"gpt4_eval.py","file_url":"https://github.com/jingkang50/funqa/blob/HEAD/gpt4_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b1ee5d8a47a7d92a","mcp_get_code":{"code_sha256":"b1ee5d8a47a7d92a"}},{"arxiv_id":"2306.11911","paper":"/paper/lnl-k-learning-with-noisy-labels-and-noise","title":"LNL+K: Enhancing Learning with Noisy Labels Through Noise Source Knowledge Integration","date":"2023-06-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sunnysiqi/lnl_k","path":"adaptation_methods/fine_k.py","file_url":"https://github.com/sunnysiqi/lnl_k/blob/HEAD/adaptation_methods/fine_k.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"56ccb4fcec6f5a98","mcp_get_code":{"code_sha256":"56ccb4fcec6f5a98"}},{"arxiv_id":"2306.03438","paper":"/paper/large-language-models-of-code-fail-at","title":"Large Language Models of Code Fail at Completing Code with Potential Bugs","date":"2023-06-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"amazon-science/buggy-code-completion","path":"src/infiller/infill_line.py","file_url":"https://github.com/amazon-science/buggy-code-completion/blob/HEAD/src/infiller/infill_line.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"5f28bf9a2c267875","mcp_get_code":{"code_sha256":"5f28bf9a2c267875"}},{"arxiv_id":"2305.12707","paper":"/paper/quantifying-association-capabilities-of-large","title":"Quantifying Association Capabilities of Large Language Models and Its Implications on Privacy Leakage","date":"2023-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hanyins/lm_association_quantification","path":"LAMA/analysis.py","file_url":"https://github.com/hanyins/lm_association_quantification/blob/HEAD/LAMA/analysis.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"54f850b6751cfb2e","mcp_get_code":{"code_sha256":"54f850b6751cfb2e"}},{"arxiv_id":"2302.00796","paper":"/paper/unsupervised-entity-alignment-for-temporal","title":"Unsupervised Entity Alignment for Temporal Knowledge Graphs","date":"2023-02-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zju-daily/dualmatch","path":"processing.py","file_url":"https://github.com/zju-daily/dualmatch/blob/HEAD/processing.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"68824ba6e717496e","mcp_get_code":{"code_sha256":"68824ba6e717496e"}},{"arxiv_id":"2206.08220","paper":"/paper/functional-output-regression-with-infimal","title":"Functional Output Regression with Infimal Convolution: Exploring the Huber and $ε$-insensitive Losses","date":"2022-06-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"allambert/foreg","path":"model_selection/model_selection.py","file_url":"https://github.com/allambert/foreg/blob/HEAD/model_selection/model_selection.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"00e351b580b1143b","mcp_get_code":{"code_sha256":"00e351b580b1143b"}},{"arxiv_id":"2205.09739","paper":"/paper/diverse-weight-averaging-for-out-of","title":"Diverse Weight Averaging for Out-of-Distribution Generalization","date":"2022-05-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alexrame/diwa","path":"domainbed/lib/misc.py","file_url":"https://github.com/alexrame/diwa/blob/HEAD/domainbed/lib/misc.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"837d97171609bbbf","mcp_get_code":{"code_sha256":"837d97171609bbbf"}},{"arxiv_id":"2204.06507","paper":"/paper/out-of-distribution-detection-with-deep","title":"Out-of-Distribution Detection with Deep Nearest Neighbors","date":"2022-04-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"deeplearning-wisc/knn-ood","path":"util/score.py","file_url":"https://github.com/deeplearning-wisc/knn-ood/blob/HEAD/util/score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b874b3a0ce06fd5f","mcp_get_code":{"code_sha256":"b874b3a0ce06fd5f"}},{"arxiv_id":"2201.12091","paper":"/paper/linear-adversarial-concept-erasure","title":"Linear Adversarial Concept Erasure","date":"2022-01-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shauli-ravfogel/rlace-icml","path":"rlace.py","file_url":"https://github.com/shauli-ravfogel/rlace-icml/blob/HEAD/rlace.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d7edaf170e8ed046","mcp_get_code":{"code_sha256":"d7edaf170e8ed046"}},{"arxiv_id":"2201.12091","paper":"/paper/linear-adversarial-concept-erasure","title":"Linear Adversarial Concept Erasure","date":"2022-01-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shauli-ravfogel/rlace-icml","path":"rlace.py","file_url":"https://github.com/shauli-ravfogel/rlace-icml/blob/HEAD/rlace.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b765918151ddbd4e","mcp_get_code":{"code_sha256":"b765918151ddbd4e"}},{"arxiv_id":"2201.12091","paper":"/paper/linear-adversarial-concept-erasure","title":"Linear Adversarial Concept Erasure","date":"2022-01-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shauli-ravfogel/adv-kernel-removal","path":"relaxed_inlp.py","file_url":"https://github.com/shauli-ravfogel/adv-kernel-removal/blob/HEAD/relaxed_inlp.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2e9f7545f8882b0b","mcp_get_code":{"code_sha256":"2e9f7545f8882b0b"}},{"arxiv_id":"2111.07677","paper":"/paper/fastflow-unsupervised-anomaly-detection-and","title":"FastFlow: Unsupervised Anomaly Detection and Localization via 2D Normalizing Flows","date":"2021-11-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"RistoranteRist/FastFlow","path":"utils.py","file_url":"https://github.com/RistoranteRist/FastFlow/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2f2d58311dd3969d","mcp_get_code":{"code_sha256":"2f2d58311dd3969d"}},{"arxiv_id":"2107.01904","paper":"/paper/ensemble-and-auxiliary-tasks-for-data","title":"Ensemble and Auxiliary Tasks for Data-Efficient Deep Reinforcement Learning","date":"2021-07-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"NUS-LID/RENAULT","path":"benchmark.py","file_url":"https://github.com/NUS-LID/RENAULT/blob/HEAD/benchmark.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"640097e0a5b3d620","mcp_get_code":{"code_sha256":"640097e0a5b3d620"}},{"arxiv_id":"2102.11628","paper":"/paper/winning-ticket-in-noisy-image-classification","title":"FINE Samples for Learning with Noisy Labels","date":"2021-02-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Kthyeon/FINE_official","path":"dynamic_selection/selection/svd_classifier.py","file_url":"https://github.com/Kthyeon/FINE_official/blob/HEAD/dynamic_selection/selection/svd_classifier.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1b10ac39415f471f","mcp_get_code":{"code_sha256":"1b10ac39415f471f"}},{"arxiv_id":"1910.04376","paper":"/paper/rlcard-a-toolkit-for-reinforcement-learning","title":"RLCard: A Toolkit for Reinforcement Learning in Card Games","date":"2019-10-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"datamllab/rlcard","path":"rlcard/envs/blackjack.py","file_url":"https://github.com/datamllab/rlcard/blob/HEAD/rlcard/envs/blackjack.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6c08fd179035271c","mcp_get_code":{"code_sha256":"6c08fd179035271c"}},{"arxiv_id":"1908.07500","paper":"/paper/image-synthesis-from-reconfigurable-layout","title":"Image Synthesis From Reconfigurable Layout and Style","date":"2019-08-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"stanifrolov/attrlostgan","path":"eval/attr_f1.py","file_url":"https://github.com/stanifrolov/attrlostgan/blob/HEAD/eval/attr_f1.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f19ef3041e7db2b2","mcp_get_code":{"code_sha256":"f19ef3041e7db2b2"}},{"arxiv_id":"1703.01008","paper":"/paper/end-to-end-task-completion-neural-dialogue","title":"End-to-End Task-Completion Neural Dialogue Systems","date":"2017-03-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AtmaHou/UserSimulator","path":"src/collect_result.py","file_url":"https://github.com/AtmaHou/UserSimulator/blob/HEAD/src/collect_result.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"cdefc0c373bb5645","mcp_get_code":{"code_sha256":"cdefc0c373bb5645"}},{"arxiv_id":"1611.08050","paper":"/paper/realtime-multi-person-2d-pose-estimation","title":"Realtime Multi-Person 2D Pose Estimation using Part Affinity Fields","date":"2016-11-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ibm/max-human-pose-estimator","path":"core/tf_pose/pafprocess/pafprocess.py","file_url":"https://github.com/ibm/max-human-pose-estimator/blob/HEAD/core/tf_pose/pafprocess/pafprocess.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"418d24f644276625","mcp_get_code":{"code_sha256":"418d24f644276625"}},{"arxiv_id":"aaai_34690","paper":null,"title":"arXiv:aaai_34690","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"ZongyueQin/DSBD","path":"evaluation.py","file_url":"https://github.com/ZongyueQin/DSBD/blob/HEAD/evaluation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"214e835658c1d3c9","mcp_get_code":{"code_sha256":"214e835658c1d3c9"}},{"arxiv_id":"2025.findings-acl.931","paper":null,"title":"arXiv:2025.findings-acl.931","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"ikergarcia1996/T-Projection","path":"label_projection.py","file_url":"https://github.com/ikergarcia1996/T-Projection/blob/HEAD/label_projection.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"fafc9230eec0ad80","mcp_get_code":{"code_sha256":"fafc9230eec0ad80"}},{"arxiv_id":"2025.findings-acl.1384","paper":null,"title":"arXiv:2025.findings-acl.1384","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"Muennighoff/sgpt","path":"crossencoder/beir/openai_search_endpoint_functionality.py","file_url":"https://github.com/Muennighoff/sgpt/blob/HEAD/crossencoder/beir/openai_search_endpoint_functionality.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e71e0d710acd1e64","mcp_get_code":{"code_sha256":"e71e0d710acd1e64"}}]}