{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/score","entry":"score","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":54,"n_papers_ran":22,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":57,"n_samples_ran":22,"n_samples_fingerprinted":1,"n_places":60,"n_places_pointer_only":23,"by_status":{"ran_honours":5,"ran_violates":0,"ran_draft_wrong":3,"ran_fixture":3,"ran":11,"unverified":35},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.13279","paper":"/paper/arxiv-2609-13279","title":"The MODA General Attribute Suite: A Four-Track Evaluation Benchmark for Fashion Attribute Extraction","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"hopit-ai/Moda_ner","path":"suite/text/score.py","file_url":"https://github.com/hopit-ai/Moda_ner/blob/HEAD/suite/text/score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7d080e2859a10e18","mcp_get_code":{"code_sha256":"7d080e2859a10e18"}},{"arxiv_id":"2606.18910","paper":"/paper/arxiv-2606-18910","title":"REVES: REvision and VErification-Augmented Training for Test-Time Scaling","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"yxliu02/REVES","path":"evaluator/eval_travelplanner.py","file_url":"https://github.com/yxliu02/REVES/blob/HEAD/evaluator/eval_travelplanner.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2dddcb48090ef99f","mcp_get_code":{"code_sha256":"2dddcb48090ef99f"}},{"arxiv_id":"2606.13818","paper":"/paper/arxiv-2606-13818","title":"PAC-Chernoff Bounds: Understanding Generalization in the Interpolation Regime","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"Ludvins/FixedMeanGaussianProcesses","path":"bayesipy/utils/metrics.py","file_url":"https://github.com/Ludvins/FixedMeanGaussianProcesses/blob/HEAD/bayesipy/utils/metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9110c7d87661f572","mcp_get_code":{"code_sha256":"9110c7d87661f572"}},{"arxiv_id":"2605.12857","paper":"/paper/arxiv-2605-12857","title":"ChipMATE: Multi-Agent Training via Reinforcement Learning for Enhanced RTL Generation","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"zhongkaiyu/ChipMATE","path":"paper_repro/chipbench/chipbench_score.py","file_url":"https://github.com/zhongkaiyu/ChipMATE/blob/HEAD/paper_repro/chipbench/chipbench_score.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"dd0f65afa280200f","mcp_get_code":{"code_sha256":"dd0f65afa280200f"}},{"arxiv_id":"2605.12857","paper":"/paper/arxiv-2605-12857","title":"ChipMATE: Multi-Agent Training via Reinforcement Learning for Enhanced RTL Generation","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"zhongkaiyu/ChipMATE","path":"paper_repro/chipbench/chipbench_framework.py","file_url":"https://github.com/zhongkaiyu/ChipMATE/blob/HEAD/paper_repro/chipbench/chipbench_framework.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"c6869481cba9dde6","mcp_get_code":{"code_sha256":"c6869481cba9dde6"}},{"arxiv_id":"2605.12857","paper":"/paper/arxiv-2605-12857","title":"ChipMATE: Multi-Agent Training via Reinforcement Learning for Enhanced RTL Generation","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"zhongkaiyu/ChipMATE","path":"paper_repro/rtllm/rtllm_framework.py","file_url":"https://github.com/zhongkaiyu/ChipMATE/blob/HEAD/paper_repro/rtllm/rtllm_framework.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"a222280252a2db47","mcp_get_code":{"code_sha256":"a222280252a2db47"}},{"arxiv_id":"2605.09623","paper":"/paper/arxiv-2605-09623","title":"Adaptive AI Task Partitioning and Safe Offloading in Heterogeneous Edge-Cloud Continuum Akuen Akoi Deng ⋆[0009-0007-6228-3340] , Eimantas Butkus ⋆[0009-0001-5647-0779] , Alfreds Lapkovskis [0009-0003-4424-949X] , and","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"Akuien/DNN-partitioning-and-Oflloading-framework-REAP","path":"adaptive-framework/split_infer.py","file_url":"https://github.com/Akuien/DNN-partitioning-and-Oflloading-framework-REAP/blob/HEAD/adaptive-framework/split_infer.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"72847bdb72a3845e","mcp_get_code":{"code_sha256":"72847bdb72a3845e"}},{"arxiv_id":"2605.01336","paper":"/paper/arxiv-2605-01336","title":"A Multi-View Media Profiling Suite: Resources, Evaluation, and Analysis","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"codelucas/newspaper","path":"newspaper/nlp.py","file_url":"https://github.com/codelucas/newspaper/blob/HEAD/newspaper/nlp.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"491f58b8741fe8d7","mcp_get_code":{"code_sha256":"491f58b8741fe8d7"}},{"arxiv_id":"2604.08559","paper":"/paper/arxiv-2604-08559","title":"Medical Reasoning with Large Language Models: A Survey and MR-Bench","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"RXH04-USTC/Medical-Reasoning-Survey","path":"src_eval/scorer.py","file_url":"https://github.com/RXH04-USTC/Medical-Reasoning-Survey/blob/HEAD/src_eval/scorer.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"abe344037ab0d857","mcp_get_code":{"code_sha256":"abe344037ab0d857"}},{"arxiv_id":"2601.02163","paper":"/paper/arxiv-2601-02163","title":"EverMemOS: A Self-Organizing Memory Operating System for Structured Long-Horizon Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"EverMind-AI/EverMemOS","path":"benchmarks/metrics/core.py","file_url":"https://github.com/EverMind-AI/EverMemOS/blob/HEAD/benchmarks/metrics/core.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0fcb7854e27f1cad","mcp_get_code":{"code_sha256":"0fcb7854e27f1cad"}},{"arxiv_id":"2601.02163","paper":"/paper/arxiv-2601-02163","title":"EverMemOS: A Self-Organizing Memory Operating System for Structured Long-Horizon Reasoning","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"EverMind-AI/EverMemOS","path":"benchmarks/metrics/ir.py","file_url":"https://github.com/EverMind-AI/EverMemOS/blob/HEAD/benchmarks/metrics/ir.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f05168abaac86222","mcp_get_code":{"code_sha256":"f05168abaac86222"}},{"arxiv_id":"2512.01048","paper":"/paper/arxiv-2512-01048","title":"TROVE: Discovering Error-Inducing Static Feature Biases in Temporal Vision-Language Models","date":null,"month_inferred_from_arxiv_id":"2025-12","title_source":"syntology","repo":"Stanford-AIMI/TRoVe","path":"src/trove/score.py","file_url":"https://github.com/Stanford-AIMI/TRoVe/blob/HEAD/src/trove/score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"10b39a57b20324c8","mcp_get_code":{"code_sha256":"10b39a57b20324c8"}},{"arxiv_id":"2508.18379","paper":"/paper/arxiv-2508-18379","title":"REALM: Recursive Relevance Modeling for LLM-based Document Re-Ranking *","date":null,"month_inferred_from_arxiv_id":"2025-08","title_source":"syntology","repo":"Joeyw02/REALM","path":"src/algorithm.py","file_url":"https://github.com/Joeyw02/REALM/blob/HEAD/src/algorithm.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"051a1bed8cba2a19","mcp_get_code":{"code_sha256":"051a1bed8cba2a19"}},{"arxiv_id":"2507.00322","paper":null,"title":"arXiv:2507.00322","date":null,"month_inferred_from_arxiv_id":"2025-07","title_source":null,"repo":"EleutherAI/pythia","path":"predictable-memorization/eval_memorization.py","file_url":"https://github.com/EleutherAI/pythia/blob/HEAD/predictable-memorization/eval_memorization.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"81bc28dc8aceb9c2","mcp_get_code":{"code_sha256":"81bc28dc8aceb9c2"}},{"arxiv_id":"2506.18841","paper":"/paper/longwriter-zero-mastering-ultra-long-text","title":"LongWriter-Zero: Mastering Ultra-Long Text Generation via Reinforcement Learning","date":"2025-06-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thudm/longwriter","path":"evaluation/eval_length.py","file_url":"https://github.com/thudm/longwriter/blob/HEAD/evaluation/eval_length.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"653c358686bbb406","mcp_get_code":{"code_sha256":"653c358686bbb406"}},{"arxiv_id":"2505.16620","paper":"/paper/causaldynamics-a-large-scale-benchmark-for","title":"CausalDynamics: A large-scale benchmark for structural discovery of dynamical causal models","date":"2025-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kausable/CausalDynamics","path":"src/causaldynamics/score.py","file_url":"https://github.com/kausable/CausalDynamics/blob/HEAD/src/causaldynamics/score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"fe027f1e8d858d64","mcp_get_code":{"code_sha256":"fe027f1e8d858d64"}},{"arxiv_id":"2505.14828","paper":"/paper/deep-koopman-operator-framework-for-causal","title":"Deep Koopman operator framework for causal discovery in nonlinear dynamical systems","date":"2025-05-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"juannat7/kausal","path":"kausal/utils.py","file_url":"https://github.com/juannat7/kausal/blob/HEAD/kausal/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f1286fc26e8a683a","mcp_get_code":{"code_sha256":"f1286fc26e8a683a"}},{"arxiv_id":"2504.11844","paper":"/paper/evaluating-the-goal-directedness-of-large","title":"Evaluating the Goal-Directedness of Large Language Models","date":"2025-04-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Crista23/goal_directedness_llms","path":"tasks/full_task.py","file_url":"https://github.com/Crista23/goal_directedness_llms/blob/HEAD/tasks/full_task.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bb22736701cd8662","mcp_get_code":{"code_sha256":"bb22736701cd8662"}},{"arxiv_id":"2410.02284","paper":"/paper/correlation-and-navigation-in-the-vocabulary","title":"Correlation and Navigation in the Vocabulary Key Representation Space of Language Models","date":"2024-10-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"KomeijiForce/KeyNavi","path":"zerogen.py","file_url":"https://github.com/KomeijiForce/KeyNavi/blob/HEAD/zerogen.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"19516e6551d27215","mcp_get_code":{"code_sha256":"19516e6551d27215"}},{"arxiv_id":"2409.19177","paper":"/paper/evidence-is-all-you-need-ordering-imaging","title":"Evidence Is All You Need: Ordering Imaging Studies via Language Model Alignment with the ACR Appropriateness Criteria","date":"2024-09-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"michael-s-yao/radGPT","path":"radgpt/utils.py","file_url":"https://github.com/michael-s-yao/radGPT/blob/HEAD/radgpt/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f69f9d0af8972175","mcp_get_code":{"code_sha256":"f69f9d0af8972175"}},{"arxiv_id":"2407.12927","paper":"/paper/text-and-feature-based-models-for-compound","title":"Textualized and Feature-based Models for Compound Multimodal Emotion Recognition in the Wild","date":"2024-07-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nicolas-richet/feature-vs-text-compound-emotion","path":"model.py","file_url":"https://github.com/nicolas-richet/feature-vs-text-compound-emotion/blob/HEAD/model.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"fc5685919b105cb5","mcp_get_code":{"code_sha256":"fc5685919b105cb5"}},{"arxiv_id":"2404.09788","paper":"/paper/shape-arithmetic-expressions-advancing","title":"Shape Arithmetic Expressions: Advancing Scientific Discovery Beyond Closed-Form Equations","date":"2024-04-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"krzysztof-kacprzyk/shares","path":"experiments/benchmarks.py","file_url":"https://github.com/krzysztof-kacprzyk/shares/blob/HEAD/experiments/benchmarks.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cb821c0bc00a2598","mcp_get_code":{"code_sha256":"cb821c0bc00a2598"}},{"arxiv_id":"2312.03207","paper":"/paper/satellite-imagery-and-ai-a-new-era-in-ocean","title":"Satellite Imagery and AI: A New Era in Ocean Conservation, from Research to Deployment and Impact","date":"2023-12-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"allenai/vessel-detection-sentinels","path":"src/training/metric.py","file_url":"https://github.com/allenai/vessel-detection-sentinels/blob/HEAD/src/training/metric.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"eb4e28dc5f3a8176","mcp_get_code":{"code_sha256":"eb4e28dc5f3a8176"}},{"arxiv_id":"2310.13189","paper":"/paper/fast-and-accurate-factual-inconsistency","title":"Fast and Accurate Factual Inconsistency Detection Over Long Documents","date":"2023-10-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"asappresearch/scale-score","path":"scale_score/scorer.py","file_url":"https://github.com/asappresearch/scale-score/blob/HEAD/scale_score/scorer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0d59d2ed3cc16146","mcp_get_code":{"code_sha256":"0d59d2ed3cc16146"}},{"arxiv_id":"2310.12611","paper":"/paper/identifying-and-adapting-transformer","title":"Identifying and Adapting Transformer-Components Responsible for Gender Bias in an English Language Model","date":"2023-10-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"iabhijith/bias-causal-analysis","path":"evaluation/blimp.py","file_url":"https://github.com/iabhijith/bias-causal-analysis/blob/HEAD/evaluation/blimp.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e7c03f0573dfb130","mcp_get_code":{"code_sha256":"e7c03f0573dfb130"}},{"arxiv_id":"2310.08319","paper":"/paper/fine-tuning-llama-for-multi-stage-text","title":"Fine-Tuning LLaMA for Multi-Stage Text Retrieval","date":"2023-10-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"texttron/tevatron","path":"src/tevatron/eval/metrics.py","file_url":"https://github.com/texttron/tevatron/blob/HEAD/src/tevatron/eval/metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"5eb05d1d51ec1fa5","mcp_get_code":{"code_sha256":"5eb05d1d51ec1fa5"}},{"arxiv_id":"2310.05401","paper":"/paper/entropy-mcmc-sampling-from-flat-basins-with","title":"Entropy-MCMC: Sampling from Flat Basins with Ease","date":"2023-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lblaoke/emcmc","path":"metric.py","file_url":"https://github.com/lblaoke/emcmc/blob/HEAD/metric.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"76887d631d72b05f","mcp_get_code":{"code_sha256":"76887d631d72b05f"}},{"arxiv_id":"2308.11606","paper":"/paper/storybench-a-multifaceted-benchmark-for-1","title":"StoryBench: A Multifaceted Benchmark for Continuous Story Visualization","date":"2023-08-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"google/storybench","path":"metrics/vtm_clip.py","file_url":"https://github.com/google/storybench/blob/HEAD/metrics/vtm_clip.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b482b6a0ac227ae9","mcp_get_code":{"code_sha256":"b482b6a0ac227ae9"}},{"arxiv_id":"2308.02773","paper":"/paper/educhat-a-large-scale-language-model-based","title":"EduChat: A Large-Scale Language Model-based Chatbot System for Intelligent Education","date":"2023-08-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"icalk-nlp/educhat","path":"demo/score_utils.py","file_url":"https://github.com/icalk-nlp/educhat/blob/HEAD/demo/score_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"097528b8a72d71a7","mcp_get_code":{"code_sha256":"097528b8a72d71a7"}},{"arxiv_id":"2307.03027","paper":"/paper/improving-retrieval-augmented-large-language","title":"Improving Retrieval-Augmented Large Language Models via Data Importance Learning","date":"2023-07-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"amsterdata/ragbooster","path":"python/ragbooster/core.py","file_url":"https://github.com/amsterdata/ragbooster/blob/HEAD/python/ragbooster/core.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"cfd204a25911c17c","mcp_get_code":{"code_sha256":"cfd204a25911c17c"}},{"arxiv_id":"2304.11158","paper":"/paper/emergent-and-predictable-memorization-in-1","title":"Emergent and Predictable Memorization in Large Language Models","date":"2023-04-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"eleutherai/pythia","path":"predictable-memorization/eval_memorization.py","file_url":"https://github.com/eleutherai/pythia/blob/HEAD/predictable-memorization/eval_memorization.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"81bc28dc8aceb9c2","mcp_get_code":{"code_sha256":"81bc28dc8aceb9c2"}},{"arxiv_id":"2212.11672","paper":"/paper/trustworthy-social-bias-measurement","title":"Trustworthy Social Bias Measurement","date":"2022-12-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rishibommasani/biasmeasures","path":"convergent_validity.py","file_url":"https://github.com/rishibommasani/biasmeasures/blob/HEAD/convergent_validity.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0569862de44357f1","mcp_get_code":{"code_sha256":"0569862de44357f1"}},{"arxiv_id":"2210.03992","paper":"/paper/generative-language-models-for-paragraph","title":"Generative Language Models for Paragraph-Level Question Generation","date":"2022-10-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"asahi417/lm-question-generation","path":"lmqg/automatic_evaluation_tool/bert_score/score.py","file_url":"https://github.com/asahi417/lm-question-generation/blob/HEAD/lmqg/automatic_evaluation_tool/bert_score/score.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a57915bf6208b453","mcp_get_code":{"code_sha256":"a57915bf6208b453"}},{"arxiv_id":"2209.13654","paper":"/paper/embarrassingly-easy-document-level-mt-metrics","title":"Embarrassingly Easy Document-Level MT Metrics: How to Convert Any Pretrained Metric Into a Document-Level Metric","date":"2022-09-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"amazon-science/doc-mt-metrics","path":"bert_score/bert_score/score.py","file_url":"https://github.com/amazon-science/doc-mt-metrics/blob/HEAD/bert_score/bert_score/score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"efd9b25cfcd27f97","mcp_get_code":{"code_sha256":"efd9b25cfcd27f97"}},{"arxiv_id":"2205.08514","paper":"/paper/recovering-private-text-in-federated-learning","title":"Recovering Private Text in Federated Learning of Language Models","date":"2022-05-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"princeton-sysml/film","path":"reorder.py","file_url":"https://github.com/princeton-sysml/film/blob/HEAD/reorder.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"CC0-1.0","inline_ok":true,"code_sha256_prefix":"b902d2253e571934","mcp_get_code":{"code_sha256":"b902d2253e571934"}},{"arxiv_id":"2202.08479","paper":"/paper/revisiting-the-evaluation-metrics-of","title":"On the Evaluation Metrics for Paraphrase Generation","date":"2022-02-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shadowkiller33/parascore","path":"bert_score/score.py","file_url":"https://github.com/shadowkiller33/parascore/blob/HEAD/bert_score/score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8f1719ca230ca245","mcp_get_code":{"code_sha256":"8f1719ca230ca245"}},{"arxiv_id":"2111.02813","paper":"/paper/wavefake-a-data-set-to-facilitate-audio","title":"WaveFake: A Data Set to Facilitate Audio Deepfake Detection","date":"2021-11-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rub-syssec/wavefake","path":"dfadetect/models/gaussian_mixture_model.py","file_url":"https://github.com/rub-syssec/wavefake/blob/HEAD/dfadetect/models/gaussian_mixture_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"371c6c7eedd45605","mcp_get_code":{"code_sha256":"371c6c7eedd45605"}},{"arxiv_id":"2110.08288","paper":"/paper/explaining-deep-learning-of-galaxy-morphology","title":"Explaining deep learning of galaxy morphology with saliency mapping","date":null,"month_inferred_from_arxiv_id":"2021-10","title_source":"archive","repo":"prabhbhambra13/xai_bar_lengths","path":"generate_heatmaps.py","file_url":"https://github.com/prabhbhambra13/xai_bar_lengths/blob/HEAD/generate_heatmaps.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"296ac592eba01d02","mcp_get_code":{"code_sha256":"296ac592eba01d02"}},{"arxiv_id":"2109.11503","paper":"/paper/finding-a-balanced-degree-of-automation-for","title":"Finding a Balanced Degree of Automation for Summary Evaluation","date":"2021-09-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ZhangShiyue/Lite2-3Pyramid","path":"metric/score.py","file_url":"https://github.com/ZhangShiyue/Lite2-3Pyramid/blob/HEAD/metric/score.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"56276e9a4bd27b29","mcp_get_code":{"code_sha256":"56276e9a4bd27b29"}},{"arxiv_id":"2109.04867","paper":"/paper/studying-word-order-through-iterative","title":"Studying word order through iterative shuffling","date":"2021-09-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"malkin1729/ibis","path":"code/ibis.py","file_url":"https://github.com/malkin1729/ibis/blob/HEAD/code/ibis.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"dea9d346ab79356f","mcp_get_code":{"code_sha256":"dea9d346ab79356f"}},{"arxiv_id":"2109.04711","paper":"/paper/pre-train-or-annotate-domain-adaptation-with","title":"Pre-train or Annotate? Domain Adaptation with a Constrained Budget","date":"2021-09-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"allenai/kb","path":"bin/tacred_scorer.py","file_url":"https://github.com/allenai/kb/blob/HEAD/bin/tacred_scorer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"9cb44938e773d88b","mcp_get_code":{"code_sha256":"9cb44938e773d88b"}},{"arxiv_id":"2109.04470","paper":"/paper/truth-discovery-in-sequence-labels-from","title":"Truth Discovery in Sequence Labels from Crowds","date":"2021-09-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nasimisu/truth-discovery-in-sequence-labels-from-crowds","path":"calculations.py","file_url":"https://github.com/nasimisu/truth-discovery-in-sequence-labels-from-crowds/blob/HEAD/calculations.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"35705fd6f9774602","mcp_get_code":{"code_sha256":"35705fd6f9774602"}},{"arxiv_id":"2106.02208","paper":"/paper/berttune-fine-tuning-neural-machine","title":"BERTTune: Fine-Tuning Neural Machine Translation with BERTScore","date":"2021-06-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ijauregiCMCRC/fairseq-bert-loss","path":"fairseq/bert_score/score.py","file_url":"https://github.com/ijauregiCMCRC/fairseq-bert-loss/blob/HEAD/fairseq/bert_score/score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"0a91572da431f7d0","mcp_get_code":{"code_sha256":"0a91572da431f7d0"}},{"arxiv_id":"2012.06421","paper":"/paper/when-is-memorization-of-irrelevant-training","title":"When is Memorization of Irrelevant Training Data Necessary for High-Accuracy Learning?","date":"2020-12-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gavinrbrown1/training-data-memorization","path":"attacks.py","file_url":"https://github.com/gavinrbrown1/training-data-memorization/blob/HEAD/attacks.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f80d3b29688c4d7a","mcp_get_code":{"code_sha256":"f80d3b29688c4d7a"}},{"arxiv_id":"2010.07518","paper":"/paper/automatic-analysis-and-influence-of","title":"Automatic Analysis and Influence of Hierarchical Structure on Melody, Rhythm and Harmony in Popular Music","date":null,"month_inferred_from_arxiv_id":"2020-10","title_source":"archive","repo":"Dsqvival/hierarchical-structure-analysis","path":"preprocessing/key_finding.py","file_url":"https://github.com/Dsqvival/hierarchical-structure-analysis/blob/HEAD/preprocessing/key_finding.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e46c2500ec92d4a2","mcp_get_code":{"code_sha256":"e46c2500ec92d4a2"}},{"arxiv_id":"2006.12719","paper":"/paper/unsupervised-evaluation-of-interactive-dialog","title":"Unsupervised Evaluation of Interactive Dialog with DialoGPT","date":"2020-06-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shikib/fed","path":"fed.py","file_url":"https://github.com/shikib/fed/blob/HEAD/fed.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e5b4d588ad5ce486","mcp_get_code":{"code_sha256":"e5b4d588ad5ce486"}},{"arxiv_id":"2005.10481","paper":"/paper/aows-adaptive-and-optimal-network-width","title":"AOWS: Adaptive and optimal network width search with latency constraints","date":"2020-05-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bermanmaxim/AOWS","path":"viterbi.py","file_url":"https://github.com/bermanmaxim/AOWS/blob/HEAD/viterbi.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"dfc753382f18458e","mcp_get_code":{"code_sha256":"dfc753382f18458e"}},{"arxiv_id":"2004.14523","paper":"/paper/exploiting-sentence-order-in-document","title":"Exploiting Sentence Order in Document Alignment","date":"2020-04-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"facebookresearch/LASER","path":"source/mine_bitexts.py","file_url":"https://github.com/facebookresearch/LASER/blob/HEAD/source/mine_bitexts.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"ce7491b4961154f3","mcp_get_code":{"code_sha256":"ce7491b4961154f3"}},{"arxiv_id":"1906.07510","paper":"/paper/attention-guided-graph-convolutional-networks","title":"Attention Guided Graph Convolutional Networks for Relation Extraction","date":"2019-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Cartus/AGGCN_TACRED","path":"utils/scorer.py","file_url":"https://github.com/Cartus/AGGCN_TACRED/blob/HEAD/utils/scorer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9cb44938e773d88b","mcp_get_code":{"code_sha256":"9cb44938e773d88b"}},{"arxiv_id":"1906.07510","paper":"/paper/attention-guided-graph-convolutional-networks","title":"Attention Guided Graph Convolutional Networks for Relation Extraction","date":"2019-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Cartus/AGGCN_TACRED","path":"semeval/utils/scorer.py","file_url":"https://github.com/Cartus/AGGCN_TACRED/blob/HEAD/semeval/utils/scorer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c4b42e34eb5cc2e9","mcp_get_code":{"code_sha256":"c4b42e34eb5cc2e9"}},{"arxiv_id":"1906.02762","paper":"/paper/understanding-and-improving-transformer-from","title":"Understanding and Improving Transformer From a Multi-Particle Dynamic System Point of View","date":"2019-06-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhuohan123/macaron-net","path":"bert/macaron-scripts/bert/concat_short_sentences.py","file_url":"https://github.com/zhuohan123/macaron-net/blob/HEAD/bert/macaron-scripts/bert/concat_short_sentences.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"43404249e690d259","mcp_get_code":{"code_sha256":"43404249e690d259"}},{"arxiv_id":"1812.06162","paper":"/paper/an-empirical-model-of-large-batch-training","title":"An Empirical Model of Large-Batch Training","date":"2018-12-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"davidandym/task-conflict-in-text-to-text-learners","path":"src/deca_metrics.py","file_url":"https://github.com/davidandym/task-conflict-in-text-to-text-learners/blob/HEAD/src/deca_metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e81a8138a9405edf","mcp_get_code":{"code_sha256":"e81a8138a9405edf"}},{"arxiv_id":"1806.08730","paper":"/paper/the-natural-language-decathlon-multitask","title":"The Natural Language Decathlon: Multitask Learning as Question Answering","date":"2018-06-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"salesforce/decaNLP","path":"metrics.py","file_url":"https://github.com/salesforce/decaNLP/blob/HEAD/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"e81a8138a9405edf","mcp_get_code":{"code_sha256":"e81a8138a9405edf"}},{"arxiv_id":"1711.05363","paper":"/paper/kernel-conditional-exponential-family","title":"Kernel Conditional Exponential Family","date":"2017-11-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MichaelArbel/KCEF","path":"KCEF/functions.py","file_url":"https://github.com/MichaelArbel/KCEF/blob/HEAD/KCEF/functions.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"841efea8b489cbff","mcp_get_code":{"code_sha256":"841efea8b489cbff"}},{"arxiv_id":"1508.01745","paper":"/paper/semantically-conditioned-lstm-based-natural","title":"Semantically Conditioned LSTM-based Natural Language Generation for Spoken Dialogue Systems","date":"2015-08-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mrcmoresi/sc-lstm","path":"run_woz3.py","file_url":"https://github.com/mrcmoresi/sc-lstm/blob/HEAD/run_woz3.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"faf421c0fe195b81","mcp_get_code":{"code_sha256":"faf421c0fe195b81"}},{"arxiv_id":"1508.01745","paper":"/paper/semantically-conditioned-lstm-based-natural","title":"Semantically Conditioned LSTM-based Natural Language Generation for Spoken Dialogue Systems","date":"2015-08-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"andy194673/nlg-sclstm-multiwoz","path":"run_woz3.py","file_url":"https://github.com/andy194673/nlg-sclstm-multiwoz/blob/HEAD/run_woz3.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2d9e4c619f62021f","mcp_get_code":{"code_sha256":"2d9e4c619f62021f"}},{"arxiv_id":"2024.findings-acl.848","paper":null,"title":"arXiv:2024.findings-acl.848","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"CAS-SIAT-ConsistencyAI/NUMCoT","path":"src/code/experience_first/run_medium.py","file_url":"https://github.com/CAS-SIAT-ConsistencyAI/NUMCoT/blob/HEAD/src/code/experience_first/run_medium.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"CC0-1.0","inline_ok":true,"code_sha256_prefix":"205a0d52ead816d9","mcp_get_code":{"code_sha256":"205a0d52ead816d9"}},{"arxiv_id":"2024.findings-acl.848","paper":null,"title":"arXiv:2024.findings-acl.848","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"CAS-SIAT-ConsistencyAI/NUMCoT","path":"src/code/experience_second/run_unit_measurement.py","file_url":"https://github.com/CAS-SIAT-ConsistencyAI/NUMCoT/blob/HEAD/src/code/experience_second/run_unit_measurement.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"CC0-1.0","inline_ok":true,"code_sha256_prefix":"a06a8861b5c1c21e","mcp_get_code":{"code_sha256":"a06a8861b5c1c21e"}},{"arxiv_id":"2023.findings-acl.807","paper":null,"title":"arXiv:2023.findings-acl.807","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"amueller/word_cloud","path":"wordcloud/tokenization.py","file_url":"https://github.com/amueller/word_cloud/blob/HEAD/wordcloud/tokenization.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f23a61928a841fa0","mcp_get_code":{"code_sha256":"f23a61928a841fa0"}},{"arxiv_id":"2021.findings-emnlp.152","paper":null,"title":"arXiv:2021.findings-emnlp.152","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"ruidan/IMN-E2E-ABSA","path":"code/evaluation.py","file_url":"https://github.com/ruidan/IMN-E2E-ABSA/blob/HEAD/code/evaluation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c293c9d2e5e4fde1","mcp_get_code":{"code_sha256":"c293c9d2e5e4fde1"}}]}