{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/exact-match-score","entry":"exact_match_score","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":82,"n_papers_ran":65,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":35,"n_samples_ran":19,"n_samples_fingerprinted":17,"n_places":87,"n_places_pointer_only":32,"by_status":{"ran_honours":0,"ran_violates":4,"ran_draft_wrong":0,"ran_fixture":0,"ran":15,"unverified":16},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.00513","paper":"/paper/arxiv-2609-00513","title":"ISO-RAG: Isoperimetric Noise Control for Retrieval-Augmented Generation","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"ZaiizaiZHANG/ISO-RAG","path":"analyze_predictions.py","file_url":"https://github.com/ZaiizaiZHANG/ISO-RAG/blob/HEAD/analyze_predictions.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"66804d06fa601b9c","mcp_get_code":{"code_sha256":"66804d06fa601b9c"}},{"arxiv_id":"2609.00513","paper":"/paper/arxiv-2609-00513","title":"ISO-RAG: Isoperimetric Noise Control for Retrieval-Augmented Generation","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"ZaiizaiZHANG/ISO-RAG","path":"analyze_retrieval_qa_transfer.py","file_url":"https://github.com/ZaiizaiZHANG/ISO-RAG/blob/HEAD/analyze_retrieval_qa_transfer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"209e457134cf7e77","mcp_get_code":{"code_sha256":"209e457134cf7e77"}},{"arxiv_id":"2604.05818","paper":"/paper/arxiv-2604-05818","title":"WikiSeeker: Rethinking the Role of Vision-Language Models in Knowledge-Based Visual Question Answering","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"zhuyjan/WikiSeeker","path":"utils/infoseek_evaluation_utils.py","file_url":"https://github.com/zhuyjan/WikiSeeker/blob/HEAD/utils/infoseek_evaluation_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7055aa97f2bde50e","mcp_get_code":{"code_sha256":"7055aa97f2bde50e"}},{"arxiv_id":"2603.15892","paper":"/paper/arxiv-2603-15892","title":"Temporal Conflicts in LLMs: Reproducibility Insights from Unifying DYNAMICQA and MULAN","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"terrierteam/temporal_conflicts","path":"fact_mutability/analysis/f1_score.py","file_url":"https://github.com/terrierteam/temporal_conflicts/blob/HEAD/fact_mutability/analysis/f1_score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"fd5f06fcb775f9e0","mcp_get_code":{"code_sha256":"fd5f06fcb775f9e0"}},{"arxiv_id":"2602.05728","paper":"/paper/arxiv-2602-05728","title":"CompactRAG: Reducing LLM Calls and Token Overhead in Multi-Hop Question Answering","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"How-Young-X/CompactRAG","path":"src/metrics/F1Eval.py","file_url":"https://github.com/How-Young-X/CompactRAG/blob/HEAD/src/metrics/F1Eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1a90d2afb9055e47","mcp_get_code":{"code_sha256":"1a90d2afb9055e47"}},{"arxiv_id":"2601.03908","paper":"/paper/arxiv-2601-03908","title":"Decide Then Retrieve: A Training-Free Framework with Uncertainty-Guided Triggering and Dual-Path Retrieval","date":"2026-01-07","month_inferred_from_arxiv_id":null,"title_source":"syntology","repo":"ChenWangHKU/DTR","path":"evaluation/metrics.py","file_url":"https://github.com/ChenWangHKU/DTR/blob/HEAD/evaluation/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"108d116eb5bdf952","mcp_get_code":{"code_sha256":"108d116eb5bdf952"}},{"arxiv_id":"2601.03042","paper":"/paper/arxiv-2601-03042","title":"BaseCal: Unsupervised Confidence Calibration via Base Model Signals","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"Tan-Hexiang/BaseCal","path":"gen_metric/em.py","file_url":"https://github.com/Tan-Hexiang/BaseCal/blob/HEAD/gen_metric/em.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8aa9ea4da99420bd","mcp_get_code":{"code_sha256":"8aa9ea4da99420bd"}},{"arxiv_id":"2510.13494","paper":"/paper/arxiv-2510-13494","title":"LiteraryQA: Towards Effective Evaluation of Long-document Narrative QA","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"SapienzaNLP/LiteraryQA","path":"literaryqa/ngram_metrics.py","file_url":"https://github.com/SapienzaNLP/LiteraryQA/blob/HEAD/literaryqa/ngram_metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b2e99f6b9eea4ae3","mcp_get_code":{"code_sha256":"b2e99f6b9eea4ae3"}},{"arxiv_id":"2506.21215","paper":"/paper/unveiling-causal-reasoning-in-large-language","title":"Unveiling Causal Reasoning in Large Language Models: Reality or Mirage?","date":"2025-06-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Haoang97/CausalProbe-2024","path":"metrics.py","file_url":"https://github.com/Haoang97/CausalProbe-2024/blob/HEAD/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d97ab73d08dfa3b9","mcp_get_code":{"code_sha256":"d97ab73d08dfa3b9"}},{"arxiv_id":"2505.17005","paper":"/paper/r1-searcher-incentivizing-the-dynamic","title":"R1-Searcher++: Incentivizing the Dynamic Knowledge Acquisition of LLMs via Reinforcement Learning","date":"2025-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"RUCAIBox/R1-Searcher","path":"train/reward_server_qwen_zero.py","file_url":"https://github.com/RUCAIBox/R1-Searcher/blob/HEAD/train/reward_server_qwen_zero.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1332ed2c182add91","mcp_get_code":{"code_sha256":"1332ed2c182add91"}},{"arxiv_id":"2412.06206","paper":"/paper/sirerag-indexing-similar-and-related","title":"SiReRAG: Indexing Similar and Related Information for Multihop Reasoning","date":"2024-12-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"SalesforceAIResearch/SiReRAG","path":"evaluate_2wiki.py","file_url":"https://github.com/SalesforceAIResearch/SiReRAG/blob/HEAD/evaluate_2wiki.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9d0dc82a4491f803","mcp_get_code":{"code_sha256":"9d0dc82a4491f803"}},{"arxiv_id":"2410.04139","paper":"/paper/from-reading-to-compressing-exploring-the","title":"From Reading to Compressing: Exploring the Multi-document Reader for Prompt Compression","date":"2024-10-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"eunseongc/r2c","path":"LLM_inference.py","file_url":"https://github.com/eunseongc/r2c/blob/HEAD/LLM_inference.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"31f795a3e0c1be69","mcp_get_code":{"code_sha256":"31f795a3e0c1be69"}},{"arxiv_id":"2406.17245","paper":"/paper/unlocking-continual-learning-abilities-in","title":"Unlocking Continual Learning Abilities in Language Models","date":"2024-06-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wenyudu/migu","path":"src/compute_metrics.py","file_url":"https://github.com/wenyudu/migu/blob/HEAD/src/compute_metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e168ce39d04b76f1","mcp_get_code":{"code_sha256":"e168ce39d04b76f1"}},{"arxiv_id":"2406.12382","paper":"/paper/from-instance-training-to-instruction","title":"From Instance Training to Instruction Learning: Task Adapters Generation from Instructions","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Xnhyacinth/TAGI","path":"src/compute_metrics.py","file_url":"https://github.com/Xnhyacinth/TAGI/blob/HEAD/src/compute_metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e168ce39d04b76f1","mcp_get_code":{"code_sha256":"e168ce39d04b76f1"}},{"arxiv_id":"2406.09072","paper":"/paper/living-in-the-moment-can-large-language","title":"Living in the Moment: Can Large Language Models Grasp Co-Temporal Reasoning?","date":"2024-06-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhaochen0110/cotempqa","path":"config.py","file_url":"https://github.com/zhaochen0110/cotempqa/blob/HEAD/config.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1d5bd4c4648b1e9d","mcp_get_code":{"code_sha256":"1d5bd4c4648b1e9d"}},{"arxiv_id":"2406.06326","paper":"/paper/self-tuning-instructing-llms-to-effectively","title":"Self-Tuning: Instructing LLMs to Effectively Acquire New Knowledge through Self-Teaching","date":"2024-06-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"2c66ac5f9374f1f7","mcp_get_code":{"code_sha256":"2c66ac5f9374f1f7"}},{"arxiv_id":"2406.04823","paper":"/paper/berts-are-generative-in-context-learners","title":"BERTs are Generative In-Context Learners","date":"2024-06-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ltgoslo/bert-in-context","path":"glue/record.py","file_url":"https://github.com/ltgoslo/bert-in-context/blob/HEAD/glue/record.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f6c275d6a18330a9","mcp_get_code":{"code_sha256":"f6c275d6a18330a9"}},{"arxiv_id":"2405.20978","paper":"/paper/enhancing-noise-robustness-of-retrieval","title":"Enhancing Noise Robustness of Retrieval-Augmented Language Models with Adaptive Adversarial Training","date":"2024-05-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"calubkk/RAAT","path":"tuner/metrics/em_f1.py","file_url":"https://github.com/calubkk/RAAT/blob/HEAD/tuner/metrics/em_f1.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0e88ab2d4fdaca3c","mcp_get_code":{"code_sha256":"0e88ab2d4fdaca3c"}},{"arxiv_id":"2405.16821","paper":"/paper/perturbation-restrained-sequential-model","title":"Perturbation-Restrained Sequential Model Editing","date":"2024-05-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mjy1111/PRUNE","path":"edit_load.py","file_url":"https://github.com/mjy1111/PRUNE/blob/HEAD/edit_load.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4be72e172f90cb2c","mcp_get_code":{"code_sha256":"4be72e172f90cb2c"}},{"arxiv_id":"2404.18824","paper":"/paper/benchmarking-benchmark-leakage-in-large","title":"Benchmarking Benchmark Leakage in Large Language Models","date":"2024-04-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gair-nlp/benbench","path":"src/metric_utils.py","file_url":"https://github.com/gair-nlp/benbench/blob/HEAD/src/metric_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6828c333fed5ca6d","mcp_get_code":{"code_sha256":"6828c333fed5ca6d"}},{"arxiv_id":"2404.12145","paper":"/paper/from-form-s-to-meaning-probing-the-semantic","title":"From Form(s) to Meaning: Probing the Semantic Depths of Language Models Using Multisense Consistency","date":"2024-04-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"facebookresearch/multisense_consistency","path":"utils/eval_metrics.py","file_url":"https://github.com/facebookresearch/multisense_consistency/blob/HEAD/utils/eval_metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"8325d0eddc238131","mcp_get_code":{"code_sha256":"8325d0eddc238131"}},{"arxiv_id":"2402.18150","paper":"/paper/unsupervised-information-refinement-training","title":"Unsupervised Information Refinement Training of Large Language Models for Retrieval-Augmented Generation","date":"2024-02-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xsc1234/info-rag","path":"training/train_info_rag.py","file_url":"https://github.com/xsc1234/info-rag/blob/HEAD/training/train_info_rag.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"34916672e3ddaf0c","mcp_get_code":{"code_sha256":"34916672e3ddaf0c"}},{"arxiv_id":"2402.13991","paper":"/paper/analysing-the-impact-of-sequence-composition","title":"Analysing The Impact of Sequence Composition on Language Model Pre-Training","date":"2024-02-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yuzhaouoe/pretraining-data-packing","path":"evaluation/eval_utils.py","file_url":"https://github.com/yuzhaouoe/pretraining-data-packing/blob/HEAD/evaluation/eval_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"efa3643614e9c5f5","mcp_get_code":{"code_sha256":"efa3643614e9c5f5"}},{"arxiv_id":"2402.12052","paper":"/paper/small-models-big-insights-leveraging-slim","title":"Small Models, Big Insights: Leveraging Slim Proxy Models To Decide When and What to Retrieve for LLMs","date":"2024-02-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"plageon/slimplm","path":"SKR/skr.py","file_url":"https://github.com/plageon/slimplm/blob/HEAD/SKR/skr.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0be284da9bf9ca86","mcp_get_code":{"code_sha256":"0be284da9bf9ca86"}},{"arxiv_id":"2402.11893","paper":"/paper/discerning-and-resolving-knowledge-conflicts","title":"Discerning and Resolving Knowledge Conflicts through Adaptive Decoding with Contextual Information-Entropy Constraint","date":"2024-02-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"bf75673211903e14","mcp_get_code":{"code_sha256":"bf75673211903e14"}},{"arxiv_id":"2402.02503","paper":"/paper/gerea-question-aware-prompt-captions-for","title":"GeReA: Question-Aware Prompt Captions for Knowledge-based Visual Question Answering","date":"2024-02-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"upper9527/gerea","path":"leaderboard_evaluation.py","file_url":"https://github.com/upper9527/gerea/blob/HEAD/leaderboard_evaluation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"108d116eb5bdf952","mcp_get_code":{"code_sha256":"108d116eb5bdf952"}},{"arxiv_id":"2402.01032","paper":"/paper/repeat-after-me-transformers-are-better-than","title":"Repeat After Me: Transformers are Better than State Space Models at Copying","date":"2024-02-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sjelassi/transformers_ssm_copy","path":"pretrained_exps/qa_evaluation_utils.py","file_url":"https://github.com/sjelassi/transformers_ssm_copy/blob/HEAD/pretrained_exps/qa_evaluation_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"67b1c1bbca4316ab","mcp_get_code":{"code_sha256":"67b1c1bbca4316ab"}},{"arxiv_id":"2401.12585","paper":"/paper/slang-new-concept-comprehension-of-large","title":"SLANG: New Concept Comprehension of Large Language Models","date":"2024-01-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"meirtz/focusonslang-toolbox","path":"eval_urban_f1.py","file_url":"https://github.com/meirtz/focusonslang-toolbox/blob/HEAD/eval_urban_f1.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"978ce66ddf4e7b47","mcp_get_code":{"code_sha256":"978ce66ddf4e7b47"}},{"arxiv_id":"2312.09039","paper":"/paper/tap4llm-table-provider-on-sampling-augmenting","title":"TAP4LLM: Table Provider on Sampling, Augmenting, and Packing Semi-structured Data for Large Language Model Reasoning","date":"2023-12-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Y-Sui/GPT4Table","path":"table_meets_llm/eval/evaluate_benchmark.py","file_url":"https://github.com/Y-Sui/GPT4Table/blob/HEAD/table_meets_llm/eval/evaluate_benchmark.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"79f3ecb504ea2b1e","mcp_get_code":{"code_sha256":"79f3ecb504ea2b1e"}},{"arxiv_id":"2311.03734","paper":"/paper/leveraging-structured-information-for","title":"Leveraging Structured Information for Explainable Multi-hop Question Answering and Reasoning","date":"2023-11-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bcdnlp/structure-qa","path":"src/hotpot_evaluate.py","file_url":"https://github.com/bcdnlp/structure-qa/blob/HEAD/src/hotpot_evaluate.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"9d0dc82a4491f803","mcp_get_code":{"code_sha256":"9d0dc82a4491f803"}},{"arxiv_id":"2311.00288","paper":"/paper/active-instruction-tuning-improving-cross","title":"Active Instruction Tuning: Improving Cross-Task Generalization by Training on Prompt Sensitive Tasks","date":"2023-11-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pluslabnlp/active-it","path":"ActiveIT/src/compute_metrics.py","file_url":"https://github.com/pluslabnlp/active-it/blob/HEAD/ActiveIT/src/compute_metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e168ce39d04b76f1","mcp_get_code":{"code_sha256":"e168ce39d04b76f1"}},{"arxiv_id":"2310.14152","paper":"/paper/orthogonal-subspace-learning-for-language","title":"Orthogonal Subspace Learning for Language Model Continual Learning","date":"2023-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cmnfriend/o-lora","path":"src/compute_metrics.py","file_url":"https://github.com/cmnfriend/o-lora/blob/HEAD/src/compute_metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e168ce39d04b76f1","mcp_get_code":{"code_sha256":"e168ce39d04b76f1"}},{"arxiv_id":"2310.11511","paper":"/paper/self-rag-learning-to-retrieve-generate-and","title":"Self-RAG: Learning to Retrieve, Generate, and Critique through Self-Reflection","date":"2023-10-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AkariAsai/self-rag","path":"retrieval_lm/metrics.py","file_url":"https://github.com/AkariAsai/self-rag/blob/HEAD/retrieval_lm/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d97ab73d08dfa3b9","mcp_get_code":{"code_sha256":"d97ab73d08dfa3b9"}},{"arxiv_id":"2309.07822","paper":"/paper/catfood-counterfactual-augmented-training-for","title":"CATfOOD: Counterfactual Augmented Training for Improving Out-of-Domain Performance and Calibration","date":"2023-09-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ukplab/catfood","path":"src/shortcuts/evaluate.py","file_url":"https://github.com/ukplab/catfood/blob/HEAD/src/shortcuts/evaluate.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e35cb1727750ade1","mcp_get_code":{"code_sha256":"e35cb1727750ade1"}},{"arxiv_id":"2309.07822","paper":"/paper/catfood-counterfactual-augmented-training-for","title":"CATfOOD: Counterfactual Augmented Training for Improving Out-of-Domain Performance and Calibration","date":"2023-09-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ukplab/catfood","path":"src/calibration/baseline/calibration_metrics.py","file_url":"https://github.com/ukplab/catfood/blob/HEAD/src/calibration/baseline/calibration_metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ee7ee45b82d4949f","mcp_get_code":{"code_sha256":"ee7ee45b82d4949f"}},{"arxiv_id":"2308.16475","paper":"/paper/transformer-compression-via-subspace","title":"$\\rm SP^3$: Enhancing Structured Pruning via PCA Projection","date":"2023-08-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hyx1999/sp3","path":"bert/metric/squad.py","file_url":"https://github.com/hyx1999/sp3/blob/HEAD/bert/metric/squad.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bf75673211903e14","mcp_get_code":{"code_sha256":"bf75673211903e14"}},{"arxiv_id":"2308.10819","paper":"/paper/do-you-really-follow-me-adversarial","title":"Evaluating the Instruction-Following Robustness of Large Language Models to Prompt Injection","date":"2023-08-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"leezekun/instruction-following-robustness-eval","path":"qa_utils.py","file_url":"https://github.com/leezekun/instruction-following-robustness-eval/blob/HEAD/qa_utils.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bf75673211903e14","mcp_get_code":{"code_sha256":"bf75673211903e14"}},{"arxiv_id":"2305.15387","paper":"/paper/peek-across-improving-multi-document-modeling","title":"Peek Across: Improving Multi-Document Modeling via Cross-Document Question-Answering","date":"2023-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aviclu/peekacross","path":"trainer_seq2seq_qa.py","file_url":"https://github.com/aviclu/peekacross/blob/HEAD/trainer_seq2seq_qa.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8fad2994edad57e5","mcp_get_code":{"code_sha256":"8fad2994edad57e5"}},{"arxiv_id":"2305.12421","paper":"/paper/evaluating-open-qa-evaluation-1","title":"Evaluating Open-QA Evaluation","date":"2023-05-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wangcunxiang/QA-Eval","path":"lexical_match.py","file_url":"https://github.com/wangcunxiang/QA-Eval/blob/HEAD/lexical_match.py","status":"unverified","verification_level":0,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c4ae86237858e5d6","mcp_get_code":{"code_sha256":"c4ae86237858e5d6"}},{"arxiv_id":"2305.00944","paper":"/paper/poisoning-language-models-during-instruction","title":"Poisoning Language Models During Instruction Tuning","date":"2023-05-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alexwan0/poisoning-instruction-tuned-models","path":"src/compute_metrics.py","file_url":"https://github.com/alexwan0/poisoning-instruction-tuned-models/blob/HEAD/src/compute_metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e168ce39d04b76f1","mcp_get_code":{"code_sha256":"e168ce39d04b76f1"}},{"arxiv_id":"2303.08774","paper":"/paper/gpt-4-technical-report-1","title":"GPT-4 Technical Report","date":"2023-03-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AUCOHL/RTL-Repo","path":"src/utils.py","file_url":"https://github.com/AUCOHL/RTL-Repo/blob/HEAD/src/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2d1ad042f17b54ac","mcp_get_code":{"code_sha256":"2d1ad042f17b54ac"}},{"arxiv_id":"2302.14691","paper":"/paper/in-context-instruction-learning","title":"Investigating the Effectiveness of Task-Agnostic Prefix Prompt for Instruction Following","date":"2023-02-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"seonghyeonye/icil","path":"src/compute_metrics.py","file_url":"https://github.com/seonghyeonye/icil/blob/HEAD/src/compute_metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e168ce39d04b76f1","mcp_get_code":{"code_sha256":"e168ce39d04b76f1"}},{"arxiv_id":"2302.11713","paper":"/paper/can-pre-trained-vision-and-language-models","title":"Can Pre-trained Vision and Language Models Answer Visual Information-Seeking Questions?","date":"2023-02-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"edchengg/infoseek_eval","path":"infoseek_eval.py","file_url":"https://github.com/edchengg/infoseek_eval/blob/HEAD/infoseek_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7055aa97f2bde50e","mcp_get_code":{"code_sha256":"7055aa97f2bde50e"}},{"arxiv_id":"2212.03813","paper":"/paper/robustness-of-learning-from-task-instructions","title":"Robustness of Learning from Task Instructions","date":"2022-12-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jiashenggu/tk-instruct","path":"src/compute_metrics.py","file_url":"https://github.com/jiashenggu/tk-instruct/blob/HEAD/src/compute_metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e168ce39d04b76f1","mcp_get_code":{"code_sha256":"e168ce39d04b76f1"}},{"arxiv_id":"2210.10861","paper":"/paper/qa-domain-adaptation-using-hidden-space","title":"QA Domain Adaptation using Hidden Space Augmentation and Self-Supervised Contrastive Adaptation","date":"2022-10-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"f6c275d6a18330a9","mcp_get_code":{"code_sha256":"f6c275d6a18330a9"}},{"arxiv_id":"2204.07705","paper":"/paper/benchmarking-generalization-via-in-context","title":"Super-NaturalInstructions: Generalization via Declarative Instructions on 1600+ NLP Tasks","date":"2022-04-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"renzelou/pick-rank","path":"src/compute_metrics.py","file_url":"https://github.com/renzelou/pick-rank/blob/HEAD/src/compute_metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e168ce39d04b76f1","mcp_get_code":{"code_sha256":"e168ce39d04b76f1"}},{"arxiv_id":"2109.13880","paper":"/paper/single-dataset-experts-for-multi-dataset","title":"Single-dataset Experts for Multi-dataset Question Answering","date":"2021-09-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"princeton-nlp/MADE","path":"src/utils/metrics.py","file_url":"https://github.com/princeton-nlp/MADE/blob/HEAD/src/utils/metrics.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bf75673211903e14","mcp_get_code":{"code_sha256":"bf75673211903e14"}},{"arxiv_id":"2109.08133","paper":"/paper/phrase-retrieval-learns-passage-retrieval-too","title":"Phrase Retrieval Learns Passage Retrieval, Too","date":"2021-09-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"princeton-nlp/DensePhrases","path":"densephrases/utils/eval_utils.py","file_url":"https://github.com/princeton-nlp/DensePhrases/blob/HEAD/densephrases/utils/eval_utils.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"9d0dc82a4491f803","mcp_get_code":{"code_sha256":"9d0dc82a4491f803"}},{"arxiv_id":"2109.04912","paper":"/paper/reasonbert-pre-trained-to-reason-with-distant","title":"ReasonBERT: Pre-trained to Reason with Distant Supervision","date":"2021-09-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sunlab-osu/reasonbert","path":"model/metric.py","file_url":"https://github.com/sunlab-osu/reasonbert/blob/HEAD/model/metric.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f6c275d6a18330a9","mcp_get_code":{"code_sha256":"f6c275d6a18330a9"}},{"arxiv_id":"2105.02692","paper":"/paper/learning-to-perturb-word-embeddings-for-out","title":"Learning to Perturb Word Embeddings for Out-of-distribution QA","date":"2021-05-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"seanie12/SWEP","path":"mrqa_utils.py","file_url":"https://github.com/seanie12/SWEP/blob/HEAD/mrqa_utils.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bf75673211903e14","mcp_get_code":{"code_sha256":"bf75673211903e14"}},{"arxiv_id":"2104.00640","paper":"/paper/evidence-based-verification-for-real-world","title":"AmbiFC: Fact-Checking Ambiguous Claims with Evidence","date":"2021-04-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"knot-fit-but/claimdissector","path":"src/common/eval_utils.py","file_url":"https://github.com/knot-fit-but/claimdissector/blob/HEAD/src/common/eval_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"efa3643614e9c5f5","mcp_get_code":{"code_sha256":"efa3643614e9c5f5"}},{"arxiv_id":"2101.10421","paper":"/paper/english-machine-reading-comprehension","title":"English Machine Reading Comprehension Datasets: A Survey","date":"2021-01-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"2c66ac5f9374f1f7","mcp_get_code":{"code_sha256":"2c66ac5f9374f1f7"}},{"arxiv_id":"2011.01060","paper":"/paper/constructing-a-multi-hop-qa-dataset-for","title":"Constructing A Multi-hop QA Dataset for Comprehensive Evaluation of Reasoning Steps","date":"2020-11-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Alab-NII/2wikimultihop","path":"2wikimultihop_evaluate.py","file_url":"https://github.com/Alab-NII/2wikimultihop/blob/HEAD/2wikimultihop_evaluate.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"9d0dc82a4491f803","mcp_get_code":{"code_sha256":"9d0dc82a4491f803"}},{"arxiv_id":"2010.16021","paper":"/paper/cliniqg4qa-generating-diverse-questions-for","title":"CliniQG4QA: Generating Diverse Questions for Domain Adaptation of Clinical Question Answering","date":"2020-10-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sunlab-osu/CliniQG4QA","path":"QA/evaluate-v1.1_human_generated.py","file_url":"https://github.com/sunlab-osu/CliniQG4QA/blob/HEAD/QA/evaluate-v1.1_human_generated.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bf75673211903e14","mcp_get_code":{"code_sha256":"bf75673211903e14"}},{"arxiv_id":"2010.12527","paper":"/paper/retrieve-rerank-read-then-iterate-answering","title":"Answering Open-Domain Questions of Varying Reasoning Steps from Text","date":"2020-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"f6c275d6a18330a9","mcp_get_code":{"code_sha256":"f6c275d6a18330a9"}},{"arxiv_id":"2008.02637","paper":"/paper/question-and-answer-test-train-overlap-in","title":"Question and Answer Test-Train Overlap in Open-Domain Question Answering Datasets","date":"2020-08-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"facebookresearch/QA-Overlap","path":"evaluate.py","file_url":"https://github.com/facebookresearch/QA-Overlap/blob/HEAD/evaluate.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f6c275d6a18330a9","mcp_get_code":{"code_sha256":"f6c275d6a18330a9"}},{"arxiv_id":"2006.12467","paper":"/paper/limits-to-depth-efficiencies-of-self","title":"The Depth-to-Width Interplay in Self-Attention","date":"2020-06-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"uf-hobi-informatics-lab/GatorTron","path":"finetuning/qa/evaluate-v1.1.py","file_url":"https://github.com/uf-hobi-informatics-lab/GatorTron/blob/HEAD/finetuning/qa/evaluate-v1.1.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bf75673211903e14","mcp_get_code":{"code_sha256":"bf75673211903e14"}},{"arxiv_id":"1911.10470","paper":"/paper/learning-to-retrieve-reasoning-paths-over-1","title":"Learning to Retrieve Reasoning Paths over Wikipedia Graph for Question Answering","date":"2019-11-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AkariAsai/learning_to_retrieve_reasoning_paths","path":"eval_utils.py","file_url":"https://github.com/AkariAsai/learning_to_retrieve_reasoning_paths/blob/HEAD/eval_utils.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f6c275d6a18330a9","mcp_get_code":{"code_sha256":"f6c275d6a18330a9"}},{"arxiv_id":"1911.02896","paper":"/paper/contextualized-sparse-representation-with-1","title":"Contextualized Sparse Representations for Real-Time Open-Domain Question Answering","date":"2019-11-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jhyuklee/sparc","path":"evaluate-v1.1.py","file_url":"https://github.com/jhyuklee/sparc/blob/HEAD/evaluate-v1.1.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f6c275d6a18330a9","mcp_get_code":{"code_sha256":"f6c275d6a18330a9"}},{"arxiv_id":"1911.02896","paper":"/paper/contextualized-sparse-representation-with-1","title":"Contextualized Sparse Representations for Real-Time Open-Domain Question Answering","date":"2019-11-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jhyuklee/sparc","path":"eval_utils.py","file_url":"https://github.com/jhyuklee/sparc/blob/HEAD/eval_utils.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"9d0dc82a4491f803","mcp_get_code":{"code_sha256":"9d0dc82a4491f803"}},{"arxiv_id":"1910.14520","paper":"/paper/do-multi-hop-readers-dream-of-reasoning","title":"Do Multi-hop Readers Dream of Reasoning Chains?","date":"2019-10-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"helloeve/bert-co-matching","path":"evaluate-v1.1-original.py","file_url":"https://github.com/helloeve/bert-co-matching/blob/HEAD/evaluate-v1.1-original.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f6c275d6a18330a9","mcp_get_code":{"code_sha256":"f6c275d6a18330a9"}},{"arxiv_id":"1910.14520","paper":"/paper/do-multi-hop-readers-dream-of-reasoning","title":"Do Multi-hop Readers Dream of Reasoning Chains?","date":"2019-10-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"helloeve/bert-co-matching","path":"evaluate-v1.1.py","file_url":"https://github.com/helloeve/bert-co-matching/blob/HEAD/evaluate-v1.1.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d299ec45effa875c","mcp_get_code":{"code_sha256":"d299ec45effa875c"}},{"arxiv_id":"1910.09753","paper":"/paper/mrqa-2019-shared-task-evaluating","title":"MRQA 2019 Shared Task: Evaluating Generalization in Reading Comprehension","date":"2019-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mrqa/MRQA-Shared-Task-2019","path":"mrqa_official_eval.py","file_url":"https://github.com/mrqa/MRQA-Shared-Task-2019/blob/HEAD/mrqa_official_eval.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f6c275d6a18330a9","mcp_get_code":{"code_sha256":"f6c275d6a18330a9"}},{"arxiv_id":"1909.06356","paper":"/paper/addressing-semantic-drift-in-question","title":"Addressing Semantic Drift in Question Generation for Semi-Supervised Question Answering","date":"2019-09-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ZhangShiyue/QGforQA","path":"LIB/EVAL/evaluate.py","file_url":"https://github.com/ZhangShiyue/QGforQA/blob/HEAD/LIB/EVAL/evaluate.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f6c275d6a18330a9","mcp_get_code":{"code_sha256":"f6c275d6a18330a9"}},{"arxiv_id":"1909.05803","paper":"/paper/self-assembling-modular-networks-for","title":"Self-Assembling Modular Networks for Interpretable Multi-Hop Reasoning","date":"2019-09-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jiangycTarheel/NMN-MultiHopQA","path":"hotpotqa/evaluate-v1.1.py","file_url":"https://github.com/jiangycTarheel/NMN-MultiHopQA/blob/HEAD/hotpotqa/evaluate-v1.1.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f6c275d6a18330a9","mcp_get_code":{"code_sha256":"f6c275d6a18330a9"}},{"arxiv_id":"1906.05807","paper":"/paper/real-time-open-domain-question-answering-with","title":"Real-Time Open-Domain Question Answering with Dense-Sparse Phrase Index","date":"2019-06-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"uwnlp/denspi","path":"evaluate-v1.1.py","file_url":"https://github.com/uwnlp/denspi/blob/HEAD/evaluate-v1.1.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f6c275d6a18330a9","mcp_get_code":{"code_sha256":"f6c275d6a18330a9"}},{"arxiv_id":"1906.05394","paper":"/paper/neural-arabic-question-answering","title":"Neural Arabic Question Answering","date":"2019-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"husseinmozannar/SOQAL","path":"baselines_reading/evaluate_baselines.py","file_url":"https://github.com/husseinmozannar/SOQAL/blob/HEAD/baselines_reading/evaluate_baselines.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e9b032891a5228a4","mcp_get_code":{"code_sha256":"e9b032891a5228a4"}},{"arxiv_id":"1905.13453","paper":"/paper/multiqa-an-empirical-investigation-of","title":"MultiQA: An Empirical Investigation of Generalization and Transfer in Reading Comprehension","date":"2019-05-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alontalmor/multiqa","path":"common/official_eval.py","file_url":"https://github.com/alontalmor/multiqa/blob/HEAD/common/official_eval.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f6c275d6a18330a9","mcp_get_code":{"code_sha256":"f6c275d6a18330a9"}},{"arxiv_id":"1905.05460","paper":"/paper/cognitive-graph-for-multi-hop-reading","title":"Cognitive Graph for Multi-Hop Reading Comprehension at Scale","date":"2019-05-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"THUDM/CogQA","path":"hotpot_evaluate_v1.py","file_url":"https://github.com/THUDM/CogQA/blob/HEAD/hotpot_evaluate_v1.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9d0dc82a4491f803","mcp_get_code":{"code_sha256":"9d0dc82a4491f803"}},{"arxiv_id":"1810.04805","paper":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Microsoft/AzureML-BERT","path":"finetune/evaluate_squad.py","file_url":"https://github.com/Microsoft/AzureML-BERT/blob/HEAD/finetune/evaluate_squad.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f6c275d6a18330a9","mcp_get_code":{"code_sha256":"f6c275d6a18330a9"}},{"arxiv_id":"1809.09600","paper":"/paper/hotpotqa-a-dataset-for-diverse-explainable","title":"HotpotQA: A Dataset for Diverse, Explainable Multi-hop Question Answering","date":"2018-09-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hotpotqa/hotpot","path":"hotpot_evaluate_v1.py","file_url":"https://github.com/hotpotqa/hotpot/blob/HEAD/hotpot_evaluate_v1.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"9d0dc82a4491f803","mcp_get_code":{"code_sha256":"9d0dc82a4491f803"}},{"arxiv_id":"1804.07927","paper":"/paper/duorc-towards-complex-language-understanding","title":"DuoRC: Towards Complex Language Understanding with Paraphrased Reading Comprehension","date":"2018-04-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"duorc/duorc","path":"evaluate.py","file_url":"https://github.com/duorc/duorc/blob/HEAD/evaluate.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f6c275d6a18330a9","mcp_get_code":{"code_sha256":"f6c275d6a18330a9"}},{"arxiv_id":"1804.07726","paper":"/paper/phrase-indexed-question-answering-a-new","title":"Phrase-Indexed Question Answering: A New Challenge for Scalable Document Comprehension","date":"2018-04-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"uwnlp/piqa","path":"squad/piqa_evaluate.py","file_url":"https://github.com/uwnlp/piqa/blob/HEAD/squad/piqa_evaluate.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f6c275d6a18330a9","mcp_get_code":{"code_sha256":"f6c275d6a18330a9"}},{"arxiv_id":"1711.05116","paper":"/paper/evidence-aggregation-for-answer-re-ranking-in","title":"Evidence Aggregation for Answer Re-Ranking in Open-Domain Question Answering","date":"2017-11-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shuohangwang/mprc","path":"trainedmodel/evaluation/quasart/evaluate-v1.1.py","file_url":"https://github.com/shuohangwang/mprc/blob/HEAD/trainedmodel/evaluation/quasart/evaluate-v1.1.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f6c275d6a18330a9","mcp_get_code":{"code_sha256":"f6c275d6a18330a9"}},{"arxiv_id":"1711.05116","paper":"/paper/evidence-aggregation-for-answer-re-ranking-in","title":"Evidence Aggregation for Answer Re-Ranking in Open-Domain Question Answering","date":"2017-11-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shuohangwang/mprc","path":"trainedmodel/evaluation/unftriviaqa/triviaqa_evaluation.py","file_url":"https://github.com/shuohangwang/mprc/blob/HEAD/trainedmodel/evaluation/unftriviaqa/triviaqa_evaluation.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2c66ac5f9374f1f7","mcp_get_code":{"code_sha256":"2c66ac5f9374f1f7"}},{"arxiv_id":"1705.03551","paper":"/paper/triviaqa-a-large-scale-distantly-supervised","title":"TriviaQA: A Large Scale Distantly Supervised Challenge Dataset for Reading Comprehension","date":"2017-05-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mandarjoshi90/triviaqa","path":"evaluation/triviaqa_evaluation.py","file_url":"https://github.com/mandarjoshi90/triviaqa/blob/HEAD/evaluation/triviaqa_evaluation.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2c66ac5f9374f1f7","mcp_get_code":{"code_sha256":"2c66ac5f9374f1f7"}},{"arxiv_id":"1611.01603","paper":"/paper/bidirectional-attention-flow-for-machine","title":"Bidirectional Attention Flow for Machine Comprehension","date":"2016-11-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ghus75/Question_Answering","path":"code/evaluate.py","file_url":"https://github.com/ghus75/Question_Answering/blob/HEAD/code/evaluate.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f6c275d6a18330a9","mcp_get_code":{"code_sha256":"f6c275d6a18330a9"}},{"arxiv_id":"1608.07905","paper":"/paper/machine-comprehension-using-match-lstm-and","title":"Machine Comprehension Using Match-LSTM and Answer Pointer","date":"2016-08-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"f6c275d6a18330a9","mcp_get_code":{"code_sha256":"f6c275d6a18330a9"}},{"arxiv_id":"1606.05250","paper":"/paper/squad-100000-questions-for-machine","title":"SQuAD: 100,000+ Questions for Machine Comprehension of Text","date":"2016-06-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"f6c275d6a18330a9","mcp_get_code":{"code_sha256":"f6c275d6a18330a9"}},{"arxiv_id":"1508.06615","paper":"/paper/character-aware-neural-language-models","title":"Character-Aware Neural Language Models","date":"2015-08-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"NLPLearn/QANet","path":"evaluate-v1.1.py","file_url":"https://github.com/NLPLearn/QANet/blob/HEAD/evaluate-v1.1.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f6c275d6a18330a9","mcp_get_code":{"code_sha256":"f6c275d6a18330a9"}},{"arxiv_id":"aaai_34690","paper":null,"title":"arXiv:aaai_34690","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"ZongyueQin/DSBD","path":"sampling/utils.py","file_url":"https://github.com/ZongyueQin/DSBD/blob/HEAD/sampling/utils.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bf75673211903e14","mcp_get_code":{"code_sha256":"bf75673211903e14"}},{"arxiv_id":"2025.acl-long.191","paper":null,"title":"arXiv:2025.acl-long.191","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"UKPLab/acl2025-diverse-cot","path":"src/hotpotqa_evaluation.py","file_url":"https://github.com/UKPLab/acl2025-diverse-cot/blob/HEAD/src/hotpotqa_evaluation.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"9d0dc82a4491f803","mcp_get_code":{"code_sha256":"9d0dc82a4491f803"}},{"arxiv_id":"2024.findings-emnlp.379","paper":null,"title":"arXiv:2024.findings-emnlp.379","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"wenyudu/MIGU","path":"src/compute_metrics.py","file_url":"https://github.com/wenyudu/MIGU/blob/HEAD/src/compute_metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e168ce39d04b76f1","mcp_get_code":{"code_sha256":"e168ce39d04b76f1"}},{"arxiv_id":"2023.findings-emnlp.836","paper":null,"title":"arXiv:2023.findings-emnlp.836","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"IBM/ensemble-instruct","path":"ensemble_instruct/ensemble_output.py","file_url":"https://github.com/IBM/ensemble-instruct/blob/HEAD/ensemble_instruct/ensemble_output.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3c8cf46311df61a2","mcp_get_code":{"code_sha256":"3c8cf46311df61a2"}},{"arxiv_id":"2023.findings-emnlp.835","paper":null,"title":"arXiv:2023.findings-emnlp.835","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"THU-KEG/ProbTree","path":"src/2wiki/RoHT/evaluate.py","file_url":"https://github.com/THU-KEG/ProbTree/blob/HEAD/src/2wiki/RoHT/evaluate.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9d0dc82a4491f803","mcp_get_code":{"code_sha256":"9d0dc82a4491f803"}},{"arxiv_id":"2023.emnlp-main.803","paper":null,"title":"arXiv:2023.emnlp-main.803","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"yisunlp/Anti-CF","path":"utils/squad_evaluate.py","file_url":"https://github.com/yisunlp/Anti-CF/blob/HEAD/utils/squad_evaluate.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f6c275d6a18330a9","mcp_get_code":{"code_sha256":"f6c275d6a18330a9"}},{"arxiv_id":"2021.acl-long.48","paper":null,"title":"arXiv:2021.acl-long.48","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"PluviophileYU/COSY","path":"XQA/src/evaluate_v1_1.py","file_url":"https://github.com/PluviophileYU/COSY/blob/HEAD/XQA/src/evaluate_v1_1.py","status":"ran_violates","verification_level":1,"contract_check":"VIOLATES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f6c275d6a18330a9","mcp_get_code":{"code_sha256":"f6c275d6a18330a9"}}]}