{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/f1-score","entry":"f1_score","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":121,"n_papers_ran":86,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":73,"n_samples_ran":40,"n_samples_fingerprinted":27,"n_places":127,"n_places_pointer_only":43,"by_status":{"ran_honours":10,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":4,"ran":26,"unverified":33},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.00513","paper":"/paper/arxiv-2609-00513","title":"ISO-RAG: Isoperimetric Noise Control for Retrieval-Augmented Generation","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"ZaiizaiZHANG/ISO-RAG","path":"analyze_predictions.py","file_url":"https://github.com/ZaiizaiZHANG/ISO-RAG/blob/HEAD/analyze_predictions.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6666d2397110c519","mcp_get_code":{"code_sha256":"6666d2397110c519"}},{"arxiv_id":"2609.00513","paper":"/paper/arxiv-2609-00513","title":"ISO-RAG: Isoperimetric Noise Control for Retrieval-Augmented Generation","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"ZaiizaiZHANG/ISO-RAG","path":"analyze_retrieval_qa_transfer.py","file_url":"https://github.com/ZaiizaiZHANG/ISO-RAG/blob/HEAD/analyze_retrieval_qa_transfer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"04e609efd10e1c78","mcp_get_code":{"code_sha256":"04e609efd10e1c78"}},{"arxiv_id":"2608.30399","paper":"/paper/arxiv-2608-30399","title":"SemPOI-RL: Aligning LLM Semantic Reasoning for Interpretable Out-of-Town POI Sequential Generation","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"Wind-Flipped/SemPOI-RL","path":"code/metrics.py","file_url":"https://github.com/Wind-Flipped/SemPOI-RL/blob/HEAD/code/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"49807bc3c4da0a97","mcp_get_code":{"code_sha256":"49807bc3c4da0a97"}},{"arxiv_id":"2608.01904","paper":"/paper/arxiv-2608-01904","title":"CoEvoKG: Co-Evolving Knowledge Graphs with Self-Evolving Search Agents","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"lazzy1225/CoEvoKG","path":"coevokg/reward/score/search_eval_score.py","file_url":"https://github.com/lazzy1225/CoEvoKG/blob/HEAD/coevokg/reward/score/search_eval_score.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d8e7ceb7ac0087e6","mcp_get_code":{"code_sha256":"d8e7ceb7ac0087e6"}},{"arxiv_id":"2606.08397","paper":"/paper/arxiv-2606-08397","title":"TrustMargin: Training-Free Arbitration between Parametric Memory and Retrieved Evidence in Large Language Models","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"mojixu/TrustMargin","path":"src/trustmargin.py","file_url":"https://github.com/mojixu/TrustMargin/blob/HEAD/src/trustmargin.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4366274c4d602004","mcp_get_code":{"code_sha256":"4366274c4d602004"}},{"arxiv_id":"2605.18565","paper":"/paper/arxiv-2605-18565","title":"MINTEVAL: Evaluating Memory under Multi-Target Interference in Long-Horizon Agent Systems","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"amy-hyunji/MINTEval","path":"src/mem_alpha/memalpha/llm_agent/metrics.py","file_url":"https://github.com/amy-hyunji/MINTEval/blob/HEAD/src/mem_alpha/memalpha/llm_agent/metrics.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9fba68468dc933e9","mcp_get_code":{"code_sha256":"9fba68468dc933e9"}},{"arxiv_id":"2603.15892","paper":"/paper/arxiv-2603-15892","title":"Temporal Conflicts in LLMs: Reproducibility Insights from Unifying DYNAMICQA and MULAN","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"terrierteam/temporal_conflicts","path":"fact_mutability/analysis/f1_score.py","file_url":"https://github.com/terrierteam/temporal_conflicts/blob/HEAD/fact_mutability/analysis/f1_score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"96e55856ac1587bd","mcp_get_code":{"code_sha256":"96e55856ac1587bd"}},{"arxiv_id":"2603.04759","paper":"/paper/arxiv-2603-04759","title":"Stacked from One: Multi-Scale Self-Injection for Context Window Extension","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"Clement25/SharedLLM","path":"utils.py","file_url":"https://github.com/Clement25/SharedLLM/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"27f9fcd2b79fea52","mcp_get_code":{"code_sha256":"27f9fcd2b79fea52"}},{"arxiv_id":"2602.13551","paper":"/paper/arxiv-2602-13551","title":"Small Reward Models via Backward Inference","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"yikee/FLIP","path":"metrics.py","file_url":"https://github.com/yikee/FLIP/blob/HEAD/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f5ae0a44fc4a94b0","mcp_get_code":{"code_sha256":"f5ae0a44fc4a94b0"}},{"arxiv_id":"2602.05728","paper":"/paper/arxiv-2602-05728","title":"CompactRAG: Reducing LLM Calls and Token Overhead in Multi-Hop Question Answering","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"How-Young-X/CompactRAG","path":"src/metrics/F1Eval.py","file_url":"https://github.com/How-Young-X/CompactRAG/blob/HEAD/src/metrics/F1Eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a4de989b76f916f5","mcp_get_code":{"code_sha256":"a4de989b76f916f5"}},{"arxiv_id":"2601.22047","paper":"/paper/arxiv-2601-22047","title":"On the Paradoxical Interference between Instruction-Following and Task Solving","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"kijlk/IF-Interference","path":"src/math_and_qa/qa_utils.py","file_url":"https://github.com/kijlk/IF-Interference/blob/HEAD/src/math_and_qa/qa_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"01329cd140ce1ed8","mcp_get_code":{"code_sha256":"01329cd140ce1ed8"}},{"arxiv_id":"2601.21064","paper":"/paper/arxiv-2601-21064","title":"Textual Equilibrium Propagation for Deep Compound AI Systems","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"MinghuiChen43/TEP","path":"src/tep/metrics/f1.py","file_url":"https://github.com/MinghuiChen43/TEP/blob/HEAD/src/tep/metrics/f1.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6ce524d71a9b9d78","mcp_get_code":{"code_sha256":"6ce524d71a9b9d78"}},{"arxiv_id":"2601.09402","paper":"/paper/arxiv-2601-09402","title":"SEEK: Steering LLM Reasoning for RAG via Internal Reasoning Sketches","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"OpenBMB/PAGER","path":"src/evaluate_infer.py","file_url":"https://github.com/OpenBMB/PAGER/blob/HEAD/src/evaluate_infer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"53d9c134452ad7d1","mcp_get_code":{"code_sha256":"53d9c134452ad7d1"}},{"arxiv_id":"2601.09028","paper":"/paper/arxiv-2601-09028","title":"OpenDecoder: Open Large Language Model Decoding to Incorporate Document Quality in RAG","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"fengranMark/OpenDecoder","path":"src/evaluation.py","file_url":"https://github.com/fengranMark/OpenDecoder/blob/HEAD/src/evaluation.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e08cad149be1d1ab","mcp_get_code":{"code_sha256":"e08cad149be1d1ab"}},{"arxiv_id":"2601.07894","paper":"/paper/arxiv-2601-07894","title":"Revealing the Attention Floating Mechanism in Masked Diffusion Models","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"NEUIR/Attention-Floating","path":"evaluate/evaluate_dream.py","file_url":"https://github.com/NEUIR/Attention-Floating/blob/HEAD/evaluate/evaluate_dream.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cb0727ddfdbb00d2","mcp_get_code":{"code_sha256":"cb0727ddfdbb00d2"}},{"arxiv_id":"2601.05903","paper":"/paper/arxiv-2601-05903","title":"HAPS: Hierarchical LLM Routing with Joint Architecture and Parameter Search","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"zihangtian/HAPS","path":"HotpotQA/joint_rl/rl_batch.py","file_url":"https://github.com/zihangtian/HAPS/blob/HEAD/HotpotQA/joint_rl/rl_batch.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c871481cb5ee3d92","mcp_get_code":{"code_sha256":"c871481cb5ee3d92"}},{"arxiv_id":"2510.13494","paper":"/paper/arxiv-2510-13494","title":"LiteraryQA: Towards Effective Evaluation of Long-document Narrative QA","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"SapienzaNLP/LiteraryQA","path":"literaryqa/ngram_metrics.py","file_url":"https://github.com/SapienzaNLP/LiteraryQA/blob/HEAD/literaryqa/ngram_metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"757c1276cb093efc","mcp_get_code":{"code_sha256":"757c1276cb093efc"}},{"arxiv_id":"2505.17005","paper":"/paper/r1-searcher-incentivizing-the-dynamic","title":"R1-Searcher++: Incentivizing the Dynamic Knowledge Acquisition of LLMs via Reinforcement Learning","date":"2025-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"RUCAIBox/R1-Searcher","path":"evaluation/metric_calc_rule.py","file_url":"https://github.com/RUCAIBox/R1-Searcher/blob/HEAD/evaluation/metric_calc_rule.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"53d9c134452ad7d1","mcp_get_code":{"code_sha256":"53d9c134452ad7d1"}},{"arxiv_id":"2504.03101","paper":"/paper/single-pass-document-scanning-for-question","title":"Single-Pass Document Scanning for Question Answering","date":"2025-04-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mambaretriever/mambaretriever","path":"rag_pipeline/metrics.py","file_url":"https://github.com/mambaretriever/mambaretriever/blob/HEAD/rag_pipeline/metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"477258f8938c0fd2","mcp_get_code":{"code_sha256":"477258f8938c0fd2"}},{"arxiv_id":"2502.15975","paper":"/paper/sparsity-may-be-all-you-need-sparse-random","title":"Sparsity May Be All You Need: Sparse Random Parameter Adaptation","date":"2025-02-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"IBM/SpaRTA","path":"sparta/utils.py","file_url":"https://github.com/IBM/SpaRTA/blob/HEAD/sparta/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bf5cd77dea835a0c","mcp_get_code":{"code_sha256":"bf5cd77dea835a0c"}},{"arxiv_id":"2412.17336","paper":"/paper/apex-2-adaptive-and-extreme-summarization-for","title":"APEX$^2$: Adaptive and Extreme Summarization for Personalized Knowledge Graphs","date":"2024-12-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"iDEA-iSAIL-Lab-UIUC/APEX","path":"code/src/metrics.py","file_url":"https://github.com/iDEA-iSAIL-Lab-UIUC/APEX/blob/HEAD/code/src/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1de880067d0e1282","mcp_get_code":{"code_sha256":"1de880067d0e1282"}},{"arxiv_id":"2412.06206","paper":"/paper/sirerag-indexing-similar-and-related","title":"SiReRAG: Indexing Similar and Related Information for Multihop Reasoning","date":"2024-12-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"SalesforceAIResearch/SiReRAG","path":"evaluate_2wiki.py","file_url":"https://github.com/SalesforceAIResearch/SiReRAG/blob/HEAD/evaluate_2wiki.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"4eebef4ef9c478a1","mcp_get_code":{"code_sha256":"4eebef4ef9c478a1"}},{"arxiv_id":"2410.18050","paper":"/paper/longrag-a-dual-perspective-retrieval","title":"LongRAG: A Dual-Perspective Retrieval-Augmented Generation Paradigm for Long-Context Question Answering","date":"2024-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"QingFei1/LongRAG","path":"src/metric.py","file_url":"https://github.com/QingFei1/LongRAG/blob/HEAD/src/metric.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d8e7ceb7ac0087e6","mcp_get_code":{"code_sha256":"d8e7ceb7ac0087e6"}},{"arxiv_id":"2410.07652","paper":"/paper/stableprompt-automatic-prompt-tuning-using","title":"StablePrompt: Automatic Prompt Tuning using Reinforcement Learning for Large Language Models","date":"2024-10-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kmc0207/Stableprompt","path":"utils.py","file_url":"https://github.com/kmc0207/Stableprompt/blob/HEAD/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f019b3b51ce43942","mcp_get_code":{"code_sha256":"f019b3b51ce43942"}},{"arxiv_id":"2410.02694","paper":"/paper/helmet-how-to-evaluate-long-context-language","title":"HELMET: How to Evaluate Long-Context Language Models Effectively and Thoroughly","date":"2024-10-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"princeton-nlp/helmet","path":"utils.py","file_url":"https://github.com/princeton-nlp/helmet/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"27f9fcd2b79fea52","mcp_get_code":{"code_sha256":"27f9fcd2b79fea52"}},{"arxiv_id":"2410.01671","paper":"/paper/bridging-context-gaps-leveraging-coreference","title":"Bridging Context Gaps: Leveraging Coreference Resolution for Long Contextual Understanding","date":"2024-10-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"OceannTwT/LQCA","path":"predict.py","file_url":"https://github.com/OceannTwT/LQCA/blob/HEAD/predict.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2e4cd3f8b7c0198c","mcp_get_code":{"code_sha256":"2e4cd3f8b7c0198c"}},{"arxiv_id":"2408.11745","paper":"/paper/focusllm-scaling-llm-s-context-by-parallel","title":"FocusLLM: Precise Understanding of Long Context by Dynamic Condensing","date":"2024-08-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"leezythu/focusllm","path":"infbench_src/compute_scores.py","file_url":"https://github.com/leezythu/focusllm/blob/HEAD/infbench_src/compute_scores.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"43daed1a27177d83","mcp_get_code":{"code_sha256":"43daed1a27177d83"}},{"arxiv_id":"2407.12784","paper":"/paper/agentpoison-red-teaming-llm-agents-via","title":"AgentPoison: Red-teaming LLM Agents via Poisoning Memory or Knowledge Bases","date":"2024-07-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"BillChan226/AgentPoison","path":"ReAct/wrappers.py","file_url":"https://github.com/BillChan226/AgentPoison/blob/HEAD/ReAct/wrappers.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"32a5f1733d9a8971","mcp_get_code":{"code_sha256":"32a5f1733d9a8971"}},{"arxiv_id":"2407.09450","paper":"/paper/human-like-episodic-memory-for-infinite","title":"Human-like Episodic Memory for Infinite Context LLMs","date":"2024-07-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"em-llm/EM-LLM-model","path":"benchmark/infinitebench_eval.py","file_url":"https://github.com/em-llm/EM-LLM-model/blob/HEAD/benchmark/infinitebench_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2eb701a1cd15182a","mcp_get_code":{"code_sha256":"2eb701a1cd15182a"}},{"arxiv_id":"2407.06192","paper":"/paper/multi-object-hallucination-in-vision-language","title":"Multi-Object Hallucination in Vision-Language Models","date":"2024-07-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sled-group/moh","path":"utils.py","file_url":"https://github.com/sled-group/moh/blob/HEAD/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d464b547251e80a2","mcp_get_code":{"code_sha256":"d464b547251e80a2"}},{"arxiv_id":"2406.18532","paper":"/paper/symbolic-learning-enables-self-evolving","title":"Symbolic Learning Enables Self-Evolving Agents","date":"2024-06-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aiwaves-cn/agents","path":"src/agents/datasets/hotpotqa.py","file_url":"https://github.com/aiwaves-cn/agents/blob/HEAD/src/agents/datasets/hotpotqa.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"32a5f1733d9a8971","mcp_get_code":{"code_sha256":"32a5f1733d9a8971"}},{"arxiv_id":"2406.11612","paper":"/paper/long-code-arena-a-set-of-benchmarks-for-long","title":"Long Code Arena: a Set of Benchmarks for Long-Context Code Models","date":"2024-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jetbrains-research/lca-baselines","path":"bug_localization/src/baselines/metrics/quality_metrics.py","file_url":"https://github.com/jetbrains-research/lca-baselines/blob/HEAD/bug_localization/src/baselines/metrics/quality_metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e1f0a26bc2cc3531","mcp_get_code":{"code_sha256":"e1f0a26bc2cc3531"}},{"arxiv_id":"2406.07528","paper":"/paper/quickllama-query-aware-inference-acceleration","title":"QuickLLaMA: Query-aware Inference Acceleration for Large Language Models","date":"2024-06-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dvlab-research/q-llm","path":"benchmark/infinitebench_eval.py","file_url":"https://github.com/dvlab-research/q-llm/blob/HEAD/benchmark/infinitebench_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2eb701a1cd15182a","mcp_get_code":{"code_sha256":"2eb701a1cd15182a"}},{"arxiv_id":"2406.06326","paper":"/paper/self-tuning-instructing-llms-to-effectively","title":"Self-Tuning: Instructing LLMs to Effectively Acquire New Knowledge through Self-Teaching","date":"2024-06-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhangxy-2019/Effective-Knowledge-Injection","path":"evaluation/csqa_vllm_inference.py","file_url":"https://github.com/zhangxy-2019/Effective-Knowledge-Injection/blob/HEAD/evaluation/csqa_vllm_inference.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"b7db8a24fdb86534","mcp_get_code":{"code_sha256":"b7db8a24fdb86534"}},{"arxiv_id":"2406.04823","paper":"/paper/berts-are-generative-in-context-learners","title":"BERTs are Generative In-Context Learners","date":"2024-06-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ltgoslo/bert-in-context","path":"glue/record.py","file_url":"https://github.com/ltgoslo/bert-in-context/blob/HEAD/glue/record.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d9296df040618579","mcp_get_code":{"code_sha256":"d9296df040618579"}},{"arxiv_id":"2405.20978","paper":"/paper/enhancing-noise-robustness-of-retrieval","title":"Enhancing Noise Robustness of Retrieval-Augmented Language Models with Adaptive Adversarial Training","date":"2024-05-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"calubkk/RAAT","path":"tuner/metrics/em_f1.py","file_url":"https://github.com/calubkk/RAAT/blob/HEAD/tuner/metrics/em_f1.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a38180bf25a2409b","mcp_get_code":{"code_sha256":"a38180bf25a2409b"}},{"arxiv_id":"2405.19119","paper":"/paper/can-graph-learning-improve-task-planning","title":"Can Graph Learning Improve Planning in LLM-based Agents?","date":"2024-05-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wxxshirley/gnn4taskplan","path":"evaluate.py","file_url":"https://github.com/wxxshirley/gnn4taskplan/blob/HEAD/evaluate.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8908fcb355c0a7cb","mcp_get_code":{"code_sha256":"8908fcb355c0a7cb"}},{"arxiv_id":"2405.17613","paper":"/paper/a-framework-for-multi-modal-learning-jointly","title":"Jointly Modeling Inter- & Intra-Modality Dependencies for Multi-modal Learning","date":"2024-05-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"divyam3897/i2m2","path":"avmnist_and_mimic/eval_scripts/performance.py","file_url":"https://github.com/divyam3897/i2m2/blob/HEAD/avmnist_and_mimic/eval_scripts/performance.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"191c253b532edecf","mcp_get_code":{"code_sha256":"191c253b532edecf"}},{"arxiv_id":"2404.16811","paper":"/paper/make-your-llm-fully-utilize-the-context","title":"Make Your LLM Fully Utilize the Context","date":"2024-04-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/FILM","path":"real_world_long/metrics.py","file_url":"https://github.com/microsoft/FILM/blob/HEAD/real_world_long/metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"477258f8938c0fd2","mcp_get_code":{"code_sha256":"477258f8938c0fd2"}},{"arxiv_id":"2404.12145","paper":"/paper/from-form-s-to-meaning-probing-the-semantic","title":"From Form(s) to Meaning: Probing the Semantic Depths of Language Models Using Multisense Consistency","date":"2024-04-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"facebookresearch/multisense_consistency","path":"utils/eval_metrics.py","file_url":"https://github.com/facebookresearch/multisense_consistency/blob/HEAD/utils/eval_metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"a51c9ee8226e3742","mcp_get_code":{"code_sha256":"a51c9ee8226e3742"}},{"arxiv_id":"2403.12393","paper":"/paper/dr3-ask-large-language-models-not-to-give-off","title":"Dr3: Ask Large Language Models Not to Give Off-Topic Answers in Open Domain Multi-Hop Question Answering","date":"2024-03-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gy915/dr3","path":"Dr3/metrics.py","file_url":"https://github.com/gy915/dr3/blob/HEAD/Dr3/metrics.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"32a5f1733d9a8971","mcp_get_code":{"code_sha256":"32a5f1733d9a8971"}},{"arxiv_id":"2403.06840","paper":"/paper/ra-isf-learning-to-answer-and-understand-from","title":"RA-ISF: Learning to Answer and Understand from Retrieval Augmentation via Iterative Self-Feedback","date":"2024-03-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"oceanntwt/ra-isf","path":"main_gpt.py","file_url":"https://github.com/oceanntwt/ra-isf/blob/HEAD/main_gpt.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e79de4f5e7a0318c","mcp_get_code":{"code_sha256":"e79de4f5e7a0318c"}},{"arxiv_id":"2403.04317","paper":"/paper/online-adaptation-of-language-models-with-a","title":"Online Adaptation of Language Models with a Memory of Amortized Contexts","date":"2024-03-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"fdd9a4d20f5a9d6f","mcp_get_code":{"code_sha256":"fdd9a4d20f5a9d6f"}},{"arxiv_id":"2403.03514","paper":"/paper/clongeval-a-chinese-benchmark-for-evaluating","title":"CLongEval: A Chinese Benchmark for Evaluating Long-Context Large Language Models","date":"2024-03-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zexuanqiu/clongeval","path":"metrics.py","file_url":"https://github.com/zexuanqiu/clongeval/blob/HEAD/metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"477258f8938c0fd2","mcp_get_code":{"code_sha256":"477258f8938c0fd2"}},{"arxiv_id":"2402.18048","paper":"/paper/characterizing-truthfulness-in-large-language","title":"Characterizing Truthfulness in Large Language Model Generations with Local Intrinsic Dimension","date":"2024-02-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"fanyin3639/lid-hallucinationdetection","path":"src/metrics.py","file_url":"https://github.com/fanyin3639/lid-hallucinationdetection/blob/HEAD/src/metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9a71eba082ec07de","mcp_get_code":{"code_sha256":"9a71eba082ec07de"}},{"arxiv_id":"2402.17012","paper":"/paper/pandora-s-white-box-increased-training-data","title":"Pandora's White-Box: Precise Training Data Detection and Extraction in Large Language Models","date":"2024-02-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"iamgroot42/mimir","path":"mimir/attacks/attack_utils.py","file_url":"https://github.com/iamgroot42/mimir/blob/HEAD/mimir/attacks/attack_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2ef8c3a1db651c55","mcp_get_code":{"code_sha256":"2ef8c3a1db651c55"}},{"arxiv_id":"2402.16617","paper":"/paper/long-context-language-modeling-with-parallel","title":"Long-Context Language Modeling with Parallel Context Encoding","date":"2024-02-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"princeton-nlp/cepe","path":"utils.py","file_url":"https://github.com/princeton-nlp/cepe/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"27f9fcd2b79fea52","mcp_get_code":{"code_sha256":"27f9fcd2b79fea52"}},{"arxiv_id":"2402.12052","paper":"/paper/small-models-big-insights-leveraging-slim","title":"Small Models, Big Insights: Leveraging Slim Proxy Models To Decide When and What to Retrieve for LLMs","date":"2024-02-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"plageon/slimplm","path":"SKR/skr.py","file_url":"https://github.com/plageon/slimplm/blob/HEAD/SKR/skr.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"42ae926df6227a65","mcp_get_code":{"code_sha256":"42ae926df6227a65"}},{"arxiv_id":"2402.11893","paper":"/paper/discerning-and-resolving-knowledge-conflicts","title":"Discerning and Resolving Knowledge Conflicts through Adaptive Decoding with Contextual Information-Entropy Constraint","date":"2024-02-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"stacy027/coiecd","path":"evaluate.py","file_url":"https://github.com/stacy027/coiecd/blob/HEAD/evaluate.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a8ef93fef742e557","mcp_get_code":{"code_sha256":"a8ef93fef742e557"}},{"arxiv_id":"2402.11534","paper":"/paper/preact-predicting-future-in-react-enhances","title":"PreAct: Prediction Enhances Agent's Planning Ability","date":"2024-02-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"fu-dayuan/preact","path":"LanguageAgentTreeSearch/hotpot/wrappers.py","file_url":"https://github.com/fu-dayuan/preact/blob/HEAD/LanguageAgentTreeSearch/hotpot/wrappers.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"32a5f1733d9a8971","mcp_get_code":{"code_sha256":"32a5f1733d9a8971"}},{"arxiv_id":"2402.04617","paper":"/paper/infllm-unveiling-the-intrinsic-capacity-of","title":"InfLLM: Training-Free Long-Context Extrapolation for LLMs with an Efficient Context Memory","date":"2024-02-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thunlp/infllm","path":"benchmark/infinitebench_eval.py","file_url":"https://github.com/thunlp/infllm/blob/HEAD/benchmark/infinitebench_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2eb701a1cd15182a","mcp_get_code":{"code_sha256":"2eb701a1cd15182a"}},{"arxiv_id":"2402.01032","paper":"/paper/repeat-after-me-transformers-are-better-than","title":"Repeat After Me: Transformers are Better than State Space Models at Copying","date":"2024-02-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sjelassi/transformers_ssm_copy","path":"pretrained_exps/qa_evaluation_utils.py","file_url":"https://github.com/sjelassi/transformers_ssm_copy/blob/HEAD/pretrained_exps/qa_evaluation_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"933e94e11554ccb1","mcp_get_code":{"code_sha256":"933e94e11554ccb1"}},{"arxiv_id":"2401.12585","paper":"/paper/slang-new-concept-comprehension-of-large","title":"SLANG: New Concept Comprehension of Large Language Models","date":"2024-01-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"meirtz/focusonslang-toolbox","path":"eval_urban_f1.py","file_url":"https://github.com/meirtz/focusonslang-toolbox/blob/HEAD/eval_urban_f1.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e6e54258949655c8","mcp_get_code":{"code_sha256":"e6e54258949655c8"}},{"arxiv_id":"2401.11374","paper":"/paper/language-models-as-hierarchy-encoders","title":"Language Models as Hierarchy Encoders","date":"2024-01-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"krr-oxford/hierarchytransformers","path":"src/hierarchy_transformers/evaluation/metrics.py","file_url":"https://github.com/krr-oxford/hierarchytransformers/blob/HEAD/src/hierarchy_transformers/evaluation/metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"018eaf80bea2d0e0","mcp_get_code":{"code_sha256":"018eaf80bea2d0e0"}},{"arxiv_id":"2401.04518","paper":"/paper/the-critique-of-critique","title":"The Critique of Critique","date":"2024-01-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gair-nlp/metacritique","path":"meta_critique/cal_meta_scores.py","file_url":"https://github.com/gair-nlp/metacritique/blob/HEAD/meta_critique/cal_meta_scores.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"9b5da268fc9d2533","mcp_get_code":{"code_sha256":"9b5da268fc9d2533"}},{"arxiv_id":"2311.12022","paper":"/paper/gpqa-a-graduate-level-google-proof-q-a","title":"GPQA: A Graduate-Level Google-Proof Q&A Benchmark","date":"2023-11-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"idavidrein/gpqa","path":"baselines/open_book.py","file_url":"https://github.com/idavidrein/gpqa/blob/HEAD/baselines/open_book.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bf93745bda87d8d6","mcp_get_code":{"code_sha256":"bf93745bda87d8d6"}},{"arxiv_id":"2311.03734","paper":"/paper/leveraging-structured-information-for","title":"Leveraging Structured Information for Explainable Multi-hop Question Answering and Reasoning","date":"2023-11-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bcdnlp/structure-qa","path":"src/hotpot_evaluate.py","file_url":"https://github.com/bcdnlp/structure-qa/blob/HEAD/src/hotpot_evaluate.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"32a5f1733d9a8971","mcp_get_code":{"code_sha256":"32a5f1733d9a8971"}},{"arxiv_id":"2310.15903","paper":"/paper/neural-collapse-in-multi-label-learning-with","title":"Neural Collapse in Multi-label Learning with Pick-all-label Loss","date":"2023-10-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"heimine/nc_mlab","path":"utility.py","file_url":"https://github.com/heimine/nc_mlab/blob/HEAD/utility.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"535fb70574a55b79","mcp_get_code":{"code_sha256":"535fb70574a55b79"}},{"arxiv_id":"2310.14566","paper":"/paper/hallusionbench-you-see-what-you-think-or-you","title":"HallusionBench: An Advanced Diagnostic Suite for Entangled Language Hallucination and Visual Illusion in Large Vision-Language Models","date":"2023-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zli12321/qa_metrics","path":"qa_metrics/f1.py","file_url":"https://github.com/zli12321/qa_metrics/blob/HEAD/qa_metrics/f1.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"032dc5fee649f52c","mcp_get_code":{"code_sha256":"032dc5fee649f52c"}},{"arxiv_id":"2310.13671","paper":"/paper/let-s-synthesize-step-by-step-iterative","title":"Let's Synthesize Step by Step: Iterative Dataset Synthesis with Large Language Models by Extrapolating Errors from Small Models","date":"2023-10-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rickyskywalker/synthesis_step-by-step_official","path":"ModelTraining/trainHelperQA.py","file_url":"https://github.com/rickyskywalker/synthesis_step-by-step_official/blob/HEAD/ModelTraining/trainHelperQA.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1cccb768f5164e2b","mcp_get_code":{"code_sha256":"1cccb768f5164e2b"}},{"arxiv_id":"2310.04406","paper":"/paper/language-agent-tree-search-unifies-reasoning","title":"Language Agent Tree Search Unifies Reasoning Acting and Planning in Language Models","date":"2023-10-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"andyz245/LanguageAgentTreeSearch","path":"hotpot/wrappers.py","file_url":"https://github.com/andyz245/LanguageAgentTreeSearch/blob/HEAD/hotpot/wrappers.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"32a5f1733d9a8971","mcp_get_code":{"code_sha256":"32a5f1733d9a8971"}},{"arxiv_id":"2309.07822","paper":"/paper/catfood-counterfactual-augmented-training-for","title":"CATfOOD: Counterfactual Augmented Training for Improving Out-of-Domain Performance and Calibration","date":"2023-09-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ukplab/catfood","path":"src/shortcuts/evaluate.py","file_url":"https://github.com/ukplab/catfood/blob/HEAD/src/shortcuts/evaluate.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5e88f388713168c7","mcp_get_code":{"code_sha256":"5e88f388713168c7"}},{"arxiv_id":"2309.07822","paper":"/paper/catfood-counterfactual-augmented-training-for","title":"CATfOOD: Counterfactual Augmented Training for Improving Out-of-Domain Performance and Calibration","date":"2023-09-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ukplab/catfood","path":"src/calibration/baseline/calibration_metrics.py","file_url":"https://github.com/ukplab/catfood/blob/HEAD/src/calibration/baseline/calibration_metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8cf315af1221f8a1","mcp_get_code":{"code_sha256":"8cf315af1221f8a1"}},{"arxiv_id":"2308.16475","paper":"/paper/transformer-compression-via-subspace","title":"$\\rm SP^3$: Enhancing Structured Pruning via PCA Projection","date":"2023-08-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hyx1999/sp3","path":"bert/metric/squad.py","file_url":"https://github.com/hyx1999/sp3/blob/HEAD/bert/metric/squad.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cdf4b8c0fec3da02","mcp_get_code":{"code_sha256":"cdf4b8c0fec3da02"}},{"arxiv_id":"2305.15387","paper":"/paper/peek-across-improving-multi-document-modeling","title":"Peek Across: Improving Multi-Document Modeling via Cross-Document Question-Answering","date":"2023-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aviclu/peekacross","path":"trainer_seq2seq_qa.py","file_url":"https://github.com/aviclu/peekacross/blob/HEAD/trainer_seq2seq_qa.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e3ad4a101c45630f","mcp_get_code":{"code_sha256":"e3ad4a101c45630f"}},{"arxiv_id":"2305.06984","paper":"/paper/evaluating-open-domain-question-answering-in","title":"Evaluating Open-Domain Question Answering in the Era of Large Language Models","date":"2023-05-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ehsk/openqa-eval","path":"src/oqaeval/squad_evaluate.py","file_url":"https://github.com/ehsk/openqa-eval/blob/HEAD/src/oqaeval/squad_evaluate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8fd21d86f0ccdd14","mcp_get_code":{"code_sha256":"8fd21d86f0ccdd14"}},{"arxiv_id":"2212.01241","paper":"/paper/analyzing-the-hardware-software-implications","title":"MMBench: Benchmarking End-to-End Multi-modal DNNs and Understanding Their Hardware-Software Implications","date":null,"month_inferred_from_arxiv_id":"2022-12","title_source":"archive","repo":"xfhelen/mmbench","path":"models/eval_scripts/performance.py","file_url":"https://github.com/xfhelen/mmbench/blob/HEAD/models/eval_scripts/performance.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"191c253b532edecf","mcp_get_code":{"code_sha256":"191c253b532edecf"}},{"arxiv_id":"2210.10861","paper":"/paper/qa-domain-adaptation-using-hidden-space","title":"QA Domain Adaptation using Hidden Space Augmentation and Self-Supervised Contrastive Adaptation","date":"2022-10-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"2112c433b9c6d343","mcp_get_code":{"code_sha256":"2112c433b9c6d343"}},{"arxiv_id":"2210.03629","paper":"/paper/react-synergizing-reasoning-and-acting-in","title":"ReAct: Synergizing Reasoning and Acting in Language Models","date":"2022-10-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ysymyth/ReAct","path":"wrappers.py","file_url":"https://github.com/ysymyth/ReAct/blob/HEAD/wrappers.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"32a5f1733d9a8971","mcp_get_code":{"code_sha256":"32a5f1733d9a8971"}},{"arxiv_id":"2204.04046","paper":"/paper/kcd-knowledge-walks-and-textual-cues-enhanced-1","title":"KCD: Knowledge Walks and Textual Cues Enhanced Political Perspective Detection in News Media","date":"2022-04-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wenqian-zhang/kcd","path":"main/allmain/KSD_GatedRGCN.py","file_url":"https://github.com/wenqian-zhang/kcd/blob/HEAD/main/allmain/KSD_GatedRGCN.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"16cbde9056915afd","mcp_get_code":{"code_sha256":"16cbde9056915afd"}},{"arxiv_id":"2202.07586","paper":"/paper/deep-generative-model-with-hierarchical","title":"Deep Generative model with Hierarchical Latent Factors for Time Series Anomaly Detection","date":"2022-02-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cchallu/dghl","path":"src/utils/utils.py","file_url":"https://github.com/cchallu/dghl/blob/HEAD/src/utils/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"93baef73fc0030f8","mcp_get_code":{"code_sha256":"93baef73fc0030f8"}},{"arxiv_id":"2109.13880","paper":"/paper/single-dataset-experts-for-multi-dataset","title":"Single-dataset Experts for Multi-dataset Question Answering","date":"2021-09-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"princeton-nlp/MADE","path":"src/utils/metrics.py","file_url":"https://github.com/princeton-nlp/MADE/blob/HEAD/src/utils/metrics.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a8ef93fef742e557","mcp_get_code":{"code_sha256":"a8ef93fef742e557"}},{"arxiv_id":"2109.08133","paper":"/paper/phrase-retrieval-learns-passage-retrieval-too","title":"Phrase Retrieval Learns Passage Retrieval, Too","date":"2021-09-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"princeton-nlp/DensePhrases","path":"densephrases/utils/eval_utils.py","file_url":"https://github.com/princeton-nlp/DensePhrases/blob/HEAD/densephrases/utils/eval_utils.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"32a5f1733d9a8971","mcp_get_code":{"code_sha256":"32a5f1733d9a8971"}},{"arxiv_id":"2109.04912","paper":"/paper/reasonbert-pre-trained-to-reason-with-distant","title":"ReasonBERT: Pre-trained to Reason with Distant Supervision","date":"2021-09-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sunlab-osu/reasonbert","path":"model/metric.py","file_url":"https://github.com/sunlab-osu/reasonbert/blob/HEAD/model/metric.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"fdd9a4d20f5a9d6f","mcp_get_code":{"code_sha256":"fdd9a4d20f5a9d6f"}},{"arxiv_id":"2108.11179","paper":"/paper/recall-k-surrogate-loss-with-large-batches","title":"Recall@k Surrogate Loss with Large Batches and Similarity Mixup","date":"2021-08-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yash0307/RecallatK_surrogate","path":"src/auxiliaries_nofaiss.py","file_url":"https://github.com/yash0307/RecallatK_surrogate/blob/HEAD/src/auxiliaries_nofaiss.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b0205d59d4a00b5a","mcp_get_code":{"code_sha256":"b0205d59d4a00b5a"}},{"arxiv_id":"2105.14024","paper":"/paper/near-optimal-multi-perturbation-experimental","title":"Near-Optimal Multi-Perturbation Experimental Design for Causal Structure Learning","date":"2021-05-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ssethz/multi-perturbation-ed","path":"represent.py","file_url":"https://github.com/ssethz/multi-perturbation-ed/blob/HEAD/represent.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c8472f6604828bf4","mcp_get_code":{"code_sha256":"c8472f6604828bf4"}},{"arxiv_id":"2105.11259","paper":"/paper/ptr-prompt-tuning-with-rules-for-text","title":"PTR: Prompt Tuning with Rules for Text Classification","date":"2021-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thunlp/PTR","path":"src/run_prompt.py","file_url":"https://github.com/thunlp/PTR/blob/HEAD/src/run_prompt.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0d40c216b798acb4","mcp_get_code":{"code_sha256":"0d40c216b798acb4"}},{"arxiv_id":"2105.02692","paper":"/paper/learning-to-perturb-word-embeddings-for-out","title":"Learning to Perturb Word Embeddings for Out-of-distribution QA","date":"2021-05-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"seanie12/SWEP","path":"mrqa_utils.py","file_url":"https://github.com/seanie12/SWEP/blob/HEAD/mrqa_utils.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cdf4b8c0fec3da02","mcp_get_code":{"code_sha256":"cdf4b8c0fec3da02"}},{"arxiv_id":"2104.14839","paper":"/paper/the-factual-inconsistency-problem-in","title":"The Factual Inconsistency Problem in Abstractive Text Summarization: A Survey","date":"2021-04-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"huffon/factsumm","path":"factsumm/utils/utils.py","file_url":"https://github.com/huffon/factsumm/blob/HEAD/factsumm/utils/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"74f6c7af6784dc33","mcp_get_code":{"code_sha256":"74f6c7af6784dc33"}},{"arxiv_id":"2104.08202","paper":"/paper/q-2-evaluating-factual-consistency-in","title":"$Q^{2}$: Evaluating Factual Consistency in Knowledge-Grounded Dialogues via Question Generation and Question Answering","date":"2021-04-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"orhonovich/q-squared","path":"pipeline/score.py","file_url":"https://github.com/orhonovich/q-squared/blob/HEAD/pipeline/score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e5df2ff8c5c66f6a","mcp_get_code":{"code_sha256":"e5df2ff8c5c66f6a"}},{"arxiv_id":"2104.00640","paper":"/paper/evidence-based-verification-for-real-world","title":"AmbiFC: Fact-Checking Ambiguous Claims with Evidence","date":"2021-04-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"knot-fit-but/claimdissector","path":"src/common/eval_utils.py","file_url":"https://github.com/knot-fit-but/claimdissector/blob/HEAD/src/common/eval_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6b3ccf76356a23b7","mcp_get_code":{"code_sha256":"6b3ccf76356a23b7"}},{"arxiv_id":"2101.10421","paper":"/paper/english-machine-reading-comprehension","title":"English Machine Reading Comprehension Datasets: A Survey","date":"2021-01-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"5d0a37a7c59decf8","mcp_get_code":{"code_sha256":"5d0a37a7c59decf8"}},{"arxiv_id":"2011.01060","paper":"/paper/constructing-a-multi-hop-qa-dataset-for","title":"Constructing A Multi-hop QA Dataset for Comprehensive Evaluation of Reasoning Steps","date":"2020-11-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Alab-NII/2wikimultihop","path":"2wikimultihop_evaluate.py","file_url":"https://github.com/Alab-NII/2wikimultihop/blob/HEAD/2wikimultihop_evaluate.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"32a5f1733d9a8971","mcp_get_code":{"code_sha256":"32a5f1733d9a8971"}},{"arxiv_id":"2010.16021","paper":"/paper/cliniqg4qa-generating-diverse-questions-for","title":"CliniQG4QA: Generating Diverse Questions for Domain Adaptation of Clinical Question Answering","date":"2020-10-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sunlab-osu/CliniQG4QA","path":"QA/evaluate-v1.1_human_generated.py","file_url":"https://github.com/sunlab-osu/CliniQG4QA/blob/HEAD/QA/evaluate-v1.1_human_generated.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cdf4b8c0fec3da02","mcp_get_code":{"code_sha256":"cdf4b8c0fec3da02"}},{"arxiv_id":"2010.12527","paper":"/paper/retrieve-rerank-read-then-iterate-answering","title":"Answering Open-Domain Questions of Varying Reasoning Steps from Text","date":"2020-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"2112c433b9c6d343","mcp_get_code":{"code_sha256":"2112c433b9c6d343"}},{"arxiv_id":"2008.02637","paper":"/paper/question-and-answer-test-train-overlap-in","title":"Question and Answer Test-Train Overlap in Open-Domain Question Answering Datasets","date":"2020-08-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"facebookresearch/QA-Overlap","path":"evaluate.py","file_url":"https://github.com/facebookresearch/QA-Overlap/blob/HEAD/evaluate.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2112c433b9c6d343","mcp_get_code":{"code_sha256":"2112c433b9c6d343"}},{"arxiv_id":"2007.05481","paper":"/paper/starflow-a-spatiotemporal-recurrent-cell-for","title":"STaRFlow: A SpatioTemporal Recurrent Cell for Lightweight Multi-Frame Optical Flow Estimation","date":"2020-07-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pgodet/star_flow","path":"losses.py","file_url":"https://github.com/pgodet/star_flow/blob/HEAD/losses.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e351a73a9141e59d","mcp_get_code":{"code_sha256":"e351a73a9141e59d"}},{"arxiv_id":"2006.12467","paper":"/paper/limits-to-depth-efficiencies-of-self","title":"The Depth-to-Width Interplay in Self-Attention","date":"2020-06-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"uf-hobi-informatics-lab/GatorTron","path":"finetuning/qa/evaluate-v1.1.py","file_url":"https://github.com/uf-hobi-informatics-lab/GatorTron/blob/HEAD/finetuning/qa/evaluate-v1.1.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cdf4b8c0fec3da02","mcp_get_code":{"code_sha256":"cdf4b8c0fec3da02"}},{"arxiv_id":"2003.11113","paper":"/paper/pads-policy-adapted-sampling-for-visual","title":"PADS: Policy-Adapted Sampling for Visual Similarity Learning","date":"2020-03-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Confusezius/CVPR2020_PADS","path":"auxiliaries.py","file_url":"https://github.com/Confusezius/CVPR2020_PADS/blob/HEAD/auxiliaries.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"667f22002df30035","mcp_get_code":{"code_sha256":"667f22002df30035"}},{"arxiv_id":"2002.08473","paper":"/paper/revisiting-training-strategies-and","title":"Revisiting Training Strategies and Generalization Performance in Deep Metric Learning","date":"2020-02-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Confusezius/Deep-Metric-Learning-Baselines","path":"auxiliaries.py","file_url":"https://github.com/Confusezius/Deep-Metric-Learning-Baselines/blob/HEAD/auxiliaries.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d629034ac6132f80","mcp_get_code":{"code_sha256":"d629034ac6132f80"}},{"arxiv_id":"1911.10470","paper":"/paper/learning-to-retrieve-reasoning-paths-over-1","title":"Learning to Retrieve Reasoning Paths over Wikipedia Graph for Question Answering","date":"2019-11-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AkariAsai/learning_to_retrieve_reasoning_paths","path":"eval_utils.py","file_url":"https://github.com/AkariAsai/learning_to_retrieve_reasoning_paths/blob/HEAD/eval_utils.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2112c433b9c6d343","mcp_get_code":{"code_sha256":"2112c433b9c6d343"}},{"arxiv_id":"1911.02896","paper":"/paper/contextualized-sparse-representation-with-1","title":"Contextualized Sparse Representations for Real-Time Open-Domain Question Answering","date":"2019-11-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jhyuklee/sparc","path":"evaluate-v1.1.py","file_url":"https://github.com/jhyuklee/sparc/blob/HEAD/evaluate-v1.1.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2112c433b9c6d343","mcp_get_code":{"code_sha256":"2112c433b9c6d343"}},{"arxiv_id":"1911.02896","paper":"/paper/contextualized-sparse-representation-with-1","title":"Contextualized Sparse Representations for Real-Time Open-Domain Question Answering","date":"2019-11-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jhyuklee/sparc","path":"eval_utils.py","file_url":"https://github.com/jhyuklee/sparc/blob/HEAD/eval_utils.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"32a5f1733d9a8971","mcp_get_code":{"code_sha256":"32a5f1733d9a8971"}},{"arxiv_id":"1910.14520","paper":"/paper/do-multi-hop-readers-dream-of-reasoning","title":"Do Multi-hop Readers Dream of Reasoning Chains?","date":"2019-10-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"helloeve/bert-co-matching","path":"evaluate-v1.1-original.py","file_url":"https://github.com/helloeve/bert-co-matching/blob/HEAD/evaluate-v1.1-original.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2112c433b9c6d343","mcp_get_code":{"code_sha256":"2112c433b9c6d343"}},{"arxiv_id":"1910.14520","paper":"/paper/do-multi-hop-readers-dream-of-reasoning","title":"Do Multi-hop Readers Dream of Reasoning Chains?","date":"2019-10-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"helloeve/bert-co-matching","path":"evaluate-v1.1.py","file_url":"https://github.com/helloeve/bert-co-matching/blob/HEAD/evaluate-v1.1.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c30a45593c8f216f","mcp_get_code":{"code_sha256":"c30a45593c8f216f"}},{"arxiv_id":"1910.09753","paper":"/paper/mrqa-2019-shared-task-evaluating","title":"MRQA 2019 Shared Task: Evaluating Generalization in Reading Comprehension","date":"2019-10-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mrqa/MRQA-Shared-Task-2019","path":"mrqa_official_eval.py","file_url":"https://github.com/mrqa/MRQA-Shared-Task-2019/blob/HEAD/mrqa_official_eval.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"fdd9a4d20f5a9d6f","mcp_get_code":{"code_sha256":"fdd9a4d20f5a9d6f"}},{"arxiv_id":"1909.06356","paper":"/paper/addressing-semantic-drift-in-question","title":"Addressing Semantic Drift in Question Generation for Semi-Supervised Question Answering","date":"2019-09-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ZhangShiyue/QGforQA","path":"LIB/EVAL/evaluate.py","file_url":"https://github.com/ZhangShiyue/QGforQA/blob/HEAD/LIB/EVAL/evaluate.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2112c433b9c6d343","mcp_get_code":{"code_sha256":"2112c433b9c6d343"}},{"arxiv_id":"1909.05803","paper":"/paper/self-assembling-modular-networks-for","title":"Self-Assembling Modular Networks for Interpretable Multi-Hop Reasoning","date":"2019-09-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jiangycTarheel/NMN-MultiHopQA","path":"hotpotqa/evaluate-v1.1.py","file_url":"https://github.com/jiangycTarheel/NMN-MultiHopQA/blob/HEAD/hotpotqa/evaluate-v1.1.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2112c433b9c6d343","mcp_get_code":{"code_sha256":"2112c433b9c6d343"}},{"arxiv_id":"1909.01610","paper":"/paper/answers-unite-unsupervised-metrics-for","title":"Answers Unite! Unsupervised Metrics for Reinforced Summarization Models","date":"2019-09-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"recitalAI/summa-qa","path":"summaqa/f1_squad.py","file_url":"https://github.com/recitalAI/summa-qa/blob/HEAD/summaqa/f1_squad.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2112c433b9c6d343","mcp_get_code":{"code_sha256":"2112c433b9c6d343"}},{"arxiv_id":"1906.05807","paper":"/paper/real-time-open-domain-question-answering-with","title":"Real-Time Open-Domain Question Answering with Dense-Sparse Phrase Index","date":"2019-06-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"uwnlp/denspi","path":"evaluate-v1.1.py","file_url":"https://github.com/uwnlp/denspi/blob/HEAD/evaluate-v1.1.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2112c433b9c6d343","mcp_get_code":{"code_sha256":"2112c433b9c6d343"}},{"arxiv_id":"1906.05394","paper":"/paper/neural-arabic-question-answering","title":"Neural Arabic Question Answering","date":"2019-06-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"husseinmozannar/SOQAL","path":"baselines_reading/evaluate_baselines.py","file_url":"https://github.com/husseinmozannar/SOQAL/blob/HEAD/baselines_reading/evaluate_baselines.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d917b8d6ed9310b1","mcp_get_code":{"code_sha256":"d917b8d6ed9310b1"}},{"arxiv_id":"1905.13453","paper":"/paper/multiqa-an-empirical-investigation-of","title":"MultiQA: An Empirical Investigation of Generalization and Transfer in Reading Comprehension","date":"2019-05-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alontalmor/multiqa","path":"common/official_eval.py","file_url":"https://github.com/alontalmor/multiqa/blob/HEAD/common/official_eval.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"fdd9a4d20f5a9d6f","mcp_get_code":{"code_sha256":"fdd9a4d20f5a9d6f"}},{"arxiv_id":"1905.05460","paper":"/paper/cognitive-graph-for-multi-hop-reading","title":"Cognitive Graph for Multi-Hop Reading Comprehension at Scale","date":"2019-05-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"THUDM/CogQA","path":"hotpot_evaluate_v1.py","file_url":"https://github.com/THUDM/CogQA/blob/HEAD/hotpot_evaluate_v1.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"32a5f1733d9a8971","mcp_get_code":{"code_sha256":"32a5f1733d9a8971"}},{"arxiv_id":"1904.05290","paper":"/paper/iterative-residual-refinement-for-joint","title":"Iterative Residual Refinement for Joint Optical Flow and Occlusion Estimation","date":"2019-04-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"visinf/irr","path":"losses.py","file_url":"https://github.com/visinf/irr/blob/HEAD/losses.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e351a73a9141e59d","mcp_get_code":{"code_sha256":"e351a73a9141e59d"}},{"arxiv_id":"1810.04805","paper":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Microsoft/AzureML-BERT","path":"finetune/evaluate_squad.py","file_url":"https://github.com/Microsoft/AzureML-BERT/blob/HEAD/finetune/evaluate_squad.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2112c433b9c6d343","mcp_get_code":{"code_sha256":"2112c433b9c6d343"}},{"arxiv_id":"1810.04805","paper":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"louis-udm/ner-bert-crf","path":"NER_BERT_CRF.py","file_url":"https://github.com/louis-udm/ner-bert-crf/blob/HEAD/NER_BERT_CRF.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3cbc191d5ba7a810","mcp_get_code":{"code_sha256":"3cbc191d5ba7a810"}},{"arxiv_id":"1809.09600","paper":"/paper/hotpotqa-a-dataset-for-diverse-explainable","title":"HotpotQA: A Dataset for Diverse, Explainable Multi-hop Question Answering","date":"2018-09-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hotpotqa/hotpot","path":"hotpot_evaluate_v1.py","file_url":"https://github.com/hotpotqa/hotpot/blob/HEAD/hotpot_evaluate_v1.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"32a5f1733d9a8971","mcp_get_code":{"code_sha256":"32a5f1733d9a8971"}},{"arxiv_id":"1808.10399","paper":"/paper/modeling-empathy-and-distress-in-reaction-to","title":"Modeling Empathy and Distress in Reaction to News Stories","date":"2018-08-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wwbp/empathic_reactions","path":"modeling/main/crossvalidation/experiment.py","file_url":"https://github.com/wwbp/empathic_reactions/blob/HEAD/modeling/main/crossvalidation/experiment.py","status":"ran_fixture","verification_level":1,"contract_check":"RAISES","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1de7d76b6643c22f","mcp_get_code":{"code_sha256":"1de7d76b6643c22f"}},{"arxiv_id":"1804.07927","paper":"/paper/duorc-towards-complex-language-understanding","title":"DuoRC: Towards Complex Language Understanding with Paraphrased Reading Comprehension","date":"2018-04-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"duorc/duorc","path":"evaluate.py","file_url":"https://github.com/duorc/duorc/blob/HEAD/evaluate.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2112c433b9c6d343","mcp_get_code":{"code_sha256":"2112c433b9c6d343"}},{"arxiv_id":"1804.07726","paper":"/paper/phrase-indexed-question-answering-a-new","title":"Phrase-Indexed Question Answering: A New Challenge for Scalable Document Comprehension","date":"2018-04-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"uwnlp/piqa","path":"squad/piqa_evaluate.py","file_url":"https://github.com/uwnlp/piqa/blob/HEAD/squad/piqa_evaluate.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2112c433b9c6d343","mcp_get_code":{"code_sha256":"2112c433b9c6d343"}},{"arxiv_id":"1711.05116","paper":"/paper/evidence-aggregation-for-answer-re-ranking-in","title":"Evidence Aggregation for Answer Re-Ranking in Open-Domain Question Answering","date":"2017-11-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shuohangwang/mprc","path":"trainedmodel/evaluation/quasart/evaluate-v1.1.py","file_url":"https://github.com/shuohangwang/mprc/blob/HEAD/trainedmodel/evaluation/quasart/evaluate-v1.1.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2112c433b9c6d343","mcp_get_code":{"code_sha256":"2112c433b9c6d343"}},{"arxiv_id":"1711.05116","paper":"/paper/evidence-aggregation-for-answer-re-ranking-in","title":"Evidence Aggregation for Answer Re-Ranking in Open-Domain Question Answering","date":"2017-11-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shuohangwang/mprc","path":"trainedmodel/evaluation/unftriviaqa/triviaqa_evaluation.py","file_url":"https://github.com/shuohangwang/mprc/blob/HEAD/trainedmodel/evaluation/unftriviaqa/triviaqa_evaluation.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"5d0a37a7c59decf8","mcp_get_code":{"code_sha256":"5d0a37a7c59decf8"}},{"arxiv_id":"1705.03551","paper":"/paper/triviaqa-a-large-scale-distantly-supervised","title":"TriviaQA: A Large Scale Distantly Supervised Challenge Dataset for Reading Comprehension","date":"2017-05-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mandarjoshi90/triviaqa","path":"evaluation/triviaqa_evaluation.py","file_url":"https://github.com/mandarjoshi90/triviaqa/blob/HEAD/evaluation/triviaqa_evaluation.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"5d0a37a7c59decf8","mcp_get_code":{"code_sha256":"5d0a37a7c59decf8"}},{"arxiv_id":"1703.06870","paper":"/paper/mask-r-cnn","title":"Mask R-CNN","date":"2017-03-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AKASH2907/bird-species-classification","path":"evaluate.py","file_url":"https://github.com/AKASH2907/bird-species-classification/blob/HEAD/evaluate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7a27a90793b810e0","mcp_get_code":{"code_sha256":"7a27a90793b810e0"}},{"arxiv_id":"1702.01417","paper":"/paper/all-but-the-top-simple-and-effective","title":"All-but-the-Top: Simple and Effective Postprocessing for Word Representations","date":"2017-02-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lgalke/vec4ir","path":"vec4ir/base.py","file_url":"https://github.com/lgalke/vec4ir/blob/HEAD/vec4ir/base.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4e503c5d0be1ac1c","mcp_get_code":{"code_sha256":"4e503c5d0be1ac1c"}},{"arxiv_id":"1611.01603","paper":"/paper/bidirectional-attention-flow-for-machine","title":"Bidirectional Attention Flow for Machine Comprehension","date":"2016-11-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"allenai/bi-att-flow","path":"squad/evaluate-v1.1.py","file_url":"https://github.com/allenai/bi-att-flow/blob/HEAD/squad/evaluate-v1.1.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2112c433b9c6d343","mcp_get_code":{"code_sha256":"2112c433b9c6d343"}},{"arxiv_id":"1608.07905","paper":"/paper/machine-comprehension-using-match-lstm-and","title":"Machine Comprehension Using Match-LSTM and Answer Pointer","date":"2016-08-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"2112c433b9c6d343","mcp_get_code":{"code_sha256":"2112c433b9c6d343"}},{"arxiv_id":"1606.05250","paper":"/paper/squad-100000-questions-for-machine","title":"SQuAD: 100,000+ Questions for Machine Comprehension of Text","date":"2016-06-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"2112c433b9c6d343","mcp_get_code":{"code_sha256":"2112c433b9c6d343"}},{"arxiv_id":"1508.06615","paper":"/paper/character-aware-neural-language-models","title":"Character-Aware Neural Language Models","date":"2015-08-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"NLPLearn/QANet","path":"evaluate-v1.1.py","file_url":"https://github.com/NLPLearn/QANet/blob/HEAD/evaluate-v1.1.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2112c433b9c6d343","mcp_get_code":{"code_sha256":"2112c433b9c6d343"}},{"arxiv_id":"Patel_Recallk_Surrogate_Loss_With_Large_Batches_and_Similarity_Mixup_CVPR_2022_paper","paper":null,"title":"arXiv:Patel_Recallk_Surrogate_Loss_With_Large_Batches_and_Similarity_Mixup_CVPR_2022_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"yash0307/RecallatK","path":"src/auxiliaries_nofaiss.py","file_url":"https://github.com/yash0307/RecallatK/blob/HEAD/src/auxiliaries_nofaiss.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b0205d59d4a00b5a","mcp_get_code":{"code_sha256":"b0205d59d4a00b5a"}},{"arxiv_id":"2025.acl-long.191","paper":null,"title":"arXiv:2025.acl-long.191","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"UKPLab/acl2025-diverse-cot","path":"src/hotpotqa_evaluation.py","file_url":"https://github.com/UKPLab/acl2025-diverse-cot/blob/HEAD/src/hotpotqa_evaluation.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"32a5f1733d9a8971","mcp_get_code":{"code_sha256":"32a5f1733d9a8971"}},{"arxiv_id":"2025.acl-long.1500","paper":null,"title":"arXiv:2025.acl-long.1500","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"leezythu/FocusLLM","path":"infbench_src/compute_scores.py","file_url":"https://github.com/leezythu/FocusLLM/blob/HEAD/infbench_src/compute_scores.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"43daed1a27177d83","mcp_get_code":{"code_sha256":"43daed1a27177d83"}},{"arxiv_id":"2024.findings-emnlp.230","paper":null,"title":"arXiv:2024.findings-emnlp.230","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"zexuanqiu/CLongEval","path":"metrics.py","file_url":"https://github.com/zexuanqiu/CLongEval/blob/HEAD/metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"477258f8938c0fd2","mcp_get_code":{"code_sha256":"477258f8938c0fd2"}},{"arxiv_id":"2024.findings-acl.975","paper":null,"title":"arXiv:2024.findings-acl.975","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"CLU-UML/MedDec","path":"eval_gen.py","file_url":"https://github.com/CLU-UML/MedDec/blob/HEAD/eval_gen.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"dd90568980882938","mcp_get_code":{"code_sha256":"dd90568980882938"}},{"arxiv_id":"2023.findings-emnlp.835","paper":null,"title":"arXiv:2023.findings-emnlp.835","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"THU-KEG/ProbTree","path":"src/2wiki/RoHT/evaluate.py","file_url":"https://github.com/THU-KEG/ProbTree/blob/HEAD/src/2wiki/RoHT/evaluate.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"32a5f1733d9a8971","mcp_get_code":{"code_sha256":"32a5f1733d9a8971"}},{"arxiv_id":"2023.emnlp-main.803","paper":null,"title":"arXiv:2023.emnlp-main.803","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"yisunlp/Anti-CF","path":"utils/squad_evaluate.py","file_url":"https://github.com/yisunlp/Anti-CF/blob/HEAD/utils/squad_evaluate.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2112c433b9c6d343","mcp_get_code":{"code_sha256":"2112c433b9c6d343"}},{"arxiv_id":"2021.acl-long.48","paper":null,"title":"arXiv:2021.acl-long.48","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"PluviophileYU/COSY","path":"XQA/src/evaluate_v1_1.py","file_url":"https://github.com/PluviophileYU/COSY/blob/HEAD/XQA/src/evaluate_v1_1.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2112c433b9c6d343","mcp_get_code":{"code_sha256":"2112c433b9c6d343"}}]}