{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/metric-max-over-ground-truths","entry":"metric_max_over_ground_truths","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-25T09:33:49+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":12,"n_papers_ran":9,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":4,"n_samples_ran":2,"n_samples_fingerprinted":0,"n_places":12,"n_places_pointer_only":3,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":1,"ran_fixture":0,"ran":1,"unverified":2},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2604.05818","paper":"/paper/arxiv-2604-05818","title":"WikiSeeker: Rethinking the Role of Vision-Language Models in Knowledge-Based Visual Question Answering","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"zhuyjan/WikiSeeker","path":"utils/infoseek_evaluation_utils.py","file_url":"https://github.com/zhuyjan/WikiSeeker/blob/HEAD/utils/infoseek_evaluation_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"921382b550348607","mcp_get_code":{"code_sha256":"921382b550348607"}},{"arxiv_id":"2506.21215","paper":"/paper/unveiling-causal-reasoning-in-large-language","title":"Unveiling Causal Reasoning in Large Language Models: Reality or Mirage?","date":"2025-06-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Haoang97/CausalProbe-2024","path":"metrics.py","file_url":"https://github.com/Haoang97/CausalProbe-2024/blob/HEAD/metrics.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6829ceb72569fa12","mcp_get_code":{"code_sha256":"6829ceb72569fa12"}},{"arxiv_id":"2406.11357","paper":"/paper/textit-refiner-restructure-retrieval-content","title":"Refiner: Restructure Retrieval Content Efficiently to Advance Question-Answering Capabilities","date":"2024-06-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"allen-li1231/refiner-rag","path":"metrics.py","file_url":"https://github.com/allen-li1231/refiner-rag/blob/HEAD/metrics.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"CC-BY-4.0","inline_ok":false,"code_sha256_prefix":"6829ceb72569fa12","mcp_get_code":{"code_sha256":"6829ceb72569fa12"}},{"arxiv_id":"2402.13991","paper":"/paper/analysing-the-impact-of-sequence-composition","title":"Analysing The Impact of Sequence Composition on Language Model Pre-Training","date":"2024-02-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yuzhaouoe/pretraining-data-packing","path":"evaluation/eval_utils.py","file_url":"https://github.com/yuzhaouoe/pretraining-data-packing/blob/HEAD/evaluation/eval_utils.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6829ceb72569fa12","mcp_get_code":{"code_sha256":"6829ceb72569fa12"}},{"arxiv_id":"2402.03216","paper":"/paper/bge-m3-embedding-multi-lingual-multi","title":"BGE M3-Embedding: Multi-Lingual, Multi-Functionality, Multi-Granularity Text Embeddings Through Self-Knowledge Distillation","date":"2024-02-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"allen-li1231/treehop-rag","path":"metrics.py","file_url":"https://github.com/allen-li1231/treehop-rag/blob/HEAD/metrics.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6829ceb72569fa12","mcp_get_code":{"code_sha256":"6829ceb72569fa12"}},{"arxiv_id":"2310.11511","paper":"/paper/self-rag-learning-to-retrieve-generate-and","title":"Self-RAG: Learning to Retrieve, Generate, and Critique through Self-Reflection","date":"2023-10-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"AkariAsai/self-rag","path":"retrieval_lm/metrics.py","file_url":"https://github.com/AkariAsai/self-rag/blob/HEAD/retrieval_lm/metrics.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6829ceb72569fa12","mcp_get_code":{"code_sha256":"6829ceb72569fa12"}},{"arxiv_id":"2305.01938","paper":"/paper/doc2soargraph-discrete-reasoning-over","title":"Doc2SoarGraph: Discrete Reasoning over Visually-Rich Table-Text Documents via Semantic-Oriented Hierarchical Graphs","date":"2023-05-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"fengbinzhu/doc2soargraph","path":"tatqa_metric.py","file_url":"https://github.com/fengbinzhu/doc2soargraph/blob/HEAD/tatqa_metric.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"03692148eacb9928","mcp_get_code":{"code_sha256":"03692148eacb9928"}},{"arxiv_id":"2302.11713","paper":"/paper/can-pre-trained-vision-and-language-models","title":"Can Pre-trained Vision and Language Models Answer Visual Information-Seeking Questions?","date":"2023-02-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"edchengg/infoseek_eval","path":"infoseek_eval.py","file_url":"https://github.com/edchengg/infoseek_eval/blob/HEAD/infoseek_eval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"921382b550348607","mcp_get_code":{"code_sha256":"921382b550348607"}},{"arxiv_id":"2210.10861","paper":"/paper/qa-domain-adaptation-using-hidden-space","title":"QA Domain Adaptation using Hidden Space Augmentation and Self-Supervised Contrastive Adaptation","date":"2022-10-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"6829ceb72569fa12","mcp_get_code":{"code_sha256":"6829ceb72569fa12"}},{"arxiv_id":"2105.07624","paper":"/paper/tat-qa-a-question-answering-benchmark-on-a","title":"TAT-QA: A Question Answering Benchmark on a Hybrid of Tabular and Textual Content in Finance","date":"2021-05-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"NExTplusplus/TAT-QA","path":"tatqa_metric.py","file_url":"https://github.com/NExTplusplus/TAT-QA/blob/HEAD/tatqa_metric.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"03692148eacb9928","mcp_get_code":{"code_sha256":"03692148eacb9928"}},{"arxiv_id":"2104.08773","paper":"/paper/natural-instructions-benchmarking","title":"Cross-Task Generalization via Natural Language Crowdsourcing Instructions","date":"2021-04-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"allenai/natural-instructions-v1","path":"src/evaluation.py","file_url":"https://github.com/allenai/natural-instructions-v1/blob/HEAD/src/evaluation.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"6829ceb72569fa12","mcp_get_code":{"code_sha256":"6829ceb72569fa12"}},{"arxiv_id":"aaai_34690","paper":null,"title":"arXiv:aaai_34690","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"ZongyueQin/DSBD","path":"sampling/utils.py","file_url":"https://github.com/ZongyueQin/DSBD/blob/HEAD/sampling/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"75706fc9216378b9","mcp_get_code":{"code_sha256":"75706fc9216378b9"}}]}