{"url":"/task/long-context-understanding","name":"Long-Context Understanding","slug":"long-context-understanding","description_markdown":null,"categories":[{"name":"Natural Language Processing","url":"/area/natural-language-processing"}],"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","slug_source":"archive_url"},"counts":{"papers_tagged":81,"papers_with_code":54,"benchmarks":5,"benchmark_tables_in_archive":5,"benchmark_tables_shown":5,"benchmark_tables_withheld_as_spam":0,"benchmark_definition":"a leaderboard table with at least one row; benchmark_tables_shown also counts the zero-row tables; benchmark_tables_in_archive adds the tables withheld as spam","datasets":4,"subtasks":0,"parent_tasks":0},"benchmarks":[{"leaderboard":"/sota/long-context-understanding-on-mmneedle","slug":"long-context-understanding-on-mmneedle","dataset":"MMNeedle","dataset_url":"/dataset/mmneedle","rows_in_archive":12,"metrics":["1 Image, 4*4 Stitching, Exact Accuracy","1 Image, 8*8 Stitching, Exact Accuracy","1 Image, 2*2 Stitching, Exact Accuracy","10 Images, 1*1 Stitching, Exact Accuracy","10 Images, 2*2 Stitching, Exact Accuracy","10 Images, 4*4 Stitching, Exact Accuracy","10 Images, 8*8 Stitching, Exact Accuracy"],"first_row_in_archive_order":{"model":"GPT-4o","paper_title":"GPT-4 Technical Report","paper_url":"/paper/gpt-4-technical-report-1","paper_date":"2023-03-15","arxiv_id":"2303.08774","code_links":[{"title":"openai/evals","url":"https://github.com/openai/evals"},{"title":"shmsw25/factscore","url":"https://github.com/shmsw25/factscore"},{"title":"unispac/visual-adversarial-examples-jailbreak-large-language-models","url":"https://github.com/unispac/visual-adversarial-examples-jailbreak-large-language-models"},{"title":"gpt4life/alpagasus","url":"https://github.com/gpt4life/alpagasus"},{"title":"emrgnt-cmplxty/zero-shot-replication","url":"https://github.com/emrgnt-cmplxty/zero-shot-replication"},{"title":"ethz-privsec/superhuman-ai-consistency","url":"https://github.com/ethz-privsec/superhuman-ai-consistency"},{"title":"ethz-spylab/superhuman-ai-consistency","url":"https://github.com/ethz-spylab/superhuman-ai-consistency"},{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"},{"title":"AUCOHL/RTL-Repo","url":"https://github.com/AUCOHL/RTL-Repo"},{"title":"zach-zhiling-zheng/reticular_chemist","url":"https://github.com/zach-zhiling-zheng/reticular_chemist"},{"title":"lflage/openfactscore","url":"https://github.com/lflage/openfactscore"}],"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":1}}},{"leaderboard":"/sota/long-context-understanding-on-ada-leval","slug":"long-context-understanding-on-ada-leval","dataset":"Ada-LEval (BestAnswer)","dataset_url":null,"rows_in_archive":10,"metrics":["1k","2k","4k","6k","8k","12k","16k","32k","64k","128k"],"first_row_in_archive_order":{"model":"GPT-4-Turbo-1106","paper_title":"GPT-4 Technical Report","paper_url":"/paper/gpt-4-technical-report-1","paper_date":"2023-03-15","arxiv_id":"2303.08774","code_links":[{"title":"openai/evals","url":"https://github.com/openai/evals"},{"title":"shmsw25/factscore","url":"https://github.com/shmsw25/factscore"},{"title":"unispac/visual-adversarial-examples-jailbreak-large-language-models","url":"https://github.com/unispac/visual-adversarial-examples-jailbreak-large-language-models"},{"title":"gpt4life/alpagasus","url":"https://github.com/gpt4life/alpagasus"},{"title":"emrgnt-cmplxty/zero-shot-replication","url":"https://github.com/emrgnt-cmplxty/zero-shot-replication"},{"title":"ethz-privsec/superhuman-ai-consistency","url":"https://github.com/ethz-privsec/superhuman-ai-consistency"},{"title":"ethz-spylab/superhuman-ai-consistency","url":"https://github.com/ethz-spylab/superhuman-ai-consistency"},{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"},{"title":"AUCOHL/RTL-Repo","url":"https://github.com/AUCOHL/RTL-Repo"},{"title":"zach-zhiling-zheng/reticular_chemist","url":"https://github.com/zach-zhiling-zheng/reticular_chemist"},{"title":"lflage/openfactscore","url":"https://github.com/lflage/openfactscore"}],"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":1}}},{"leaderboard":"/sota/long-context-understanding-on-ada-leval-tsort","slug":"long-context-understanding-on-ada-leval-tsort","dataset":"Ada-LEval (TSort)","dataset_url":null,"rows_in_archive":10,"metrics":["2k","4k","8k","16k","32k","64k","128k"],"first_row_in_archive_order":{"model":"GPT-4-Turbo-1106","paper_title":"GPT-4 Technical Report","paper_url":"/paper/gpt-4-technical-report-1","paper_date":"2023-03-15","arxiv_id":"2303.08774","code_links":[{"title":"openai/evals","url":"https://github.com/openai/evals"},{"title":"shmsw25/factscore","url":"https://github.com/shmsw25/factscore"},{"title":"unispac/visual-adversarial-examples-jailbreak-large-language-models","url":"https://github.com/unispac/visual-adversarial-examples-jailbreak-large-language-models"},{"title":"gpt4life/alpagasus","url":"https://github.com/gpt4life/alpagasus"},{"title":"emrgnt-cmplxty/zero-shot-replication","url":"https://github.com/emrgnt-cmplxty/zero-shot-replication"},{"title":"ethz-privsec/superhuman-ai-consistency","url":"https://github.com/ethz-privsec/superhuman-ai-consistency"},{"title":"ethz-spylab/superhuman-ai-consistency","url":"https://github.com/ethz-spylab/superhuman-ai-consistency"},{"title":"eternityyw/tram-benchmark","url":"https://github.com/eternityyw/tram-benchmark"},{"title":"AUCOHL/RTL-Repo","url":"https://github.com/AUCOHL/RTL-Repo"},{"title":"zach-zhiling-zheng/reticular_chemist","url":"https://github.com/zach-zhiling-zheng/reticular_chemist"},{"title":"lflage/openfactscore","url":"https://github.com/lflage/openfactscore"}],"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":1}}},{"leaderboard":"/sota/long-context-understanding-on-l-eval","slug":"long-context-understanding-on-l-eval","dataset":"L-Eval","dataset_url":"/dataset/l-eval","rows_in_archive":4,"metrics":["Average Score"],"first_row_in_archive_order":{"model":"GALI(Llama3-8b-ins-4k-to-16k)","paper_title":"A Training-Free Length Extrapolation Approach for LLMs: Greedy Attention Logit Interpolation (GALI)","paper_url":"/paper/a-training-free-length-extrapolation-approach","paper_date":"2025-02-04","arxiv_id":"2502.02659","code_links":[{"title":"academycityl/gali","url":"https://github.com/academycityl/gali"}],"syntology":null}},{"leaderboard":"/sota/long-context-understanding-on-longbench","slug":"long-context-understanding-on-longbench","dataset":"LongBench","dataset_url":"/dataset/longbench","rows_in_archive":3,"metrics":["Average Score"],"first_row_in_archive_order":{"model":"GALI(Llama3-8b-ins-4k-to-16k)","paper_title":"A Training-Free Length Extrapolation Approach for LLMs: Greedy Attention Logit Interpolation (GALI)","paper_url":"/paper/a-training-free-length-extrapolation-approach","paper_date":"2025-02-04","arxiv_id":"2502.02659","code_links":[{"title":"academycityl/gali","url":"https://github.com/academycityl/gali"}],"syntology":null}}],"datasets":[{"url":"/dataset/l-eval","name":"L-Eval","full_name":"","num_papers_in_archive":43},{"url":"/dataset/mmneedle","name":"MMNeedle","full_name":"Multimodal Needle in a Haystack","num_papers_in_archive":12},{"url":"/dataset/longbench","name":"LongBench","full_name":"","num_papers_in_archive":8},{"url":"/dataset/nolima","name":"NoLiMa","full_name":"NoLiMa: Long-Context Evaluation Beyond Literal Matching","num_papers_in_archive":1}],"subtasks":[],"parent_tasks":[],"papers":{"order":"repositories listed in the archive (desc), then date (desc); the archive holds no stars","population":"papers tagged with this task that list at least one repository in the archive","shown":30,"of":54,"tagged_in_all":81,"items":[{"url":"/paper/judging-llm-as-a-judge-with-mt-bench-and-1","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","date":"2023-06-09","arxiv_id":"2306.05685","repositories_listed":11,"syntology":{"n":12,"n_ran":9,"n_unverified":3,"n_pointer_only":0}},{"url":"/paper/gpt-4-technical-report-1","title":"GPT-4 Technical Report","date":"2023-03-15","arxiv_id":"2303.08774","repositories_listed":11,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":1}},{"url":"/paper/glm-130b-an-open-bilingual-pre-trained-model","title":"GLM-130B: An Open Bilingual Pre-trained Model","date":"2022-10-05","arxiv_id":"2210.02414","repositories_listed":9,"syntology":{"n":21,"n_ran":13,"n_unverified":8,"n_pointer_only":0}},{"url":"/paper/ruler-what-s-the-real-context-size-of-your","title":"RULER: What's the Real Context Size of Your Long-Context Language Models?","date":"2024-04-09","arxiv_id":"2404.06654","repositories_listed":6,"syntology":{"n":5,"n_ran":5,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/gated-delta-networks-improving-mamba2-with","title":"Gated Delta Networks: Improving Mamba2 with Delta Rule","date":"2024-12-09","arxiv_id":"2412.06464","repositories_listed":4,"syntology":{"n":13,"n_ran":11,"n_unverified":2,"n_pointer_only":7}},{"url":"/paper/cogvlm-visual-expert-for-pretrained-language","title":"CogVLM: Visual Expert for Pretrained Language Models","date":"2023-11-06","arxiv_id":"2311.03079","repositories_listed":4,"syntology":null},{"url":"/paper/instructblip-towards-general-purpose-vision","title":"InstructBLIP: Towards General-purpose Vision-Language Models with Instruction Tuning","date":"2023-05-11","arxiv_id":"2305.06500","repositories_listed":4,"syntology":null},{"url":"/paper/fables-evaluating-faithfulness-and-content","title":"FABLES: Evaluating faithfulness and content selection in book-length summarization","date":"2024-04-01","arxiv_id":"2404.01261","repositories_listed":3,"syntology":{"n":3,"n_ran":3,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/internlm2-technical-report","title":"InternLM2 Technical Report","date":"2024-03-26","arxiv_id":"2403.17297","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/longbench-a-bilingual-multitask-benchmark-for","title":"LongBench: A Bilingual, Multitask Benchmark for Long Context Understanding","date":"2023-08-28","arxiv_id":"2308.14508","repositories_listed":3,"syntology":{"n":1,"n_ran":1,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/l-citeeval-do-long-context-models-truly","title":"L-CiteEval: Do Long-Context Models Truly Leverage Context for Responding?","date":"2024-10-03","arxiv_id":"2410.02115","repositories_listed":2,"syntology":null},{"url":"/paper/long-context-llms-struggle-with-long-in","title":"Long-context LLMs Struggle with Long In-context Learning","date":"2024-04-02","arxiv_id":"2404.02060","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_unverified":0,"n_pointer_only":0}},{"url":"/paper/mplug-owl2-revolutionizing-multi-modal-large","title":"mPLUG-Owl2: Revolutionizing Multi-modal Large Language Model with Modality Collaboration","date":"2023-11-07","arxiv_id":"2311.04257","repositories_listed":2,"syntology":{"n":3,"n_ran":1,"n_unverified":2,"n_pointer_only":3}},{"url":"/paper/s3eval-a-synthetic-scalable-systematic","title":"S3Eval: A Synthetic, Scalable, Systematic Evaluation Suite for Large Language Models","date":"2023-10-23","arxiv_id":"2310.15147","repositories_listed":2,"syntology":{"n":26,"n_ran":21,"n_unverified":5,"n_pointer_only":26}},{"url":"/paper/ref-long-benchmarking-the-long-context","title":"Ref-Long: Benchmarking the Long-context Referencing Capability of Long-context Language Models","date":"2025-07-13","arxiv_id":"2507.09506","repositories_listed":1,"syntology":null},{"url":"/paper/cache-me-if-you-can-how-many-kvs-do-you-need","title":"Cache Me If You Can: How Many KVs Do You Need for Effective Long-Context LMs?","date":"2025-06-20","arxiv_id":"2506.17121","repositories_listed":1,"syntology":null},{"url":"/paper/dam-dynamic-attention-mask-for-long-context","title":"DAM: Dynamic Attention Mask for Long-Context Large Language Model Inference Acceleration","date":"2025-06-06","arxiv_id":"2506.11104","repositories_listed":1,"syntology":null},{"url":"/paper/mesanet-sequence-modeling-by-locally-optimal","title":"MesaNet: Sequence Modeling by Locally Optimal Test-Time Training","date":"2025-06-05","arxiv_id":"2506.05233","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_unverified":1,"n_pointer_only":0}},{"url":"/paper/specextend-a-drop-in-enhancement-for","title":"SpecExtend: A Drop-in Enhancement for Speculative Decoding of Long Sequences","date":"2025-05-27","arxiv_id":"2505.20776","repositories_listed":1,"syntology":null},{"url":"/paper/minilongbench-the-low-cost-long-context","title":"MiniLongBench: The Low-cost Long Context Understanding Benchmark for Large Language Models","date":"2025-05-26","arxiv_id":"2505.19959","repositories_listed":1,"syntology":null},{"url":"/paper/can-compressed-llms-truly-act-an-empirical","title":"Can Compressed LLMs Truly Act? An Empirical Evaluation of Agentic Capabilities in LLM Compression","date":"2025-05-26","arxiv_id":"2505.19433","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_unverified":0,"n_pointer_only":2}},{"url":"/paper/too-long-didn-t-model-decomposing-llm-long","title":"Too Long, Didn't Model: Decomposing LLM Long-Context Understanding With Novels","date":"2025-05-20","arxiv_id":"2505.14925","repositories_listed":1,"syntology":null},{"url":"/paper/livelongbench-tackling-long-context","title":"LiveLongBench: Tackling Long-Context Understanding for Spoken Texts from Live Streams","date":"2025-04-24","arxiv_id":"2504.17366","repositories_listed":1,"syntology":null},{"url":"/paper/longmamba-enhancing-mamba-s-long-context","title":"LongMamba: Enhancing Mamba's Long Context Capabilities via Training-Free Receptive Field Enlargement","date":"2025-04-22","arxiv_id":"2504.16053","repositories_listed":1,"syntology":null},{"url":"/paper/kimi-vl-technical-report","title":"Kimi-VL Technical Report","date":"2025-04-10","arxiv_id":"2504.07491","repositories_listed":1,"syntology":null},{"url":"/paper/curie-evaluating-llms-on-multitask-scientific","title":"CURIE: Evaluating LLMs On Multitask Scientific Long Context Understanding and Reasoning","date":"2025-03-14","arxiv_id":"2503.13517","repositories_listed":1,"syntology":null},{"url":"/paper/token-weighting-for-long-range-language","title":"Token Weighting for Long-Range Language Modeling","date":"2025-03-12","arxiv_id":"2503.09202","repositories_listed":1,"syntology":null},{"url":"/paper/longprolip-a-probabilistic-vision-language","title":"LongProLIP: A Probabilistic Vision-Language Model with Long Context Text","date":"2025-03-11","arxiv_id":"2503.08048","repositories_listed":1,"syntology":{"n":16,"n_ran":7,"n_unverified":9,"n_pointer_only":16}},{"url":"/paper/self-taught-agentic-long-context","title":"Self-Taught Agentic Long Context Understanding","date":"2025-02-21","arxiv_id":"2502.15920","repositories_listed":1,"syntology":null},{"url":"/paper/scalar-scientific-citation-based-live","title":"SCALAR: Scientific Citation-based Live Assessment of Long-context Academic Reasoning","date":"2025-02-19","arxiv_id":"2502.13753","repositories_listed":1,"syntology":null}],"syntology_records":14,"syntology_note":"a paper without a record is not a recorded non-run: it may lack an arXiv id or simply be absent from the graph layer"},"description_links":{"kept":0,"unwrapped_to_text":0,"bare_urls_linked":0,"relative_images_dropped":0,"rule":"internal links are kept only when the target slug exists in the catalog"},"syntology":{"read_at":"2026-09-25T09:33:49+00:00","claim":"Per-sample execution status on synthesized fixtures ('ran N of M samples'); not a correctness claim and not a ranking signal.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"}},"not_shown":{"libraries":"the archive has no per-task library table","trend_sparklines":"the Trend column of the benchmarks table was a rendered image; it is not in the archive","social_and_latest_sorts":"stars and social signals are not in the archive"}}