{"url":"/sota/long-context-understanding-on-ada-leval","task":{"name":"Long-Context Understanding","url":"/task/long-context-understanding","note":null},"dataset":{"name":"Ada-LEval (BestAnswer)","url":null},"category":"Natural Language Processing","categories":["Natural Language Processing"],"category_note":null,"description":null,"description_from":null,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["1k","2k","4k","6k","8k","12k","16k","32k","64k","128k"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"1k":null,"2k":null,"4k":null,"6k":null,"8k":null,"12k":null,"16k":null,"32k":null,"64k":null,"128k":null}},"counts":{"rows":10,"rows_with_code":8,"rows_with_paper_page":8,"rows_dated":8,"rows_using_additional_data":0},"rows":[{"rank_in_archive_order":1,"model":"GPT-4-Turbo-1106","metrics":{"128k":"0.0","12k":"49.5","16k":"44.0","1k":"74.0","2k":"73.5","32k":"16.0","4k":"67.5","64k":"0.0","6k":"59.5","8k":"53.5"},"uses_additional_data":false,"paper_date":"2023-03-15","paper":"/paper/gpt-4-technical-report-1","paper_url":"https://arxiv.org/abs/2303.08774v5","paper_title":"GPT-4 Technical Report","code":"https://github.com/openai/evals","n_code_links":11,"syntology":{"n_ran":2,"n_unverified":3,"n_samples":5,"n_pointer_only_licence":1}},{"rank_in_archive_order":2,"model":"GPT-4-Turbo-0125","metrics":{"128k":"0.0","12k":"52.0","16k":"44.5","1k":"73.5","2k":"73.5","32k":"30.0","4k":"65.5","64k":"0.0","6k":"63.0","8k":"56.5"},"uses_additional_data":false,"paper_date":"2023-03-15","paper":"/paper/gpt-4-technical-report-1","paper_url":"https://arxiv.org/abs/2303.08774v5","paper_title":"GPT-4 Technical Report","code":"https://github.com/openai/evals","n_code_links":11,"syntology":{"n_ran":2,"n_unverified":3,"n_samples":5,"n_pointer_only_licence":1}},{"rank_in_archive_order":3,"model":"Claude-2","metrics":{"12k":"12.0","16k":"11.0","1k":"65.0","2k":"43.5","32k":"4.0","4k":"23.5","64k":"0.0","6k":"15.0","8k":"17.0"},"uses_additional_data":false,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":4,"model":"GPT-3.5-Turbo-1106","metrics":{"12k":"2.5","16k":"2.5","1k":"61.5","2k":"48.5","4k":"41.5","6k":"29.5","8k":"17.0"},"uses_additional_data":false,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":5,"model":"InternLM2-7b","metrics":{"128k":"0.0","12k":"2.0","16k":"0.8","1k":"58.6","2k":"49.5","32k":"0.5","4k":"33.9","64k":"0.5","6k":"12.3","8k":"13.4"},"uses_additional_data":false,"paper_date":"2024-03-26","paper":"/paper/internlm2-technical-report","paper_url":"https://arxiv.org/abs/2403.17297v1","paper_title":"InternLM2 Technical Report","code":"https://github.com/internlm/internlm","n_code_links":3,"syntology":{"n_ran":1,"n_unverified":0,"n_samples":1,"n_pointer_only_licence":0}},{"rank_in_archive_order":6,"model":"Vicuna-13b-v1.5-16k","metrics":{"12k":"1.4","16k":"0.9","1k":"53.4","2k":"29.2","4k":"13.1","6k":"4.3","8k":"2.2"},"uses_additional_data":false,"paper_date":"2023-06-09","paper":"/paper/judging-llm-as-a-judge-with-mt-bench-and-1","paper_url":"https://arxiv.org/abs/2306.05685v4","paper_title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","code":"https://github.com/lm-sys/fastchat","n_code_links":11,"syntology":{"n_ran":0,"n_unverified":12,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":7,"model":"ChatGLM3-6b-32k","metrics":{"12k":"0.9","16k":"0.5","1k":"39.8","2k":"18.8","4k":"9.0","6k":"5.0","8k":"3.4"},"uses_additional_data":false,"paper_date":"2022-10-05","paper":"/paper/glm-130b-an-open-bilingual-pre-trained-model","paper_url":"https://arxiv.org/abs/2210.02414v2","paper_title":"GLM-130B: An Open Bilingual Pre-trained Model","code":"https://github.com/thudm/chatglm2-6b","n_code_links":9,"syntology":{"n_ran":5,"n_unverified":16,"n_samples":21,"n_pointer_only_licence":0}},{"rank_in_archive_order":8,"model":"Vicuna-7b-v1.5-16k","metrics":{"12k":"1.9","16k":"1.0","1k":"37.0","2k":"11.1","4k":"5.8","6k":"3.2","8k":"1.8"},"uses_additional_data":false,"paper_date":"2023-06-09","paper":"/paper/judging-llm-as-a-judge-with-mt-bench-and-1","paper_url":"https://arxiv.org/abs/2306.05685v4","paper_title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","code":"https://github.com/lm-sys/fastchat","n_code_links":11,"syntology":{"n_ran":0,"n_unverified":12,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":9,"model":"LongChat-7b-v1.5-32k","metrics":{"12k":"1.6","16k":"0.8","1k":"32.4","2k":"10.7","4k":"5.7","6k":"3.1","8k":"1.9"},"uses_additional_data":false,"paper_date":"2023-06-09","paper":"/paper/judging-llm-as-a-judge-with-mt-bench-and-1","paper_url":"https://arxiv.org/abs/2306.05685v4","paper_title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","code":"https://github.com/lm-sys/fastchat","n_code_links":11,"syntology":{"n_ran":0,"n_unverified":12,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":10,"model":"ChatGLM2-6b-32k","metrics":{"12k":"0.0","16k":"0.3","1k":"31.2","2k":"10.9","4k":"4.5","6k":"1.6","8k":"1.6"},"uses_additional_data":false,"paper_date":"2022-10-05","paper":"/paper/glm-130b-an-open-bilingual-pre-trained-model","paper_url":"https://arxiv.org/abs/2210.02414v2","paper_title":"GLM-130B: An Open Bilingual Pre-trained Model","code":"https://github.com/thudm/chatglm2-6b","n_code_links":9,"syntology":{"n_ran":5,"n_unverified":16,"n_samples":21,"n_pointer_only_licence":0}}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":8,"rows_with_any_sample_ran":5,"distinct_papers_with_graph_line":4,"distinct_papers_with_any_sample_ran":3,"samples_over_distinct_papers":{"n_ran":8,"n_unverified":31,"n_samples":39,"n_pointer_only_licence":1,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":15,"n_unverified":74,"n_samples":89,"n_pointer_only_licence":2,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}