{"url":"/sota/on-implicitqa","task":{"name":null,"url":null,"note":"task not recorded in the archive for this table"},"dataset":{"name":"ImplicitQA","url":"/dataset/implicitqa"},"category":null,"categories":["Adversarial","Audio","Computer Code","Computer Vision","Medical","Methodology","Miscellaneous","Music","Natural Language Processing","Reasoning","Speech"],"category_note":"the archive's category list for this table covers most areas; treated as no area assigned","description":null,"description_from":null,"source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["Average Accuracy","Macro Average Accuracy"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"Average Accuracy":"higher","Macro Average Accuracy":"higher"}},"counts":{"rows":7,"rows_with_code":4,"rows_with_paper_page":5,"rows_dated":5,"rows_using_additional_data":0},"rows":[{"rank_in_archive_order":1,"model":"GPT O3","metrics":{"Average Accuracy":"64.1","Macro Average Accuracy":"68.6"},"uses_additional_data":false,"paper_date":"2025-06-26","paper":"/paper/implicitqa-going-beyond-frames-towards","paper_url":"https://arxiv.org/abs/2506.21742v1","paper_title":"ImplicitQA: Going beyond frames towards Implicit Video Reasoning","code":"https://github.com/UCF-CRCV/ImplicitQA","n_code_links":1,"syntology":null},{"rank_in_archive_order":2,"model":"GPT 4.1","metrics":{"Average Accuracy":"54.3","Macro Average Accuracy":"58.6"},"uses_additional_data":false,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":3,"model":"Qwen2 VL - 7B","metrics":{"Average Accuracy":"44.9","Macro Average Accuracy":"46.0"},"uses_additional_data":false,"paper_date":"2024-09-18","paper":"/paper/qwen2-vl-enhancing-vision-language-model-s","paper_url":"https://arxiv.org/abs/2409.12191v2","paper_title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","code":"https://github.com/qwenlm/qwen2-vl","n_code_links":8,"syntology":{"n_ran":8,"n_unverified":4,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":4,"model":"LLaVA-OneVision - 7B","metrics":{"Average Accuracy":"43.4","Macro Average Accuracy":"46.4"},"uses_additional_data":false,"paper_date":"2024-08-06","paper":"/paper/llava-onevision-easy-visual-task-transfer","paper_url":"https://arxiv.org/abs/2408.03326v3","paper_title":"LLaVA-OneVision: Easy Visual Task Transfer","code":"https://github.com/evolvinglmms-lab/lmms-eval","n_code_links":2,"syntology":null},{"rank_in_archive_order":5,"model":"Qwen 2.5 VL - 7B","metrics":{"Average Accuracy":"42.8","Macro Average Accuracy":"46.1"},"uses_additional_data":false,"paper_date":"2025-02-19","paper":"/paper/qwen2-5-vl-technical-report","paper_url":"https://arxiv.org/abs/2502.13923v1","paper_title":"Qwen2.5-VL Technical Report","code":"https://github.com/qwenlm/qwen2-vl","n_code_links":4,"syntology":{"n_ran":2,"n_unverified":1,"n_samples":3,"n_pointer_only_licence":0}},{"rank_in_archive_order":6,"model":"LLaVA-Video - 7B","metrics":{"Average Accuracy":"42.1","Macro Average Accuracy":"46.3"},"uses_additional_data":false,"paper_date":"2024-10-03","paper":"/paper/video-instruction-tuning-with-synthetic-data","paper_url":"https://arxiv.org/abs/2410.02713v2","paper_title":"Video Instruction Tuning With Synthetic Data","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":7,"model":"LLaVA-Next-Video - 7B","metrics":{"Average Accuracy":"33.9","Macro Average Accuracy":"37.5"},"uses_additional_data":false,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":2,"rows_with_any_sample_ran":2,"distinct_papers_with_graph_line":2,"distinct_papers_with_any_sample_ran":2,"samples_over_distinct_papers":{"n_ran":10,"n_unverified":5,"n_samples":15,"n_pointer_only_licence":0,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":10,"n_unverified":5,"n_samples":15,"n_pointer_only_licence":0,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}