{"url":"/sota/temporal-relation-extraction-on-vinoground","task":{"name":"Temporal Relation Extraction","url":"/task/temporal-relation-extraction","note":null},"dataset":{"name":"Vinoground","url":"/dataset/vinoground"},"category":"Natural Language Processing","categories":["Natural Language Processing"],"category_note":null,"description":"Temporal relation extraction systems aim to identify and classify the temporal relation between a pair of entities provided in a text. For instance, in the sentence \"Bob sent a message to Alice while she was leaving her birthday party.\" one can infer that the actions \"sent\" and \"leaving\" entails a temporal relation that can be described as \"simultaneous\".","description_from":"task","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","rank":"the archive's row order at snapshot; not re-ranked","rows_end_at":"2025-07-28","rows_withheld_as_spam":0,"metric_values":"the archive's strings, untouched"},"metrics":["Text Score","Video Score","Group Score"],"metric_direction":{"note":"inferred from the metric name only (the archive records no direction); null = not inferred, chart draws points only","by_metric":{"Text Score":"higher","Video Score":"higher","Group Score":"higher"}},"counts":{"rows":24,"rows_with_code":16,"rows_with_paper_page":16,"rows_dated":16,"rows_using_additional_data":0},"rows":[{"rank_in_archive_order":1,"model":"GPT-4o (CoT)","metrics":{"Group Score":"35","Text Score":"59.2","Video Score":"51"},"uses_additional_data":false,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":2,"model":"GPT-4o","metrics":{"Group Score":"24.6","Text Score":"54","Video Score":"38.2"},"uses_additional_data":false,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":3,"model":"Qwen2-VL-72B","metrics":{"Group Score":"17.4","Text Score":"50.4","Video Score":"32.6"},"uses_additional_data":false,"paper_date":"2024-09-18","paper":"/paper/qwen2-vl-enhancing-vision-language-model-s","paper_url":"https://arxiv.org/abs/2409.12191v2","paper_title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","code":"https://github.com/qwenlm/qwen2-vl","n_code_links":8,"syntology":{"n_ran":8,"n_unverified":4,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":4,"model":"LLaVA-OneVision-Qwen2-72B","metrics":{"Group Score":"21.8","Text Score":"48.4","Video Score":"35.2"},"uses_additional_data":false,"paper_date":"2024-08-06","paper":"/paper/llava-onevision-easy-visual-task-transfer","paper_url":"https://arxiv.org/abs/2408.03326v3","paper_title":"LLaVA-OneVision: Easy Visual Task Transfer","code":"https://github.com/evolvinglmms-lab/lmms-eval","n_code_links":2,"syntology":null},{"rank_in_archive_order":5,"model":"LLaVA-OneVision-Qwen2-7B","metrics":{"Group Score":"14.6","Text Score":"41.6","Video Score":"29.4"},"uses_additional_data":false,"paper_date":"2024-08-06","paper":"/paper/llava-onevision-easy-visual-task-transfer","paper_url":"https://arxiv.org/abs/2408.03326v3","paper_title":"LLaVA-OneVision: Easy Visual Task Transfer","code":"https://github.com/evolvinglmms-lab/lmms-eval","n_code_links":2,"syntology":null},{"rank_in_archive_order":6,"model":"Qwen2-VL-7B","metrics":{"Group Score":"15.2","Text Score":"40.2","Video Score":"32.4"},"uses_additional_data":false,"paper_date":"2024-09-18","paper":"/paper/qwen2-vl-enhancing-vision-language-model-s","paper_url":"https://arxiv.org/abs/2409.12191v2","paper_title":"Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution","code":"https://github.com/qwenlm/qwen2-vl","n_code_links":8,"syntology":{"n_ran":8,"n_unverified":4,"n_samples":12,"n_pointer_only_licence":0}},{"rank_in_archive_order":7,"model":"Gemini-1.5-Pro (CoT)","metrics":{"Group Score":"12.4","Text Score":"37","Video Score":"27.6"},"uses_additional_data":false,"paper_date":"2024-03-08","paper":"/paper/gemini-1-5-unlocking-multimodal-understanding","paper_url":"https://arxiv.org/abs/2403.05530v5","paper_title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","code":"https://github.com/dlvuldet/primevul","n_code_links":1,"syntology":null},{"rank_in_archive_order":8,"model":"VideoLLaMA2-72B","metrics":{"Group Score":"8.4","Text Score":"36.2","Video Score":"21.8"},"uses_additional_data":false,"paper_date":"2024-06-11","paper":"/paper/videollama-2-advancing-spatial-temporal","paper_url":"https://arxiv.org/abs/2406.07476v3","paper_title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","code":"https://github.com/damo-nlp-sg/videollama2","n_code_links":3,"syntology":{"n_ran":7,"n_unverified":10,"n_samples":17,"n_pointer_only_licence":0}},{"rank_in_archive_order":9,"model":"Gemini-1.5-Pro","metrics":{"Group Score":"10.2","Text Score":"35.8","Video Score":"22.6"},"uses_additional_data":false,"paper_date":"2024-03-08","paper":"/paper/gemini-1-5-unlocking-multimodal-understanding","paper_url":"https://arxiv.org/abs/2403.05530v5","paper_title":"Gemini 1.5: Unlocking multimodal understanding across millions of tokens of context","code":"https://github.com/dlvuldet/primevul","n_code_links":1,"syntology":null},{"rank_in_archive_order":10,"model":"Claude 3.5 Sonnet","metrics":{"Group Score":"10.6","Text Score":"32.8","Video Score":"28.8"},"uses_additional_data":false,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":11,"model":"MiniCPM-2.6","metrics":{"Group Score":"11.2","Text Score":"32.6","Video Score":"29.2"},"uses_additional_data":false,"paper_date":"2024-08-03","paper":"/paper/2408-01800","paper_url":"https://arxiv.org/abs/2408.01800v1","paper_title":"MiniCPM-V: A GPT-4V Level MLLM on Your Phone","code":"https://github.com/openbmb/minicpm-v","n_code_links":2,"syntology":{"n_ran":9,"n_unverified":5,"n_samples":14,"n_pointer_only_licence":0}},{"rank_in_archive_order":12,"model":"InternLM-XC-2.5 (CoT)","metrics":{"Group Score":"9","Text Score":"30.8","Video Score":"28.4"},"uses_additional_data":false,"paper_date":"2024-07-03","paper":"/paper/internlm-xcomposer-2-5-a-versatile-large","paper_url":"https://arxiv.org/abs/2407.03320v1","paper_title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","code":"https://github.com/internlm/internlm-xcomposer","n_code_links":1,"syntology":{"n_ran":2,"n_unverified":0,"n_samples":2,"n_pointer_only_licence":2}},{"rank_in_archive_order":13,"model":"InternLM-XC-2.5","metrics":{"Group Score":"9.6","Text Score":"28.8","Video Score":"27.8"},"uses_additional_data":false,"paper_date":"2024-07-03","paper":"/paper/internlm-xcomposer-2-5-a-versatile-large","paper_url":"https://arxiv.org/abs/2407.03320v1","paper_title":"InternLM-XComposer-2.5: A Versatile Large Vision Language Model Supporting Long-Contextual Input and Output","code":"https://github.com/internlm/internlm-xcomposer","n_code_links":1,"syntology":{"n_ran":2,"n_unverified":0,"n_samples":2,"n_pointer_only_licence":2}},{"rank_in_archive_order":14,"model":"LLaVA-NeXT-Video-34B (CoT)","metrics":{"Group Score":"5.2","Text Score":"25.8","Video Score":"22.2"},"uses_additional_data":false,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":15,"model":"Video-LLaVA-7B","metrics":{"Group Score":"6.6","Text Score":"24.8","Video Score":"25.8"},"uses_additional_data":false,"paper_date":"2023-11-16","paper":"/paper/video-llava-learning-united-visual-1","paper_url":"https://arxiv.org/abs/2311.10122v3","paper_title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","code":"https://github.com/PKU-YuanGroup/Video-LLaVA","n_code_links":6,"syntology":{"n_ran":4,"n_unverified":3,"n_samples":7,"n_pointer_only_licence":1}},{"rank_in_archive_order":16,"model":"Phi-3.5-Vision","metrics":{"Group Score":"6.2","Text Score":"24","Video Score":"22.4"},"uses_additional_data":false,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":17,"model":"MA-LMM-Vicuna-7B","metrics":{"Group Score":"6.8","Text Score":"23.8","Video Score":"25.6"},"uses_additional_data":false,"paper_date":"2024-04-08","paper":"/paper/ma-lmm-memory-augmented-large-multimodal","paper_url":"https://arxiv.org/abs/2404.05726v2","paper_title":"MA-LMM: Memory-Augmented Large Multimodal Model for Long-Term Video Understanding","code":"https://github.com/boheumd/MA-LMM","n_code_links":1,"syntology":{"n_ran":7,"n_unverified":2,"n_samples":9,"n_pointer_only_licence":0}},{"rank_in_archive_order":18,"model":"LLaVA-NeXT-Video-34B","metrics":{"Group Score":"3.8","Text Score":"23","Video Score":"21.2"},"uses_additional_data":false,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":19,"model":"LLaVA-NeXT-Video-7B (CoT)","metrics":{"Group Score":"6.8","Text Score":"21.8","Video Score":"26.2"},"uses_additional_data":false,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":20,"model":"LLaVA-NeXT-Video-7B","metrics":{"Group Score":"6.2","Text Score":"21.8","Video Score":"25.6"},"uses_additional_data":false,"paper_date":null,"paper":null,"paper_url":null,"paper_title":"","code":null,"n_code_links":0,"syntology":null},{"rank_in_archive_order":21,"model":"VTimeLLM","metrics":{"Group Score":"5.2","Text Score":"19.4","Video Score":"27"},"uses_additional_data":false,"paper_date":"2023-11-30","paper":"/paper/vtimellm-empower-llm-to-grasp-video-moments","paper_url":"https://arxiv.org/abs/2311.18445v1","paper_title":"VTimeLLM: Empower LLM to Grasp Video Moments","code":"https://github.com/huangb23/vtimellm","n_code_links":1,"syntology":{"n_ran":5,"n_unverified":6,"n_samples":11,"n_pointer_only_licence":11}},{"rank_in_archive_order":22,"model":"VideoCLIP","metrics":{"Group Score":"1.2","Text Score":"17","Video Score":"2.8"},"uses_additional_data":false,"paper_date":"2021-09-28","paper":"/paper/videoclip-contrastive-pre-training-for-zero","paper_url":"https://arxiv.org/abs/2109.14084v2","paper_title":"VideoCLIP: Contrastive Pre-training for Zero-shot Video-Text Understanding","code":"https://github.com/facebookresearch/fairseq","n_code_links":2,"syntology":null},{"rank_in_archive_order":23,"model":"LanguageBind","metrics":{"Group Score":"1.2","Text Score":"10.6","Video Score":"5"},"uses_additional_data":false,"paper_date":"2023-10-03","paper":"/paper/languagebind-extending-video-language","paper_url":"https://arxiv.org/abs/2310.01852v7","paper_title":"LanguageBind: Extending Video-Language Pretraining to N-modality by Language-based Semantic Alignment","code":"https://github.com/PKU-YuanGroup/Video-LLaVA","n_code_links":6,"syntology":{"n_ran":7,"n_unverified":7,"n_samples":14,"n_pointer_only_licence":0}},{"rank_in_archive_order":24,"model":"ImageBind","metrics":{"Group Score":"0.6","Text Score":"9.4","Video Score":"3.4"},"uses_additional_data":false,"paper_date":"2023-05-09","paper":"/paper/imagebind-one-embedding-space-to-bind-them","paper_url":"https://arxiv.org/abs/2305.05665v2","paper_title":"ImageBind: One Embedding Space To Bind Them All","code":"https://github.com/facebookresearch/imagebind","n_code_links":3,"syntology":{"n_ran":24,"n_unverified":10,"n_samples":34,"n_pointer_only_licence":32}}],"since_archive":{"present":false,"note":"No Syntology-extracted rows are published in this build."},"syntology":{"read_at":"2026-09-24T18:15:14+00:00","claim":"Per row: N of M harvested code samples from that row's paper executed on a synthesized fixture; the other M-N are unverified. Not a reproduction of the row's number; not a correctness claim. n_pointer_only_licence counts samples the site points at rather than redistributes (a licence axis, independent of ran/unverified).","rows_with_graph_line":11,"rows_with_any_sample_ran":11,"distinct_papers_with_graph_line":9,"distinct_papers_with_any_sample_ran":9,"samples_over_distinct_papers":{"n_ran":73,"n_unverified":47,"n_samples":120,"n_pointer_only_licence":46,"note":"each paper (arXiv id) counted once, however many rows it is behind; this is the page-level figure"},"samples_row_weighted":{"n_ran":83,"n_unverified":51,"n_samples":134,"n_pointer_only_licence":48,"note":"row-weighted: a paper behind several rows is counted once per row; inflated relative to samples_over_distinct_papers by design, kept for readers summing the per-row syntology blocks"}}}