{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/video-grounding/papers/ran/1","list_of":"/task/video-grounding","task":"Video Grounding","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":1,"rows_per_page":100,"rows":[1,23],"of":23,"counts":{"archive_papers_tagged":114,"with_a_code_link":56,"where_syntology_ran_a_sample":23,"not_listed_spam_title":0,"listed":114,"listed_where_code_ran":23,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":21,"every_run_a_failure_of_syntologys_instrument":2,"listed_with_a_run_with_no_instrument_failure":21,"listed_every_run_a_failure_of_syntologys_instrument":2,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/video-grounding/papers/ran/1","prev":null,"next":null,"papers":[{"url":"/paper/reinforcement-learning-tuning-for-videollms","slug":"reinforcement-learning-tuning-for-videollms","title":"Reinforcement Learning Tuning for VideoLLMs: Reward Design and Data Efficiency","date":"2025-06-02","arxiv_id":"2506.01908","repositories_listed":1,"syntology":{"n":17,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/reinforcement-learning-tuning-for-videollms#ran","syntology_url":"https://syntology.ai/paper/2506.01908","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.01908"}},"official":{"repos":["appletea233/temporal-r1"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/object-shot-enhanced-grounding-network-for","slug":"object-shot-enhanced-grounding-network-for","title":"Object-Shot Enhanced Grounding Network for Egocentric Video","date":"2025-05-07","arxiv_id":"2505.04270","repositories_listed":1,"syntology":{"n":17,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":10,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/object-shot-enhanced-grounding-network-for#ran","syntology_url":"https://syntology.ai/paper/2505.04270","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.04270"}},"official":{"repos":["yisen-feng/osgnet"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":10,"ran_from_kinds":["official"]}}},{"url":"/paper/timezero-temporal-video-grounding-with","slug":"timezero-temporal-video-grounding-with","title":"TimeZero: Temporal Video Grounding with Reasoning-Guided LVLM","date":"2025-03-17","arxiv_id":"2503.13377","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/timezero-temporal-video-grounding-with#ran","syntology_url":"https://syntology.ai/paper/2503.13377","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.13377"}},"official":{"repos":["www-ye/timezero"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/tarsier2-advancing-large-vision-language","slug":"tarsier2-advancing-large-vision-language","title":"Tarsier2: Advancing Large Vision-Language Models from Detailed Video Description to Comprehensive Video Understanding","date":"2025-01-14","arxiv_id":"2501.07888","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tarsier2-advancing-large-vision-language#ran","syntology_url":"https://syntology.ai/paper/2501.07888","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.07888"}},"official":{"repos":["bytedance/tarsier"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/llava-st-a-multimodal-large-language-model","slug":"llava-st-a-multimodal-large-language-model","title":"LLaVA-ST: A Multimodal Large Language Model for Fine-Grained Spatial-Temporal Understanding","date":"2025-01-14","arxiv_id":"2501.08282","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":2,"n_ran_checked":2,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/llava-st-a-multimodal-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2501.08282","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.08282"}},"official":{"repos":["appletea233/llava-st"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/videollm-knows-when-to-speak-enhancing-time","slug":"videollm-knows-when-to-speak-enhancing-time","title":"VideoLLM Knows When to Speak: Enhancing Time-Sensitive Video Comprehension with Video-Text Duet Interaction Format","date":"2024-11-27","arxiv_id":"2411.17991","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/videollm-knows-when-to-speak-enhancing-time#ran","syntology_url":"https://syntology.ai/paper/2411.17991","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.17991"}},"official":{"repos":["yellow-binary-tree/mmduet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prior-knowledge-integration-via-llm-encoding","slug":"prior-knowledge-integration-via-llm-encoding","title":"Prior Knowledge Integration via LLM Encoding and Pseudo Event Regulation for Video Moment Retrieval","date":"2024-07-21","arxiv_id":"2407.15051","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/prior-knowledge-integration-via-llm-encoding#ran","syntology_url":"https://syntology.ai/paper/2407.15051","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15051"}},"official":{"repos":["fletcherjiang/llmepet"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/snag-scalable-and-accurate-video-grounding","slug":"snag-scalable-and-accurate-video-grounding","title":"SnAG: Scalable and Accurate Video Grounding","date":"2024-04-02","arxiv_id":"2404.02257","repositories_listed":1,"syntology":{"n":20,"n_ran":13,"n_constructed":0,"n_ran_checked":11,"n_instrument":2,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":16,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 2 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/snag-scalable-and-accurate-video-grounding#ran","syntology_url":"https://syntology.ai/paper/2404.02257","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.02257"}},"official":null}},{"url":"/paper/hawkeye-training-video-text-llms-for","slug":"hawkeye-training-video-text-llms-for","title":"HawkEye: Training Video-Text LLMs for Grounding Text in Videos","date":"2024-03-15","arxiv_id":"2403.10228","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hawkeye-training-video-text-llms-for#ran","syntology_url":"https://syntology.ai/paper/2403.10228","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.10228"}},"official":{"repos":["yellow-binary-tree/hawkeye"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/context-guided-spatio-temporal-video","slug":"context-guided-spatio-temporal-video","title":"Context-Guided Spatio-Temporal Video Grounding","date":"2024-01-03","arxiv_id":"2401.01578","repositories_listed":2,"syntology":{"n":34,"n_ran":22,"n_constructed":0,"n_ran_checked":17,"n_instrument":5,"n_unverified":12,"n_honours":0,"n_violates":1,"n_no_contract":16,"n_pointer_only":34,"phrase":"22 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 1 violated, 16 with no contract checked; 5 where Syntology's instrument failed) · 12 unverified","sample_list":"/paper/context-guided-spatio-temporal-video#ran","syntology_url":"https://syntology.ai/paper/2401.01578","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.01578"}},"official":{"repos":["henglan/cgstvg"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":12,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/grounded-question-answering-in-long","slug":"grounded-question-answering-in-long","title":"Grounded Question-Answering in Long Egocentric Videos","date":"2023-12-11","arxiv_id":"2312.06505","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/grounded-question-answering-in-long#ran","syntology_url":"https://syntology.ai/paper/2312.06505","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06505"}},"official":{"repos":["becomebright/groundvqa"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vtimellm-empower-llm-to-grasp-video-moments","slug":"vtimellm-empower-llm-to-grasp-video-moments","title":"VTimeLLM: Empower LLM to Grasp Video Moments","date":"2023-11-30","arxiv_id":"2311.18445","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":5,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":11,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/vtimellm-empower-llm-to-grasp-video-moments#ran","syntology_url":"https://syntology.ai/paper/2311.18445","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.18445"}},"official":{"repos":["huangb23/vtimellm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/bridging-the-gap-a-unified-video","slug":"bridging-the-gap-a-unified-video","title":"Bridging the Gap: A Unified Video Comprehension Framework for Moment Retrieval and Highlight Detection","date":"2023-11-28","arxiv_id":"2311.16464","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/bridging-the-gap-a-unified-video#ran","syntology_url":"https://syntology.ai/paper/2311.16464","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.16464"}},"official":{"repos":["easonxiao-888/uvcom"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/pg-video-llava-pixel-grounding-large-video","slug":"pg-video-llava-pixel-grounding-large-video","title":"PG-Video-LLaVA: Pixel Grounding Large Video-Language Models","date":"2023-11-22","arxiv_id":"2311.13435","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/pg-video-llava-pixel-grounding-large-video#ran","syntology_url":"https://syntology.ai/paper/2311.13435","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.13435"}},"official":{"repos":["mbzuai-oryx/video-llava"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/can-i-trust-your-answer-visually-grounded","slug":"can-i-trust-your-answer-visually-grounded","title":"Can I Trust Your Answer? Visually Grounded Video Question Answering","date":"2023-09-04","arxiv_id":"2309.01327","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/can-i-trust-your-answer-visually-grounded#ran","syntology_url":"https://syntology.ai/paper/2309.01327","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2309.01327"}},"official":{"repos":["doc-doc/next-gqa"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/knowing-where-to-focus-event-aware","slug":"knowing-where-to-focus-event-aware","title":"Knowing Where to Focus: Event-aware Transformer for Video Grounding","date":"2023-08-14","arxiv_id":"2308.06947","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":1,"n_ran_checked":3,"n_instrument":1,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"4 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/knowing-where-to-focus-event-aware#ran","syntology_url":"https://syntology.ai/paper/2308.06947","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.06947"}},"official":{"repos":["jinhyunj/eatr"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/query-dependent-video-representation-for","slug":"query-dependent-video-representation-for","title":"Query-Dependent Video Representation for Moment Retrieval and Highlight Detection","date":"2023-03-24","arxiv_id":"2303.13874","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/query-dependent-video-representation-for#ran","syntology_url":"https://syntology.ai/paper/2303.13874","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.13874"}},"official":{"repos":["wjun0830/qd-detr"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/text-visual-prompting-for-efficient-2d","slug":"text-visual-prompting-for-efficient-2d","title":"Text-Visual Prompting for Efficient 2D Temporal Video Grounding","date":"2023-03-09","arxiv_id":"2303.04995","repositories_listed":1,"syntology":{"n":16,"n_ran":14,"n_constructed":0,"n_ran_checked":12,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":4,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/text-visual-prompting-for-efficient-2d#ran","syntology_url":"https://syntology.ai/paper/2303.04995","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2303.04995"}},"official":{"repos":["intel/TVP"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/embracing-consistency-a-one-stage-approach","slug":"embracing-consistency-a-one-stage-approach","title":"Embracing Consistency: A One-Stage Approach for Spatio-Temporal Video Grounding","date":"2022-09-27","arxiv_id":"2209.13306","repositories_listed":1,"syntology":{"n":18,"n_ran":11,"n_constructed":9,"n_ran_checked":10,"n_instrument":1,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 9 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/embracing-consistency-a-one-stage-approach#ran","syntology_url":"https://syntology.ai/paper/2209.13306","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.13306"}},"official":{"repos":["jy0205/stcat"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":9,"n_ran_no_instrument_failure":10,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/video-guided-curriculum-learning-for-spoken","slug":"video-guided-curriculum-learning-for-spoken","title":"Video-Guided Curriculum Learning for Spoken Video Grounding","date":"2022-09-01","arxiv_id":"2209.00277","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/video-guided-curriculum-learning-for-spoken#ran","syntology_url":"https://syntology.ai/paper/2209.00277","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.00277"}},"official":{"repos":["marmot-xy/spoken-video-grounding"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/tubedetr-spatio-temporal-video-grounding-with","slug":"tubedetr-spatio-temporal-video-grounding-with","title":"TubeDETR: Spatio-Temporal Video Grounding with Transformers","date":"2022-03-30","arxiv_id":"2203.16434","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/tubedetr-spatio-temporal-video-grounding-with#ran","syntology_url":"https://syntology.ai/paper/2203.16434","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.16434"}},"official":{"repos":["antoyang/TubeDETR"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/vlg-net-video-language-graph-matching-network","slug":"vlg-net-video-language-graph-matching-network","title":"VLG-Net: Video-Language Graph Matching Network for Video Grounding","date":"2020-11-19","arxiv_id":"2011.10132","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vlg-net-video-language-graph-matching-network#ran","syntology_url":"https://syntology.ai/paper/2011.10132","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.10132"}},"official":null}},{"url":"/paper/human-centric-spatio-temporal-video-grounding","slug":"human-centric-spatio-temporal-video-grounding","title":"Human-centric Spatio-Temporal Video Grounding With Visual Transformers","date":"2020-11-10","arxiv_id":"2011.05049","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/human-centric-spatio-temporal-video-grounding#ran","syntology_url":"https://syntology.ai/paper/2011.05049","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2011.05049"}},"official":{"repos":["tzhhhh123/HC-STVG"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"53667ace590d8eb832f099bb2f758ade1c9b3d9b5da618f2791a360f9364e290","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}