{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/box-cxcywh-to-xyxy","entry":"box_cxcywh_to_xyxy","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":60,"n_papers_ran":53,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":12,"n_samples_ran":7,"n_samples_fingerprinted":6,"n_places":64,"n_places_pointer_only":27,"by_status":{"ran_honours":3,"ran_violates":0,"ran_draft_wrong":1,"ran_fixture":0,"ran":3,"unverified":5},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.02318","paper":"/paper/arxiv-2609-02318","title":"YesTrack: Referring Multi-Object Tracking via MLLM-based Yes/No Verification","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"ggbondrighthere24/YesTrack","path":"utils/box_ops.py","file_url":"https://github.com/ggbondrighthere24/YesTrack/blob/HEAD/utils/box_ops.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"08b30db38a23e292","mcp_get_code":{"code_sha256":"08b30db38a23e292"}},{"arxiv_id":"2601.01908","paper":"/paper/arxiv-2601-01908","title":"Nodule-DETR: A Novel DETR Architecture with Frequency-Channel Attention for Ultrasound Thyroid Nodule Detection","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"wjj1wjj/Nodule-DETR","path":"Nodule-DETR/detr_See.py","file_url":"https://github.com/wjj1wjj/Nodule-DETR/blob/HEAD/Nodule-DETR/detr_See.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d8ff64e15d6e7507","mcp_get_code":{"code_sha256":"d8ff64e15d6e7507"}},{"arxiv_id":"2509.04833","paper":"/paper/arxiv-2509-04833","title":"PropVG: End-to-End Proposal-Driven Visual Grounding with Multi-Granularity Discrimination","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"Dmmm1997/PropVG","path":"propvg/layers/box_ops.py","file_url":"https://github.com/Dmmm1997/PropVG/blob/HEAD/propvg/layers/box_ops.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"856982244cffd7e3","mcp_get_code":{"code_sha256":"856982244cffd7e3"}},{"arxiv_id":"2506.05890","paper":"/paper/unleashing-the-potential-of-consistency-1","title":"Unleashing the Potential of Consistency Learning for Detecting and Grounding Multi-Modal Media Manipulation","date":"2025-06-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"liyih/CSCL","path":"code/MultiModal-DeepFake-main/models/box_ops.py","file_url":"https://github.com/liyih/CSCL/blob/HEAD/code/MultiModal-DeepFake-main/models/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2502.19842","paper":"/paper/clip-under-the-microscope-a-fine-grained","title":"CLIP Under the Microscope: A Fine-Grained Analysis of Multi-Object Representation","date":"2025-02-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2501.06710","paper":"/paper/multi-task-visual-grounding-with-coarse-to","title":"Multi-task Visual Grounding with Coarse-to-Fine Consistency Constraints","date":"2025-01-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dmmm1997/c3vg","path":"c3vg/core/utils.py","file_url":"https://github.com/dmmm1997/c3vg/blob/HEAD/c3vg/core/utils.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2501.06710","paper":"/paper/multi-task-visual-grounding-with-coarse-to","title":"Multi-task Visual Grounding with Coarse-to-Fine Consistency Constraints","date":"2025-01-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dmmm1997/c3vg","path":"c3vg/layers/box_ops.py","file_url":"https://github.com/dmmm1997/c3vg/blob/HEAD/c3vg/layers/box_ops.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"856982244cffd7e3","mcp_get_code":{"code_sha256":"856982244cffd7e3"}},{"arxiv_id":"2412.15691","paper":"/paper/exploiting-multimodal-spatial-temporal","title":"Exploiting Multimodal Spatial-temporal Patterns for Video Object Tracking","date":"2024-12-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nju-pcalab/sttrack","path":"lib/utils/box_ops.py","file_url":"https://github.com/nju-pcalab/sttrack/blob/HEAD/lib/utils/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2412.04234","paper":"/paper/deim-detr-with-improved-matching-for-fast","title":"DEIM: DETR with Improved Matching for Fast Convergence","date":"2024-12-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shihuahuang95/deim","path":"engine/deim/box_ops.py","file_url":"https://github.com/shihuahuang95/deim/blob/HEAD/engine/deim/box_ops.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"cb48d7481c17a1f7","mcp_get_code":{"code_sha256":"cb48d7481c17a1f7"}},{"arxiv_id":"2411.17606","paper":"/paper/hyperseg-towards-universal-visual","title":"HyperSeg: Towards Universal Visual Segmentation with Large Language Model","date":"2024-11-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"congvvc/HyperSeg","path":"hyperseg/model/tracker/box_ops.py","file_url":"https://github.com/congvvc/HyperSeg/blob/HEAD/hyperseg/model/tracker/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2411.11919","paper":"/paper/vl-uncertainty-detecting-hallucination-in","title":"VL-Uncertainty: Detecting Hallucination in Large Vision-Language Model via Uncertainty Estimation","date":"2024-11-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"JT-Sun/Filtering-WoRA","path":"models/box_ops.py","file_url":"https://github.com/JT-Sun/Filtering-WoRA/blob/HEAD/models/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2411.10293","paper":"/paper/retr-multi-view-radar-detection-transformer","title":"RETR: Multi-View Radar Detection Transformer for Indoor Perception","date":"2024-11-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"merlresearch/radar-detection-transformer","path":"src/models/module_retr/box_ops.py","file_url":"https://github.com/merlresearch/radar-detection-transformer/blob/HEAD/src/models/module_retr/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"AGPL-3.0","inline_ok":false,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2410.23904","paper":"/paper/ez-hoi-vlm-adaptation-via-guided-prompt","title":"EZ-HOI: VLM Adaptation via Guided Prompt Learning for Zero-Shot HOI Detection","date":"2024-10-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ChelsieLei/EZ-HOI","path":"ops.py","file_url":"https://github.com/ChelsieLei/EZ-HOI/blob/HEAD/ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2409.17531","paper":"/paper/simvg-a-simple-framework-for-visual-grounding","title":"SimVG: A Simple Framework for Visual Grounding with Decoupled Multi-modal Fusion","date":"2024-09-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dmmm1997/simvg","path":"simvg/core/utils.py","file_url":"https://github.com/dmmm1997/simvg/blob/HEAD/simvg/core/utils.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2408.12246","paper":"/paper/ova-detr-open-vocabulary-aerial-object","title":"OVA-DETR: Open Vocabulary Aerial Object Detection Using Image-Text Alignment and Fusion","date":"2024-08-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"GT-Wei/RT-OVAD","path":"src/zoo/itc_ovad/box_ops.py","file_url":"https://github.com/GT-Wei/RT-OVAD/blob/HEAD/src/zoo/itc_ovad/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2408.10487","paper":"/paper/mambaevt-event-stream-based-visual-object","title":"MambaEVT: Event Stream based Visual Object Tracking using State Space Model","date":"2024-08-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"event-ahu/mambaevt","path":"lib/utils/box_ops.py","file_url":"https://github.com/event-ahu/mambaevt/blob/HEAD/lib/utils/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2408.02484","paper":"/paper/2408-02484","title":"Exploring Conditional Multi-Modal Prompts for Zero-shot HOI Detection","date":"2024-08-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ltttpku/cmmp","path":"ops.py","file_url":"https://github.com/ltttpku/cmmp/blob/HEAD/ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2408.00714","paper":"/paper/2408-00714","title":"SAM 2: Segment Anything in Images and Videos","date":"2024-08-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"louisfinner/him2sam","path":"lib/utils/box_ops.py","file_url":"https://github.com/louisfinner/him2sam/blob/HEAD/lib/utils/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2407.07402","paper":"/paper/actionvos-actions-as-prompts-for-video-object","title":"ActionVOS: Actions as Prompts for Video Object Segmentation","date":"2024-07-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ut-vision/ActionVOS","path":"RF_ActionVOS/inference_actionvos.py","file_url":"https://github.com/ut-vision/ActionVOS/blob/HEAD/RF_ActionVOS/inference_actionvos.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ef1a3e10a9dbf4a9","mcp_get_code":{"code_sha256":"ef1a3e10a9dbf4a9"}},{"arxiv_id":"2404.18174","paper":"/paper/mamba-fetrack-frame-event-tracking-via-state","title":"Mamba-FETrack: Frame-Event Tracking via State Space Model","date":"2024-04-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"event-ahu/mamba_fetrack","path":"Mamba_FETrack/lib/utils/box_ops.py","file_url":"https://github.com/event-ahu/mamba_fetrack/blob/HEAD/Mamba_FETrack/lib/utils/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2404.08506","paper":"/paper/lasagna-language-based-segmentation-assistant","title":"LaSagnA: Language-based Segmentation Assistant for Complex Queries","date":"2024-04-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"congvvc/lasagna","path":"model/matcher.py","file_url":"https://github.com/congvvc/lasagna/blob/HEAD/model/matcher.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2403.19407","paper":"/paper/towards-temporally-consistent-referring-video","title":"Temporally Consistent Referring Video Object Segmentation with Hybrid Memory","date":"2024-03-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bo-miao/HTR","path":"inference_davis.py","file_url":"https://github.com/bo-miao/HTR/blob/HEAD/inference_davis.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ef1a3e10a9dbf4a9","mcp_get_code":{"code_sha256":"ef1a3e10a9dbf4a9"}},{"arxiv_id":"2403.12042","paper":"/paper/exploring-pre-trained-text-to-video-diffusion","title":"Exploring Pre-trained Text-to-Video Diffusion Models for Referring Video Object Segmentation","date":"2024-03-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"buxiangzhiren/vd-it","path":"inference_davis.py","file_url":"https://github.com/buxiangzhiren/vd-it/blob/HEAD/inference_davis.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ef1a3e10a9dbf4a9","mcp_get_code":{"code_sha256":"ef1a3e10a9dbf4a9"}},{"arxiv_id":"2402.03094","paper":"/paper/cross-domain-few-shot-object-detection-via","title":"Cross-Domain Few-Shot Object Detection via Enhanced Open-Set Object Detector","date":"2024-02-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lovelyqian/CDFSOD-benchmark","path":"lib/regionprop.py","file_url":"https://github.com/lovelyqian/CDFSOD-benchmark/blob/HEAD/lib/regionprop.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"009912c75c8c77cf","mcp_get_code":{"code_sha256":"009912c75c8c77cf"}},{"arxiv_id":"2310.10071","paper":"/paper/zoomtrack-target-aware-non-uniform-resizing-1","title":"ZoomTrack: Target-aware Non-uniform Resizing for Efficient Visual Tracking","date":"2023-10-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Kou-99/ZoomTrack","path":"lib/utils/box_ops.py","file_url":"https://github.com/Kou-99/ZoomTrack/blob/HEAD/lib/utils/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2309.14611","paper":"/paper/event-stream-based-visual-object-tracking-a","title":"Event Stream-based Visual Object Tracking: A High-Resolution Benchmark Dataset and A Novel Baseline","date":"2023-09-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"event-ahu/coesot","path":"CEUTrack/lib/utils/box_ops.py","file_url":"https://github.com/event-ahu/coesot/blob/HEAD/CEUTrack/lib/utils/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2309.14203","paper":"/paper/detecting-and-grounding-multi-modal-media-1","title":"Detecting and Grounding Multi-Modal Media Manipulation and Beyond","date":"2023-09-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rshaojimmy/multimodal-deepfake","path":"models/box_ops.py","file_url":"https://github.com/rshaojimmy/multimodal-deepfake/blob/HEAD/models/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2309.12969","paper":"/paper/detect-every-thing-with-few-examples","title":"Detect Everything with Few Examples","date":"2023-09-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mlzxy/devit","path":"lib/regionprop.py","file_url":"https://github.com/mlzxy/devit/blob/HEAD/lib/regionprop.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"009912c75c8c77cf","mcp_get_code":{"code_sha256":"009912c75c8c77cf"}},{"arxiv_id":"2309.03874","paper":"/paper/box-based-refinement-for-weakly-supervised","title":"Box-based Refinement for Weakly Supervised and Unsupervised Localization Tasks","date":"2023-09-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"eyalgomel/box-based-refinement","path":"detr/util/box_ops.py","file_url":"https://github.com/eyalgomel/box-based-refinement/blob/HEAD/detr/util/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2309.02691","paper":"/paper/a-joint-study-of-phrase-grounding-and-task","title":"A Joint Study of Phrase Grounding and Task Performance in Vision and Language Models","date":"2023-09-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lil-lab/phrase_grounding","path":"src/evals/postprocessors.py","file_url":"https://github.com/lil-lab/phrase_grounding/blob/HEAD/src/evals/postprocessors.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2309.02691","paper":"/paper/a-joint-study-of-phrase-grounding-and-task","title":"A Joint Study of Phrase Grounding and Task Performance in Vision and Language Models","date":"2023-09-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lil-lab/phrase_grounding","path":"src/datasets/coco_format_dataset.py","file_url":"https://github.com/lil-lab/phrase_grounding/blob/HEAD/src/datasets/coco_format_dataset.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ac92d3ba92f8d1ef","mcp_get_code":{"code_sha256":"ac92d3ba92f8d1ef"}},{"arxiv_id":"2308.11322","paper":"/paper/citetracker-correlating-image-and-text-for","title":"CiteTracker: Correlating Image and Text for Visual Tracking","date":"2023-08-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"NorahGreen/CiteTracker","path":"lib/utils/box_ops.py","file_url":"https://github.com/NorahGreen/CiteTracker/blob/HEAD/lib/utils/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2307.13537","paper":"/paper/spectrum-guided-multi-granularity-referring","title":"Spectrum-guided Multi-granularity Referring Video Object Segmentation","date":"2023-07-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bo-miao/sgmg","path":"inference_davis.py","file_url":"https://github.com/bo-miao/sgmg/blob/HEAD/inference_davis.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ef1a3e10a9dbf4a9","mcp_get_code":{"code_sha256":"ef1a3e10a9dbf4a9"}},{"arxiv_id":"2307.12616","paper":"/paper/ctvis-consistent-training-for-online-video","title":"CTVIS: Consistent Training for Online Video Instance Segmentation","date":"2023-07-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kainingying/ctvis","path":"ctvis/utils/utils.py","file_url":"https://github.com/kainingying/ctvis/blob/HEAD/ctvis/utils/utils.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2307.12612","paper":"/paper/less-is-more-focus-attention-for-efficient","title":"Less is More: Focus Attention for Efficient DETR","date":"2023-07-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"linxid/Focus-DETR","path":"models/focus_detr/box_ops.py","file_url":"https://github.com/linxid/Focus-DETR/blob/HEAD/models/focus_detr/box_ops.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"074307e51146fa09","mcp_get_code":{"code_sha256":"074307e51146fa09"}},{"arxiv_id":"2307.09356","paper":"/paper/onlinerefer-a-simple-online-baseline-for","title":"OnlineRefer: A Simple Online Baseline for Referring Video Object Segmentation","date":"2023-07-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wudongming97/onlinerefer","path":"inference_davis_online.py","file_url":"https://github.com/wudongming97/onlinerefer/blob/HEAD/inference_davis_online.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ef1a3e10a9dbf4a9","mcp_get_code":{"code_sha256":"ef1a3e10a9dbf4a9"}},{"arxiv_id":"2307.08249","paper":"/paper/random-boxes-are-open-world-object-detectors","title":"Random Boxes Are Open-world Object Detectors","date":"2023-07-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"scuwyh2000/RandBox","path":"randbox/util/box_ops.py","file_url":"https://github.com/scuwyh2000/RandBox/blob/HEAD/randbox/util/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2305.15896","paper":"/paper/mixformerv2-efficient-fully-transformer-1","title":"MixFormerV2: Efficient Fully Transformer Tracking","date":"2023-05-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mcg-nju/mixformerv2","path":"lib/utils/box_ops.py","file_url":"https://github.com/mcg-nju/mixformerv2/blob/HEAD/lib/utils/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2304.04742","paper":"/paper/detection-transformer-with-stable-matching","title":"Detection Transformer with Stable Matching","date":"2023-04-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"IDEA-Research/detrex","path":"detrex/modeling/matcher/modified_matcher.py","file_url":"https://github.com/IDEA-Research/detrex/blob/HEAD/detrex/modeling/matcher/modified_matcher.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"MISDECLARED","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"856982244cffd7e3","mcp_get_code":{"code_sha256":"856982244cffd7e3"}},{"arxiv_id":"2212.02773","paper":"/paper/diffusioninst-diffusion-model-for-instance","title":"DiffusionInst: Diffusion Model for Instance Segmentation","date":"2022-12-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alipay/diffusion-model-for-instance-segmentation","path":"diffusioninst/util/box_ops.py","file_url":"https://github.com/alipay/diffusion-model-for-instance-segmentation/blob/HEAD/diffusioninst/util/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2208.08965","paper":"/paper/gsrformer-grounded-situation-recognition","title":"GSRFormer: Grounded Situation Recognition Transformer with Alternate Semantic Attention Refinement","date":"2022-08-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhiqic/gsrformer","path":"util/box_ops.py","file_url":"https://github.com/zhiqic/gsrformer/blob/HEAD/util/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2207.07078","paper":"/paper/towards-grand-unification-of-object-tracking","title":"Towards Grand Unification of Object Tracking","date":"2022-07-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2207.05293","paper":"/paper/towards-hard-positive-query-mining-for-detr","title":"Towards Hard-Positive Query Mining for DETR-based Human-Object Interaction Detection","date":"2022-07-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MuchHair/HQM","path":"models/Hard_Sample/HQM/hoi_HQM.py","file_url":"https://github.com/MuchHair/HQM/blob/HEAD/models/Hard_Sample/HQM/hoi_HQM.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ae8f0c20ba273e10","mcp_get_code":{"code_sha256":"ae8f0c20ba273e10"}},{"arxiv_id":"2206.00621","paper":"/paper/cross-view-language-modeling-towards-unified","title":"Cross-View Language Modeling: Towards Unified Cross-Lingual Cross-Modal Pre-training","date":"2022-06-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zengyan-97/cclm","path":"models/box_ops.py","file_url":"https://github.com/zengyan-97/cclm/blob/HEAD/models/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"BSD-3-Clause","inline_ok":false,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2203.01666","paper":"/paper/correlation-aware-deep-tracking","title":"Correlation-Aware Deep Tracking","date":"2022-03-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"phiphiphi31/SBT","path":"lib/models/sbt/transt_loss/box_ops.py","file_url":"https://github.com/phiphiphi31/SBT/blob/HEAD/lib/models/sbt/transt_loss/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2201.00487","paper":"/paper/language-as-queries-for-referring-video","title":"Language as Queries for Referring Video Object Segmentation","date":"2022-01-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wjn922/ReferFormer","path":"inference_davis.py","file_url":"https://github.com/wjn922/ReferFormer/blob/HEAD/inference_davis.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"ef1a3e10a9dbf4a9","mcp_get_code":{"code_sha256":"ef1a3e10a9dbf4a9"}},{"arxiv_id":"2112.05375","paper":"/paper/rethinking-the-two-stage-framework-for","title":"Rethinking the Two-Stage Framework for Grounded Situation Recognition","date":"2021-12-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kellyiss/situformer","path":"util/box_ops.py","file_url":"https://github.com/kellyiss/situformer/blob/HEAD/util/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2112.01838","paper":"/paper/efficient-two-stage-detection-of-human-object","title":"Efficient Two-Stage Detection of Human-Object Interactions with a Novel Unary-Pairwise Transformer","date":"2021-12-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"fredzzhang/upt","path":"ops.py","file_url":"https://github.com/fredzzhang/upt/blob/HEAD/ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2111.10135","paper":"/paper/grounded-situation-recognition-with","title":"Grounded Situation Recognition with Transformers","date":"2021-11-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jhcho99/gsrtr","path":"util/box_ops.py","file_url":"https://github.com/jhcho99/gsrtr/blob/HEAD/util/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2110.00061","paper":"/paper/scientific-evidence-extraction","title":"PubTables-1M: Towards comprehensive table extraction from unstructured documents","date":"2021-09-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/table-transformer","path":"src/inference.py","file_url":"https://github.com/microsoft/table-transformer/blob/HEAD/src/inference.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"52d4b812bdc17687","mcp_get_code":{"code_sha256":"52d4b812bdc17687"}},{"arxiv_id":"2110.00061","paper":"/paper/scientific-evidence-extraction","title":"PubTables-1M: Towards comprehensive table extraction from unstructured documents","date":"2021-09-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"phamquiluan/table-transformer","path":"src/transforms.py","file_url":"https://github.com/phamquiluan/table-transformer/blob/HEAD/src/transforms.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2109.01066","paper":"/paper/4d-net-for-learned-multi-modal-alignment","title":"4D-Net for Learned Multi-Modal Alignment","date":"2021-09-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"chanlilong/4D_NET_pytorch","path":"utils/box_ops.py","file_url":"https://github.com/chanlilong/4D_NET_pytorch/blob/HEAD/utils/box_ops.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a8982a32c4283e7b","mcp_get_code":{"code_sha256":"a8982a32c4283e7b"}},{"arxiv_id":"2104.00969","paper":"/paper/tuber-tube-transformer-for-action-detection","title":"TubeR: Tubelet Transformer for Video Action Detection","date":"2021-04-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"amazon-science/tubelet-transformer","path":"models/transformer/util/box_ops.py","file_url":"https://github.com/amazon-science/tubelet-transformer/blob/HEAD/models/transformer/util/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2103.12115","paper":"/paper/end-to-end-trainable-multi-instance-pose","title":"End-to-End Trainable Multi-Instance Pose Estimation with Transformers","date":"2021-03-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"amathislab/poet","path":"util/box_ops.py","file_url":"https://github.com/amathislab/poet/blob/HEAD/util/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2103.12115","paper":"/paper/end-to-end-trainable-multi-instance-pose","title":"End-to-End Trainable Multi-Instance Pose Estimation with Transformers","date":"2021-03-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pranoyr/pose-estimation-with-transformers","path":"inference.py","file_url":"https://github.com/pranoyr/pose-estimation-with-transformers/blob/HEAD/inference.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"ef1a3e10a9dbf4a9","mcp_get_code":{"code_sha256":"ef1a3e10a9dbf4a9"}},{"arxiv_id":"2011.10881","paper":"/paper/rethinking-transformer-based-set-prediction","title":"Rethinking Transformer-based Set Prediction for Object Detection","date":"2020-11-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"edward-sun/tsp-detection","path":"rcnn/rcnn_heads.py","file_url":"https://github.com/edward-sun/tsp-detection/blob/HEAD/rcnn/rcnn_heads.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"2010.04159","paper":"/paper/deformable-detr-deformable-transformers-for-1","title":"Deformable DETR: Deformable Transformers for End-to-End Object Detection","date":"2020-10-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lyqcom/detr","path":"src/DETR/matcher_np.py","file_url":"https://github.com/lyqcom/detr/blob/HEAD/src/DETR/matcher_np.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"074307e51146fa09","mcp_get_code":{"code_sha256":"074307e51146fa09"}},{"arxiv_id":"2005.12872","paper":"/paper/end-to-end-object-detection-with-transformers","title":"End-to-End Object Detection with Transformers","date":"2020-05-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LKLQQ/detr","path":"src/box_ops.py","file_url":"https://github.com/LKLQQ/detr/blob/HEAD/src/box_ops.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"074307e51146fa09","mcp_get_code":{"code_sha256":"074307e51146fa09"}},{"arxiv_id":"1811.07628","paper":"/paper/atom-accurate-tracking-by-overlap","title":"ATOM: Accurate Tracking by Overlap Maximization","date":"2018-11-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xuefeng-zhu5/cdaat","path":"lib/utils/box_ops.py","file_url":"https://github.com/xuefeng-zhu5/cdaat/blob/HEAD/lib/utils/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"ijcai2024_0084","paper":null,"title":"arXiv:ijcai2024_0084","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"116508/CF-Deformable-DETR","path":"VisualResult.py","file_url":"https://github.com/116508/CF-Deformable-DETR/blob/HEAD/VisualResult.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ef1a3e10a9dbf4a9","mcp_get_code":{"code_sha256":"ef1a3e10a9dbf4a9"}},{"arxiv_id":"aaai_28465","paper":null,"title":"arXiv:aaai_28465","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"OpenGVLab/MUTR","path":"inference_davis.py","file_url":"https://github.com/OpenGVLab/MUTR/blob/HEAD/inference_davis.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ef1a3e10a9dbf4a9","mcp_get_code":{"code_sha256":"ef1a3e10a9dbf4a9"}},{"arxiv_id":"Zhou_When_Pixel_Difference_Patterns_Meet_ViT_PiDiViT_for_Few-Shot_Object_ICCV_2025_paper","paper":null,"title":"arXiv:Zhou_When_Pixel_Difference_Patterns_Meet_ViT_PiDiViT_for_Few-Shot_Object_ICCV_2025_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"Seaz9/PiDiViT","path":"lib/regionprop.py","file_url":"https://github.com/Seaz9/PiDiViT/blob/HEAD/lib/regionprop.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"009912c75c8c77cf","mcp_get_code":{"code_sha256":"009912c75c8c77cf"}},{"arxiv_id":"Yuan_CAT_A_Unified_Click-and-Track_Framework_for_Realistic_Tracking_ICCV_2025_paper","paper":null,"title":"arXiv:Yuan_CAT_A_Unified_Click-and-Track_Framework_for_Realistic_Tracking_ICCV_2025_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"ysyuann/CAT","path":"lib/utils/box_ops.py","file_url":"https://github.com/ysyuann/CAT/blob/HEAD/lib/utils/box_ops.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"invariant","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e0a06ded5d4f6c3c","mcp_get_code":{"code_sha256":"e0a06ded5d4f6c3c"}},{"arxiv_id":"Pan_Wnet_Audio-Guided_Video_Object_Segmentation_via_Wavelet-Based_Cross-Modal_Denoising_Networks_CVPR_2022_paper","paper":null,"title":"arXiv:Pan_Wnet_Audio-Guided_Video_Object_Segmentation_via_Wavelet-Based_Cross-Modal_Denoising_Networks_CVPR_2022_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"asudahkzj/Wnet","path":"inference_a2d.py","file_url":"https://github.com/asudahkzj/Wnet/blob/HEAD/inference_a2d.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ef1a3e10a9dbf4a9","mcp_get_code":{"code_sha256":"ef1a3e10a9dbf4a9"}}]}