{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/build-position-encoding","entry":"build_position_encoding","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":42,"n_papers_ran":23,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":30,"n_samples_ran":17,"n_samples_fingerprinted":0,"n_places":42,"n_places_pointer_only":15,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":11,"ran_fixture":0,"ran":6,"unverified":13},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2510.17218","paper":"/paper/arxiv-2510-17218","title":"When One Moment Isn't Enough: Multi-Moment Retrieval with Cross-Moment Interactions","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"Zhuo-Cao/QV-M2","path":"FlashMMR/position_encoding.py","file_url":"https://github.com/Zhuo-Cao/QV-M2/blob/HEAD/FlashMMR/position_encoding.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1affbf903fa3204b","mcp_get_code":{"code_sha256":"1affbf903fa3204b"}},{"arxiv_id":"2507.19993","paper":null,"title":"arXiv:2507.19993","date":null,"month_inferred_from_arxiv_id":"2025-07","title_source":null,"repo":"Howardkhh/FROSS","path":"EGTR/model/deformable_detr.py","file_url":"https://github.com/Howardkhh/FROSS/blob/HEAD/EGTR/model/deformable_detr.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"72ceaeb9390c2de9","mcp_get_code":{"code_sha256":"72ceaeb9390c2de9"}},{"arxiv_id":"2505.11726","paper":"/paper/disambiguating-reference-in-visually-grounded","title":"Disambiguating Reference in Visually Grounded Dialogues through Joint Modeling of Textual and Multimodal Semantic Structures","date":"2025-05-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ashkamath/mdetr","path":"models/position_encoding.py","file_url":"https://github.com/ashkamath/mdetr/blob/HEAD/models/position_encoding.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"cea3f92defccfb9c","mcp_get_code":{"code_sha256":"cea3f92defccfb9c"}},{"arxiv_id":"2503.09402","paper":"/paper/vlog-video-language-models-by-generative","title":"VLog: Video-Language Models by Generative Retrieval of Narration Vocabulary","date":"2025-03-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"showlab/VLog","path":"VLog/model/models.py","file_url":"https://github.com/showlab/VLog/blob/HEAD/VLog/model/models.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"020b7f70a236fb97","mcp_get_code":{"code_sha256":"020b7f70a236fb97"}},{"arxiv_id":"2501.18954","paper":"/paper/llmdet-learning-strong-open-vocabulary-object","title":"LLMDet: Learning Strong Open-Vocabulary Object Detectors under the Supervision of Large Language Models","date":"2025-01-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"iSEE-Laboratory/LLMDet","path":"hf_model/modeling_grounding_dino.py","file_url":"https://github.com/iSEE-Laboratory/LLMDet/blob/HEAD/hf_model/modeling_grounding_dino.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"d23bffc29f903730","mcp_get_code":{"code_sha256":"d23bffc29f903730"}},{"arxiv_id":"2501.07305","paper":"/paper/the-devil-is-in-the-spurious-correlation","title":"The Devil is in the Spurious Correlation: Boosting Moment Retrieval via Temporal Dynamic Learning","date":"2025-01-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xyangzhou/TD-DETR","path":"td_detr/position_encoding.py","file_url":"https://github.com/xyangzhou/TD-DETR/blob/HEAD/td_detr/position_encoding.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d6ca656c20df137a","mcp_get_code":{"code_sha256":"d6ca656c20df137a"}},{"arxiv_id":"2501.05644","paper":"/paper/interpretable-enzyme-function-prediction-via","title":"Interpretable Enzyme Function Prediction via Residue-Level Detection","date":"2025-01-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yangzhao1230/protdetr","path":"models/position_encoding.py","file_url":"https://github.com/yangzhao1230/protdetr/blob/HEAD/models/position_encoding.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ea84c6853134f470","mcp_get_code":{"code_sha256":"ea84c6853134f470"}},{"arxiv_id":"2412.13441","paper":"/paper/flashvtg-feature-layering-and-adaptive-score","title":"FlashVTG: Feature Layering and Adaptive Score Handling Network for Video Temporal Grounding","date":"2024-12-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhuo-cao/flashvtg","path":"FlashVTG/position_encoding.py","file_url":"https://github.com/zhuo-cao/flashvtg/blob/HEAD/FlashVTG/position_encoding.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1affbf903fa3204b","mcp_get_code":{"code_sha256":"1affbf903fa3204b"}},{"arxiv_id":"2407.15051","paper":"/paper/prior-knowledge-integration-via-llm-encoding","title":"Prior Knowledge Integration via LLM Encoding and Pseudo Event Regulation for Video Moment Retrieval","date":"2024-07-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"fletcherjiang/llmepet","path":"llm_epet/position_encoding.py","file_url":"https://github.com/fletcherjiang/llmepet/blob/HEAD/llm_epet/position_encoding.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"79c51f8fc390ab83","mcp_get_code":{"code_sha256":"79c51f8fc390ab83"}},{"arxiv_id":"2407.05118","paper":"/paper/shine-saliency-aware-hierarchical-negative","title":"SHINE: Saliency-aware HIerarchical NEgative Ranking for Compositional Temporal Grounding","date":"2024-07-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zxccade/SHINE","path":"shine/position_encoding.py","file_url":"https://github.com/zxccade/SHINE/blob/HEAD/shine/position_encoding.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"d6ca656c20df137a","mcp_get_code":{"code_sha256":"d6ca656c20df137a"}},{"arxiv_id":"2407.03200","paper":"/paper/segvg-transferring-object-bounding-box-to","title":"SegVG: Transferring Object Bounding Box to Segmentation for Visual Grounding","date":"2024-07-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"WeitaiKang/SegVG","path":"models/SegVG.py","file_url":"https://github.com/WeitaiKang/SegVG/blob/HEAD/models/SegVG.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"153726d9d215fcbd","mcp_get_code":{"code_sha256":"153726d9d215fcbd"}},{"arxiv_id":"2404.09263","paper":"/paper/task-driven-exploration-decoupling-and-inter","title":"Task-Driven Exploration: Decoupling and Inter-Task Feedback for Joint Moment Retrieval and Highlight Detection","date":"2024-04-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"EdenGabriel/TaskWeave","path":"taskweave/position_encoding.py","file_url":"https://github.com/EdenGabriel/TaskWeave/blob/HEAD/taskweave/position_encoding.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d6ca656c20df137a","mcp_get_code":{"code_sha256":"d6ca656c20df137a"}},{"arxiv_id":"2404.02072","paper":"/paper/egtr-extracting-graph-from-transformer-for","title":"EGTR: Extracting Graph from Transformer for Scene Graph Generation","date":"2024-04-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"naver-ai/egtr","path":"model/deformable_detr.py","file_url":"https://github.com/naver-ai/egtr/blob/HEAD/model/deformable_detr.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"72ceaeb9390c2de9","mcp_get_code":{"code_sha256":"72ceaeb9390c2de9"}},{"arxiv_id":"2403.01813","paper":"/paper/a-simple-baseline-for-efficient-hand-mesh","title":"A Simple Baseline for Efficient Hand Mesh Reconstruction","date":"2024-03-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"patiencefromzhou/simplehand","path":"models/position_embedding.py","file_url":"https://github.com/patiencefromzhou/simplehand/blob/HEAD/models/position_embedding.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e75bcdbf04de72fc","mcp_get_code":{"code_sha256":"e75bcdbf04de72fc"}},{"arxiv_id":"2401.02309","paper":"/paper/tr-detr-task-reciprocal-transformer-for-joint","title":"TR-DETR: Task-Reciprocal Transformer for Joint Moment Retrieval and Highlight Detection","date":"2024-01-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mingyao1120/tr-detr","path":"tr_detr/position_encoding.py","file_url":"https://github.com/mingyao1120/tr-detr/blob/HEAD/tr_detr/position_encoding.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d6ca656c20df137a","mcp_get_code":{"code_sha256":"d6ca656c20df137a"}},{"arxiv_id":"2401.01578","paper":"/paper/context-guided-spatio-temporal-video","title":"Context-Guided Spatio-Temporal Video Grounding","date":"2024-01-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"HengLan/CGSTVG","path":"models/pipeline.py","file_url":"https://github.com/HengLan/CGSTVG/blob/HEAD/models/pipeline.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e936bfbe7328d661","mcp_get_code":{"code_sha256":"e936bfbe7328d661"}},{"arxiv_id":"2312.00083","paper":"/paper/bam-detr-boundary-aligned-moment-detection","title":"BAM-DETR: Boundary-Aligned Moment Detection Transformer for Temporal Sentence Grounding in Videos","date":"2023-11-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Pilhyeon/BAM-DETR","path":"bam_detr/position_encoding.py","file_url":"https://github.com/Pilhyeon/BAM-DETR/blob/HEAD/bam_detr/position_encoding.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"e8a160c4a8f899b5","mcp_get_code":{"code_sha256":"e8a160c4a8f899b5"}},{"arxiv_id":"2311.16464","paper":"/paper/bridging-the-gap-a-unified-video","title":"Bridging the Gap: A Unified Video Comprehension Framework for Moment Retrieval and Highlight Detection","date":"2023-11-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"easonxiao-888/uvcom","path":"uvcom/position_encoding.py","file_url":"https://github.com/easonxiao-888/uvcom/blob/HEAD/uvcom/position_encoding.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d6ca656c20df137a","mcp_get_code":{"code_sha256":"d6ca656c20df137a"}},{"arxiv_id":"2311.08835","paper":"/paper/correlation-guided-query-dependency","title":"Correlation-Guided Query-Dependency Calibration for Video Temporal Grounding","date":"2023-11-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wjun0830/cgdetr","path":"cg_detr/position_encoding.py","file_url":"https://github.com/wjun0830/cgdetr/blob/HEAD/cg_detr/position_encoding.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1affbf903fa3204b","mcp_get_code":{"code_sha256":"1affbf903fa3204b"}},{"arxiv_id":"2308.15512","paper":"/paper/shatter-and-gather-learning-referring-image","title":"Shatter and Gather: Learning Referring Image Segmentation with Text Supervision","date":"2023-08-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kdwonn/SaG","path":"model/cross_modal_attention.py","file_url":"https://github.com/kdwonn/SaG/blob/HEAD/model/cross_modal_attention.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1508f01cf4f224d9","mcp_get_code":{"code_sha256":"1508f01cf4f224d9"}},{"arxiv_id":"2308.13814","paper":"/paper/point-query-quadtree-for-crowd-counting","title":"Point-Query Quadtree for Crowd Counting, Localization, and More","date":"2023-08-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cxliu0/PET","path":"models/pet.py","file_url":"https://github.com/cxliu0/PET/blob/HEAD/models/pet.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f57a0aab388112d0","mcp_get_code":{"code_sha256":"f57a0aab388112d0"}},{"arxiv_id":"2308.06947","paper":"/paper/knowing-where-to-focus-event-aware","title":"Knowing Where to Focus: Event-aware Transformer for Video Grounding","date":"2023-08-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jinhyunj/eatr","path":"models/position_encoding.py","file_url":"https://github.com/jinhyunj/eatr/blob/HEAD/models/position_encoding.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b4196f787bec2e3e","mcp_get_code":{"code_sha256":"b4196f787bec2e3e"}},{"arxiv_id":"2308.02463","paper":"/paper/towards-generalist-foundation-model-for","title":"Towards Generalist Foundation Model for Radiology by Leveraging Web-scale 2D&3D Medical Data","date":"2023-08-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"chaoyi-wu/radfm","path":"Quick_demo/Model/RadFM/position_encoding.py","file_url":"https://github.com/chaoyi-wu/radfm/blob/HEAD/Quick_demo/Model/RadFM/position_encoding.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c8f7b3c548c9d56a","mcp_get_code":{"code_sha256":"c8f7b3c548c9d56a"}},{"arxiv_id":"2307.16715","paper":"/paper/univtg-towards-unified-video-language","title":"UniVTG: Towards Unified Video-Language Temporal Grounding","date":"2023-07-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"showlab/univtg","path":"model/position_encoding.py","file_url":"https://github.com/showlab/univtg/blob/HEAD/model/position_encoding.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2912d21d62514265","mcp_get_code":{"code_sha256":"2912d21d62514265"}},{"arxiv_id":"2307.02869","paper":"/paper/momentdiff-generative-video-moment-retrieval","title":"MomentDiff: Generative Video Moment Retrieval from Random to Real","date":"2023-07-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"imccretrieval/momentdiff","path":"momentdiff/position_encoding.py","file_url":"https://github.com/imccretrieval/momentdiff/blob/HEAD/momentdiff/position_encoding.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d6ca656c20df137a","mcp_get_code":{"code_sha256":"d6ca656c20df137a"}},{"arxiv_id":"2305.00434","paper":"/paper/evreal-towards-a-comprehensive-benchmark-and","title":"EVREAL: Towards a Comprehensive Benchmark and Analysis Suite for Event-based Video Reconstruction","date":"2023-04-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ercanburak/EVREAL","path":"model/eitr/position_encoding.py","file_url":"https://github.com/ercanburak/EVREAL/blob/HEAD/model/eitr/position_encoding.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bfc2d3b5c69ffb3b","mcp_get_code":{"code_sha256":"bfc2d3b5c69ffb3b"}},{"arxiv_id":"2304.10131","paper":"/paper/learning-bottleneck-concepts-in-image","title":"Learning Bottleneck Concepts in Image Classification","date":"2023-04-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wbw520/botcl","path":"model/reconstruct/model_main.py","file_url":"https://github.com/wbw520/botcl/blob/HEAD/model/reconstruct/model_main.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"59886ab4f3558c49","mcp_get_code":{"code_sha256":"59886ab4f3558c49"}},{"arxiv_id":"2304.09502","paper":"/paper/sampling-is-matter-point-guided-3d-human-mesh-1","title":"Sampling is Matter: Point-guided 3D Human Mesh Reconstruction","date":"2023-04-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"DCVL-3D/PointHMR_release","path":"src/modeling/model/network.py","file_url":"https://github.com/DCVL-3D/PointHMR_release/blob/HEAD/src/modeling/model/network.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c62bd3602bcf3ba2","mcp_get_code":{"code_sha256":"c62bd3602bcf3ba2"}},{"arxiv_id":"2210.12658","paper":"/paper/extending-phrase-grounding-with-pronouns-in","title":"Extending Phrase Grounding with Pronouns in Visual Dialogues","date":"2022-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"izhx/Phrase-Grounding-with-Pronoun","path":"code/src/mdetr/position_encoding.py","file_url":"https://github.com/izhx/Phrase-Grounding-with-Pronoun/blob/HEAD/code/src/mdetr/position_encoding.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1b6e04ba3ceda006","mcp_get_code":{"code_sha256":"1b6e04ba3ceda006"}},{"arxiv_id":"2209.13306","paper":"/paper/embracing-consistency-a-one-stage-approach","title":"Embracing Consistency: A One-Stage Approach for Spatio-Temporal Video Grounding","date":"2022-09-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jy0205/stcat","path":"models/pipeline.py","file_url":"https://github.com/jy0205/stcat/blob/HEAD/models/pipeline.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ebf7ea1f0aba8315","mcp_get_code":{"code_sha256":"ebf7ea1f0aba8315"}},{"arxiv_id":"2209.02242","paper":"/paper/ptseformer-progressive-temporal-spatial","title":"PTSEFormer: Progressive Temporal-Spatial Enhanced TransFormer Towards Video Object Detection","date":"2022-09-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Hon-Wong/PTSEFormer","path":"src/models/model_builder.py","file_url":"https://github.com/Hon-Wong/PTSEFormer/blob/HEAD/src/models/model_builder.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f2cdbc05e138fb89","mcp_get_code":{"code_sha256":"f2cdbc05e138fb89"}},{"arxiv_id":"2207.13820","paper":"/paper/cross-attention-of-disentangled-modalities","title":"Cross-Attention of Disentangled Modalities for 3D Human Mesh Recovery with Transformers","date":"2022-07-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"postech-ami/fastmetro","path":"src/modeling/model/modeling_fastmetro.py","file_url":"https://github.com/postech-ami/fastmetro/blob/HEAD/src/modeling/model/modeling_fastmetro.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4407bd06305a2f6c","mcp_get_code":{"code_sha256":"4407bd06305a2f6c"}},{"arxiv_id":"2207.13325","paper":"/paper/siri-a-simple-selective-retraining-mechanism","title":"SiRi: A Simple Selective Retraining Mechanism for Transformer-based Visual Grounding","date":"2022-07-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"qumengxue/siri-vg","path":"models/position_encoding.py","file_url":"https://github.com/qumengxue/siri-vg/blob/HEAD/models/position_encoding.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"cea3f92defccfb9c","mcp_get_code":{"code_sha256":"cea3f92defccfb9c"}},{"arxiv_id":"2111.12085","paper":"/paper/crossing-the-format-boundary-of-text-and","title":"UniTAB: Unifying Text and Box Outputs for Grounded Vision-Language Modeling","date":"2021-11-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/UniTAB","path":"models/position_encoding.py","file_url":"https://github.com/microsoft/UniTAB/blob/HEAD/models/position_encoding.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cea3f92defccfb9c","mcp_get_code":{"code_sha256":"cea3f92defccfb9c"}},{"arxiv_id":"2110.04722","paper":"/paper/transformer-based-dual-relation-graph-for-1","title":"Transformer-based Dual Relation Graph for Multi-label Image Recognition","date":"2021-10-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"iCVTEAM/TDRG","path":"models/TDRG.py","file_url":"https://github.com/iCVTEAM/TDRG/blob/HEAD/models/TDRG.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"626421743e751cfd","mcp_get_code":{"code_sha256":"626421743e751cfd"}},{"arxiv_id":"2108.11084","paper":"/paper/efficient-transformer-for-single-image-super","title":"Transformer for Single Image Super-Resolution","date":"2021-08-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"luissen/esrt","path":"util/position.py","file_url":"https://github.com/luissen/esrt/blob/HEAD/util/position.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"61eaa43ff60ccf54","mcp_get_code":{"code_sha256":"61eaa43ff60ccf54"}},{"arxiv_id":"2104.12763","paper":"/paper/mdetr-modulated-detection-for-end-to-end","title":"MDETR -- Modulated Detection for End-to-End Multi-Modal Understanding","date":"2021-04-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"b-faye/lightmdetr","path":"models/position_encoding.py","file_url":"https://github.com/b-faye/lightmdetr/blob/HEAD/models/position_encoding.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"cea3f92defccfb9c","mcp_get_code":{"code_sha256":"cea3f92defccfb9c"}},{"arxiv_id":"2104.04369","paper":"/paper/video-aided-unsupervised-grammar-induction","title":"Video-aided Unsupervised Grammar Induction","date":"2021-04-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Sy-Zhang/MMC-PCFG","path":"lib/model/vpcfg/position_encoding.py","file_url":"https://github.com/Sy-Zhang/MMC-PCFG/blob/HEAD/lib/model/vpcfg/position_encoding.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ea9874d5c6513212","mcp_get_code":{"code_sha256":"ea9874d5c6513212"}},{"arxiv_id":"2103.11468","paper":"/paper/learning-multi-scene-absolute-pose-regression","title":"Learning Multi-Scene Absolute Pose Regression with Transformers","date":"2021-03-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yolish/c2f-ms-transformer","path":"models/transposenet/C2FEMSTransPoseNet.py","file_url":"https://github.com/yolish/c2f-ms-transformer/blob/HEAD/models/transposenet/C2FEMSTransPoseNet.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9cbd4a5c6cc2d670","mcp_get_code":{"code_sha256":"9cbd4a5c6cc2d670"}},{"arxiv_id":"aaai_29295","paper":null,"title":"arXiv:aaai_29295","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"Jack24658735/FedLGT","path":"models/utils.py","file_url":"https://github.com/Jack24658735/FedLGT/blob/HEAD/models/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a809b63d1fb6ac5c","mcp_get_code":{"code_sha256":"a809b63d1fb6ac5c"}},{"arxiv_id":"aaai_27766","paper":null,"title":"arXiv:aaai_27766","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"7tl7qns7ch/IPOT","path":"models/ipot/position_encoding.py","file_url":"https://github.com/7tl7qns7ch/IPOT/blob/HEAD/models/ipot/position_encoding.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"92655046480cb7fb","mcp_get_code":{"code_sha256":"92655046480cb7fb"}},{"arxiv_id":"Xiao_Bridging_the_Gap_A_Unified_Video_Comprehension_Framework_for_Moment_CVPR_2024_paper","paper":null,"title":"arXiv:Xiao_Bridging_the_Gap_A_Unified_Video_Comprehension_Framework_for_Moment_CVPR_2024_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"EasonXiao-888/UVCOM","path":"uvcom/position_encoding.py","file_url":"https://github.com/EasonXiao-888/UVCOM/blob/HEAD/uvcom/position_encoding.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d6ca656c20df137a","mcp_get_code":{"code_sha256":"d6ca656c20df137a"}}]}