{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/build-transformer","entry":"build_transformer","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":62,"n_papers_ran":0,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":59,"n_samples_ran":0,"n_samples_fingerprinted":0,"n_places":66,"n_places_pointer_only":20,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":0,"unverified":59},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2601.12079","paper":"/paper/arxiv-2601-12079","title":"EmoLat: Text-driven Image Sentiment Transfer via Emotion Latent Space","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"JingVIPLab/EmoLat","path":"model/transformer.py","file_url":"https://github.com/JingVIPLab/EmoLat/blob/HEAD/model/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4223bbb30a11ebd5","mcp_get_code":{"code_sha256":"4223bbb30a11ebd5"}},{"arxiv_id":"2601.02289","paper":"/paper/arxiv-2601-02289","title":"Rank-based Geographical Regularization: Revisiting Contrastive Self-Supervised Learning for Multispectral Remote Sensing Imagery","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"tomburgert/georank","path":"models/scalemae_backbone.py","file_url":"https://github.com/tomburgert/georank/blob/HEAD/models/scalemae_backbone.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c22ce1de0a795380","mcp_get_code":{"code_sha256":"c22ce1de0a795380"}},{"arxiv_id":"2511.07819","paper":"/paper/arxiv-2511-07819","title":"Human Motion Synthesis in 3D Scenes via Unified Scene Semantic Occupancy","date":null,"month_inferred_from_arxiv_id":"2025-11","title_source":"syntology","repo":"jingyugong/SSOMotion","path":"model/transformer.py","file_url":"https://github.com/jingyugong/SSOMotion/blob/HEAD/model/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5aad82953ecc92f5","mcp_get_code":{"code_sha256":"5aad82953ecc92f5"}},{"arxiv_id":"2505.11726","paper":"/paper/disambiguating-reference-in-visually-grounded","title":"Disambiguating Reference in Visually Grounded Dialogues through Joint Modeling of Textual and Multimodal Semantic Structures","date":"2025-05-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ashkamath/mdetr","path":"models/transformer.py","file_url":"https://github.com/ashkamath/mdetr/blob/HEAD/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0796c9e7b3ba5e23","mcp_get_code":{"code_sha256":"0796c9e7b3ba5e23"}},{"arxiv_id":"2503.09402","paper":"/paper/vlog-video-language-models-by-generative","title":"VLog: Video-Language Models by Generative Retrieval of Narration Vocabulary","date":"2025-03-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"showlab/VLog","path":"VLog/model/models.py","file_url":"https://github.com/showlab/VLog/blob/HEAD/VLog/model/models.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c8d145f15e26a93d","mcp_get_code":{"code_sha256":"c8d145f15e26a93d"}},{"arxiv_id":"2501.07305","paper":"/paper/the-devil-is-in-the-spurious-correlation","title":"The Devil is in the Spurious Correlation: Boosting Moment Retrieval via Temporal Dynamic Learning","date":"2025-01-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xyangzhou/TD-DETR","path":"td_detr/transformer.py","file_url":"https://github.com/xyangzhou/TD-DETR/blob/HEAD/td_detr/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"595b7178ce4c0616","mcp_get_code":{"code_sha256":"595b7178ce4c0616"}},{"arxiv_id":"2501.05644","paper":"/paper/interpretable-enzyme-function-prediction-via","title":"Interpretable Enzyme Function Prediction via Residue-Level Detection","date":"2025-01-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yangzhao1230/protdetr","path":"models/transformer.py","file_url":"https://github.com/yangzhao1230/protdetr/blob/HEAD/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d8c3057c1f1df693","mcp_get_code":{"code_sha256":"d8c3057c1f1df693"}},{"arxiv_id":"2412.13441","paper":"/paper/flashvtg-feature-layering-and-adaptive-score","title":"FlashVTG: Feature Layering and Adaptive Score Handling Network for Video Temporal Grounding","date":"2024-12-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhuo-cao/flashvtg","path":"FlashVTG/transformer.py","file_url":"https://github.com/zhuo-cao/flashvtg/blob/HEAD/FlashVTG/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2359afb511e0c62c","mcp_get_code":{"code_sha256":"2359afb511e0c62c"}},{"arxiv_id":"2409.10385","paper":"/paper/mamba-st-state-space-model-for-efficient","title":"Mamba-ST: State Space Model for Efficient Style Transfer","date":"2024-09-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"filippobotti/mambast","path":"models/mamba.py","file_url":"https://github.com/filippobotti/mambast/blob/HEAD/models/mamba.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"163a3014ed2dbf50","mcp_get_code":{"code_sha256":"163a3014ed2dbf50"}},{"arxiv_id":"2409.04999","paper":"/paper/visual-grounding-with-multi-modal-conditional","title":"Visual Grounding with Multi-modal Conditional Adaptation","date":"2024-09-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mr-bigworth/mmca","path":"models_mmca_vector_based/visual_model/transformer.py","file_url":"https://github.com/mr-bigworth/mmca/blob/HEAD/models_mmca_vector_based/visual_model/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"dd06382c371fad2d","mcp_get_code":{"code_sha256":"dd06382c371fad2d"}},{"arxiv_id":"2408.07500","paper":"/paper/cross-platform-video-person-reid-a-new","title":"Cross-Platform Video Person ReID: A New Benchmark Dataset and Adaptation Approach","date":"2024-08-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"FHR-L/VSLA-CLIP","path":"model/make_model_clipvideoreid_reidadapter_pbp.py","file_url":"https://github.com/FHR-L/VSLA-CLIP/blob/HEAD/model/make_model_clipvideoreid_reidadapter_pbp.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a8f5824659641940","mcp_get_code":{"code_sha256":"a8f5824659641940"}},{"arxiv_id":"2408.01044","paper":"/paper/2408-01044","title":"Boosting Gaze Object Prediction via Pixel-level Supervision from Vision Foundation Model","date":"2024-08-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jinyang06/SamGOP","path":"maskGOP/transformer.py","file_url":"https://github.com/jinyang06/SamGOP/blob/HEAD/maskGOP/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f52bf9a389e50bbc","mcp_get_code":{"code_sha256":"f52bf9a389e50bbc"}},{"arxiv_id":"2407.12366","paper":"/paper/navgpt-2-unleashing-navigational-reasoning","title":"NavGPT-2: Unleashing Navigational Reasoning Capability for Large Vision-Language Models","date":"2024-07-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"GengzeZhou/NavGPT-2","path":"map_nav_src/models/transformer.py","file_url":"https://github.com/GengzeZhou/NavGPT-2/blob/HEAD/map_nav_src/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"df2919fd1dad1d43","mcp_get_code":{"code_sha256":"df2919fd1dad1d43"}},{"arxiv_id":"2407.08133","paper":"/paper/nonverbal-interaction-detection","title":"Nonverbal Interaction Detection","date":"2024-07-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"weijianan1/nvi","path":"NVI-DEHR/models/transformer.py","file_url":"https://github.com/weijianan1/nvi/blob/HEAD/NVI-DEHR/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5c4a9bab338ee34e","mcp_get_code":{"code_sha256":"5c4a9bab338ee34e"}},{"arxiv_id":"2407.03200","paper":"/paper/segvg-transferring-object-bounding-box-to","title":"SegVG: Transferring Object Bounding Box to Segmentation for Visual Grounding","date":"2024-07-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"WeitaiKang/SegVG","path":"models/SegVG.py","file_url":"https://github.com/WeitaiKang/SegVG/blob/HEAD/models/SegVG.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9008729eeca3ae5e","mcp_get_code":{"code_sha256":"9008729eeca3ae5e"}},{"arxiv_id":"2406.03459","paper":"/paper/lw-detr-a-transformer-replacement-to-yolo-for","title":"LW-DETR: A Transformer Replacement to YOLO for Real-Time Detection","date":"2024-06-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"atten4vis/lw-detr","path":"models/transformer.py","file_url":"https://github.com/atten4vis/lw-detr/blob/HEAD/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ec7a90cc24d6630d","mcp_get_code":{"code_sha256":"ec7a90cc24d6630d"}},{"arxiv_id":"2404.01725","paper":"/paper/disentangled-pre-training-for-human-object","title":"Disentangled Pre-training for Human-Object Interaction Detection","date":"2024-04-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xingaoli/DP-HOI","path":"models/transformer.py","file_url":"https://github.com/xingaoli/DP-HOI/blob/HEAD/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"fdc5e45115a1f362","mcp_get_code":{"code_sha256":"fdc5e45115a1f362"}},{"arxiv_id":"2403.05021","paper":"/paper/beyond-mot-semantic-multi-object-tracking","title":"Beyond MOT: Semantic Multi-Object Tracking","date":"2024-03-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Nathan-Li123/SMOTer","path":"smoter/modeling/roi_heads/transformer.py","file_url":"https://github.com/Nathan-Li123/SMOTer/blob/HEAD/smoter/modeling/roi_heads/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"88dddc90d2044da0","mcp_get_code":{"code_sha256":"88dddc90d2044da0"}},{"arxiv_id":"2402.17062","paper":"/paper/hoisdf-constraining-3d-hand-object-pose","title":"HOISDF: Constraining 3D Hand-Object Pose Estimation with Global Signed Distance Fields","date":"2024-02-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"amathislab/hoisdf","path":"common/nets/transformer.py","file_url":"https://github.com/amathislab/hoisdf/blob/HEAD/common/nets/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"47b8b020f555d270","mcp_get_code":{"code_sha256":"47b8b020f555d270"}},{"arxiv_id":"2401.01529","paper":"/paper/glance-and-focus-memory-prompting-for-multi-1","title":"Glance and Focus: Memory Prompting for Multi-Event Video Question Answering","date":"2024-01-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ByZ0e/Glance-Focus","path":"model/transformer_gf.py","file_url":"https://github.com/ByZ0e/Glance-Focus/blob/HEAD/model/transformer_gf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"e66b5e8b24dfea32","mcp_get_code":{"code_sha256":"e66b5e8b24dfea32"}},{"arxiv_id":"2312.02010","paper":"/paper/towards-learning-a-generalist-model-for","title":"Towards Learning a Generalist Model for Embodied Navigation","date":"2023-12-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lavi-lab/navillm","path":"models/detr_transformer.py","file_url":"https://github.com/lavi-lab/navillm/blob/HEAD/models/detr_transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"df2919fd1dad1d43","mcp_get_code":{"code_sha256":"df2919fd1dad1d43"}},{"arxiv_id":"2309.01692","paper":"/paper/mask-attention-free-transformer-for-3d","title":"Mask-Attention-Free Transformer for 3D Instance Segmentation","date":"2023-09-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dvlab-research/mask-attention-free-transformer","path":"maft/model/transformer.py","file_url":"https://github.com/dvlab-research/mask-attention-free-transformer/blob/HEAD/maft/model/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"31be13e9daaa9da9","mcp_get_code":{"code_sha256":"31be13e9daaa9da9"}},{"arxiv_id":"2308.12604","paper":"/paper/promptmrg-diagnosis-driven-prompts-for","title":"PromptMRG: Diagnosis-Driven Prompts for Medical Report Generation","date":"2023-08-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jhb86253817/promptmrg","path":"models/transformer.py","file_url":"https://github.com/jhb86253817/promptmrg/blob/HEAD/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"6c667e1b7257c930","mcp_get_code":{"code_sha256":"6c667e1b7257c930"}},{"arxiv_id":"2308.09351","paper":"/paper/rlipv2-fast-scaling-of-relational-language","title":"RLIPv2: Fast Scaling of Relational Language-Image Pre-training","date":"2023-08-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jacobyuan7/ocn-hoi-benchmark","path":"models/transformer.py","file_url":"https://github.com/jacobyuan7/ocn-hoi-benchmark/blob/HEAD/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"9a524c433c47ed10","mcp_get_code":{"code_sha256":"9a524c433c47ed10"}},{"arxiv_id":"2308.06112","paper":"/paper/lip2vec-efficient-and-robust-visual-speech","title":"Lip2Vec: Efficient and Robust Visual Speech Recognition via Latent-to-Latent Visual to Audio Representation Mapping","date":"2023-08-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"YasserdahouML/Lip2Vec","path":"models/transformer.py","file_url":"https://github.com/YasserdahouML/Lip2Vec/blob/HEAD/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5c86d34cae9ef6a1","mcp_get_code":{"code_sha256":"5c86d34cae9ef6a1"}},{"arxiv_id":"2308.04758","paper":"/paper/bird-s-eye-view-scene-graph-for-vision","title":"Bird's-Eye-View Scene Graph for Vision-Language Navigation","date":"2023-08-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"defaultrui/bev-scene-graph","path":"bsg_vln/map_nav_src/models/transformer.py","file_url":"https://github.com/defaultrui/bev-scene-graph/blob/HEAD/bsg_vln/map_nav_src/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"df2919fd1dad1d43","mcp_get_code":{"code_sha256":"df2919fd1dad1d43"}},{"arxiv_id":"2307.16715","paper":"/paper/univtg-towards-unified-video-language","title":"UniVTG: Towards Unified Video-Language Temporal Grounding","date":"2023-07-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"showlab/univtg","path":"model/transformer.py","file_url":"https://github.com/showlab/univtg/blob/HEAD/model/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a03d2e1a023a5c63","mcp_get_code":{"code_sha256":"a03d2e1a023a5c63"}},{"arxiv_id":"2307.16715","paper":"/paper/univtg-towards-unified-video-language","title":"UniVTG: Towards Unified Video-Language Temporal Grounding","date":"2023-07-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"showlab/univtg","path":"model/transformer_encoder_droppath.py","file_url":"https://github.com/showlab/univtg/blob/HEAD/model/transformer_encoder_droppath.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"141ae5fb5b4041f0","mcp_get_code":{"code_sha256":"141ae5fb5b4041f0"}},{"arxiv_id":"2307.12239","paper":"/paper/dq-det-learning-dynamic-query-combinations","title":"Learning Dynamic Query Combinations for Transformer-based Object Detection and Segmentation","date":"2023-07-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bytedance/DQ-Det","path":"Cond-DETR-DQ/models/transformer.py","file_url":"https://github.com/bytedance/DQ-Det/blob/HEAD/Cond-DETR-DQ/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"86f65ed94e4398da","mcp_get_code":{"code_sha256":"86f65ed94e4398da"}},{"arxiv_id":"2305.13921","paper":"/paper/compositional-text-to-image-synthesis-with","title":"Compositional Text-to-Image Synthesis with Attention Map Control of Diffusion Models","date":"2023-05-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"OPPO-Mente-Lab/attention-mask-control","path":"boxnet_models/transformer.py","file_url":"https://github.com/OPPO-Mente-Lab/attention-mask-control/blob/HEAD/boxnet_models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9083e7dd88fe1570","mcp_get_code":{"code_sha256":"9083e7dd88fe1570"}},{"arxiv_id":"2305.00434","paper":"/paper/evreal-towards-a-comprehensive-benchmark-and","title":"EVREAL: Towards a Comprehensive Benchmark and Analysis Suite for Event-based Video Reconstruction","date":"2023-04-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ercanburak/EVREAL","path":"model/eitr/transformer.py","file_url":"https://github.com/ercanburak/EVREAL/blob/HEAD/model/eitr/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"858bdc32c51ebdef","mcp_get_code":{"code_sha256":"858bdc32c51ebdef"}},{"arxiv_id":"2303.14736","paper":"/paper/disentangling-writer-and-character-styles-for","title":"Disentangling Writer and Character Styles for Handwriting Generation","date":"2023-03-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dailenson/SDT","path":"models/transformer.py","file_url":"https://github.com/dailenson/SDT/blob/HEAD/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d3c210804250a1dc","mcp_get_code":{"code_sha256":"d3c210804250a1dc"}},{"arxiv_id":"2210.05668","paper":"/paper/understanding-embodied-reference-with-touch","title":"Understanding Embodied Reference with Touch-Line Transformer","date":"2022-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yang-li-2000/understanding-embodied-reference-with-touch-line-transformer","path":"models/transformer.py","file_url":"https://github.com/yang-li-2000/understanding-embodied-reference-with-touch-line-transformer/blob/HEAD/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1e7c78d96968a055","mcp_get_code":{"code_sha256":"1e7c78d96968a055"}},{"arxiv_id":"2210.05668","paper":"/paper/understanding-embodied-reference-with-touch","title":"Understanding Embodied Reference with Touch-Line Transformer","date":"2022-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yang-li-2000/understanding-embodied-reference-with-touch-line-transformer","path":"models/transformer_ori.py","file_url":"https://github.com/yang-li-2000/understanding-embodied-reference-with-touch-line-transformer/blob/HEAD/models/transformer_ori.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b0beffbe8deaa71a","mcp_get_code":{"code_sha256":"b0beffbe8deaa71a"}},{"arxiv_id":"2210.05176","paper":"/paper/fine-grained-image-style-transfer-with-visual","title":"Fine-Grained Image Style Transfer with Visual Transformers","date":"2022-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"researchmm/sttr","path":"models_istt/transformer_nonorm_flx.py","file_url":"https://github.com/researchmm/sttr/blob/HEAD/models_istt/transformer_nonorm_flx.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a7bb87542c72ca8e","mcp_get_code":{"code_sha256":"a7bb87542c72ca8e"}},{"arxiv_id":"2209.12213","paper":"/paper/eco-tr-efficient-correspondences-finding-via","title":"ECO-TR: Efficient Correspondences Finding Via Coarse-to-Fine Refinement","date":"2022-09-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dltan7/ECO-TR","path":"src/models/ecotr_modules/transformer.py","file_url":"https://github.com/dltan7/ECO-TR/blob/HEAD/src/models/ecotr_modules/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8ee33b336f178e2f","mcp_get_code":{"code_sha256":"8ee33b336f178e2f"}},{"arxiv_id":"2209.01814","paper":"/paper/rlip-relational-language-image-pre-training","title":"RLIP: Relational Language-Image Pre-training for Human-Object Interaction Detection","date":"2022-09-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"JacobYuan7/RLIP","path":"models/ParSetransformer.py","file_url":"https://github.com/JacobYuan7/RLIP/blob/HEAD/models/ParSetransformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1db67394391c79fa","mcp_get_code":{"code_sha256":"1db67394391c79fa"}},{"arxiv_id":"2207.13820","paper":"/paper/cross-attention-of-disentangled-modalities","title":"Cross-Attention of Disentangled Modalities for 3D Human Mesh Recovery with Transformers","date":"2022-07-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"postech-ami/fastmetro","path":"src/modeling/model/modeling_fastmetro.py","file_url":"https://github.com/postech-ami/fastmetro/blob/HEAD/src/modeling/model/modeling_fastmetro.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c6caa9a5fb3f06a7","mcp_get_code":{"code_sha256":"c6caa9a5fb3f06a7"}},{"arxiv_id":"2207.13325","paper":"/paper/siri-a-simple-selective-retraining-mechanism","title":"SiRi: A Simple Selective Retraining Mechanism for Transformer-based Visual Grounding","date":"2022-07-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"qumengxue/siri-vg","path":"models/transformer.py","file_url":"https://github.com/qumengxue/siri-vg/blob/HEAD/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f8cea2a98b3268dd","mcp_get_code":{"code_sha256":"f8cea2a98b3268dd"}},{"arxiv_id":"2112.05375","paper":"/paper/rethinking-the-two-stage-framework-for","title":"Rethinking the Two-Stage Framework for Grounded Situation Recognition","date":"2021-12-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kellyiss/situformer","path":"models/transformer.py","file_url":"https://github.com/kellyiss/situformer/blob/HEAD/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"da89ce279ae764b6","mcp_get_code":{"code_sha256":"da89ce279ae764b6"}},{"arxiv_id":"2111.12085","paper":"/paper/crossing-the-format-boundary-of-text-and","title":"UniTAB: Unifying Text and Box Outputs for Grounded Vision-Language Modeling","date":"2021-11-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/UniTAB","path":"models/transformer_unitab.py","file_url":"https://github.com/microsoft/UniTAB/blob/HEAD/models/transformer_unitab.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0b02672adf92e20a","mcp_get_code":{"code_sha256":"0b02672adf92e20a"}},{"arxiv_id":"2111.10135","paper":"/paper/grounded-situation-recognition-with","title":"Grounded Situation Recognition with Transformers","date":"2021-11-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jhcho99/gsrtr","path":"models/transformer.py","file_url":"https://github.com/jhcho99/gsrtr/blob/HEAD/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"9888762109072b2a","mcp_get_code":{"code_sha256":"9888762109072b2a"}},{"arxiv_id":"2110.04722","paper":"/paper/transformer-based-dual-relation-graph-for-1","title":"Transformer-based Dual Relation Graph for Multi-label Image Recognition","date":"2021-10-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"iCVTEAM/TDRG","path":"models/TDRG.py","file_url":"https://github.com/iCVTEAM/TDRG/blob/HEAD/models/TDRG.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2a6ad2bf42c596fb","mcp_get_code":{"code_sha256":"2a6ad2bf42c596fb"}},{"arxiv_id":"2110.03864","paper":"/paper/boundary-aware-transformers-for-skin-lesion","title":"Boundary-aware Transformers for Skin Lesion Segmentation","date":"2021-10-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jcwang123/BA-Transformer","path":"src/transformer.py","file_url":"https://github.com/jcwang123/BA-Transformer/blob/HEAD/src/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1b1d552ee72944ff","mcp_get_code":{"code_sha256":"1b1d552ee72944ff"}},{"arxiv_id":"2110.00061","paper":"/paper/scientific-evidence-extraction","title":"PubTables-1M: Towards comprehensive table extraction from unstructured documents","date":"2021-09-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"phamquiluan/table-transformer","path":"detr/models/transformer.py","file_url":"https://github.com/phamquiluan/table-transformer/blob/HEAD/detr/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"302ec916d91b12a8","mcp_get_code":{"code_sha256":"302ec916d91b12a8"}},{"arxiv_id":"2109.10852","paper":"/paper/pix2seq-a-language-modeling-framework-for","title":"Pix2seq: A Language Modeling Framework for Object Detection","date":"2021-09-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gaopengcuhk/Unofficial-Pix2Seq","path":"models/transformer.py","file_url":"https://github.com/gaopengcuhk/Unofficial-Pix2Seq/blob/HEAD/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"cd17dcaf812b820d","mcp_get_code":{"code_sha256":"cd17dcaf812b820d"}},{"arxiv_id":"2108.12630","paper":"/paper/groupformer-group-activity-recognition-with","title":"GroupFormer: Group Activity Recognition with Clustered Spatial-Temporal Transformer","date":"2021-08-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xueyee/groupformer","path":"group/models/transformer.py","file_url":"https://github.com/xueyee/groupformer/blob/HEAD/group/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"302ec916d91b12a8","mcp_get_code":{"code_sha256":"302ec916d91b12a8"}},{"arxiv_id":"2108.10723","paper":"/paper/improving-3d-object-detection-with-channel","title":"Improving 3D Object Detection with Channel-wise Transformer","date":"2021-08-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hlsheng1/ct3d","path":"pcdet/models/roi_heads/ct3d_head.py","file_url":"https://github.com/hlsheng1/ct3d/blob/HEAD/pcdet/models/roi_heads/ct3d_head.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2b5d92421f68b895","mcp_get_code":{"code_sha256":"2b5d92421f68b895"}},{"arxiv_id":"2108.06152","paper":"/paper/conditional-detr-for-fast-training","title":"Conditional DETR for Fast Training Convergence","date":"2021-08-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"atten4vis/conditionaldetr","path":"models/transformer.py","file_url":"https://github.com/atten4vis/conditionaldetr/blob/HEAD/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"bf149877606cbfba","mcp_get_code":{"code_sha256":"bf149877606cbfba"}},{"arxiv_id":"2106.04550","paper":"/paper/detreg-unsupervised-pretraining-with-region","title":"DETReg: Unsupervised Pretraining with Region Priors for Object Detection","date":"2021-06-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"amirbar/detreg","path":"models/transformer.py","file_url":"https://github.com/amirbar/detreg/blob/HEAD/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ccd0edc0cca7b92e","mcp_get_code":{"code_sha256":"ccd0edc0cca7b92e"}},{"arxiv_id":"2104.12763","paper":"/paper/mdetr-modulated-detection-for-end-to-end","title":"MDETR -- Modulated Detection for End-to-End Multi-Modal Understanding","date":"2021-04-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"b-faye/lightmdetr","path":"models/transformer.py","file_url":"https://github.com/b-faye/lightmdetr/blob/HEAD/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"09e538653a528ac8","mcp_get_code":{"code_sha256":"09e538653a528ac8"}},{"arxiv_id":"2104.12763","paper":"/paper/mdetr-modulated-detection-for-end-to-end","title":"MDETR -- Modulated Detection for End-to-End Multi-Modal Understanding","date":"2021-04-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"b-faye/lightmdetr","path":"models/transformer_plus.py","file_url":"https://github.com/b-faye/lightmdetr/blob/HEAD/models/transformer_plus.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"751cd5726698116b","mcp_get_code":{"code_sha256":"751cd5726698116b"}},{"arxiv_id":"2104.11896","paper":"/paper/m3detr-multi-representation-multi-scale","title":"M3DeTR: Multi-representation, Multi-scale, Mutual-relation 3D Object Detection with Transformers","date":"2021-04-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rayguan97/M3DeTR","path":"pcdet/models/backbones_2d/transformer.py","file_url":"https://github.com/rayguan97/M3DeTR/blob/HEAD/pcdet/models/backbones_2d/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"87f4c841962a967b","mcp_get_code":{"code_sha256":"87f4c841962a967b"}},{"arxiv_id":"2104.08541","paper":"/paper/transvg-end-to-end-visual-grounding-with","title":"TransVG: End-to-End Visual Grounding with Transformers","date":"2021-04-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nku-shengzheliu/Pytorch-TransVG","path":"models/transformer.py","file_url":"https://github.com/nku-shengzheliu/Pytorch-TransVG/blob/HEAD/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"87992961bac26fc7","mcp_get_code":{"code_sha256":"87992961bac26fc7"}},{"arxiv_id":"2104.04369","paper":"/paper/video-aided-unsupervised-grammar-induction","title":"Video-aided Unsupervised Grammar Induction","date":"2021-04-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Sy-Zhang/MMC-PCFG","path":"lib/model/vpcfg/transformer.py","file_url":"https://github.com/Sy-Zhang/MMC-PCFG/blob/HEAD/lib/model/vpcfg/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"302ec916d91b12a8","mcp_get_code":{"code_sha256":"302ec916d91b12a8"}},{"arxiv_id":"2104.00969","paper":"/paper/tuber-tube-transformer-for-action-detection","title":"TubeR: Tubelet Transformer for Video Action Detection","date":"2021-04-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"amazon-science/tubelet-transformer","path":"models/transformer/transformer.py","file_url":"https://github.com/amazon-science/tubelet-transformer/blob/HEAD/models/transformer/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7b94aa61b4d5ecac","mcp_get_code":{"code_sha256":"7b94aa61b4d5ecac"}},{"arxiv_id":"2103.12115","paper":"/paper/end-to-end-trainable-multi-instance-pose","title":"End-to-End Trainable Multi-Instance Pose Estimation with Transformers","date":"2021-03-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pranoyr/pose-estimation-with-transformers","path":"models/transformer.py","file_url":"https://github.com/pranoyr/pose-estimation-with-transformers/blob/HEAD/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"302ec916d91b12a8","mcp_get_code":{"code_sha256":"302ec916d91b12a8"}},{"arxiv_id":"2103.12115","paper":"/paper/end-to-end-trainable-multi-instance-pose","title":"End-to-End Trainable Multi-Instance Pose Estimation with Transformers","date":"2021-03-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"amathislab/poet","path":"models/transformer.py","file_url":"https://github.com/amathislab/poet/blob/HEAD/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"de542ddb0a808b05","mcp_get_code":{"code_sha256":"de542ddb0a808b05"}},{"arxiv_id":"2005.12872","paper":"/paper/end-to-end-object-detection-with-transformers","title":"End-to-End Object Detection with Transformers","date":"2020-05-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"facebookresearch/detr","path":"models/transformer.py","file_url":"https://github.com/facebookresearch/detr/blob/HEAD/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"302ec916d91b12a8","mcp_get_code":{"code_sha256":"302ec916d91b12a8"}},{"arxiv_id":"1910.06611","paper":"/paper/enhancing-the-transformer-with-explicit-1","title":"Enhancing the Transformer with Explicit Relational Encoding for Math Problem Solving","date":"2019-10-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ischlag/TP-Transformer","path":"models/transformer.py","file_url":"https://github.com/ischlag/TP-Transformer/blob/HEAD/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ce9444c4e1d8e4a6","mcp_get_code":{"code_sha256":"ce9444c4e1d8e4a6"}},{"arxiv_id":"aaai_28500","paper":null,"title":"arXiv:aaai_28500","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"AsuradaYuci/TF-CLIP","path":"model/make_model_clipreid.py","file_url":"https://github.com/AsuradaYuci/TF-CLIP/blob/HEAD/model/make_model_clipreid.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b14287d0108cb653","mcp_get_code":{"code_sha256":"b14287d0108cb653"}},{"arxiv_id":"Zheng_Towards_Learning_a_Generalist_Model_for_Embodied_Navigation_CVPR_2024_paper","paper":null,"title":"arXiv:Zheng_Towards_Learning_a_Generalist_Model_for_Embodied_Navigation_CVPR_2024_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"LaVi-Lab/NaviLLM","path":"models/detr_transformer.py","file_url":"https://github.com/LaVi-Lab/NaviLLM/blob/HEAD/models/detr_transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"df2919fd1dad1d43","mcp_get_code":{"code_sha256":"df2919fd1dad1d43"}},{"arxiv_id":"Ye_Hierarchical_Modular_Network_for_Video_Captioning_CVPR_2022_paper","paper":null,"title":"arXiv:Ye_Hierarchical_Modular_Network_for_Video_Captioning_CVPR_2022_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"MarcusNerva/HMN","path":"models/encoders/transformer.py","file_url":"https://github.com/MarcusNerva/HMN/blob/HEAD/models/encoders/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3adf1f93bdafc63e","mcp_get_code":{"code_sha256":"3adf1f93bdafc63e"}},{"arxiv_id":"Tang_Progressive_Attention_on_Multi-Level_Dense_Difference_Maps_for_Generic_Event_CVPR_2022_paper","paper":null,"title":"arXiv:Tang_Progressive_Attention_on_Multi-Level_Dense_Difference_Maps_for_Generic_Event_CVPR_2022_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"MCG-NJU/DDM","path":"DDM-Net/modeling/transformer.py","file_url":"https://github.com/MCG-NJU/DDM/blob/HEAD/DDM-Net/modeling/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4341942277ed876c","mcp_get_code":{"code_sha256":"4341942277ed876c"}},{"arxiv_id":"Dang_FASTer_Focal_token_Acquiring-and-Scaling_Transformer_for_Long-term_3D_Objection_Detection_CVPR_2025_paper","paper":null,"title":"arXiv:Dang_FASTer_Focal_token_Acquiring-and-Scaling_Transformer_for_Long-term_3D_Objection_Detection_CVPR_2025_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"MSunDYY/FASTer","path":"pcdet/models/model_utils/faster_utils.py","file_url":"https://github.com/MSunDYY/FASTer/blob/HEAD/pcdet/models/model_utils/faster_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"5939165c82dcb087","mcp_get_code":{"code_sha256":"5939165c82dcb087"}},{"arxiv_id":"Chen_Recurrent_Glimpse-Based_Decoder_for_Detection_With_Transformer_CVPR_2022_paper","paper":null,"title":"arXiv:Chen_Recurrent_Glimpse-Based_Decoder_for_Detection_With_Transformer_CVPR_2022_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"zhechen/Deformable-DETR-REGO","path":"models/transformer.py","file_url":"https://github.com/zhechen/Deformable-DETR-REGO/blob/HEAD/models/transformer.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":false,"code_sha256_prefix":"e4cf3178d4a1e4c0","mcp_get_code":{"code_sha256":"e4cf3178d4a1e4c0"}}]}