{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/generate-caption","entry":"generate_caption","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":21,"n_papers_ran":16,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":7,"n_samples_ran":3,"n_samples_fingerprinted":2,"n_places":21,"n_places_pointer_only":9,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":3,"unverified":4},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2606.22723","paper":"/paper/arxiv-2606-22723","title":"BLUEX v2: Benchmarking LLMs on Open-Ended Questions from Brazilian University Entrance Exams","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"TropicAI-Research/BLUEXv2","path":"dataset_pipeline/generate_captions.py","file_url":"https://github.com/TropicAI-Research/BLUEXv2/blob/HEAD/dataset_pipeline/generate_captions.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f8e511927033900d","mcp_get_code":{"code_sha256":"f8e511927033900d"}},{"arxiv_id":"2502.13363","paper":"/paper/pretrained-image-text-models-are-secretly","title":"Pretrained Image-Text Models are Secretly Video Captioners","date":"2025-02-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"chunhuizng/mllm-video-captioner","path":"app/caption.py","file_url":"https://github.com/chunhuizng/mllm-video-captioner/blob/HEAD/app/caption.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":false,"code_sha256_prefix":"56f02812d66a0d11","mcp_get_code":{"code_sha256":"56f02812d66a0d11"}},{"arxiv_id":"2411.19331","paper":"/paper/talking-to-dino-bridging-self-supervised","title":"Talking to DINO: Bridging Self-Supervised Vision Backbones with Language for Open-Vocabulary Segmentation","date":"2024-11-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lorebianchi98/Talk2DINO","path":"dino_extraction_v2.py","file_url":"https://github.com/lorebianchi98/Talk2DINO/blob/HEAD/dino_extraction_v2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e56697ee94f100ea","mcp_get_code":{"code_sha256":"e56697ee94f100ea"}},{"arxiv_id":"2407.14505","paper":"/paper/t2v-compbench-a-comprehensive-benchmark-for","title":"T2V-CompBench: A Comprehensive Benchmark for Compositional Text-to-video Generation","date":"2024-07-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"KaiyueSun98/T2V-CompBench","path":"Grounded-Segment-Anything/automatic_label_demo.py","file_url":"https://github.com/KaiyueSun98/T2V-CompBench/blob/HEAD/Grounded-Segment-Anything/automatic_label_demo.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d893c774b1f34088","mcp_get_code":{"code_sha256":"d893c774b1f34088"}},{"arxiv_id":"2407.05282","paper":"/paper/ultraedit-instruction-based-fine-grained","title":"UltraEdit: Instruction-based Fine-Grained Image Editing at Scale","date":"2024-07-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pkunlp-icler/ultraedit","path":"data_generation/Grounded-Segment-Anything/automatic_label_demo.py","file_url":"https://github.com/pkunlp-icler/ultraedit/blob/HEAD/data_generation/Grounded-Segment-Anything/automatic_label_demo.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d893c774b1f34088","mcp_get_code":{"code_sha256":"d893c774b1f34088"}},{"arxiv_id":"2407.00788","paper":"/paper/instantstyle-plus-style-transfer-with-content","title":"InstantStyle-Plus: Style Transfer with Content-Preserving in Text-to-Image Generation","date":"2024-06-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"instantx-research/instantstyle-plus","path":"infer_style.py","file_url":"https://github.com/instantx-research/instantstyle-plus/blob/HEAD/infer_style.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f47234275dee8d5c","mcp_get_code":{"code_sha256":"f47234275dee8d5c"}},{"arxiv_id":"2406.13807","paper":"/paper/alanavlm-a-multimodal-embodied-ai-foundation","title":"AlanaVLM: A Multimodal Embodied AI Foundation Model for Egocentric Video Understanding","date":"2024-06-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alanaai/evud","path":"hm3d/prepare_hm3d_dataset.py","file_url":"https://github.com/alanaai/evud/blob/HEAD/hm3d/prepare_hm3d_dataset.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d776259a0bbaf187","mcp_get_code":{"code_sha256":"d776259a0bbaf187"}},{"arxiv_id":"2405.13459","paper":"/paper/adapting-multi-modal-large-language-model-to","title":"Adapting Multi-modal Large Language Model to Concept Drift From Pre-training Onwards","date":"2024-05-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"XiaoyuYoung/ConceptDriftMLLMs","path":"app/caption.py","file_url":"https://github.com/XiaoyuYoung/ConceptDriftMLLMs/blob/HEAD/app/caption.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"56f02812d66a0d11","mcp_get_code":{"code_sha256":"56f02812d66a0d11"}},{"arxiv_id":"2404.05726","paper":"/paper/ma-lmm-memory-augmented-large-multimodal","title":"MA-LMM: Memory-Augmented Large Multimodal Model for Long-Term Video Understanding","date":"2024-04-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"boheumd/MA-LMM","path":"app/caption.py","file_url":"https://github.com/boheumd/MA-LMM/blob/HEAD/app/caption.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"56f02812d66a0d11","mcp_get_code":{"code_sha256":"56f02812d66a0d11"}},{"arxiv_id":"2404.00906","paper":"/paper/from-pixels-to-graphs-open-vocabulary-scene","title":"From Pixels to Graphs: Open-Vocabulary Scene Graph Generation with Vision-Language Models","date":"2024-04-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shtuplus/pix2grp_cvpr2024","path":"app/caption.py","file_url":"https://github.com/shtuplus/pix2grp_cvpr2024/blob/HEAD/app/caption.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"56f02812d66a0d11","mcp_get_code":{"code_sha256":"56f02812d66a0d11"}},{"arxiv_id":"2312.10300","paper":"/paper/shot2story20k-a-new-benchmark-for","title":"Shot2Story20K: A New Benchmark for Comprehensive Understanding of Multi-shot Videos","date":"2023-12-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bytedance/Shot2Story","path":"code/app/caption.py","file_url":"https://github.com/bytedance/Shot2Story/blob/HEAD/code/app/caption.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"56f02812d66a0d11","mcp_get_code":{"code_sha256":"56f02812d66a0d11"}},{"arxiv_id":"2312.07488","paper":"/paper/lmdrive-closed-loop-end-to-end-driving-with","title":"LMDrive: Closed-Loop End-to-End Driving with Large Language Models","date":"2023-12-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"opendilab/lmdrive","path":"LAVIS/app/caption.py","file_url":"https://github.com/opendilab/lmdrive/blob/HEAD/LAVIS/app/caption.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"56f02812d66a0d11","mcp_get_code":{"code_sha256":"56f02812d66a0d11"}},{"arxiv_id":"2312.04837","paper":"/paper/localized-symbolic-knowledge-distillation-for-1","title":"Localized Symbolic Knowledge Distillation for Visual Commonsense Models","date":"2023-12-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jamespark3922/lskd","path":"app/caption.py","file_url":"https://github.com/jamespark3922/lskd/blob/HEAD/app/caption.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"56f02812d66a0d11","mcp_get_code":{"code_sha256":"56f02812d66a0d11"}},{"arxiv_id":"2310.17419","paper":"/paper/antifakeprompt-prompt-tuned-vision-language","title":"AntifakePrompt: Prompt-Tuned Vision-Language Models are Fake Image Detectors","date":"2023-10-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nctu-eva-lab/antifakeprompt","path":"app/caption.py","file_url":"https://github.com/nctu-eva-lab/antifakeprompt/blob/HEAD/app/caption.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"56f02812d66a0d11","mcp_get_code":{"code_sha256":"56f02812d66a0d11"}},{"arxiv_id":"2307.07063","paper":"/paper/bootstrapping-vision-language-learning-with-1","title":"Bootstrapping Vision-Language Learning with Decoupled Language Pre-training","date":"2023-07-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yiren-jian/BLIText","path":"app/caption.py","file_url":"https://github.com/yiren-jian/BLIText/blob/HEAD/app/caption.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"56f02812d66a0d11","mcp_get_code":{"code_sha256":"56f02812d66a0d11"}},{"arxiv_id":"2305.06988","paper":"/paper/self-chained-image-language-model-for-video-1","title":"Self-Chained Image-Language Model for Video Localization and Question Answering","date":"2023-05-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yui010206/sevila","path":"app/caption.py","file_url":"https://github.com/yui010206/sevila/blob/HEAD/app/caption.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":false,"code_sha256_prefix":"56f02812d66a0d11","mcp_get_code":{"code_sha256":"56f02812d66a0d11"}},{"arxiv_id":"2305.01278","paper":"/paper/vpgtrans-transfer-visual-prompt-generator","title":"VPGTrans: Transfer Visual Prompt Generator across LLMs","date":"2023-05-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"VPGTrans/VPGTrans","path":"app/caption.py","file_url":"https://github.com/VPGTrans/VPGTrans/blob/HEAD/app/caption.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"56f02812d66a0d11","mcp_get_code":{"code_sha256":"56f02812d66a0d11"}},{"arxiv_id":"1606.07770","paper":"/paper/captioning-images-with-diverse-objects","title":"Captioning Images with Diverse Objects","date":"2016-06-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"willT97/Zero-shot-Image-Captioner","path":"noc_train.py","file_url":"https://github.com/willT97/Zero-shot-Image-Captioner/blob/HEAD/noc_train.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a075c8fcc5cc6b01","mcp_get_code":{"code_sha256":"a075c8fcc5cc6b01"}},{"arxiv_id":"2024.findings-emnlp.649","paper":null,"title":"arXiv:2024.findings-emnlp.649","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"alanaai/EVUD","path":"hm3d/prepare_hm3d_dataset.py","file_url":"https://github.com/alanaai/EVUD/blob/HEAD/hm3d/prepare_hm3d_dataset.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d776259a0bbaf187","mcp_get_code":{"code_sha256":"d776259a0bbaf187"}},{"arxiv_id":"2024.emnlp-main.88","paper":null,"title":"arXiv:2024.emnlp-main.88","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"WenjianDing/ReBo","path":"app/caption.py","file_url":"https://github.com/WenjianDing/ReBo/blob/HEAD/app/caption.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"56f02812d66a0d11","mcp_get_code":{"code_sha256":"56f02812d66a0d11"}},{"arxiv_id":"2024.emnlp-main.722","paper":null,"title":"arXiv:2024.emnlp-main.722","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"xyh97/UNICORN","path":"app/caption.py","file_url":"https://github.com/xyh97/UNICORN/blob/HEAD/app/caption.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":false,"code_sha256_prefix":"56f02812d66a0d11","mcp_get_code":{"code_sha256":"56f02812d66a0d11"}}]}