{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/zero-shot-generalization/papers/4","list_of":"/task/zero-shot-generalization","task":"Zero-shot Generalization","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":4,"pages_in_order":6,"rows_per_page":100,"rows":[301,400],"of":572,"counts":{"archive_papers_tagged":572,"with_a_code_link":301,"where_syntology_ran_a_sample":128,"not_listed_spam_title":0,"listed":572,"listed_where_code_ran":128,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":119,"every_run_a_failure_of_syntologys_instrument":9,"listed_with_a_run_with_no_instrument_failure":119,"listed_every_run_a_failure_of_syntologys_instrument":9,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/zero-shot-generalization","prev":"/task/zero-shot-generalization/papers/3","next":"/task/zero-shot-generalization/papers/5","papers":[{"url":"/paper/attention-based-natural-language-grounding-by","slug":"attention-based-natural-language-grounding-by","title":"Attention Based Natural Language Grounding by Navigating Virtual Environment","date":"2018-04-23","arxiv_id":"1804.08454","repositories_listed":1,"syntology":null},{"url":null,"slug":"samst-a-transformer-framework-based-on-sam","title":"SAMST: A Transformer framework based on SAM pseudo label filtering for remote sensing semi-supervised semantic segmentation","date":"2025-07-16","arxiv_id":"2507.11994","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-depth-foundation-model-recent-trends","title":"Towards Depth Foundation Model: Recent Trends in Vision-Based Depth Estimation","date":"2025-07-15","arxiv_id":"2507.11540","repositories_listed":0,"syntology":null},{"url":null,"slug":"posellm-enhancing-language-guided-human-pose","title":"PoseLLM: Enhancing Language-Guided Human Pose Estimation with MLP Alignment","date":"2025-07-12","arxiv_id":"2507.09139","repositories_listed":0,"syntology":null},{"url":null,"slug":"go-to-zero-towards-zero-shot-motion","title":"Go to Zero: Towards Zero-shot Motion Generation with Million-scale Data","date":"2025-07-09","arxiv_id":"2507.07095","repositories_listed":0,"syntology":null},{"url":null,"slug":"video-event-reasoning-and-prediction-by","title":"Video Event Reasoning and Prediction by Fusing World Knowledge from LLMs with Vision Foundation Models","date":"2025-07-08","arxiv_id":"2507.05822","repositories_listed":0,"syntology":null},{"url":null,"slug":"helping-clip-see-both-the-forest-and-the","title":"Helping CLIP See Both the Forest and the Trees: A Decomposition and Description Approach","date":"2025-07-04","arxiv_id":"2507.03458","repositories_listed":0,"syntology":null},{"url":null,"slug":"robustereo-robust-zero-shot-stereo-matching","title":"RobuSTereo: Robust Zero-Shot Stereo Matching under Adverse Weather","date":"2025-07-02","arxiv_id":"2507.01653","repositories_listed":0,"syntology":null},{"url":null,"slug":"vislanding-monocular-3d-perception-for-uav","title":"VisLanding: Monocular 3D Perception for UAV Safe Landing via Depth-Normal Synergy","date":"2025-06-17","arxiv_id":"2506.14525","repositories_listed":0,"syntology":null},{"url":null,"slug":"leverb-humanoid-whole-body-control-with","title":"LeVERB: Humanoid Whole-Body Control with Latent Vision-Language Instruction","date":"2025-06-16","arxiv_id":"2506.13751","repositories_listed":0,"syntology":null},{"url":null,"slug":"deal-disentangling-transformer-head","title":"DEAL: Disentangling Transformer Head Activations for LLM Steering","date":"2025-06-10","arxiv_id":"2506.08359","repositories_listed":0,"syntology":null},{"url":null,"slug":"cxr-lt-2024-a-miccai-challenge-on-long-tailed","title":"CXR-LT 2024: A MICCAI challenge on long-tailed, multi-label, and zero-shot disease classification from chest X-ray","date":"2025-06-09","arxiv_id":"2506.07984","repositories_listed":0,"syntology":null},{"url":null,"slug":"deep-equivariant-multi-agent-control-barrier","title":"Deep Equivariant Multi-Agent Control Barrier Functions","date":"2025-06-09","arxiv_id":"2506.07755","repositories_listed":0,"syntology":null},{"url":null,"slug":"zerovo-visual-odometry-with-minimal-1","title":"ZeroVO: Visual Odometry with Minimal Assumptions","date":"2025-06-09","arxiv_id":"2506.08005","repositories_listed":0,"syntology":null},{"url":null,"slug":"latent-diffusion-model-based-denoising","title":"Latent Diffusion Model Based Denoising Receiver for 6G Semantic Communication: From Stochastic Differential Theory to Application","date":"2025-06-06","arxiv_id":"2506.05710","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-synthetic-stereo-datasets-using-3d","title":"Generating Synthetic Stereo Datasets using 3D Gaussian Splatting and Expert Knowledge Transfer","date":"2025-06-05","arxiv_id":"2506.04908","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-vision-language-garment-models-for","title":"Towards Vision-Language-Garment Models For Web Knowledge Garment Understanding and Generation","date":"2025-06-05","arxiv_id":"2506.05210","repositories_listed":0,"syntology":null},{"url":null,"slug":"language-guided-multi-agent-learning-in","title":"Language-Guided Multi-Agent Learning in Simulations: A Unified Framework and Evaluation","date":"2025-06-01","arxiv_id":"2506.04251","repositories_listed":0,"syntology":null},{"url":null,"slug":"vitapes-visuotactile-position-encodings-for","title":"ViTaPEs: Visuotactile Position Encodings for Cross-Modal Alignment in Multimodal Transformers","date":"2025-05-26","arxiv_id":"2505.20032","repositories_listed":0,"syntology":null},{"url":null,"slug":"whistress-enriching-transcriptions-with","title":"WHISTRESS: Enriching Transcriptions with Sentence Stress Detection","date":"2025-05-25","arxiv_id":"2505.19103","repositories_listed":0,"syntology":null},{"url":null,"slug":"anchored-diffusion-language-model","title":"Anchored Diffusion Language Model","date":"2025-05-24","arxiv_id":"2505.18456","repositories_listed":0,"syntology":null},{"url":null,"slug":"g1-teaching-llms-to-reason-on-graphs-with","title":"G1: Teaching LLMs to Reason on Graphs with Reinforcement Learning","date":"2025-05-24","arxiv_id":"2505.18499","repositories_listed":0,"syntology":null},{"url":null,"slug":"como-learning-continuous-latent-motion-from","title":"CoMo: Learning Continuous Latent Motion from Internet Videos for Scalable Robot Learning","date":"2025-05-22","arxiv_id":"2505.17006","repositories_listed":0,"syntology":null},{"url":null,"slug":"easyinsert-a-data-efficient-and-generalizable","title":"EasyInsert: A Data-Efficient and Generalizable Insertion Policy","date":"2025-05-22","arxiv_id":"2505.16187","repositories_listed":0,"syntology":null},{"url":null,"slug":"anybody-a-benchmark-suite-for-cross","title":"AnyBody: A Benchmark Suite for Cross-Embodiment Manipulation","date":"2025-05-21","arxiv_id":"2505.14986","repositories_listed":0,"syntology":null},{"url":null,"slug":"endovla-dual-phase-vision-language-action","title":"EndoVLA: Dual-Phase Vision-Language-Action Model for Autonomous Tracking in Endoscopy","date":"2025-05-21","arxiv_id":"2505.15206","repositories_listed":0,"syntology":null},{"url":null,"slug":"gen2seg-generative-models-enable","title":"gen2seg: Generative Models Enable Generalizable Instance Segmentation","date":"2025-05-21","arxiv_id":"2505.15263","repositories_listed":0,"syntology":null},{"url":null,"slug":"orqa-a-benchmark-and-foundation-model-for","title":"ORQA: A Benchmark and Foundation Model for Holistic Operating Room Modeling","date":"2025-05-19","arxiv_id":"2505.12890","repositories_listed":0,"syntology":null},{"url":null,"slug":"aop-sam-automation-of-prompts-for-efficient","title":"AoP-SAM: Automation of Prompts for Efficient Segmentation","date":"2025-05-17","arxiv_id":"2505.11980","repositories_listed":0,"syntology":null},{"url":null,"slug":"depth-anything-with-any-prior","title":"Depth Anything with Any Prior","date":"2025-05-15","arxiv_id":"2505.10565","repositories_listed":0,"syntology":null},{"url":null,"slug":"nvspolicy-adaptive-novel-view-synthesis-for","title":"NVSPolicy: Adaptive Novel-View Synthesis for Generalizable Language-Conditioned Policy Learning","date":"2025-05-15","arxiv_id":"2505.10359","repositories_listed":0,"syntology":null},{"url":null,"slug":"denoising-and-alignment-rethinking-domain","title":"Denoising and Alignment: Rethinking Domain Generalization for Multimodal Face Anti-Spoofing","date":"2025-05-14","arxiv_id":"2505.09484","repositories_listed":0,"syntology":null},{"url":null,"slug":"visual-image-reconstruction-from-brain","title":"Visual Image Reconstruction from Brain Activity via Latent Representation","date":"2025-05-13","arxiv_id":"2505.08429","repositories_listed":0,"syntology":null},{"url":null,"slug":"towards-artificial-general-or-personalized","title":"Towards Artificial General or Personalized Intelligence? A Survey on Foundation Models for Personalized Federated Intelligence","date":"2025-05-11","arxiv_id":"2505.06907","repositories_listed":0,"syntology":null},{"url":null,"slug":"pro2sam-mask-prompt-to-sam-with-grid-points","title":"Pro2SAM: Mask Prompt to SAM with Grid Points for Weakly Supervised Object Localization","date":"2025-05-08","arxiv_id":"2505.04905","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-review-of-3d-object-detection-with-vision","title":"A Review of 3D Object Detection with Vision-Language Models","date":"2025-04-25","arxiv_id":"2504.18738","repositories_listed":0,"syntology":null},{"url":null,"slug":"text-to-decision-agent-learning-generalist","title":"Text-to-Decision Agent: Learning Generalist Policies from Natural Language Supervision","date":"2025-04-21","arxiv_id":"2504.15046","repositories_listed":0,"syntology":null},{"url":null,"slug":"evolutionary-prompt-optimization-discovers","title":"Evolutionary Prompt Optimization Discovers Emergent Multimodal Reasoning Strategies in Vision-Language Models","date":"2025-03-30","arxiv_id":"2503.23503","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-domain-generalization-of","title":"Zero-shot Domain Generalization of Foundational Models for 3D Medical Image Segmentation: An Experimental Study","date":"2025-03-28","arxiv_id":"2503.22862","repositories_listed":0,"syntology":null},{"url":null,"slug":"thinking-agents-for-zero-shot-generalization","title":"Thinking agents for zero-shot generalization to qualitatively novel tasks","date":"2025-03-25","arxiv_id":"2503.19815","repositories_listed":0,"syntology":null},{"url":null,"slug":"unpaired-object-level-sar-to-optical-image","title":"Unpaired Object-Level SAR-to-Optical Image Translation for Aircraft with Keypoints-Guided Diffusion Models","date":"2025-03-25","arxiv_id":"2503.19798","repositories_listed":0,"syntology":null},{"url":null,"slug":"aether-geometric-aware-unified-world-modeling","title":"Aether: Geometric-Aware Unified World Modeling","date":"2025-03-24","arxiv_id":"2503.18945","repositories_listed":0,"syntology":null},{"url":null,"slug":"enhancing-zero-shot-image-recognition-in","title":"Enhancing Zero-Shot Image Recognition in Vision-Language Models through Human-like Concept Guidance","date":"2025-03-20","arxiv_id":"2503.15886","repositories_listed":0,"syntology":null},{"url":"/paper/jasmine-harnessing-diffusion-prior-for-self","slug":"jasmine-harnessing-diffusion-prior-for-self","title":"Jasmine: Harnessing Diffusion Prior for Self-supervised Depth Estimation","date":"2025-03-20","arxiv_id":"2503.15905","repositories_listed":0,"syntology":null},{"url":null,"slug":"genm-3-generative-pretrained-multi-path","title":"GenM$^3$: Generative Pretrained Multi-path Motion Model for Text Conditional Human Motion Generation","date":"2025-03-19","arxiv_id":"2503.14919","repositories_listed":0,"syntology":null},{"url":null,"slug":"good-actions-succeed-bad-actions-generalize-a","title":"Good Actions Succeed, Bad Actions Generalize: A Case Study on Why RL Generalizes Better","date":"2025-03-19","arxiv_id":"2503.15693","repositories_listed":0,"syntology":null},{"url":null,"slug":"foundation-feature-driven-online-end-effector","title":"Foundation Feature-Driven Online End-Effector Pose Estimation: A Marker-Free and Learning-Free Approach","date":"2025-03-18","arxiv_id":"2503.14051","repositories_listed":0,"syntology":null},{"url":null,"slug":"compound-expression-recognition-via-large","title":"Compound Expression Recognition via Large Vision-Language Models","date":"2025-03-14","arxiv_id":"2503.11241","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-recipe-for-improving-remote-sensing-vlm","title":"A Recipe for Improving Remote Sensing VLM Zero Shot Generalization","date":"2025-03-10","arxiv_id":"2503.08722","repositories_listed":0,"syntology":null},{"url":null,"slug":"poseless-depth-free-vision-to-joint-control","title":"PoseLess: Depth-Free Vision-to-Joint Control via Direct Image Mapping with VLM","date":"2025-03-10","arxiv_id":"2503.07111","repositories_listed":0,"syntology":null},{"url":null,"slug":"otter-a-vision-language-action-model-with","title":"OTTER: A Vision-Language-Action Model with Text-Aware Visual Feature Extraction","date":"2025-03-05","arxiv_id":"2503.03734","repositories_listed":0,"syntology":null},{"url":null,"slug":"railgun-a-unified-convolutional-policy-for","title":"RAILGUN: A Unified Convolutional Policy for Multi-Agent Path Finding Across Different Environments and Tasks","date":"2025-03-04","arxiv_id":"2503.02992","repositories_listed":0,"syntology":null},{"url":null,"slug":"contrastive-learning-of-english-language-and","title":"Contrastive Learning of English Language and Crystal Graphs for Multimodal Representation of Materials Knowledge","date":"2025-02-23","arxiv_id":"2502.16451","repositories_listed":0,"syntology":null},{"url":null,"slug":"learning-from-reward-free-offline-data-a-case","title":"Learning from Reward-Free Offline Data: A Case for Planning with Latent Dynamics Models","date":"2025-02-20","arxiv_id":"2502.14819","repositories_listed":0,"syntology":null},{"url":null,"slug":"wrt-sam-foundation-model-driven-segmentation","title":"WRT-SAM: Foundation Model-Driven Segmentation for Generalized Weld Radiographic Testing","date":"2025-02-17","arxiv_id":"2502.11338","repositories_listed":0,"syntology":null},{"url":null,"slug":"salience-invariant-consistent-policy-learning","title":"Salience-Invariant Consistent Policy Learning for Generalization in Visual Reinforcement Learning","date":"2025-02-12","arxiv_id":"2502.08336","repositories_listed":0,"syntology":null},{"url":null,"slug":"mechanistic-understandings-of-representation","title":"Mechanistic Understandings of Representation Vulnerabilities and Engineering Robust Vision Transformers","date":"2025-02-07","arxiv_id":"2502.04679","repositories_listed":0,"syntology":null},{"url":null,"slug":"simsort-a-powerful-framework-for-spike","title":"SimSort: A Data-Driven Framework for Spike Sorting by Large-Scale Electrophysiology Simulation","date":"2025-02-05","arxiv_id":"2502.03198","repositories_listed":0,"syntology":null},{"url":null,"slug":"toward-task-generalization-via-memory","title":"Toward Task Generalization via Memory Augmentation in Meta-Reinforcement Learning","date":"2025-02-03","arxiv_id":"2502.01521","repositories_listed":0,"syntology":null},{"url":null,"slug":"a-zero-shot-generalization-framework-for-llm","title":"A Zero-Shot Generalization Framework for LLM-Driven Cross-Domain Sequential Recommendation","date":"2025-01-31","arxiv_id":"2501.19232","repositories_listed":0,"syntology":null},{"url":null,"slug":"flexicracknet-a-flexible-pipeline-for","title":"FlexiCrackNet: A Flexible Pipeline for Enhanced Crack Segmentation with General Features Transfered from SAM","date":"2025-01-31","arxiv_id":"2501.18855","repositories_listed":0,"syntology":null},{"url":null,"slug":"test-time-loss-landscape-adaptation-for-zero","title":"Test-time Loss Landscape Adaptation for Zero-Shot Generalization in Vision-Language Models","date":"2025-01-31","arxiv_id":"2501.18864","repositories_listed":0,"syntology":null},{"url":null,"slug":"dynaprompt-dynamic-test-time-prompt-tuning","title":"DynaPrompt: Dynamic Test-Time Prompt Tuning","date":"2025-01-27","arxiv_id":"2501.16404","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-trajectory-planning-for-signal","title":"Zero-Shot Trajectory Planning for Signal Temporal Logic Tasks","date":"2025-01-23","arxiv_id":"2501.13457","repositories_listed":0,"syntology":null},{"url":null,"slug":"state-combinatorial-generalization-in","title":"State Combinatorial Generalization In Decision Making With Conditional Diffusion Models","date":"2025-01-22","arxiv_id":"2501.13241","repositories_listed":0,"syntology":null},{"url":null,"slug":"survey-on-monocular-metric-depth-estimation","title":"Survey on Monocular Metric Depth Estimation","date":"2025-01-21","arxiv_id":"2501.11841","repositories_listed":0,"syntology":null},{"url":null,"slug":"mifnet-learning-modality-invariant-features","title":"MIFNet: Learning Modality-Invariant Features for Generalizable Multimodal Image Matching","date":"2025-01-20","arxiv_id":"2501.11299","repositories_listed":0,"syntology":null},{"url":null,"slug":"chain-of-reasoning-towards-unified","title":"Chain-of-Reasoning: Towards Unified Mathematical Reasoning in Large Language Models via a Multi-Paradigm Perspective","date":"2025-01-19","arxiv_id":"2501.11110","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-monocular-scene-flow-estimation-in","title":"Zero-Shot Monocular Scene Flow Estimation in the Wild","date":"2025-01-17","arxiv_id":"2501.10357","repositories_listed":0,"syntology":null},{"url":null,"slug":"stereogen-high-quality-stereo-image","title":"StereoGen: High-quality Stereo Image Generation from a Single Image","date":"2025-01-15","arxiv_id":"2501.08654","repositories_listed":0,"syntology":null},{"url":null,"slug":"robotic-programmer-video-instructed-policy","title":"Robotic Programmer: Video Instructed Policy Code Generation for Robotic Manipulation","date":"2025-01-08","arxiv_id":"2501.04268","repositories_listed":0,"syntology":null},{"url":null,"slug":"spot-risks-before-speaking-unraveling-safety","title":"Spot Risks Before Speaking! Unraveling Safety Attention Heads in Large Vision-Language Models","date":"2025-01-03","arxiv_id":"2501.02029","repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-out-of-distribution-generalization-of-3","title":"On the Out-Of-Distribution Generalization of Large Multimodal Models","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"on-the-zero-shot-adversarial-robustness-of","title":"On the Zero-shot Adversarial Robustness of Vision-Language Models: A Truly Zero-shot and Training-free Approach","date":"2025-01-01","arxiv_id":null,"repositories_listed":0,"syntology":null},{"url":null,"slug":"predicate-invention-from-pixels-via","title":"From Pixels to Predicates: Learning Symbolic World Models via Pretrained Vision-Language Models","date":"2024-12-31","arxiv_id":"2501.00296","repositories_listed":0,"syntology":null},{"url":null,"slug":"ec-diffuser-multi-object-manipulation-via","title":"EC-Diffuser: Multi-Object Manipulation via Entity-Centric Behavior Generation","date":"2024-12-25","arxiv_id":"2412.18907","repositories_listed":0,"syntology":null},{"url":null,"slug":"multiple-consistency-guided-test-time","title":"Multiple Consistency-guided Test-Time Adaptation for Contrastive Audio-Language Models with Unlabeled Audio","date":"2024-12-23","arxiv_id":"2412.17306","repositories_listed":0,"syntology":null},{"url":"/paper/learning-cross-task-generalities-across","slug":"learning-cross-task-generalities-across","title":"Towards Graph Foundation Models: Learning Generalities Across Graphs via Task-Trees","date":"2024-12-21","arxiv_id":"2412.16441","repositories_listed":0,"syntology":{"n":3,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 3 unverified","sample_list":"/paper/learning-cross-task-generalities-across#ran","syntology_url":"https://syntology.ai/paper/2412.16441","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.16441"}},"official":null}},{"url":null,"slug":"efficient-fine-tuning-of-single-cell","title":"Efficient Fine-Tuning of Single-Cell Foundation Models Enables Zero-Shot Molecular Perturbation Prediction","date":"2024-12-18","arxiv_id":"2412.13478","repositories_listed":0,"syntology":null},{"url":null,"slug":"marigold-dc-zero-shot-monocular-depth","title":"Marigold-DC: Zero-Shot Monocular Depth Completion with Guided Diffusion","date":"2024-12-18","arxiv_id":"2412.13389","repositories_listed":0,"syntology":null},{"url":null,"slug":"zero-shot-generalization-for-blockage","title":"Zero-Shot Generalization for Blockage Localization in mmWave Communication","date":"2024-12-18","arxiv_id":"2412.17843","repositories_listed":0,"syntology":null},{"url":null,"slug":"easyref-omni-generalized-group-image","title":"EasyRef: Omni-Generalized Group Image Reference for Diffusion Models via Multimodal LLM","date":"2024-12-12","arxiv_id":"2412.09618","repositories_listed":0,"syntology":null},{"url":null,"slug":"wifo-wireless-foundation-model-for-channel","title":"WiFo: Wireless Foundation Model for Channel Prediction","date":"2024-12-12","arxiv_id":"2412.08908","repositories_listed":0,"syntology":null},{"url":null,"slug":"disentanglement-and-compositionality-of","title":"Disentanglement and Compositionality of Letter Identity and Letter Position in Variational Auto-Encoder Vision Models","date":"2024-12-11","arxiv_id":"2412.10446","repositories_listed":0,"syntology":null},{"url":null,"slug":"lightweight-method-for-interactive-3d-medical","title":"Lightweight Method for Interactive 3D Medical Image Segmentation with Multi-Round Result Fusion","date":"2024-12-11","arxiv_id":"2412.08315","repositories_listed":0,"syntology":null},{"url":null,"slug":"s-3-synonymous-semantic-space-for-improving","title":"$S^3$: Synonymous Semantic Space for Improving Zero-Shot Generalization of Vision-Language Models","date":"2024-12-06","arxiv_id":"2412.04925","repositories_listed":0,"syntology":null},{"url":null,"slug":"clip-ping-boosting-lightweight-vision","title":"CLIP-PING: Boosting Lightweight Vision-Language Models with Proximus Intrinsic Neighbors Guidance","date":"2024-12-05","arxiv_id":"2412.03871","repositories_listed":0,"syntology":null},{"url":null,"slug":"the-matrix-infinite-horizon-world-generation","title":"The Matrix: Infinite-Horizon World Generation with Real-Time Moving Control","date":"2024-12-04","arxiv_id":"2412.03568","repositories_listed":0,"syntology":null},{"url":null,"slug":"utsd-unified-time-series-diffusion-model","title":"UTSD: Unified Time Series Diffusion Model","date":"2024-12-04","arxiv_id":"2412.03068","repositories_listed":0,"syntology":null},{"url":null,"slug":"visatronic-a-multimodal-decoder-only-model","title":"Visatronic: A Multimodal Decoder-Only Model for Speech Synthesis","date":"2024-11-26","arxiv_id":"2411.17690","repositories_listed":0,"syntology":null},{"url":null,"slug":"generating-out-of-distribution-scenarios","title":"Generating Out-Of-Distribution Scenarios Using Language Models","date":"2024-11-25","arxiv_id":"2411.16554","repositories_listed":0,"syntology":null},{"url":null,"slug":"style-pro-style-guided-prompt-learning-for","title":"Style-Pro: Style-Guided Prompt Learning for Generalizable Vision-Language Models","date":"2024-11-25","arxiv_id":"2411.16018","repositories_listed":0,"syntology":null},{"url":null,"slug":"context-aware-multimodal-pretraining","title":"Context-Aware Multimodal Pretraining","date":"2024-11-22","arxiv_id":"2411.15099","repositories_listed":0,"syntology":null},{"url":null,"slug":"height-heterogeneous-interaction-graph","title":"HEIGHT: Heterogeneous Interaction Graph Transformer for Robot Navigation in Crowded and Constrained Environments","date":"2024-11-19","arxiv_id":"2411.12150","repositories_listed":0,"syntology":null},{"url":null,"slug":"scalable-autoregressive-monocular-depth","title":"Scalable Autoregressive Monocular Depth Estimation","date":"2024-11-18","arxiv_id":"2411.11361","repositories_listed":0,"syntology":null},{"url":null,"slug":"mono2stereo-monocular-knowledge-transfer-for","title":"Mono2Stereo: Monocular Knowledge Transfer for Enhanced Stereo Matching","date":"2024-11-14","arxiv_id":"2411.09151","repositories_listed":0,"syntology":null},{"url":null,"slug":"unihoi-learning-fast-dense-and-generalizable","title":"Self-Supervised Monocular 4D Scene Reconstruction for Egocentric Videos","date":"2024-11-14","arxiv_id":"2411.09145","repositories_listed":0,"syntology":null},{"url":null,"slug":"in-the-era-of-prompt-learning-with-vision","title":"In the Era of Prompt Learning with Vision-Language Models","date":"2024-11-07","arxiv_id":"2411.04892","repositories_listed":0,"syntology":null},{"url":null,"slug":"compositional-automata-embeddings-for-goal","title":"Compositional Automata Embeddings for Goal-Conditioned Reinforcement Learning","date":"2024-10-31","arxiv_id":"2411.00205","repositories_listed":0,"syntology":null},{"url":null,"slug":"judgerank-leveraging-large-language-models","title":"JudgeRank: Leveraging Large Language Models for Reasoning-Intensive Reranking","date":"2024-10-31","arxiv_id":"2411.00142","repositories_listed":0,"syntology":null}],"record_sha256":"15acc3cd594d647badcbfa510eb221bce9167ac040815c92fa27aeea12d4e248","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}