{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/clip-sla-parameter-efficient-clip-adaptation","title":"CLIP-SLA: Parameter-Efficient CLIP Adaptation for Continuous Sign Language Recognition","arxiv_id":"2504.01666","date":"2025-04-02","proceeding":null,"authors":["Sarah Alyami","Hamzah Luqman"],"abstract":"Continuous sign language recognition (CSLR) focuses on interpreting and transcribing sequences of sign language gestures in videos. In this work, we propose CLIP sign language adaptation (CLIP-SLA), a novel CSLR framework that leverages the powerful pre-trained visual encoder from the CLIP model to sign language tasks through parameter-efficient fine-tuning (PEFT). We introduce two variants, SLA-Adapter and SLA-LoRA, which integrate PEFT modules into the CLIP visual encoder, enabling fine-tuning with minimal trainable parameters. The effectiveness of the proposed frameworks is validated on four datasets: Phoenix2014, Phoenix2014-T, CSL-Daily, and Isharah-500, where both CLIP-SLA variants outperformed several SOTA models with fewer trainable parameters. Extensive ablation studies emphasize the effectiveness and flexibility of the proposed methods with different vision-language models for CSLR. These findings showcase the potential of adapting large-scale pre-trained models for scalable and efficient CSLR, which pave the way for future advancements in sign language understanding.","url_abs":"https://arxiv.org/abs/2504.01666v1","url_pdf":"https://arxiv.org/pdf/2504.01666v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"clip-sla-parameter-efficient-clip-adaptation","repo_url":"https://github.com/snalyami/CLIP-SLA","is_official":1,"mentioned_in_paper":0,"mentioned_in_github":0,"framework":"none","reach":null}],"tasks":[{"task_slug":"sign-language-recognition","task_name":"Sign Language Recognition"},{"task_slug":"parameter-efficient-fine-tuning","task_name":"parameter-efficient fine-tuning"}],"methods":[{"method_slug":"clip","method_name":"CLIP"}],"datasets_introduced":[],"methods_introduced":[],"results":[{"leaderboard":"/sota/sign-language-recognition-on-csl-daily","task":"Sign Language Recognition","dataset":"CSL-Daily","model":"SLA-LoRA","rank_in_archive_order":3,"of":14,"metrics":{"Word Error Rate (WER)":"25.8"},"uses_additional_data":false},{"leaderboard":"/sota/sign-language-recognition-on-rwth-phoenix","task":"Sign Language Recognition","dataset":"RWTH-PHOENIX-Weather 2014","model":"SLA-Adapter","rank_in_archive_order":4,"of":22,"metrics":{"Word Error Rate (WER)":"18.8"},"uses_additional_data":false},{"leaderboard":"/sota/sign-language-recognition-on-rwth-phoenix-1","task":"Sign Language Recognition","dataset":"RWTH-PHOENIX-Weather 2014 T","model":"SLA-LoRA","rank_in_archive_order":4,"of":15,"metrics":{"Word Error Rate (WER)":"19.4"},"uses_additional_data":false}],"syntology":{"atlas_url":null,"mcp":null,"developers":"https://syntology.ai/developers"},"arxiv_metadata":null,"syntology_extracted_results":null}