{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/paper/benchmarking-vision-language-models-on","title":"Benchmarking Vision-Language Models on Optical Character Recognition in Dynamic Video Environments","arxiv_id":"2502.06445","date":"2025-02-10","proceeding":null,"authors":["Sankalp Nagaonkar","Augustya Sharma","Ashish Choithani","Ashutosh Trivedi"],"abstract":"This paper introduces an open-source benchmark for evaluating Vision-Language Models (VLMs) on Optical Character Recognition (OCR) tasks in dynamic video environments. We present a curated dataset containing 1,477 manually annotated frames spanning diverse domains, including code editors, news broadcasts, YouTube videos, and advertisements. Three state of the art VLMs - Claude-3, Gemini-1.5, and GPT-4o are benchmarked against traditional OCR systems such as EasyOCR and RapidOCR. Evaluation metrics include Word Error Rate (WER), Character Error Rate (CER), and Accuracy. Our results highlight the strengths and limitations of VLMs in video-based OCR tasks, demonstrating their potential to outperform conventional OCR models in many scenarios. However, challenges such as hallucinations, content security policies, and sensitivity to occluded or stylized text remain. The dataset and benchmarking framework are publicly available to foster further research.","url_abs":"https://arxiv.org/abs/2502.06445v1","url_pdf":"https://arxiv.org/pdf/2502.06445v1.pdf","source":{"archive":"pwc-archive (Hugging Face), CC BY-SA 4.0","snapshot":"2025-07-28","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","row_kind":"abstracts"},"code_links":[{"paper_slug":"benchmarking-vision-language-models-on","repo_url":"https://github.com/video-db/ocr-benchmark","is_official":1,"mentioned_in_paper":1,"mentioned_in_github":1,"framework":"none","reach":{"status":"ok","spdx":"MIT"}}],"tasks":[{"task_slug":"benchmarking","task_name":"Benchmarking"},{"task_slug":"optical-character-recognition","task_name":"Optical Character Recognition"},{"task_slug":"optical-character-recognition","task_name":"Optical Character Recognition (OCR)"}],"methods":[],"datasets_introduced":[{"slug":"videodb-s-ocr-benchmark-public-collection","name":"VideoDB's OCR Benchmark Public Collection","full_name":"VideoDB's OCR Benchmark Public Collection"}],"methods_introduced":[],"results":[{"leaderboard":"/sota/optical-character-recognition-ocr-on-videodb","task":"Optical Character Recognition (OCR)","dataset":"VideoDB's OCR Benchmark Public Collection","model":"GPT-4o","rank_in_archive_order":1,"of":5,"metrics":{"Average Accuracy":"76.22","Character Error Rate (CER)":"0.2378","Word Error Rate (WER)":"0.5117"},"uses_additional_data":false},{"leaderboard":"/sota/optical-character-recognition-ocr-on-videodb","task":"Optical Character Recognition (OCR)","dataset":"VideoDB's OCR Benchmark Public Collection","model":"Gemini-1.5 Pro","rank_in_archive_order":2,"of":5,"metrics":{"Average Accuracy":"76.13","Character Error Rate (CER)":"0.2387","Word Error Rate (WER)":"0.2385"},"uses_additional_data":false},{"leaderboard":"/sota/optical-character-recognition-ocr-on-videodb","task":"Optical Character Recognition (OCR)","dataset":"VideoDB's OCR Benchmark Public Collection","model":"Claude-3 Sonnet","rank_in_archive_order":3,"of":5,"metrics":{"Average Accuracy":"67.71","Character Error Rate (CER)":"0.3229","Word Error Rate (WER)":"0.4663"},"uses_additional_data":false},{"leaderboard":"/sota/optical-character-recognition-ocr-on-videodb","task":"Optical Character Recognition (OCR)","dataset":"VideoDB's OCR Benchmark Public Collection","model":"RapidOCR","rank_in_archive_order":4,"of":5,"metrics":{"Average Accuracy":"56.98","Character Error Rate (CER)":"0.7620","Word Error Rate (WER)":"0.4302"},"uses_additional_data":false},{"leaderboard":"/sota/optical-character-recognition-ocr-on-videodb","task":"Optical Character Recognition (OCR)","dataset":"VideoDB's OCR Benchmark Public Collection","model":"EasyOCR","rank_in_archive_order":5,"of":5,"metrics":{"Average Accuracy":"49.30","Character Error Rate (CER)":"0.5070","Word Error Rate (WER)":"0.8262"},"uses_additional_data":false}],"syntology":{"syntology_url":"https://syntology.ai/paper/2502.06445","atlas_url":"https://app.syntology.ai/?focus=2502.06445","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.06445"}},"developers":"https://syntology.ai/developers","read_at":"2026-09-25T09:33:49+00:00","read_at_is":"when the build read Syntology's graph, not when any sample ran","claim":"Per-sample execution status on synthesized fixtures; not a correctness claim about the paper. Samples come from repositories linked to the paper, official or community; repo_kind says which.","repos":[{"provenance":"external:paperswithcode_snapshot_2025-07-28","url":"https://github.com/video-db/ocr-benchmark","reach":{"status":"ok","spdx":"MIT"}}],"summary":{"ran":4,"unverified":1},"by_repo_kind":{"official":{"samples":5,"ran":4,"repositories":1}},"repo_kind_vocabulary":{"official":"The archive marks this repository official for the paper","named_in_paper":"The archive records that the paper mentions this repository; it is not marked official","listed":"In the archive's code links for this paper, not marked official and not recorded as mentioned in the paper","found_in_text":"Syntology found this repository in the paper's own text; whether it is the authors' implementation is not asserted","community":"Not in the archive's code links for this paper; a community repository Syntology harvested"},"n_pointer_only_for_licence":0,"samples":[{"code_sha256_prefix":"015f577fe7e3a945","entry":"create_directories","repo":"video-db/ocr-benchmark","repo_kind":"official","path":"utils.py","file_url":"https://github.com/video-db/ocr-benchmark/blob/HEAD/utils.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"015f577fe7e3a945"}},{"code_sha256_prefix":"5a715afca734373e","entry":"get_videos_from_directory","repo":"video-db/ocr-benchmark","repo_kind":"official","path":"ground_truth_preparation/create_videodb_collection.py","file_url":"https://github.com/video-db/ocr-benchmark/blob/HEAD/ground_truth_preparation/create_videodb_collection.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"5a715afca734373e"}},{"code_sha256_prefix":"1e8e72ab0fd3a12a","entry":"get_videos_from_json","repo":"video-db/ocr-benchmark","repo_kind":"official","path":"ground_truth_preparation/create_videodb_collection.py","file_url":"https://github.com/video-db/ocr-benchmark/blob/HEAD/ground_truth_preparation/create_videodb_collection.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"1e8e72ab0fd3a12a"}},{"code_sha256_prefix":"588aaa1e437f6e28","entry":"setup_logging","repo":"video-db/ocr-benchmark","repo_kind":"official","path":"utils.py","file_url":"https://github.com/video-db/ocr-benchmark/blob/HEAD/utils.py","link_basis":"harvester_set","language":"python","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"588aaa1e437f6e28"}},{"code_sha256_prefix":"455082ff52cb2cd1","entry":"get_user_input","repo":"video-db/ocr-benchmark","repo_kind":"official","path":"ground_truth_preparation/create_videodb_collection.py","file_url":"https://github.com/video-db/ocr-benchmark/blob/HEAD/ground_truth_preparation/create_videodb_collection.py","link_basis":"harvester_set","language":"python","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"mcp_get_code":{"code_sha256":"455082ff52cb2cd1"}}]},"arxiv_metadata":null,"syntology_extracted_results":null}