{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/read-jsonl-file","entry":"read_jsonl_file","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":10,"n_papers_ran":9,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":9,"n_samples_ran":8,"n_samples_fingerprinted":0,"n_places":10,"n_places_pointer_only":4,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":3,"ran_fixture":0,"ran":5,"unverified":1},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2412.18424","paper":"/paper/longdocurl-a-comprehensive-multimodal-long","title":"LongDocURL: a Comprehensive Multimodal Long Document Benchmark Integrating Understanding, Reasoning, and Locating","date":"2024-12-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dengc2023/longdocurl","path":"eval/api_models/eval_api_models.py","file_url":"https://github.com/dengc2023/longdocurl/blob/HEAD/eval/api_models/eval_api_models.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4d014a96162fe5b6","mcp_get_code":{"code_sha256":"4d014a96162fe5b6"}},{"arxiv_id":"2410.09870","paper":"/paper/chroknowledge-unveiling-chronological","title":"ChroKnowledge: Unveiling Chronological Knowledge of Language Models in Multiple Domains","date":"2024-10-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dmis-lab/chroknowledge","path":"sources/utils.py","file_url":"https://github.com/dmis-lab/chroknowledge/blob/HEAD/sources/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"25962e04a83e64c6","mcp_get_code":{"code_sha256":"25962e04a83e64c6"}},{"arxiv_id":"2407.00132","paper":"/paper/shortcutsbench-a-large-scale-real-world","title":"ShortcutsBench: A Large-Scale Real-world Benchmark for API-based Agents","date":"2024-06-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"eachsheep/shortcutsbench","path":"data_for_agent_llm/reformat_output_2_correct_format.py","file_url":"https://github.com/eachsheep/shortcutsbench/blob/HEAD/data_for_agent_llm/reformat_output_2_correct_format.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ff424f210c9cb6c1","mcp_get_code":{"code_sha256":"ff424f210c9cb6c1"}},{"arxiv_id":"2404.06139","paper":"/paper/diffharmony-latent-diffusion-model-meets","title":"DiffHarmony: Latent Diffusion Model Meets Image Harmonization","date":"2024-04-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nicecv/diffharmony","path":"src/dataset/ihd_dataset.py","file_url":"https://github.com/nicecv/diffharmony/blob/HEAD/src/dataset/ihd_dataset.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9178a3aaaf1a46b7","mcp_get_code":{"code_sha256":"9178a3aaaf1a46b7"}},{"arxiv_id":"2404.03543","paper":"/paper/codeeditorbench-evaluating-code-editing","title":"CodeEditorBench: Evaluating Code Editing Capability of Large Language Models","date":"2024-04-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"CodeEditorBench/CodeEditorBench","path":"result_postprocess.py","file_url":"https://github.com/CodeEditorBench/CodeEditorBench/blob/HEAD/result_postprocess.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"fffc4ecc055a301a","mcp_get_code":{"code_sha256":"fffc4ecc055a301a"}},{"arxiv_id":"2402.15938","paper":"/paper/generalization-or-memorization-data","title":"Generalization or Memorization: Data Contamination and Trustworthy Evaluation for Large Language Models","date":"2024-02-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yihongdong/cdd-ted4llms","path":"TED.py","file_url":"https://github.com/yihongdong/cdd-ted4llms/blob/HEAD/TED.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"70f86ab774fff1d2","mcp_get_code":{"code_sha256":"70f86ab774fff1d2"}},{"arxiv_id":"2311.04931","paper":"/paper/gpt4all-an-ecosystem-of-open-source","title":"GPT4All: An Ecosystem of Open Source Compressed Language Models","date":"2023-11-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nomic-ai/gpt4all","path":"gpt4all-training/eval_self_instruct.py","file_url":"https://github.com/nomic-ai/gpt4all/blob/HEAD/gpt4all-training/eval_self_instruct.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"92f93769061b495d","mcp_get_code":{"code_sha256":"92f93769061b495d"}},{"arxiv_id":"2311.01487","paper":"/paper/what-makes-for-good-visual-instructions","title":"What Makes for Good Visual Instructions? Synthesizing Complex Visual Reasoning Instructions for Visual Instruction Tuning","date":"2023-11-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rucaibox/comvint","path":"utils/utils.py","file_url":"https://github.com/rucaibox/comvint/blob/HEAD/utils/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"dd66639a0b4e7f4f","mcp_get_code":{"code_sha256":"dd66639a0b4e7f4f"}},{"arxiv_id":"1902.10909","paper":"/paper/bert-for-joint-intent-classification-and-slot","title":"BERT for Joint Intent Classification and Slot Filling","date":"2019-02-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alibaba-damo-academy/spokennlp","path":"mmvts/src/evaluate.py","file_url":"https://github.com/alibaba-damo-academy/spokennlp/blob/HEAD/mmvts/src/evaluate.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"90bf04bed34edf57","mcp_get_code":{"code_sha256":"90bf04bed34edf57"}},{"arxiv_id":"Zhang_Critic-V_VLM_Critics_Help_Catch_VLM_Errors_in_Multimodal_Reasoning_CVPR_2025_paper","paper":null,"title":"arXiv:Zhang_Critic-V_VLM_Critics_Help_Catch_VLM_Errors_in_Multimodal_Reasoning_CVPR_2025_paper","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"kyrieLei/Critic-V","path":"data_utils/utils/format.py","file_url":"https://github.com/kyrieLei/Critic-V/blob/HEAD/data_utils/utils/format.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"25962e04a83e64c6","mcp_get_code":{"code_sha256":"25962e04a83e64c6"}}]}