{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/count-words","entry":"count_words","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":28,"n_papers_ran":17,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":21,"n_samples_ran":11,"n_samples_fingerprinted":11,"n_places":28,"n_places_pointer_only":11,"by_status":{"ran_honours":3,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":8,"unverified":10},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2608.25478","paper":"/paper/arxiv-2608-25478","title":"VietAIDetector: An Open-Source Zero-Shot Detector for Vietnamese AI-Generated Text","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"trieuntu/VietAIDetector","path":"preprocessing/text_utils.py","file_url":"https://github.com/trieuntu/VietAIDetector/blob/HEAD/preprocessing/text_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"510c69b7c1c15674","mcp_get_code":{"code_sha256":"510c69b7c1c15674"}},{"arxiv_id":"2606.20212","paper":"/paper/arxiv-2606-20212","title":"CzechDocs: A Multiway Parallel Dataset of Formatted Documents for Minority Languages in Czechia","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"cepin19/CzechDocs","path":"count_curated_stats.py","file_url":"https://github.com/cepin19/CzechDocs/blob/HEAD/count_curated_stats.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a379d71b3bd8ea58","mcp_get_code":{"code_sha256":"a379d71b3bd8ea58"}},{"arxiv_id":"2604.22215","paper":"/paper/arxiv-2604-22215","title":"Verbal Confidence Saturation in 3-9B Open-Weight Instruction-Tuned LLMs: A Pre-Registered Psychometric Validity Screen","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"synthiumjp/koriat","path":"build_cues.py","file_url":"https://github.com/synthiumjp/koriat/blob/HEAD/build_cues.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3ede3ad4d2b8d09f","mcp_get_code":{"code_sha256":"3ede3ad4d2b8d09f"}},{"arxiv_id":"2604.16132","paper":"/paper/arxiv-2604-16132","title":"Can LLMs Understand the Impact of Trauma? Costs and Benefits of LLMs Coding the Interviews of Firearm Violence Survivors","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"jhzsquared/AIvsHumanCoding","path":"count_data.py","file_url":"https://github.com/jhzsquared/AIvsHumanCoding/blob/HEAD/count_data.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"9a0e57cbbb4a12e8","mcp_get_code":{"code_sha256":"9a0e57cbbb4a12e8"}},{"arxiv_id":"2602.12639","paper":"/paper/arxiv-2602-12639","title":"CLASE: A Hybrid Method for Chinese Legalese Stylistic Evaluation","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"rexera/CLASE","path":"linguistic_features/feature_extractor.py","file_url":"https://github.com/rexera/CLASE/blob/HEAD/linguistic_features/feature_extractor.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"9c81670a59d88906","mcp_get_code":{"code_sha256":"9c81670a59d88906"}},{"arxiv_id":"2510.00307","paper":"/paper/arxiv-2510-00307","title":"BiasBusters: Uncovering and Mitigating Tool Selection Bias in Large Language Models","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"thierry123454/tool-selection-bias","path":"5_bias_investigation/continued_pretraining/generate_data_gemini.py","file_url":"https://github.com/thierry123454/tool-selection-bias/blob/HEAD/5_bias_investigation/continued_pretraining/generate_data_gemini.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2498e76b5f8a7bd9","mcp_get_code":{"code_sha256":"2498e76b5f8a7bd9"}},{"arxiv_id":"2508.14723","paper":"/paper/arxiv-2508-14723","title":"Transplant Then Regenerate: A New Paradigm for Text Data Augmentation","date":null,"month_inferred_from_arxiv_id":"2025-08","title_source":"syntology","repo":"1024er/cbert_aug","path":"utils.py","file_url":"https://github.com/1024er/cbert_aug/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ee80d8c65de13c06","mcp_get_code":{"code_sha256":"ee80d8c65de13c06"}},{"arxiv_id":"2506.18841","paper":"/paper/longwriter-zero-mastering-ultra-long-text","title":"LongWriter-Zero: Mastering Ultra-Long Text Generation via Reinforcement Learning","date":"2025-06-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"thudm/longwriter","path":"evaluation/pred.py","file_url":"https://github.com/thudm/longwriter/blob/HEAD/evaluation/pred.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2bb5c6316f656bec","mcp_get_code":{"code_sha256":"2bb5c6316f656bec"}},{"arxiv_id":"2411.14405","paper":"/paper/marco-o1-towards-open-reasoning-models-for","title":"Marco-o1: Towards Open Reasoning Models for Open-Ended Solutions","date":"2024-11-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aidc-ai/marco-o1","path":"src/v2/src/tree_search/evaluator/ifeval.py","file_url":"https://github.com/aidc-ai/marco-o1/blob/HEAD/src/v2/src/tree_search/evaluator/ifeval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"d23b2037805b8d2b","mcp_get_code":{"code_sha256":"d23b2037805b8d2b"}},{"arxiv_id":"2410.23933","paper":"/paper/language-models-can-self-lengthen-to-generate","title":"Language Models can Self-Lengthen to Generate Long Texts","date":"2024-10-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"QwenLM/Self-Lengthen","path":"eval/length_following_eval.py","file_url":"https://github.com/QwenLM/Self-Lengthen/blob/HEAD/eval/length_following_eval.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b3334d24e49ff2b5","mcp_get_code":{"code_sha256":"b3334d24e49ff2b5"}},{"arxiv_id":"2410.15263","paper":"/paper/back-to-school-translation-using-grammar","title":"Back to School: Translation Using Grammar Books","date":"2024-10-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jonathanhus/back-to-school","path":"utilities/count_data.py","file_url":"https://github.com/jonathanhus/back-to-school/blob/HEAD/utilities/count_data.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bdd4c45f6daf0dab","mcp_get_code":{"code_sha256":"bdd4c45f6daf0dab"}},{"arxiv_id":"2410.12788","paper":"/paper/meta-chunking-learning-efficient-text","title":"Meta-Chunking: Learning Text Segmentation and Semantic Completion via Logical Perception","date":"2024-10-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"IAAR-Shanghai/Meta-Chunking","path":"meta_chunking/LongBench/LumberChunker.py","file_url":"https://github.com/IAAR-Shanghai/Meta-Chunking/blob/HEAD/meta_chunking/LongBench/LumberChunker.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a4d6434de692eabe","mcp_get_code":{"code_sha256":"a4d6434de692eabe"}},{"arxiv_id":"2410.09584","paper":"/paper/toward-general-instruction-following","title":"Toward General Instruction-Following Alignment for Retrieval-Augmented Generation","date":"2024-10-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"dongguanting/FollowRAG","path":"FollowRAG/utils/instruction_following_eval/instructions_util.py","file_url":"https://github.com/dongguanting/FollowRAG/blob/HEAD/FollowRAG/utils/instruction_following_eval/instructions_util.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cdcc85ca09b00f7e","mcp_get_code":{"code_sha256":"cdcc85ca09b00f7e"}},{"arxiv_id":"2410.08044","paper":"/paper/the-rise-of-ai-generated-content-in-wikipedia","title":"The Rise of AI-Generated Content in Wikipedia","date":"2024-10-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"brooksca3/wiki_collection","path":"misc/get_hyperlinks.py","file_url":"https://github.com/brooksca3/wiki_collection/blob/HEAD/misc/get_hyperlinks.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"95141547985bc194","mcp_get_code":{"code_sha256":"95141547985bc194"}},{"arxiv_id":"2408.01933","paper":"/paper/2408-01933","title":"DiReCT: Diagnostic Reasoning for Clinical Notes via Large Language Models","date":"2024-08-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"wbw520/DiReCT","path":"statistics.py","file_url":"https://github.com/wbw520/DiReCT/blob/HEAD/statistics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8c024efc9f977616","mcp_get_code":{"code_sha256":"8c024efc9f977616"}},{"arxiv_id":"2407.01527","paper":"/paper/kv-cache-compression-but-what-must-we-give-in","title":"KV Cache Compression, But What Must We Give in Return? A Comprehensive Benchmark of Long Context Capable Approaches","date":"2024-07-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"henryzhongsc/longctx_bench","path":"eval/passkey_utils/passkey_utils.py","file_url":"https://github.com/henryzhongsc/longctx_bench/blob/HEAD/eval/passkey_utils/passkey_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2207706bbe348d2c","mcp_get_code":{"code_sha256":"2207706bbe348d2c"}},{"arxiv_id":"2406.17526","paper":"/paper/lumberchunker-long-form-narrative-document","title":"LumberChunker: Long-Form Narrative Document Segmentation","date":"2024-06-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"joaodsmarques/lumberchunker","path":"Code/LumberChunker-Segmentation.py","file_url":"https://github.com/joaodsmarques/lumberchunker/blob/HEAD/Code/LumberChunker-Segmentation.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"70c76dd00047fdbf","mcp_get_code":{"code_sha256":"70c76dd00047fdbf"}},{"arxiv_id":"2406.17385","paper":"/paper/native-design-bias-studying-the-impact-of","title":"Native Design Bias: Studying the Impact of English Nativeness on Language Model Performance","date":"2024-06-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"manon-reusens/native_en_bias","path":"dataset_statistics.py","file_url":"https://github.com/manon-reusens/native_en_bias/blob/HEAD/dataset_statistics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0b33a7e746e505d2","mcp_get_code":{"code_sha256":"0b33a7e746e505d2"}},{"arxiv_id":"2406.12809","paper":"/paper/can-large-language-models-always-solve-easy","title":"Can Large Language Models Always Solve Easy Problems if They Can Solve Harder Ones?","date":"2024-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"QwenLM/ConsisEval","path":"instruction_following_check/instructions_util.py","file_url":"https://github.com/QwenLM/ConsisEval/blob/HEAD/instruction_following_check/instructions_util.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cdcc85ca09b00f7e","mcp_get_code":{"code_sha256":"cdcc85ca09b00f7e"}},{"arxiv_id":"2404.15846","paper":"/paper/from-complex-to-simple-enhancing-multi","title":"From Complex to Simple: Enhancing Multi-Constraint Complex Instruction Following Ability of Large Language Models","date":"2024-04-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"meowpass/followcomplexinstruction","path":"get_data/instructions_util.py","file_url":"https://github.com/meowpass/followcomplexinstruction/blob/HEAD/get_data/instructions_util.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cdcc85ca09b00f7e","mcp_get_code":{"code_sha256":"cdcc85ca09b00f7e"}},{"arxiv_id":"2401.16745","paper":"/paper/mt-eval-a-multi-turn-capabilities-evaluation","title":"MT-Eval: A Multi-Turn Capabilities Evaluation Benchmark for Large Language Models","date":"2024-01-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kwanwaichung/mt-eval","path":"utils/global_inst.py","file_url":"https://github.com/kwanwaichung/mt-eval/blob/HEAD/utils/global_inst.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cdcc85ca09b00f7e","mcp_get_code":{"code_sha256":"cdcc85ca09b00f7e"}},{"arxiv_id":"2311.07911","paper":"/paper/instruction-following-evaluation-for-large","title":"Instruction-Following Evaluation for Large Language Models","date":"2023-11-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"josejg/instruction_following_eval","path":"instruction_following_eval/instructions_util.py","file_url":"https://github.com/josejg/instruction_following_eval/blob/HEAD/instruction_following_eval/instructions_util.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"cdcc85ca09b00f7e","mcp_get_code":{"code_sha256":"cdcc85ca09b00f7e"}},{"arxiv_id":"2309.00071","paper":"/paper/yarn-efficient-context-window-extension-of","title":"YaRN: Efficient Context Window Extension of Large Language Models","date":"2023-08-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"qwenlm/qwq","path":"eval/eval/ifeval_utils/instructions_util.py","file_url":"https://github.com/qwenlm/qwq/blob/HEAD/eval/eval/ifeval_utils/instructions_util.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"cdcc85ca09b00f7e","mcp_get_code":{"code_sha256":"cdcc85ca09b00f7e"}},{"arxiv_id":"2008.09094","paper":"/paper/scruples-a-corpus-of-community-ethical","title":"Scruples: A Corpus of Community Ethical Judgments on 32,000 Real-Life Anecdotes","date":"2020-08-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"allenai/scruples","path":"src/scruples/utils.py","file_url":"https://github.com/allenai/scruples/blob/HEAD/src/scruples/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d6c73ac35e8781c2","mcp_get_code":{"code_sha256":"d6c73ac35e8781c2"}},{"arxiv_id":"2002.04784","paper":"/paper/graph-universal-adversarial-attacks-a-few-bad","title":"Graph Universal Adversarial Attacks: A Few Bad Actors Ruin Graph Learning Models","date":"2020-02-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"chisam0217/Graph-Universal-Attack","path":"deepwalk/evaluate_deepwalk.py","file_url":"https://github.com/chisam0217/Graph-Universal-Attack/blob/HEAD/deepwalk/evaluate_deepwalk.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b30d29f66472e711","mcp_get_code":{"code_sha256":"b30d29f66472e711"}},{"arxiv_id":"1908.04003","paper":"/paper/rwr-gae-random-walk-regularization-for-graph","title":"RWR-GAE: Random Walk Regularization for Graph Auto Encoders","date":"2019-08-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"MysteryVaibhav/DW-GAE","path":"deepWalk/walks.py","file_url":"https://github.com/MysteryVaibhav/DW-GAE/blob/HEAD/deepWalk/walks.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a236ee68c305670f","mcp_get_code":{"code_sha256":"a236ee68c305670f"}},{"arxiv_id":"1805.06201","paper":"/paper/contextual-augmentation-data-augmentation-by","title":"Contextual Augmentation: Data Augmentation by Words with Paradigmatic Relations","date":"2018-05-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"pfnet-research/contextual_augmentation","path":"utils.py","file_url":"https://github.com/pfnet-research/contextual_augmentation/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ee80d8c65de13c06","mcp_get_code":{"code_sha256":"ee80d8c65de13c06"}},{"arxiv_id":"2025.emnlp-main.1010","paper":null,"title":"arXiv:2025.emnlp-main.1010","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"joaopfonseca/SafeNudge","path":"ctg/instructions_util.py","file_url":"https://github.com/joaopfonseca/SafeNudge/blob/HEAD/ctg/instructions_util.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"cdcc85ca09b00f7e","mcp_get_code":{"code_sha256":"cdcc85ca09b00f7e"}}]}