{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/tokenize-function","entry":"tokenize_function","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":19,"n_papers_ran":10,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":19,"n_samples_ran":10,"n_samples_fingerprinted":0,"n_places":21,"n_places_pointer_only":6,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":4,"ran_fixture":0,"ran":6,"unverified":9},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2601.19383","paper":"/paper/arxiv-2601-19383","title":"High-quality data augmentation for code comment classification","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"ThomBors/NLBSE2026","path":"src/utils.py","file_url":"https://github.com/ThomBors/NLBSE2026/blob/HEAD/src/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"96aa26edf0c56f70","mcp_get_code":{"code_sha256":"96aa26edf0c56f70"}},{"arxiv_id":"2502.20122","paper":"/paper/self-training-elicits-concise-reasoning-in","title":"Self-Training Elicits Concise Reasoning in Large Language Models","date":"2025-02-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tergelmunkhbat/concise-reasoning","path":"src/training_utils.py","file_url":"https://github.com/tergelmunkhbat/concise-reasoning/blob/HEAD/src/training_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0b6eb4d9aa2a6729","mcp_get_code":{"code_sha256":"0b6eb4d9aa2a6729"}},{"arxiv_id":"2409.10053","paper":"/paper/householder-pseudo-rotation-a-novel-approach","title":"Householder Pseudo-Rotation: A Novel Approach to Activation Editing in LLMs with Direction-Magnitude Perspective","date":"2024-09-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"VinAIResearch/HPR","path":"activation_editing/get_activations.py","file_url":"https://github.com/VinAIResearch/HPR/blob/HEAD/activation_editing/get_activations.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"b2a2c7262c9af473","mcp_get_code":{"code_sha256":"b2a2c7262c9af473"}},{"arxiv_id":"2408.05326","paper":"/paper/a-psychology-based-unified-dynamic-framework","title":"A Psychology-based Unified Dynamic Framework for Curriculum Learning","date":"2024-08-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nd-ball/cl-irt","path":"PUDF_GLUE/PUDF_glue_DebertaV3.py","file_url":"https://github.com/nd-ball/cl-irt/blob/HEAD/PUDF_GLUE/PUDF_glue_DebertaV3.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3dc66ac0e9e83724","mcp_get_code":{"code_sha256":"3dc66ac0e9e83724"}},{"arxiv_id":"2408.05326","paper":"/paper/a-psychology-based-unified-dynamic-framework","title":"A Psychology-based Unified Dynamic Framework for Curriculum Learning","date":"2024-08-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nd-ball/cl-irt","path":"gen_difficulty/GLUE_difficulty.py","file_url":"https://github.com/nd-ball/cl-irt/blob/HEAD/gen_difficulty/GLUE_difficulty.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c19241ab56039094","mcp_get_code":{"code_sha256":"c19241ab56039094"}},{"arxiv_id":"2408.04284","paper":"/paper/llm-detectaive-a-tool-for-fine-grained","title":"LLM-DetectAIve: a Tool for Fine-Grained Machine-Generated Text Detection","date":"2024-08-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mbzuai-nlp/llm-detectaive","path":"pipeline/model_pipeline.py","file_url":"https://github.com/mbzuai-nlp/llm-detectaive/blob/HEAD/pipeline/model_pipeline.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"dbc9d07c6c1f86f7","mcp_get_code":{"code_sha256":"dbc9d07c6c1f86f7"}},{"arxiv_id":"2408.00531","paper":"/paper/resi-a-comprehensive-benchmark-for","title":"ReSi: A Comprehensive Benchmark for Representational Similarity Measures","date":"2024-08-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mklabunde/resi","path":"nlp/bert_finetune.py","file_url":"https://github.com/mklabunde/resi/blob/HEAD/nlp/bert_finetune.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"CC-BY-4.0","inline_ok":false,"code_sha256_prefix":"aa6161b70c72b969","mcp_get_code":{"code_sha256":"aa6161b70c72b969"}},{"arxiv_id":"2407.11282","paper":"/paper/uncertainty-is-fragile-manipulating","title":"Uncertainty is Fragile: Manipulating Uncertainty in Large Language Models","date":"2024-07-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"qcznlp/uncertainty_attack","path":"fine_tuning.py","file_url":"https://github.com/qcznlp/uncertainty_attack/blob/HEAD/fine_tuning.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6b4ff6a70b9692b6","mcp_get_code":{"code_sha256":"6b4ff6a70b9692b6"}},{"arxiv_id":"2406.05797","paper":"/paper/3d-molt5-towards-unified-3d-molecule-text","title":"3D-MolT5: Leveraging Discrete Structural Information for Molecule-Text Modeling","date":"2024-06-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"QizhiPei/3D-MolT5","path":"3d_molt5/utils/copied_utils.py","file_url":"https://github.com/QizhiPei/3D-MolT5/blob/HEAD/3d_molt5/utils/copied_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8a721263c3f886fd","mcp_get_code":{"code_sha256":"8a721263c3f886fd"}},{"arxiv_id":"2401.06712","paper":"/paper/few-shot-detection-of-machine-generated-text","title":"Few-Shot Detection of Machine-Generated Text using Style Representations","date":"2024-01-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"llnl/luar","path":"fewshot_iclr2024/baseline_training/finetune_roberta.py","file_url":"https://github.com/llnl/luar/blob/HEAD/fewshot_iclr2024/baseline_training/finetune_roberta.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a3e19197e69c1a89","mcp_get_code":{"code_sha256":"a3e19197e69c1a89"}},{"arxiv_id":"2312.00960","paper":"/paper/the-cost-of-compression-investigating-the","title":"The Cost of Compression: Investigating the Impact of Compression on Parametric Knowledge in Language Models","date":"2023-12-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"namburisrinath/llmcompression","path":"bert_prune.py","file_url":"https://github.com/namburisrinath/llmcompression/blob/HEAD/bert_prune.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ca23fc05f2adc8a8","mcp_get_code":{"code_sha256":"ca23fc05f2adc8a8"}},{"arxiv_id":"2311.09476","paper":"/paper/ares-an-automated-evaluation-framework-for","title":"ARES: An Automated Evaluation Framework for Retrieval-Augmented Generation Systems","date":"2023-11-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"stanford-futuredata/ares","path":"ares/LLM_as_a_Judge_Adaptation/General_Binary_Classifier.py","file_url":"https://github.com/stanford-futuredata/ares/blob/HEAD/ares/LLM_as_a_Judge_Adaptation/General_Binary_Classifier.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"042e539c6d8bd3de","mcp_get_code":{"code_sha256":"042e539c6d8bd3de"}},{"arxiv_id":"2310.13671","paper":"/paper/let-s-synthesize-step-by-step-iterative","title":"Let's Synthesize Step by Step: Iterative Dataset Synthesis with Large Language Models by Extrapolating Errors from Small Models","date":"2023-10-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rickyskywalker/synthesis_step-by-step_official","path":"ModelTraining/main_IMDb.py","file_url":"https://github.com/rickyskywalker/synthesis_step-by-step_official/blob/HEAD/ModelTraining/main_IMDb.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"c01766547414e4b9","mcp_get_code":{"code_sha256":"c01766547414e4b9"}},{"arxiv_id":"2309.02373","paper":"/paper/nanot5-a-pytorch-framework-for-pre-training","title":"nanoT5: A PyTorch Framework for Pre-training and Fine-tuning T5-style Models with Limited Resources","date":"2023-09-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"piotrnawrot/nanot5","path":"nanoT5/utils/copied_utils.py","file_url":"https://github.com/piotrnawrot/nanot5/blob/HEAD/nanoT5/utils/copied_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"8a721263c3f886fd","mcp_get_code":{"code_sha256":"8a721263c3f886fd"}},{"arxiv_id":"2305.09955","paper":"/paper/cook-empowering-general-purpose-language","title":"Knowledge Card: Filling LLMs' Knowledge Gaps with Plug-in Specialized Language Models","date":"2023-05-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"BunsenFeng/Knowledge_Card","path":"card_training.py","file_url":"https://github.com/BunsenFeng/Knowledge_Card/blob/HEAD/card_training.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"add928626b7e8654","mcp_get_code":{"code_sha256":"add928626b7e8654"}},{"arxiv_id":"2210.07352","paper":"/paper/predicting-fine-tuning-performance-with","title":"Predicting Fine-Tuning Performance with Probing","date":"2022-10-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"spoclab-ca/performance_prediction","path":"scripts/glue_classify.py","file_url":"https://github.com/spoclab-ca/performance_prediction/blob/HEAD/scripts/glue_classify.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6a5472e4dd828178","mcp_get_code":{"code_sha256":"6a5472e4dd828178"}},{"arxiv_id":"2206.15476","paper":"/paper/anoshift-a-distribution-shift-benchmark-for","title":"AnoShift: A Distribution Shift Benchmark for Unsupervised Anomaly Detection","date":"2022-06-30","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bit-ml/anoshift","path":"language_models/data_utils.py","file_url":"https://github.com/bit-ml/anoshift/blob/HEAD/language_models/data_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"184f64ac20617dd0","mcp_get_code":{"code_sha256":"184f64ac20617dd0"}},{"arxiv_id":"2106.11257","paper":"/paper/secure-distributed-training-at-scale","title":"Secure Distributed Training at Scale","date":"2021-06-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yandex-research/btard","path":"albert/experiments/tokenize_wikitext103.py","file_url":"https://github.com/yandex-research/btard/blob/HEAD/albert/experiments/tokenize_wikitext103.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d2b88e48b87e9450","mcp_get_code":{"code_sha256":"d2b88e48b87e9450"}},{"arxiv_id":"2106.10207","paper":"/paper/distributed-deep-learning-in-open","title":"Distributed Deep Learning in Open Collaborations","date":"2021-06-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yandex-research/DeDLOC","path":"albert/tokenize_wikitext103.py","file_url":"https://github.com/yandex-research/DeDLOC/blob/HEAD/albert/tokenize_wikitext103.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d2b88e48b87e9450","mcp_get_code":{"code_sha256":"d2b88e48b87e9450"}},{"arxiv_id":"2024.findings-emnlp.286","paper":null,"title":"arXiv:2024.findings-emnlp.286","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"kgarg8/Stanceformer","path":"train_automodel2.py","file_url":"https://github.com/kgarg8/Stanceformer/blob/HEAD/train_automodel2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"22184da3623db21e","mcp_get_code":{"code_sha256":"22184da3623db21e"}},{"arxiv_id":"2024.findings-emnlp.286","paper":null,"title":"arXiv:2024.findings-emnlp.286","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"kgarg8/Stanceformer","path":"ABSA/train_automodel2.py","file_url":"https://github.com/kgarg8/Stanceformer/blob/HEAD/ABSA/train_automodel2.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"28e1b38faf4dd119","mcp_get_code":{"code_sha256":"28e1b38faf4dd119"}}]}