{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/normalize-text","entry":"normalize_text","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":60,"n_papers_ran":27,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":54,"n_samples_ran":24,"n_samples_fingerprinted":24,"n_places":61,"n_places_pointer_only":18,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":7,"ran_fixture":0,"ran":17,"unverified":30},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.15229","paper":"/paper/arxiv-2609-15229","title":"Pre-PEFT Probing: Weight Statistics and Perturbation Robustness for Layer Selection in VLM Vision Encoders","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"atoz03/prepeft-probing","path":"code/src/prepeft/eval/normalize.py","file_url":"https://github.com/atoz03/prepeft-probing/blob/HEAD/code/src/prepeft/eval/normalize.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"83355a93f4ab7fad","mcp_get_code":{"code_sha256":"83355a93f4ab7fad"}},{"arxiv_id":"2609.14083","paper":"/paper/arxiv-2609-14083","title":"DEBIASING TREE-BASED VARIABLE IMPORTANCE IN MIXED DATA","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"melikechi-lab/just-add-noise","path":"applications/mixed_importance_utils.py","file_url":"https://github.com/melikechi-lab/just-add-noise/blob/HEAD/applications/mixed_importance_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8bbc006234ea32db","mcp_get_code":{"code_sha256":"8bbc006234ea32db"}},{"arxiv_id":"2609.13238","paper":"/paper/arxiv-2609-13238","title":"Clinical Reasoning Under a Partially Observed Objective in Cone Beam CT Report Generation","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"GIND123/CBCT-Clinical-Reasoner","path":"src/cbct_reasoner/text.py","file_url":"https://github.com/GIND123/CBCT-Clinical-Reasoner/blob/HEAD/src/cbct_reasoner/text.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"fd60e97a9ffd66e8","mcp_get_code":{"code_sha256":"fd60e97a9ffd66e8"}},{"arxiv_id":"2609.02255","paper":"/paper/arxiv-2609-02255","title":"T2LSC-Bench: Benchmarking Localized Semantic Control in Text-to-Image Generation","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"LLMSecResearch/T2LSC-Bench","path":"evaluation/utils.py","file_url":"https://github.com/LLMSecResearch/T2LSC-Bench/blob/HEAD/evaluation/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ee77ade1e729a87f","mcp_get_code":{"code_sha256":"ee77ade1e729a87f"}},{"arxiv_id":"2608.13786","paper":"/paper/arxiv-2608-13786","title":"Do AI chatbots find what experts would? Effects of model, user role, and sample size on study retrieval for medical questions","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"QingfangLiu/llm-evidence-retrieval-bias","path":"benchmark_tools/build_reference_indexing_from_cochrane_ris.py","file_url":"https://github.com/QingfangLiu/llm-evidence-retrieval-bias/blob/HEAD/benchmark_tools/build_reference_indexing_from_cochrane_ris.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"20c568d253ca23b7","mcp_get_code":{"code_sha256":"20c568d253ca23b7"}},{"arxiv_id":"2608.07763","paper":"/paper/arxiv-2608-07763","title":"Jako Tako or Fluent? Presenting PoVisLE: A Polish Vision-Language Evaluation","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"NASK-NLP/PoVisLE","path":"povisle/utils.py","file_url":"https://github.com/NASK-NLP/PoVisLE/blob/HEAD/povisle/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ecc81045ae9ebace","mcp_get_code":{"code_sha256":"ecc81045ae9ebace"}},{"arxiv_id":"2608.05817","paper":"/paper/arxiv-2608-05817","title":"M 3 R-Bench: A Unified Benchmark for Evidence-Grounded Multimodal Metaphor Understanding","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"hongshi4/M3R-Bench","path":"train/evaluate_sft_outputs.py","file_url":"https://github.com/hongshi4/M3R-Bench/blob/HEAD/train/evaluate_sft_outputs.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2b9ad881f4f0c4ac","mcp_get_code":{"code_sha256":"2b9ad881f4f0c4ac"}},{"arxiv_id":"2608.01953","paper":"/paper/arxiv-2608-01953","title":"Look Ahead Before You Distill: Future Trajectory Validation of Teacher Guidance for Agentic On-Policy Distillation","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"ChenChiShui/FutureBridge-OPD","path":"analysis/analyze_bridge_behavior.py","file_url":"https://github.com/ChenChiShui/FutureBridge-OPD/blob/HEAD/analysis/analyze_bridge_behavior.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f7d84e096670e78f","mcp_get_code":{"code_sha256":"f7d84e096670e78f"}},{"arxiv_id":"2607.19374","paper":"/paper/arxiv-2607-19374","title":"Euclean: Automated Geometry Problem Formalization with Unified Verification in Lean","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"tlb-22/Euclean","path":"src/codex_scheduler.py","file_url":"https://github.com/tlb-22/Euclean/blob/HEAD/src/codex_scheduler.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"55a9a37e10742ac4","mcp_get_code":{"code_sha256":"55a9a37e10742ac4"}},{"arxiv_id":"2607.18553","paper":"/paper/arxiv-2607-18553","title":"Operational Proto-Introspection in Looped Language Models Process-Quality Taps, Executable Branching, and the Readout-Control Boundary","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"VykosMolt/Branching-Looped-Transformer","path":"probes/probe_layer_tap_cached_domains.py","file_url":"https://github.com/VykosMolt/Branching-Looped-Transformer/blob/HEAD/probes/probe_layer_tap_cached_domains.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d8b04192866f751a","mcp_get_code":{"code_sha256":"d8b04192866f751a"}},{"arxiv_id":"2607.00605","paper":"/paper/arxiv-2607-00605","title":"Auditing Forgetting in Limited Memory Language Models","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"raeesiarya/LMLMAudit","path":"src/lmlm-audit/run_audit.py","file_url":"https://github.com/raeesiarya/LMLMAudit/blob/HEAD/src/lmlm-audit/run_audit.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c1f14f57e9136ea6","mcp_get_code":{"code_sha256":"c1f14f57e9136ea6"}},{"arxiv_id":"2606.29920","paper":"/paper/arxiv-2606-29920","title":"Can LLM-as-a-Judge Reliably Verify Rubrics in Agentic Scenarios?","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"THU-KEG/RuVerBench","path":"code/strategies/aggregate_voting_outputs.py","file_url":"https://github.com/THU-KEG/RuVerBench/blob/HEAD/code/strategies/aggregate_voting_outputs.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"ada78aa4aa08416f","mcp_get_code":{"code_sha256":"ada78aa4aa08416f"}},{"arxiv_id":"2606.17698","paper":"/paper/arxiv-2606-17698","title":"EComAgentBench: Benchmarking Shopping Agents on Long-Horizon Tasks with Distributed Hidden Intent","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"Morizeyao/EComAgentBench_","path":"src/generation/utils.py","file_url":"https://github.com/Morizeyao/EComAgentBench_/blob/HEAD/src/generation/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"784e6a3cfa0b903a","mcp_get_code":{"code_sha256":"784e6a3cfa0b903a"}},{"arxiv_id":"2606.03463","paper":"/paper/arxiv-2606-03463","title":"DMF: A Deterministic Memory Framework for Conversational AI Agents","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"matstech/dmf-benchmarks","path":"dmf_bench/benchmarks/locomo/rigorous.py","file_url":"https://github.com/matstech/dmf-benchmarks/blob/HEAD/dmf_bench/benchmarks/locomo/rigorous.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"ae71bb9ba2af46bf","mcp_get_code":{"code_sha256":"ae71bb9ba2af46bf"}},{"arxiv_id":"2605.26366","paper":"/paper/arxiv-2605-26366","title":"Automatic Layer Selection for Hallucination Detection","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"DesoloYw/Automatic-Layer-Selection-for-Hallucination-Detection","path":"src/metrics.py","file_url":"https://github.com/DesoloYw/Automatic-Layer-Selection-for-Hallucination-Detection/blob/HEAD/src/metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4e686c57f3e71d5f","mcp_get_code":{"code_sha256":"4e686c57f3e71d5f"}},{"arxiv_id":"2605.15467","paper":"/paper/arxiv-2605-15467","title":"MasonNLP at MEDIQA-SYNUR 2026: Retrieval-Augmented Large Language Models for Schema-Constrained Clinical Information Extraction","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"AHMRezaul/MEDIQA-SYNUR-2026","path":"llama/rag.py","file_url":"https://github.com/AHMRezaul/MEDIQA-SYNUR-2026/blob/HEAD/llama/rag.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cf405930dde4f507","mcp_get_code":{"code_sha256":"cf405930dde4f507"}},{"arxiv_id":"2604.20487","paper":"/paper/arxiv-2604-20487","title":"Knowledge Capsules: Structured Nonparametric Memory Units for LLMs","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"jubin75/KVI","path":"src/cleaning_and_dedupe.py","file_url":"https://github.com/jubin75/KVI/blob/HEAD/src/cleaning_and_dedupe.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5aa9f2fbc850bd95","mcp_get_code":{"code_sha256":"5aa9f2fbc850bd95"}},{"arxiv_id":"2604.08554","paper":"/paper/arxiv-2604-08554","title":"Drift and selection in LLM text ecosystems","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"SR123/LLM-text-ecosystems","path":"src/drift_selection/cleaning.py","file_url":"https://github.com/SR123/LLM-text-ecosystems/blob/HEAD/src/drift_selection/cleaning.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7987ba488637f64c","mcp_get_code":{"code_sha256":"7987ba488637f64c"}},{"arxiv_id":"2604.07035","paper":"/paper/arxiv-2604-07035","title":"Unified Deployment-Aware Evaluation of Open Reasoning Language Models","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"mkboch/UDAE","path":"evaluation/grader.py","file_url":"https://github.com/mkboch/UDAE/blob/HEAD/evaluation/grader.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"87a574d68dd0da52","mcp_get_code":{"code_sha256":"87a574d68dd0da52"}},{"arxiv_id":"2603.12478","paper":"/paper/arxiv-2603-12478","title":"Less Data, Faster Convergence: Goal-Driven Data Optimization for Multimodal Instruction Tuning","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"rujiewu/GDO","path":"gdo/extract_six_metrics.py","file_url":"https://github.com/rujiewu/GDO/blob/HEAD/gdo/extract_six_metrics.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"db008612f143341f","mcp_get_code":{"code_sha256":"db008612f143341f"}},{"arxiv_id":"2603.04969","paper":"/paper/arxiv-2603-04969","title":"MPCEval: A Benchmark for Multi-Party Conversation Generation","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"Owen-Yang-18/MPCEval","path":"src/local_speaker/utils/common.py","file_url":"https://github.com/Owen-Yang-18/MPCEval/blob/HEAD/src/local_speaker/utils/common.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"490e489b13b16ab5","mcp_get_code":{"code_sha256":"490e489b13b16ab5"}},{"arxiv_id":"2601.06460","paper":"/paper/arxiv-2601-06460","title":"Tone Matters: The Impact of Linguistic Tone on Hallucination in VLMs","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"bli1/tone-matters","path":"VLM_Benchmark_GitHub_Ready/evaluation/score_hybrid_600_200.py","file_url":"https://github.com/bli1/tone-matters/blob/HEAD/VLM_Benchmark_GitHub_Ready/evaluation/score_hybrid_600_200.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c535a7f4683e3452","mcp_get_code":{"code_sha256":"c535a7f4683e3452"}},{"arxiv_id":"2601.01062","paper":"/paper/arxiv-2601-01062","title":"SPoRC-VIST: A Benchmark for Evaluating Generative Natural Narrative in Vision-Language Models","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"Yunlin-Zeng/visual-podcast-VLM","path":"evaluation_50_samples/evaluate_metrics.py","file_url":"https://github.com/Yunlin-Zeng/visual-podcast-VLM/blob/HEAD/evaluation_50_samples/evaluate_metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"017711da98a034fb","mcp_get_code":{"code_sha256":"017711da98a034fb"}},{"arxiv_id":"2508.14723","paper":"/paper/arxiv-2508-14723","title":"Transplant Then Regenerate: A New Paradigm for Text Data Augmentation","date":null,"month_inferred_from_arxiv_id":"2025-08","title_source":"syntology","repo":"1024er/cbert_aug","path":"text_classification/nlp_utils.py","file_url":"https://github.com/1024er/cbert_aug/blob/HEAD/text_classification/nlp_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"be8f967431da41d7","mcp_get_code":{"code_sha256":"be8f967431da41d7"}},{"arxiv_id":"2507.02834","paper":null,"title":"arXiv:2507.02834","date":null,"month_inferred_from_arxiv_id":"2025-07","title_source":null,"repo":"dhcode-cpp/X-R1","path":"src/x_r1/rewards.py","file_url":"https://github.com/dhcode-cpp/X-R1/blob/HEAD/src/x_r1/rewards.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"11f2676b13fb9f69","mcp_get_code":{"code_sha256":"11f2676b13fb9f69"}},{"arxiv_id":"2411.12372","paper":"/paper/redpajama-an-open-dataset-for-training-large","title":"RedPajama: an Open Dataset for Training Large Language Models","date":"2024-11-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"togethercomputer/redpajama-data","path":"app/src/artifacts/utils/data_utils.py","file_url":"https://github.com/togethercomputer/redpajama-data/blob/HEAD/app/src/artifacts/utils/data_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0823f0fa0938dbfd","mcp_get_code":{"code_sha256":"0823f0fa0938dbfd"}},{"arxiv_id":"2409.02813","paper":"/paper/mmmu-pro-a-more-robust-multi-discipline","title":"MMMU-Pro: A More Robust Multi-discipline Multimodal Understanding Benchmark","date":"2024-09-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"opendatalab/pm4bench","path":"src/pm4bench/metrics.py","file_url":"https://github.com/opendatalab/pm4bench/blob/HEAD/src/pm4bench/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"abd00a80d4ff626d","mcp_get_code":{"code_sha256":"abd00a80d4ff626d"}},{"arxiv_id":"2406.04904","paper":"/paper/xtts-a-massively-multilingual-zero-shot-text","title":"XTTS: a Massively Multilingual Zero-Shot Text-to-Speech Model","date":"2024-06-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Edresson/ZS-TTS-Evaluation","path":"utils/generic_utils.py","file_url":"https://github.com/Edresson/ZS-TTS-Evaluation/blob/HEAD/utils/generic_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"15b7e710b24d5a4c","mcp_get_code":{"code_sha256":"15b7e710b24d5a4c"}},{"arxiv_id":"2406.00008","paper":"/paper/knowledgehub-an-end-to-end-tool-for-assisted","title":"KnowledgeHub: An end-to-end Tool for Assisted Scientific Discovery","date":"2024-05-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kermitt2/grobid","path":"grobid-trainer/resources/dataset/segmentation/article/light/corpus/tei/analyze_notes.py","file_url":"https://github.com/kermitt2/grobid/blob/HEAD/grobid-trainer/resources/dataset/segmentation/article/light/corpus/tei/analyze_notes.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"354d9000bf5db282","mcp_get_code":{"code_sha256":"354d9000bf5db282"}},{"arxiv_id":"2406.00008","paper":"/paper/knowledgehub-an-end-to-end-tool-for-assisted","title":"KnowledgeHub: An end-to-end Tool for Assisted Scientific Discovery","date":"2024-05-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kermitt2/grobid","path":"grobid-trainer/resources/dataset/segmentation/article/light/corpus/tei/detailed_analysis.py","file_url":"https://github.com/kermitt2/grobid/blob/HEAD/grobid-trainer/resources/dataset/segmentation/article/light/corpus/tei/detailed_analysis.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b906150b79481c79","mcp_get_code":{"code_sha256":"b906150b79481c79"}},{"arxiv_id":"2405.18357","paper":"/paper/faithful-logical-reasoning-via-symbolic-chain","title":"Faithful Logical Reasoning via Symbolic Chain-of-Thought","date":"2024-05-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Aiden0526/SymbCoT","path":"baselines/evaluation.py","file_url":"https://github.com/Aiden0526/SymbCoT/blob/HEAD/baselines/evaluation.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"71986ed6ac7b9157","mcp_get_code":{"code_sha256":"71986ed6ac7b9157"}},{"arxiv_id":"2405.10138","paper":"/paper/pl-mteb-polish-massive-text-embedding","title":"PL-MTEB: Polish Massive Text Embedding Benchmark","date":"2024-05-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rafalposwiata/pl-mteb","path":"tasks/preparation/cleaning.py","file_url":"https://github.com/rafalposwiata/pl-mteb/blob/HEAD/tasks/preparation/cleaning.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d3d2208d6468ed08","mcp_get_code":{"code_sha256":"d3d2208d6468ed08"}},{"arxiv_id":"2404.15993","paper":"/paper/uncertainty-estimation-and-quantification-for","title":"Uncertainty Estimation and Quantification for LLMs: A Simple Supervised Approach","date":"2024-04-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"LoveCatc/supervised-llm-uncertainty-estimation","path":"utils/data_entry.py","file_url":"https://github.com/LoveCatc/supervised-llm-uncertainty-estimation/blob/HEAD/utils/data_entry.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"e32f292f91eab5d8","mcp_get_code":{"code_sha256":"e32f292f91eab5d8"}},{"arxiv_id":"2402.18048","paper":"/paper/characterizing-truthfulness-in-large-language","title":"Characterizing Truthfulness in Large Language Model Generations with Local Intrinsic Dimension","date":"2024-02-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"fanyin3639/lid-hallucinationdetection","path":"src/metrics.py","file_url":"https://github.com/fanyin3639/lid-hallucinationdetection/blob/HEAD/src/metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4e686c57f3e71d5f","mcp_get_code":{"code_sha256":"4e686c57f3e71d5f"}},{"arxiv_id":"2401.03506","paper":"/paper/diarizationlm-speaker-diarization-post","title":"DiarizationLM: Speaker Diarization Post-Processing with Large Language Models","date":"2024-01-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"google/speaker-id","path":"DiarizationLM/diarizationlm/utils.py","file_url":"https://github.com/google/speaker-id/blob/HEAD/DiarizationLM/diarizationlm/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"cca62276ec792e36","mcp_get_code":{"code_sha256":"cca62276ec792e36"}},{"arxiv_id":"2311.09889","paper":"/paper/language-generation-from-human-brain","title":"Language Generation from Brain Recordings","date":"2023-11-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yeziyi1998/brain-language-generation","path":"language_generation/src/post_hoc_evaluate.py","file_url":"https://github.com/yeziyi1998/brain-language-generation/blob/HEAD/language_generation/src/post_hoc_evaluate.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"f04f39dee25fb155","mcp_get_code":{"code_sha256":"f04f39dee25fb155"}},{"arxiv_id":"2310.15047","paper":"/paper/meta-out-of-context-learning-in-neural","title":"Implicit meta-learning may lead language models to trust more reliable sources","date":"2023-10-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"krasheninnikov/internalization","path":"src/metrics.py","file_url":"https://github.com/krasheninnikov/internalization/blob/HEAD/src/metrics.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"12cb4aef8c35072c","mcp_get_code":{"code_sha256":"12cb4aef8c35072c"}},{"arxiv_id":"2310.03184","paper":"/paper/retrieval-augmented-generation-to-improve","title":"Retrieval-augmented Generation to Improve Math Question-Answering: Trade-offs Between Groundedness and Human Preference","date":"2023-10-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"digitalharborfoundation/rag-for-math-qa","path":"src/rag/retrieval.py","file_url":"https://github.com/digitalharborfoundation/rag-for-math-qa/blob/HEAD/src/rag/retrieval.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"94af878e3d7b1853","mcp_get_code":{"code_sha256":"94af878e3d7b1853"}},{"arxiv_id":"2309.08105","paper":"/paper/libriheavy-a-50000-hours-asr-corpus-with","title":"Libriheavy: a 50,000 hours ASR corpus with punctuation casing and context","date":"2023-09-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"k2-fsa/libriheavy","path":"scripts/extract_and_normalize_transcript.py","file_url":"https://github.com/k2-fsa/libriheavy/blob/HEAD/scripts/extract_and_normalize_transcript.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"5ddb3b78d7b37efe","mcp_get_code":{"code_sha256":"5ddb3b78d7b37efe"}},{"arxiv_id":"2308.06782","paper":"/paper/pentestgpt-an-llm-empowered-automatic","title":"PentestGPT: An LLM-empowered Automatic Penetration Testing Tool","date":null,"month_inferred_from_arxiv_id":"2023-08","title_source":"archive","repo":"isamu-isozaki/PentestGPT","path":"pentestgpt/utils/chroma_vector_db.py","file_url":"https://github.com/isamu-isozaki/PentestGPT/blob/HEAD/pentestgpt/utils/chroma_vector_db.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"edf2156bd952c7ed","mcp_get_code":{"code_sha256":"edf2156bd952c7ed"}},{"arxiv_id":"2305.12295","paper":"/paper/logic-lm-empowering-large-language-models","title":"Logic-LM: Empowering Large Language Models with Symbolic Solvers for Faithful Logical Reasoning","date":"2023-05-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"teacherpeterpan/logic-llm","path":"models/evaluation.py","file_url":"https://github.com/teacherpeterpan/logic-llm/blob/HEAD/models/evaluation.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"71986ed6ac7b9157","mcp_get_code":{"code_sha256":"71986ed6ac7b9157"}},{"arxiv_id":"2305.07759","paper":"/paper/tinystories-how-small-can-language-models-be","title":"TinyStories: How Small Can Language Models Be and Still Speak Coherent English?","date":"2023-05-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"vizuaraai/tiny-stories-regional","path":"analysis/BERT_BLEU_eval.py","file_url":"https://github.com/vizuaraai/tiny-stories-regional/blob/HEAD/analysis/BERT_BLEU_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"31f124b85a98aa6e","mcp_get_code":{"code_sha256":"31f124b85a98aa6e"}},{"arxiv_id":"2304.13007","paper":"/paper/answering-questions-by-meta-reasoning-over","title":"Answering Questions by Meta-Reasoning over Multiple Chains of Thought","date":"2023-04-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"oriyor/reasoning-on-cots","path":"src/pred_evaluators/evaluation.py","file_url":"https://github.com/oriyor/reasoning-on-cots/blob/HEAD/src/pred_evaluators/evaluation.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c30d6b506e3a5d6e","mcp_get_code":{"code_sha256":"c30d6b506e3a5d6e"}},{"arxiv_id":"2203.10652","paper":"/paper/continual-sequence-generation-with-adaptive","title":"Continual Sequence Generation with Adaptive Compositional Modules","date":"2022-03-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"GT-SALT/Adaptive-Compositional-Modules","path":"metrics.py","file_url":"https://github.com/GT-SALT/Adaptive-Compositional-Modules/blob/HEAD/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b971ea3b818d09b7","mcp_get_code":{"code_sha256":"b971ea3b818d09b7"}},{"arxiv_id":"2110.07803","paper":"/paper/contraqa-question-answering-under-1","title":"Attacking Open-domain Question Answering by Injecting Misinformation","date":"2021-10-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"teacherpeterpan/contraqa","path":"evaluate.py","file_url":"https://github.com/teacherpeterpan/contraqa/blob/HEAD/evaluate.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c30d6b506e3a5d6e","mcp_get_code":{"code_sha256":"c30d6b506e3a5d6e"}},{"arxiv_id":"2006.13009","paper":"/paper/iterative-deep-graph-learning-for-graph-1","title":"Iterative Deep Graph Learning for Graph Neural Networks: Better and Robust Node Embeddings","date":"2020-06-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hugochan/IDGL","path":"src/core/utils/eval_utils.py","file_url":"https://github.com/hugochan/IDGL/blob/HEAD/src/core/utils/eval_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"38ffdcfda0ea6889","mcp_get_code":{"code_sha256":"38ffdcfda0ea6889"}},{"arxiv_id":"2002.00748","paper":"/paper/asking-questions-the-human-way-scalable","title":"Asking Questions the Human Way: Scalable Question-Answer Generation from Text Corpus","date":"2020-01-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bangliu/ACS-QG","path":"DA_main.py","file_url":"https://github.com/bangliu/ACS-QG/blob/HEAD/DA_main.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d53579a101181ef7","mcp_get_code":{"code_sha256":"d53579a101181ef7"}},{"arxiv_id":"1912.07875","paper":"/paper/libri-light-a-benchmark-for-asr-with-limited","title":"Libri-Light: A Benchmark for ASR with Limited or No Supervision","date":"2019-12-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":true,"licence":null,"inline_ok":false,"code_sha256_prefix":"5ddb3b78d7b37efe","mcp_get_code":{"code_sha256":"5ddb3b78d7b37efe"}},{"arxiv_id":"1909.03329","paper":"/paper/lamal-language-modeling-is-all-you-need-for","title":"LAMOL: LAnguage MOdeling for Lifelong Language Learning","date":"2019-09-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"jojotenya/LAMOL","path":"metrics.py","file_url":"https://github.com/jojotenya/LAMOL/blob/HEAD/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b971ea3b818d09b7","mcp_get_code":{"code_sha256":"b971ea3b818d09b7"}},{"arxiv_id":"1810.04805","paper":"/paper/bert-pre-training-of-deep-bidirectional","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","date":"2018-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"autobotasia/vibert","path":"pre-training-vibert.py","file_url":"https://github.com/autobotasia/vibert/blob/HEAD/pre-training-vibert.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b4d10db840d16868","mcp_get_code":{"code_sha256":"b4d10db840d16868"}},{"arxiv_id":"1809.05679","paper":"/paper/graph-convolutional-networks-for-text","title":"Graph Convolutional Networks for Text Classification","date":"2018-09-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"koreyou/text-gcn-chainer","path":"nlp_utils.py","file_url":"https://github.com/koreyou/text-gcn-chainer/blob/HEAD/nlp_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"CC0-1.0","inline_ok":true,"code_sha256_prefix":"2950d6b582fe43a8","mcp_get_code":{"code_sha256":"2950d6b582fe43a8"}},{"arxiv_id":"openreview_XtIRCAEYoJ","paper":null,"title":"arXiv:openreview_XtIRCAEYoJ","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"WayneTomas/Artemis","path":"val/coco_detection/convert_to_coco_result.py","file_url":"https://github.com/WayneTomas/Artemis/blob/HEAD/val/coco_detection/convert_to_coco_result.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"fb5350be94c487a0","mcp_get_code":{"code_sha256":"fb5350be94c487a0"}},{"arxiv_id":"openreview_8ppVmLtA2V","paper":null,"title":"arXiv:openreview_8ppVmLtA2V","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"yanweiyue/Mem-T","path":"llm_judge.py","file_url":"https://github.com/yanweiyue/Mem-T/blob/HEAD/llm_judge.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"988bb104b8b80d1e","mcp_get_code":{"code_sha256":"988bb104b8b80d1e"}},{"arxiv_id":"aaai_29863","paper":null,"title":"arXiv:aaai_29863","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"facebookresearch/EditEval","path":"src/preprocessing.py","file_url":"https://github.com/facebookresearch/EditEval/blob/HEAD/src/preprocessing.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"CC0-1.0","inline_ok":true,"code_sha256_prefix":"223dafc9c2794bd1","mcp_get_code":{"code_sha256":"223dafc9c2794bd1"}},{"arxiv_id":"2025.findings-emnlp.1015","paper":null,"title":"arXiv:2025.findings-emnlp.1015","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"fzp0424/MT-R1-Zero","path":"eval/extract_to_eval.py","file_url":"https://github.com/fzp0424/MT-R1-Zero/blob/HEAD/eval/extract_to_eval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7715ccc1e32f7073","mcp_get_code":{"code_sha256":"7715ccc1e32f7073"}},{"arxiv_id":"2025.findings-acl.712","paper":null,"title":"arXiv:2025.findings-acl.712","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"cxcscmu/Craw4LLM","path":"document_rater.py","file_url":"https://github.com/cxcscmu/Craw4LLM/blob/HEAD/document_rater.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8d60275b3f501f63","mcp_get_code":{"code_sha256":"8d60275b3f501f63"}},{"arxiv_id":"2025.emnlp-industry.89","paper":null,"title":"arXiv:2025.emnlp-industry.89","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"sssrlll/L4","path":"metrics.py","file_url":"https://github.com/sssrlll/L4/blob/HEAD/metrics.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b971ea3b818d09b7","mcp_get_code":{"code_sha256":"b971ea3b818d09b7"}},{"arxiv_id":"2024.findings-emnlp.687","paper":null,"title":"arXiv:2024.findings-emnlp.687","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"IAAR-Shanghai/FastMem","path":"src/utils.py","file_url":"https://github.com/IAAR-Shanghai/FastMem/blob/HEAD/src/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"15d3a384e09473fb","mcp_get_code":{"code_sha256":"15d3a384e09473fb"}},{"arxiv_id":"2024.findings-eacl.35","paper":null,"title":"arXiv:2024.findings-eacl.35","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"naist-nlp/atd-mcl","path":"src/util.py","file_url":"https://github.com/naist-nlp/atd-mcl/blob/HEAD/src/util.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"77b47e3e5dacb774","mcp_get_code":{"code_sha256":"77b47e3e5dacb774"}},{"arxiv_id":"2024.acl-long.805","paper":null,"title":"arXiv:2024.acl-long.805","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"togethercomputer/RedPajama-Data","path":"app/src/artifacts/utils/data_utils.py","file_url":"https://github.com/togethercomputer/RedPajama-Data/blob/HEAD/app/src/artifacts/utils/data_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"0823f0fa0938dbfd","mcp_get_code":{"code_sha256":"0823f0fa0938dbfd"}},{"arxiv_id":"2023.acl-long.736","paper":null,"title":"arXiv:2023.acl-long.736","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"felixgwu/FastFusionNet","path":"qa/general_utils.py","file_url":"https://github.com/felixgwu/FastFusionNet/blob/HEAD/qa/general_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"dd599b6afb9aad1b","mcp_get_code":{"code_sha256":"dd599b6afb9aad1b"}}]}