{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/setup-logging","entry":"setup_logging","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":49,"n_papers_ran":11,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":50,"n_samples_ran":11,"n_samples_fingerprinted":0,"n_places":54,"n_places_pointer_only":18,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":2,"ran_fixture":0,"ran":9,"unverified":39},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2608.23936","paper":"/paper/arxiv-2608-23936","title":"MnemoDyn: Learning Resting State Dynamics from 40K FMRI sequences","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"vsingh-group/mnemodyn","path":"code/light/abide_classification.py","file_url":"https://github.com/vsingh-group/mnemodyn/blob/HEAD/code/light/abide_classification.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7ab65340af034176","mcp_get_code":{"code_sha256":"7ab65340af034176"}},{"arxiv_id":"2606.30430","paper":"/paper/arxiv-2606-30430","title":"CAN We Trust Your Results? A Cross-Dataset Study of Automotive IDS Evaluation","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"CrySyS/Cross-Dataset-Study-of-Automotive-IDS-Evaluation","path":"unified_ids/eval/core.py","file_url":"https://github.com/CrySyS/Cross-Dataset-Study-of-Automotive-IDS-Evaluation/blob/HEAD/unified_ids/eval/core.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"a5e6827fb8c56557","mcp_get_code":{"code_sha256":"a5e6827fb8c56557"}},{"arxiv_id":"2606.19821","paper":"/paper/arxiv-2606-19821","title":"TelcoAgent: A Scalable 5G Multi-KPM Forecasting With 3GPP-Grounded Explainability","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"NextG-Wireless-Lab-NC-State/TelcoAgent","path":"telcoagent/cli_utils.py","file_url":"https://github.com/NextG-Wireless-Lab-NC-State/TelcoAgent/blob/HEAD/telcoagent/cli_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"49e279fff90c238c","mcp_get_code":{"code_sha256":"49e279fff90c238c"}},{"arxiv_id":"2606.12733","paper":"/paper/arxiv-2606-12733","title":"Let's Ask Gauss: Improved One-Run Privacy Auditing","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"stoneboat/dpsgd-auditbench","path":"src/utils.py","file_url":"https://github.com/stoneboat/dpsgd-auditbench/blob/HEAD/src/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"2bfbdadb21725fa8","mcp_get_code":{"code_sha256":"2bfbdadb21725fa8"}},{"arxiv_id":"2606.08969","paper":"/paper/arxiv-2606-08969","title":"CARE: A Conformal Safety Layer for Medical Summarization","date":null,"month_inferred_from_arxiv_id":"2026-06","title_source":"syntology","repo":"som-shahlab/CARE","path":"care/utils.py","file_url":"https://github.com/som-shahlab/CARE/blob/HEAD/care/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a2e6317c9ba0c260","mcp_get_code":{"code_sha256":"a2e6317c9ba0c260"}},{"arxiv_id":"2605.28222","paper":"/paper/arxiv-2605-28222","title":"Analyzing Quality-Latency-Resource Trade-offs in a Technical Documentation RAG Assistant Using LoRA Adaptation","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"EugPal/rag-lora-tradeoffs","path":"src/data_pipeline/build_kubernetes_qa_dataset.py","file_url":"https://github.com/EugPal/rag-lora-tradeoffs/blob/HEAD/src/data_pipeline/build_kubernetes_qa_dataset.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7f6e7e7ff3d746c3","mcp_get_code":{"code_sha256":"7f6e7e7ff3d746c3"}},{"arxiv_id":"2605.24693","paper":"/paper/arxiv-2605-24693","title":"CP-Agent: A Calibrated Risk-Controlled Agent for Feedback-Driven Competitive Programming","date":null,"month_inferred_from_arxiv_id":"2026-05","title_source":"syntology","repo":"NineAbyss/CP-Agent","path":"agentflow/main_agent.py","file_url":"https://github.com/NineAbyss/CP-Agent/blob/HEAD/agentflow/main_agent.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0aeca2b1d78fe74b","mcp_get_code":{"code_sha256":"0aeca2b1d78fe74b"}},{"arxiv_id":"2604.12374","paper":"/paper/arxiv-2604-12374","title":"Nemotron 3 Super: Open, Efficient Mixture-of-Experts Hybrid Mamba-Transformer Model for Agentic Reasoning NVIDIA","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"NVIDIA-NeMo/Skills","path":"nemo_skills/utils.py","file_url":"https://github.com/NVIDIA-NeMo/Skills/blob/HEAD/nemo_skills/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4261064eb4ec7756","mcp_get_code":{"code_sha256":"4261064eb4ec7756"}},{"arxiv_id":"2604.06182","paper":"/paper/arxiv-2604-06182","title":"VenusBench-Mobile: A Challenging and User-Centric Benchmark for Mobile GUI Agents with Capability Diagnostics","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"inclusionAI/UI-Venus","path":"Venus_framework/Venus_framework_mobile/batch_runner.py","file_url":"https://github.com/inclusionAI/UI-Venus/blob/HEAD/Venus_framework/Venus_framework_mobile/batch_runner.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8c4dadfe748db74a","mcp_get_code":{"code_sha256":"8c4dadfe748db74a"}},{"arxiv_id":"2604.03779","paper":"/paper/arxiv-2604-03779","title":"CountsDiff: A Diffusion Model on the Natural Numbers for Generation and Imputation of Count-Based Data","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"rsoatto/countsdiff","path":"src/countsdiff/utils/logging.py","file_url":"https://github.com/rsoatto/countsdiff/blob/HEAD/src/countsdiff/utils/logging.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"047467be88565ffe","mcp_get_code":{"code_sha256":"047467be88565ffe"}},{"arxiv_id":"2604.02520","paper":"/paper/arxiv-2604-02520","title":"Neural posterior estimation for scalable and accurate inverse parameter inference in Li-ion batteries","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"NatLabRockies/BatFIT","path":"batfit/logging_config.py","file_url":"https://github.com/NatLabRockies/BatFIT/blob/HEAD/batfit/logging_config.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"fcf7a6d4c864cee5","mcp_get_code":{"code_sha256":"fcf7a6d4c864cee5"}},{"arxiv_id":"2603.21970","paper":"/paper/arxiv-2603-21970","title":"Parameter-Efficient Fine-Tuning for Medical Text Summarization: A Comparative Study of Lora, Prompt Tuning, and Full Fine-Tuning","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"eracoding/llm-medical-summarization","path":"src/utils/logging_utils.py","file_url":"https://github.com/eracoding/llm-medical-summarization/blob/HEAD/src/utils/logging_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"453cce5f6e38dc12","mcp_get_code":{"code_sha256":"453cce5f6e38dc12"}},{"arxiv_id":"2603.21365","paper":"/paper/arxiv-2603-21365","title":"TIDE: Token-Informed Depth Execution for Per-Token Early Exit in LLM Inference","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"RightNow-AI/TIDE","path":"python/TIDE/utils.py","file_url":"https://github.com/RightNow-AI/TIDE/blob/HEAD/python/TIDE/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d494712ddd7da224","mcp_get_code":{"code_sha256":"d494712ddd7da224"}},{"arxiv_id":"2603.18872","paper":"/paper/arxiv-2603-18872","title":"DriftGuard: Mitigating Asynchronous Data Drift in Federated Learning","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"blessonvar/DriftGuard","path":"src/driftguard/config.py","file_url":"https://github.com/blessonvar/DriftGuard/blob/HEAD/src/driftguard/config.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"121b079a948bb5d2","mcp_get_code":{"code_sha256":"121b079a948bb5d2"}},{"arxiv_id":"2602.14849","paper":"/paper/arxiv-2602-14849","title":"Atomix: Timely, Transactional Tool Use for Reliable Agentic Workflows","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"mpi-dsg/atomix","path":"src/atomix/logging.py","file_url":"https://github.com/mpi-dsg/atomix/blob/HEAD/src/atomix/logging.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"2da8052ea7b493ae","mcp_get_code":{"code_sha256":"2da8052ea7b493ae"}},{"arxiv_id":"2602.12825","paper":"/paper/arxiv-2602-12825","title":"Reliable Hierarchical Operating System Fingerprinting via Conformal Prediction","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"rubenpjove/CP-HOSfing","path":"exps/utils/io_utils.py","file_url":"https://github.com/rubenpjove/CP-HOSfing/blob/HEAD/exps/utils/io_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"51d83e84e7c5e28f","mcp_get_code":{"code_sha256":"51d83e84e7c5e28f"}},{"arxiv_id":"2602.10441","paper":"/paper/arxiv-2602-10441","title":"LakeMLB: Data Lake Machine Learning Benchmark","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"zhengwang100/LakeMLB","path":"codes/baseline/tree_models.py","file_url":"https://github.com/zhengwang100/LakeMLB/blob/HEAD/codes/baseline/tree_models.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"07e2b30309abf95c","mcp_get_code":{"code_sha256":"07e2b30309abf95c"}},{"arxiv_id":"2602.03419","paper":"/paper/arxiv-2602-03419","title":"SWE-World: Building Software Engineering Agents in Docker-Free Environments","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"RUCAIBox/SWE-World","path":"swe_world/src/r2egym/logging.py","file_url":"https://github.com/RUCAIBox/SWE-World/blob/HEAD/swe_world/src/r2egym/logging.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b6ca8cf6963ad7ee","mcp_get_code":{"code_sha256":"b6ca8cf6963ad7ee"}},{"arxiv_id":"2512.00920","paper":"/paper/arxiv-2512-00920","title":"Reward Auditor: Inference on Reward Modeling Suitability in Real-World Perturbed Scenarios","date":null,"month_inferred_from_arxiv_id":"2025-12","title_source":"syntology","repo":"hggzjx/RewardAuditor","path":"allenai_rewardbench/analysis/get_per_token_reward.py","file_url":"https://github.com/hggzjx/RewardAuditor/blob/HEAD/allenai_rewardbench/analysis/get_per_token_reward.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"de89b479b777e992","mcp_get_code":{"code_sha256":"de89b479b777e992"}},{"arxiv_id":"2510.11131","paper":"/paper/arxiv-2510-11131","title":"SocioBench: Modeling Human Behavior in Sociological Surveys with Large Language Models","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"JiaWANG-TJ/SocioBench","path":"evaluation/logger_setup.py","file_url":"https://github.com/JiaWANG-TJ/SocioBench/blob/HEAD/evaluation/logger_setup.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4a9001cfab4d077e","mcp_get_code":{"code_sha256":"4a9001cfab4d077e"}},{"arxiv_id":"2510.07768","paper":"/paper/arxiv-2510-07768","title":"ToolLibGen: Scalable Automatic Tool Creation and Aggregation for LLM Reasoning","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"SalesforceAIResearch/ToolLibGen","path":"src/ToolAggregationAgent.py","file_url":"https://github.com/SalesforceAIResearch/ToolLibGen/blob/HEAD/src/ToolAggregationAgent.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"d8b1384934a717ba","mcp_get_code":{"code_sha256":"d8b1384934a717ba"}},{"arxiv_id":"2510.00549","paper":"/paper/arxiv-2510-00549","title":"EMR-AGENT: Automating Cohort and Feature Extraction from EMR Databases","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"AITRICS/EMR-AGENT","path":"EMR-Agent/models/DIN-SQL.py","file_url":"https://github.com/AITRICS/EMR-AGENT/blob/HEAD/EMR-Agent/models/DIN-SQL.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a7a116336ab05e5f","mcp_get_code":{"code_sha256":"a7a116336ab05e5f"}},{"arxiv_id":"2510.00549","paper":"/paper/arxiv-2510-00549","title":"EMR-AGENT: Automating Cohort and Feature Extraction from EMR Databases","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"AITRICS/EMR-AGENT","path":"EMR-Agent/models/ehr_seqsql.py","file_url":"https://github.com/AITRICS/EMR-AGENT/blob/HEAD/EMR-Agent/models/ehr_seqsql.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"affeee748b50d7ef","mcp_get_code":{"code_sha256":"affeee748b50d7ef"}},{"arxiv_id":"2510.00549","paper":"/paper/arxiv-2510-00549","title":"EMR-AGENT: Automating Cohort and Feature Extraction from EMR Databases","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"AITRICS/EMR-AGENT","path":"EMR-Agent/models/ehr_sql_pluq_style.py","file_url":"https://github.com/AITRICS/EMR-AGENT/blob/HEAD/EMR-Agent/models/ehr_sql_pluq_style.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"530e4660e41c5212","mcp_get_code":{"code_sha256":"530e4660e41c5212"}},{"arxiv_id":"2510.00549","paper":"/paper/arxiv-2510-00549","title":"EMR-AGENT: Automating Cohort and Feature Extraction from EMR Databases","date":null,"month_inferred_from_arxiv_id":"2025-10","title_source":"syntology","repo":"AITRICS/EMR-AGENT","path":"EMR-Agent/models/react.py","file_url":"https://github.com/AITRICS/EMR-AGENT/blob/HEAD/EMR-Agent/models/react.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c0855d72e4fbf113","mcp_get_code":{"code_sha256":"c0855d72e4fbf113"}},{"arxiv_id":"2509.15786","paper":"/paper/arxiv-2509-15786","title":"Building Data-Driven Occupation Taxonomies: A Bottom-Up Multi-Stage Approach via Semantic Clustering and Multi-Agent Collaboration","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"aida-ugent/CLIMB","path":"src/chunk_job_postings.py","file_url":"https://github.com/aida-ugent/CLIMB/blob/HEAD/src/chunk_job_postings.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"31d85a88c212c170","mcp_get_code":{"code_sha256":"31d85a88c212c170"}},{"arxiv_id":"2509.12158","paper":"/paper/arxiv-2509-12158","title":"Pun Unintended: LLMs and the Illusion of Humor Understanding","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"alezanga/punintended","path":"utils/log.py","file_url":"https://github.com/alezanga/punintended/blob/HEAD/utils/log.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"d2ecec99293aa8f8","mcp_get_code":{"code_sha256":"d2ecec99293aa8f8"}},{"arxiv_id":"2507.10646","paper":"/paper/codeassistbench-cab-dataset-benchmarking-for","title":"CodeAssistBench (CAB): Dataset & Benchmarking for Multi-turn Chat-Based Code Assistance","date":"2025-07-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"amazon-science/CodeAssistBench","path":"script/generate_dockerfile.py","file_url":"https://github.com/amazon-science/CodeAssistBench/blob/HEAD/script/generate_dockerfile.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"841dffee01585864","mcp_get_code":{"code_sha256":"841dffee01585864"}},{"arxiv_id":"2507.10646","paper":"/paper/codeassistbench-cab-dataset-benchmarking-for","title":"CodeAssistBench (CAB): Dataset & Benchmarking for Multi-turn Chat-Based Code Assistance","date":"2025-07-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"amazon-science/CodeAssistBench","path":"script/generate_dockerfile_with_strands.py","file_url":"https://github.com/amazon-science/CodeAssistBench/blob/HEAD/script/generate_dockerfile_with_strands.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4aa375e6a657c179","mcp_get_code":{"code_sha256":"4aa375e6a657c179"}},{"arxiv_id":"2506.01937","paper":"/paper/rewardbench-2-advancing-reward-model","title":"RewardBench 2: Advancing Reward Model Evaluation","date":"2025-06-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"allenai/reward-bench","path":"analysis/get_per_token_reward.py","file_url":"https://github.com/allenai/reward-bench/blob/HEAD/analysis/get_per_token_reward.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"de89b479b777e992","mcp_get_code":{"code_sha256":"de89b479b777e992"}},{"arxiv_id":"2505.10518","paper":"/paper/multi-token-prediction-needs-registers","title":"Multi-Token Prediction Needs Registers","date":"2025-05-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"nasosger/mutor","path":"language_modeling/src/finetune.py","file_url":"https://github.com/nasosger/mutor/blob/HEAD/language_modeling/src/finetune.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a0b5557b356e0c05","mcp_get_code":{"code_sha256":"a0b5557b356e0c05"}},{"arxiv_id":"2504.16891","paper":"/paper/aimo-2-winning-solution-building-state-of-the","title":"AIMO-2 Winning Solution: Building State-of-the-Art Mathematical Reasoning Models with OpenMathReasoning dataset","date":"2025-04-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kipok/nemo-skills","path":"nemo_skills/utils.py","file_url":"https://github.com/kipok/nemo-skills/blob/HEAD/nemo_skills/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4261064eb4ec7756","mcp_get_code":{"code_sha256":"4261064eb4ec7756"}},{"arxiv_id":"2502.06445","paper":"/paper/benchmarking-vision-language-models-on","title":"Benchmarking Vision-Language Models on Optical Character Recognition in Dynamic Video Environments","date":"2025-02-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"video-db/ocr-benchmark","path":"utils.py","file_url":"https://github.com/video-db/ocr-benchmark/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"588aaa1e437f6e28","mcp_get_code":{"code_sha256":"588aaa1e437f6e28"}},{"arxiv_id":"2410.22239","paper":"/paper/discern-decoding-systematic-errors-in-natural","title":"DISCERN: Decoding Systematic Errors in Natural Language for Text Classifiers","date":"2024-10-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rrmenon10/DISCERN","path":"src/discern/refine.py","file_url":"https://github.com/rrmenon10/DISCERN/blob/HEAD/src/discern/refine.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"01a3c356acb12f97","mcp_get_code":{"code_sha256":"01a3c356acb12f97"}},{"arxiv_id":"2410.07166","paper":"/paper/embodied-agent-interface-benchmarking-llms","title":"Embodied Agent Interface: Benchmarking LLMs for Embodied Decision Making","date":"2024-10-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"embodied-agent-eval/embodied-agent-eval","path":"src/virtualhome_eval/log_config.py","file_url":"https://github.com/embodied-agent-eval/embodied-agent-eval/blob/HEAD/src/virtualhome_eval/log_config.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3fd0ee4a843d495d","mcp_get_code":{"code_sha256":"3fd0ee4a843d495d"}},{"arxiv_id":"2410.01560","paper":"/paper/openmathinstruct-2-accelerating-ai-for-math","title":"OpenMathInstruct-2: Accelerating AI for Math with Massive Open-Source Instruction Data","date":"2024-10-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"NVIDIA/NeMo-Skills","path":"nemo_skills/utils.py","file_url":"https://github.com/NVIDIA/NeMo-Skills/blob/HEAD/nemo_skills/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"4261064eb4ec7756","mcp_get_code":{"code_sha256":"4261064eb4ec7756"}},{"arxiv_id":"2409.12105","paper":"/paper/fedlf-adaptive-logit-adjustment-and-feature","title":"FedLF: Adaptive Logit Adjustment and Feature Optimization in Federated Long-Tailed Learning","date":"2024-09-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"18sym/FedLF","path":"algorithm/fedlf.py","file_url":"https://github.com/18sym/FedLF/blob/HEAD/algorithm/fedlf.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"4ebf5ed2cce22713","mcp_get_code":{"code_sha256":"4ebf5ed2cce22713"}},{"arxiv_id":"2407.02490","paper":"/paper/minference-1-0-accelerating-pre-filling-for","title":"MInference 1.0: Accelerating Pre-filling for Long-Context LLMs via Dynamic Sparse Attention","date":"2024-07-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/LLMLingua","path":"experiments/securitylingua/label_word.py","file_url":"https://github.com/microsoft/LLMLingua/blob/HEAD/experiments/securitylingua/label_word.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"4e369dc3b4632635","mcp_get_code":{"code_sha256":"4e369dc3b4632635"}},{"arxiv_id":"2407.01489","paper":"/paper/agentless-demystifying-llm-based-software","title":"Agentless: Demystifying LLM-based Software Engineering Agents","date":"2024-07-01","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"sorendunn/agentless-lite","path":"agentless_lite/util/logging.py","file_url":"https://github.com/sorendunn/agentless-lite/blob/HEAD/agentless_lite/util/logging.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c9355b8fec2e5c50","mcp_get_code":{"code_sha256":"c9355b8fec2e5c50"}},{"arxiv_id":"2406.16828","paper":"/paper/ragnarok-a-reusable-rag-framework-and","title":"Ragnarök: A Reusable RAG Framework and Baselines for TREC 2024 Retrieval-Augmented Generation Track","date":"2024-06-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"castorini/nuggetizer","path":"src/nuggetizer/cli/logging_utils.py","file_url":"https://github.com/castorini/nuggetizer/blob/HEAD/src/nuggetizer/cli/logging_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"b74163bc3735c7ee","mcp_get_code":{"code_sha256":"b74163bc3735c7ee"}},{"arxiv_id":"2406.02347","paper":"/paper/flash-diffusion-accelerating-any-conditional","title":"Flash Diffusion: Accelerating Any Conditional Diffusion Model for Few Steps Image Generation","date":"2024-06-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gojasper/flash-diffusion","path":"src/flash/trainer/utils.py","file_url":"https://github.com/gojasper/flash-diffusion/blob/HEAD/src/flash/trainer/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"a03a5220c664637d","mcp_get_code":{"code_sha256":"a03a5220c664637d"}},{"arxiv_id":"2402.17840","paper":"/paper/follow-my-instruction-and-spill-the-beans","title":"Follow My Instruction and Spill the Beans: Scalable Data Extraction from Retrieval-Augmented Generation Systems","date":"2024-02-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zhentingqi/rag-privacy","path":"utils/helpers.py","file_url":"https://github.com/zhentingqi/rag-privacy/blob/HEAD/utils/helpers.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"44a120baec5809ce","mcp_get_code":{"code_sha256":"44a120baec5809ce"}},{"arxiv_id":"2402.17427","paper":"/paper/vastgaussian-vast-3d-gaussians-for-large","title":"VastGaussian: Vast 3D Gaussians for Large Scene Reconstruction","date":"2024-02-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kangpeilun/VastGaussian","path":"train_vast.py","file_url":"https://github.com/kangpeilun/VastGaussian/blob/HEAD/train_vast.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ab6a64fe2fcad5f5","mcp_get_code":{"code_sha256":"ab6a64fe2fcad5f5"}},{"arxiv_id":"2212.02623","paper":"/paper/unifying-vision-text-and-layout-for-universal","title":"Unifying Vision, Text, and Layout for Universal Document Processing","date":"2022-12-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"DS4SD/MarkushGrapher","path":"markushgrapher/core/common/begin.py","file_url":"https://github.com/DS4SD/MarkushGrapher/blob/HEAD/markushgrapher/core/common/begin.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"fbd45e87bae891a7","mcp_get_code":{"code_sha256":"fbd45e87bae891a7"}},{"arxiv_id":"2210.10318","paper":"/paper/gaussian-bernoulli-rbms-without-tears","title":"Gaussian-Bernoulli RBMs Without Tears","date":"2022-10-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lrjconan/grbm","path":"utils.py","file_url":"https://github.com/lrjconan/grbm/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c231d83496edbc85","mcp_get_code":{"code_sha256":"c231d83496edbc85"}},{"arxiv_id":"2209.15458","paper":"/paper/towards-general-purpose-representation","title":"Towards General-Purpose Representation Learning of Polygonal Geometries","date":"2022-09-29","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gengchenmai/polygon_encoder","path":"polygoncode/polygonembed/model_utils.py","file_url":"https://github.com/gengchenmai/polygon_encoder/blob/HEAD/polygoncode/polygonembed/model_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"cbe25e27d6cdb511","mcp_get_code":{"code_sha256":"cbe25e27d6cdb511"}},{"arxiv_id":"2008.04968","paper":"/paper/campus3d-a-photogrammetry-point-cloud","title":"Campus3D: A Photogrammetry Point Cloud Benchmark for Hierarchical Understanding of Outdoor Scene","date":"2020-08-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"shinke-li/Campus3D","path":"utils/logger.py","file_url":"https://github.com/shinke-li/Campus3D/blob/HEAD/utils/logger.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d55587fc3903827a","mcp_get_code":{"code_sha256":"d55587fc3903827a"}},{"arxiv_id":"1910.00760","paper":"/paper/efficient-graph-generation-with-graph","title":"Efficient Graph Generation with Graph Recurrent Attention Networks","date":"2019-10-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lrjconan/GRAN","path":"utils/logger.py","file_url":"https://github.com/lrjconan/GRAN/blob/HEAD/utils/logger.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c231d83496edbc85","mcp_get_code":{"code_sha256":"c231d83496edbc85"}},{"arxiv_id":"1812.10907","paper":"/paper/divergence-triangle-for-joint-training-of","title":"Divergence Triangle for Joint Training of Generator Model, Energy-based Model, and Inference Model","date":"2018-12-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"enijkamp/triangle","path":"train_cifar10.py","file_url":"https://github.com/enijkamp/triangle/blob/HEAD/train_cifar10.py","status":"ran_draft_wrong","verification_level":1,"contract_check":"OUTPUT_MISDECLARED","metamorphic_tier":"deterministic","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6d587536da0a8871","mcp_get_code":{"code_sha256":"6d587536da0a8871"}},{"arxiv_id":"astro-ph/0604362","paper":"/paper/improving-cosmological-distance-measurements","title":"Improving Cosmological Distance Measurements by Reconstruction of the Baryon Acoustic Peak","date":"2006-04-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cosmodesi/pyrecon","path":"pyrecon/utils.py","file_url":"https://github.com/cosmodesi/pyrecon/blob/HEAD/pyrecon/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-3-Clause","inline_ok":true,"code_sha256_prefix":"0ff480001400d0a1","mcp_get_code":{"code_sha256":"0ff480001400d0a1"}},{"arxiv_id":"2025.findings-acl.193","paper":null,"title":"arXiv:2025.findings-acl.193","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"THUDM/SWE-Dev","path":"swedev/testcases/eval_testcases.py","file_url":"https://github.com/THUDM/SWE-Dev/blob/HEAD/swedev/testcases/eval_testcases.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"97af3d63f70bfc2d","mcp_get_code":{"code_sha256":"97af3d63f70bfc2d"}},{"arxiv_id":"2025.findings-acl.193","paper":null,"title":"arXiv:2025.findings-acl.193","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"THUDM/SWE-Dev","path":"swedev/testcases/get_testcases.py","file_url":"https://github.com/THUDM/SWE-Dev/blob/HEAD/swedev/testcases/get_testcases.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"bd5fe9b3296159c6","mcp_get_code":{"code_sha256":"bd5fe9b3296159c6"}},{"arxiv_id":"2024.findings-naacl.46","paper":null,"title":"arXiv:2024.findings-naacl.46","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"knalin55/LEEETs-Dial","path":"utils.py","file_url":"https://github.com/knalin55/LEEETs-Dial/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"b41c011e7d29125e","mcp_get_code":{"code_sha256":"b41c011e7d29125e"}},{"arxiv_id":"2024.findings-emnlp.232","paper":null,"title":"arXiv:2024.findings-emnlp.232","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"knediny/mABC","path":"utils/logger.py","file_url":"https://github.com/knediny/mABC/blob/HEAD/utils/logger.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e93129c49cbfcad4","mcp_get_code":{"code_sha256":"e93129c49cbfcad4"}}]}