{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/count-tokens","entry":"count_tokens","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":39,"n_papers_ran":11,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":39,"n_samples_ran":10,"n_samples_fingerprinted":4,"n_places":41,"n_places_pointer_only":10,"by_status":{"ran_honours":2,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":8,"unverified":29},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.15160","paper":"/paper/arxiv-2609-15160","title":"TOKEN-LEVEL ELASTIC-DEPTH LOOPED TRANSFORMERS FOR LATENT REASONING WITH DYNAMIC ROUTING UNDER REVIEW AT ICLR 2027","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"YuMingQian1234/T-LoopFormer","path":"utils.py","file_url":"https://github.com/YuMingQian1234/T-LoopFormer/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8cc31a6638b3633f","mcp_get_code":{"code_sha256":"8cc31a6638b3633f"}},{"arxiv_id":"2604.20860","paper":"/paper/arxiv-2604-20860","title":"RealRoute: Dynamic Query Routing System via Retrieve-then-Verify Paradigm","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"Joseph1951210/RealRoute","path":"pipeline/subquery_executor.py","file_url":"https://github.com/Joseph1951210/RealRoute/blob/HEAD/pipeline/subquery_executor.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"ab165e30f271182c","mcp_get_code":{"code_sha256":"ab165e30f271182c"}},{"arxiv_id":"2604.18779","paper":"/paper/arxiv-2604-18779","title":"MANGO: Multi-Agent Web Navigation via Global-View Optimization","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"VichyTong/Mango","path":"llm_web_scraper/utils/llm.py","file_url":"https://github.com/VichyTong/Mango/blob/HEAD/llm_web_scraper/utils/llm.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"04d9feb321c92fde","mcp_get_code":{"code_sha256":"04d9feb321c92fde"}},{"arxiv_id":"2603.25333","paper":"/paper/arxiv-2603-25333","title":"Adaptive Chunking: Optimizing Chunking-Method Selection for RAG","date":null,"month_inferred_from_arxiv_id":"2026-03","title_source":"syntology","repo":"ekimetrics/adaptive-chunking","path":"src/adaptive_chunking/splitters.py","file_url":"https://github.com/ekimetrics/adaptive-chunking/blob/HEAD/src/adaptive_chunking/splitters.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"936a7e0cee9f7a1c","mcp_get_code":{"code_sha256":"936a7e0cee9f7a1c"}},{"arxiv_id":"2602.03689","paper":"/paper/arxiv-2602-03689","title":"Rethinking the Reranker: Boundary-Aware Evidence Selection for Robust Retrieval-Augmented Generation","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"GasolSun36/BAR-RAG","path":"filter.py","file_url":"https://github.com/GasolSun36/BAR-RAG/blob/HEAD/filter.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0335852d729ef632","mcp_get_code":{"code_sha256":"0335852d729ef632"}},{"arxiv_id":"2601.19578","paper":"/paper/arxiv-2601-19578","title":"Yunque DeepResearch Technical Report","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"Tencent-BAC/YunqueAgent","path":"inference/base_tool.py","file_url":"https://github.com/Tencent-BAC/YunqueAgent/blob/HEAD/inference/base_tool.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"04c16d8f22c5eaad","mcp_get_code":{"code_sha256":"04c16d8f22c5eaad"}},{"arxiv_id":"2601.10960","paper":"/paper/arxiv-2601-10960","title":"Steering Language Models Before They Speak: Logit-Level Interventions","date":null,"month_inferred_from_arxiv_id":"2026-01","title_source":"syntology","repo":"hsannn/swai","path":"build_scores.py","file_url":"https://github.com/hsannn/swai/blob/HEAD/build_scores.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"c1b0786d78cff5bb","mcp_get_code":{"code_sha256":"c1b0786d78cff5bb"}},{"arxiv_id":"2509.16112","paper":"/paper/arxiv-2509-16112","title":"CodeRAG: Finding Relevant and Necessary Knowledge for Retrieval-Augmented Repository-Level Code Completion","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"KDEGroup/CodeRAG","path":"coderag/build_prompt/merge_retrieval.py","file_url":"https://github.com/KDEGroup/CodeRAG/blob/HEAD/coderag/build_prompt/merge_retrieval.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c1a55b16a582d56f","mcp_get_code":{"code_sha256":"c1a55b16a582d56f"}},{"arxiv_id":"2509.15786","paper":"/paper/arxiv-2509-15786","title":"Building Data-Driven Occupation Taxonomies: A Bottom-Up Multi-Stage Approach via Semantic Clustering and Multi-Agent Collaboration","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"aida-ugent/CLIMB","path":"src/get_prompts.py","file_url":"https://github.com/aida-ugent/CLIMB/blob/HEAD/src/get_prompts.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NOASSERTION","inline_ok":false,"code_sha256_prefix":"98bfdf443dbd6f66","mcp_get_code":{"code_sha256":"98bfdf443dbd6f66"}},{"arxiv_id":"2509.14671","paper":"/paper/arxiv-2509-14671","title":"TableDART: Dynamic Adaptive Multi-Modal Routing for Table Understanding","date":null,"month_inferred_from_arxiv_id":"2025-09","title_source":"syntology","repo":"xiaobo-xing/TableDART","path":"cost_measurement/measure_expert_costs.py","file_url":"https://github.com/xiaobo-xing/TableDART/blob/HEAD/cost_measurement/measure_expert_costs.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"2296fe2308655cb0","mcp_get_code":{"code_sha256":"2296fe2308655cb0"}},{"arxiv_id":"2507.02592","paper":"/paper/websailor-navigating-super-human-reasoning","title":"WebSailor: Navigating Super-human Reasoning for Web Agent","date":"2025-07-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alibaba-nlp/webagent","path":"WebAgent/NestBrowse/utils.py","file_url":"https://github.com/alibaba-nlp/webagent/blob/HEAD/WebAgent/NestBrowse/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"23aacba017665d93","mcp_get_code":{"code_sha256":"23aacba017665d93"}},{"arxiv_id":"2504.15466","paper":"/paper/learning-adaptive-parallel-reasoning-with","title":"Learning Adaptive Parallel Reasoning with Language Models","date":"2025-04-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"parallel-reasoning/apr","path":"src/eval/eval_sosp.py","file_url":"https://github.com/parallel-reasoning/apr/blob/HEAD/src/eval/eval_sosp.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"56c7087a3cd9e36e","mcp_get_code":{"code_sha256":"56c7087a3cd9e36e"}},{"arxiv_id":"2504.12764","paper":"/paper/graphomni-a-comprehensive-and-extendable","title":"GraphOmni: A Comprehensive and Extendable Benchmark Framework for Large Language Models on Graph-theoretic Tasks","date":"2025-04-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gai-community/graphomni","path":"eval_fun/token.py","file_url":"https://github.com/gai-community/graphomni/blob/HEAD/eval_fun/token.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"c4fa9aad367464fa","mcp_get_code":{"code_sha256":"c4fa9aad367464fa"}},{"arxiv_id":"2504.08672","paper":"/paper/genius-a-generalizable-and-purely","title":"Genius: A Generalizable and Purely Unsupervised Self-Training Framework For Advanced Reasoning","date":"2025-04-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"chang-github-00/llm-predictive-decoding","path":"agentboard/algorithms/mpc_sampling.py","file_url":"https://github.com/chang-github-00/llm-predictive-decoding/blob/HEAD/agentboard/algorithms/mpc_sampling.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"94846e60d6f6c92a","mcp_get_code":{"code_sha256":"94846e60d6f6c92a"}},{"arxiv_id":"2502.18754","paper":"/paper/agentsociety-challenge-designing-llm-agents","title":"AgentSociety Challenge: Designing LLM Agents for User Modeling and Recommendation on Web Platforms","date":"2025-02-26","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tsinghua-fib-lab/agentsocietychallenge","path":"GTsimulation/ModGTAgent.py","file_url":"https://github.com/tsinghua-fib-lab/agentsocietychallenge/blob/HEAD/GTsimulation/ModGTAgent.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"766aece7c07a8cae","mcp_get_code":{"code_sha256":"766aece7c07a8cae"}},{"arxiv_id":"2502.00757","paper":"/paper/agentbreeder-mitigating-the-ai-safety-impact","title":"AgentBreeder: Mitigating the AI Safety Impact of Multi-Agent Scaffolds via Self-Improvement","date":"2025-02-02","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"J-Rosser-UK/AgentBreeder","path":"src/api/anthropic_api.py","file_url":"https://github.com/J-Rosser-UK/AgentBreeder/blob/HEAD/src/api/anthropic_api.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"1d4a9efbfddea497","mcp_get_code":{"code_sha256":"1d4a9efbfddea497"}},{"arxiv_id":"2501.04227","paper":"/paper/agent-laboratory-using-llm-agents-as-research","title":"Agent Laboratory: Using LLM Agents as Research Assistants","date":"2025-01-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Masao-Taketani/LocalAgentLaboratory","path":"utils.py","file_url":"https://github.com/Masao-Taketani/LocalAgentLaboratory/blob/HEAD/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"023b788084b451b0","mcp_get_code":{"code_sha256":"023b788084b451b0"}},{"arxiv_id":"2412.08819","paper":"/paper/harp-a-challenging-human-annotated-math","title":"HARP: A challenging human-annotated math reasoning benchmark","date":"2024-12-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"aadityasingh/harp","path":"src/eval/costs.py","file_url":"https://github.com/aadityasingh/harp/blob/HEAD/src/eval/costs.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"da52349b06d44177","mcp_get_code":{"code_sha256":"da52349b06d44177"}},{"arxiv_id":"2411.07336","paper":"/paper/setlexsem-challenge-using-set-operations-to","title":"SetLexSem Challenge: Using Set Operations to Evaluate the Lexical and Semantic Robustness of Language Models","date":"2024-11-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"amazon-science/setlexsem-challenge","path":"setlexsem/experiment/lmapi.py","file_url":"https://github.com/amazon-science/setlexsem-challenge/blob/HEAD/setlexsem/experiment/lmapi.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"912d808705da066a","mcp_get_code":{"code_sha256":"912d808705da066a"}},{"arxiv_id":"2411.02862","paper":"/paper/the-unreasonable-effectiveness-of-llms-for","title":"The Unreasonable Effectiveness of LLMs for Query Optimization","date":"2024-11-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"peter-ai/LLMSteer","path":"models/utils.py","file_url":"https://github.com/peter-ai/LLMSteer/blob/HEAD/models/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"936c5d54cec2c170","mcp_get_code":{"code_sha256":"936c5d54cec2c170"}},{"arxiv_id":"2410.14211","paper":"/paper/paths-over-graph-knowledge-graph-enpowered","title":"Paths-over-Graph: Knowledge Graph Empowered Large Language Model Reasoning","date":"2024-10-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"SteveTANTAN/PoG","path":"PoG/utils.py","file_url":"https://github.com/SteveTANTAN/PoG/blob/HEAD/PoG/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"dcb9359b01d13cbe","mcp_get_code":{"code_sha256":"dcb9359b01d13cbe"}},{"arxiv_id":"2408.12787","paper":"/paper/llm-pbe-assessing-data-privacy-in-large","title":"LLM-PBE: Assessing Data Privacy in Large Language Models","date":"2024-08-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"QinbinLi/LLM-PBE","path":"models/togetherai.py","file_url":"https://github.com/QinbinLi/LLM-PBE/blob/HEAD/models/togetherai.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a5fbf1725fd11160","mcp_get_code":{"code_sha256":"a5fbf1725fd11160"}},{"arxiv_id":"2408.09559","paper":"/paper/hiagent-hierarchical-working-memory","title":"HiAgent: Hierarchical Working Memory Management for Solving Long-Horizon Agent Tasks with Large Language Model","date":"2024-08-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hiagent2024/hiagent","path":"agentboard/agents/ours_agent.py","file_url":"https://github.com/hiagent2024/hiagent/blob/HEAD/agentboard/agents/ours_agent.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1e64761aa30b0c32","mcp_get_code":{"code_sha256":"1e64761aa30b0c32"}},{"arxiv_id":"2406.00050","paper":"/paper/an-empirical-analysis-on-large-language","title":"An Empirical Analysis on Large Language Models in Debate Evaluation","date":"2024-05-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"xinyiliu0227/llm_debate_bias","path":"ddo_baseline_binary_1-1.py","file_url":"https://github.com/xinyiliu0227/llm_debate_bias/blob/HEAD/ddo_baseline_binary_1-1.py","status":"ran_honours","verification_level":1,"contract_check":"HONOURS","metamorphic_tier":"well_formed","behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1e64761aa30b0c32","mcp_get_code":{"code_sha256":"1e64761aa30b0c32"}},{"arxiv_id":"2403.19827","paper":"/paper/language-models-learn-rare-phenomena-from","title":"Language Models Learn Rare Phenomena from Less Rare Phenomena: The Case of the Missing AANNs","date":"2024-03-28","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kanishkamisra/aannalysis","path":"src/counterfactual_constructions.py","file_url":"https://github.com/kanishkamisra/aannalysis/blob/HEAD/src/counterfactual_constructions.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a3a3f6ce577d2d24","mcp_get_code":{"code_sha256":"a3a3f6ce577d2d24"}},{"arxiv_id":"2403.16218","paper":"/paper/coverup-coverage-guided-llm-based-test","title":"CoverUp: Effective High Coverage Test Generation for Python","date":"2024-03-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"plasma-umass/coverup","path":"src/coverup/llm.py","file_url":"https://github.com/plasma-umass/coverup/blob/HEAD/src/coverup/llm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"145acc26153dc1f4","mcp_get_code":{"code_sha256":"145acc26153dc1f4"}},{"arxiv_id":"2403.14578","paper":"/paper/rambla-a-framework-for-evaluating-the","title":"RAmBLA: A Framework for Evaluating the Reliability of LLMs as Assistants in the Biomedical Domain","date":"2024-03-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"gsk-ai/rambla","path":"rambla/models/utils.py","file_url":"https://github.com/gsk-ai/rambla/blob/HEAD/rambla/models/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"a953ecb4b9b05c6a","mcp_get_code":{"code_sha256":"a953ecb4b9b05c6a"}},{"arxiv_id":"2402.06782","paper":"/paper/debating-with-more-persuasive-llms-leads-to","title":"Debating with More Persuasive LLMs Leads to More Truthful Answers","date":"2024-02-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ucl-dark/llm_debate","path":"core/llm_api/anthropic_llm.py","file_url":"https://github.com/ucl-dark/llm_debate/blob/HEAD/core/llm_api/anthropic_llm.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"56346bcde59d7a4a","mcp_get_code":{"code_sha256":"56346bcde59d7a4a"}},{"arxiv_id":"2402.06782","paper":"/paper/debating-with-more-persuasive-llms-leads-to","title":"Debating with More Persuasive LLMs Leads to More Truthful Answers","date":"2024-02-09","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ucl-dark/llm_debate","path":"core/llm_api/openai_llm.py","file_url":"https://github.com/ucl-dark/llm_debate/blob/HEAD/core/llm_api/openai_llm.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3b94d7f86b74da2b","mcp_get_code":{"code_sha256":"3b94d7f86b74da2b"}},{"arxiv_id":"2402.05128","paper":"/paper/enhancing-textbook-question-answering-task","title":"Enhancing textual textbook question answering with large language models and retrieval augmented generation","date":"2024-02-05","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hessaalawwad/plr-tqa","path":"RAG_Pinecone.py","file_url":"https://github.com/hessaalawwad/plr-tqa/blob/HEAD/RAG_Pinecone.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"529e6478e94acc7a","mcp_get_code":{"code_sha256":"529e6478e94acc7a"}},{"arxiv_id":"2401.09432","paper":"/paper/rolecraft-glm-advancing-personalized-role","title":"RoleCraft-GLM: Advancing Personalized Role-Playing in Large Language Models","date":"2023-12-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"tml2002/rolecraft","path":"code/prompt3.py","file_url":"https://github.com/tml2002/rolecraft/blob/HEAD/code/prompt3.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"584eb7879ebbd0c1","mcp_get_code":{"code_sha256":"584eb7879ebbd0c1"}},{"arxiv_id":"2311.09805","paper":"/paper/docmath-eval-evaluating-numerical-reasoning","title":"DocMath-Eval: Evaluating Math Reasoning Capabilities of LLMs in Understanding Long and Specialized Documents","date":"2023-11-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yale-nlp/docmath-eval","path":"utils/model_input_utils.py","file_url":"https://github.com/yale-nlp/docmath-eval/blob/HEAD/utils/model_input_utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"cc3cc5ecc9833707","mcp_get_code":{"code_sha256":"cc3cc5ecc9833707"}},{"arxiv_id":"2311.08588","paper":"/paper/codescope-an-execution-based-multilingual","title":"CodeScope: An Execution-based Multilingual Multitask Multidimensional Benchmark for Evaluating LLMs on Code Understanding and Generation","date":"2023-11-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"weixiangyan/codescope","path":"automated_testing/evaluator/score.py","file_url":"https://github.com/weixiangyan/codescope/blob/HEAD/automated_testing/evaluator/score.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"20f9439429346697","mcp_get_code":{"code_sha256":"20f9439429346697"}},{"arxiv_id":"2310.03951","paper":"/paper/chain-of-natural-language-inference-for","title":"Chain of Natural Language Inference for Reducing Large Language Model Ungrounded Hallucinations","date":"2023-10-06","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/conli_hallucination","path":"CoNLI/CoNLI/modules/hallucination_detector.py","file_url":"https://github.com/microsoft/conli_hallucination/blob/HEAD/CoNLI/CoNLI/modules/hallucination_detector.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":true,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"89115682f5989ce7","mcp_get_code":{"code_sha256":"89115682f5989ce7"}},{"arxiv_id":"2305.14879","paper":"/paper/bytesized32-a-corpus-and-challenge-task-for","title":"ByteSized32: A Corpus and Challenge Task for Generating Task-Specific World Models Expressed as Text Games","date":"2023-05-24","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"cognitiveailab/BYTESIZED32","path":"bytes32/utils.py","file_url":"https://github.com/cognitiveailab/BYTESIZED32/blob/HEAD/bytes32/utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"9e4c4702c63966aa","mcp_get_code":{"code_sha256":"9e4c4702c63966aa"}},{"arxiv_id":"2212.10218","paper":"/paper/ganlm-encoder-decoder-pre-training-with-an","title":"GanLM: Encoder-Decoder Pre-training with an Auxiliary Discriminator","date":"2022-12-20","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"csjianyang/ganlm","path":"evaluation/cnn_dm.py","file_url":"https://github.com/csjianyang/ganlm/blob/HEAD/evaluation/cnn_dm.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7d0f0d30d5c3027f","mcp_get_code":{"code_sha256":"7d0f0d30d5c3027f"}},{"arxiv_id":"1805.03294","paper":"/paper/improved-training-of-end-to-end-attention","title":"Improved training of end-to-end attention models for speech recognition","date":"2018-05-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"bobchennan/espnet","path":"espnet/lm/lm_utils.py","file_url":"https://github.com/bobchennan/espnet/blob/HEAD/espnet/lm/lm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"ff899e7de6121891","mcp_get_code":{"code_sha256":"ff899e7de6121891"}},{"arxiv_id":"1805.03294","paper":"/paper/improved-training-of-end-to-end-attention","title":"Improved training of end-to-end attention models for speech recognition","date":"2018-05-08","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"creatorscan/espnet","path":"espnet/lm/lm_utils.py","file_url":"https://github.com/creatorscan/espnet/blob/HEAD/espnet/lm/lm_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"e2309fb95e3f748c","mcp_get_code":{"code_sha256":"e2309fb95e3f748c"}},{"arxiv_id":"aaai_21432","paper":null,"title":"arXiv:aaai_21432","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"microsoft/DialogLM","path":"DialogLM_UniLM/DialogLM/evaluations/eval_for_cnndm.py","file_url":"https://github.com/microsoft/DialogLM/blob/HEAD/DialogLM_UniLM/DialogLM/evaluations/eval_for_cnndm.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"7d0f0d30d5c3027f","mcp_get_code":{"code_sha256":"7d0f0d30d5c3027f"}},{"arxiv_id":"2025.findings-emnlp.479","paper":null,"title":"arXiv:2025.findings-emnlp.479","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"FoundationAgents/SPO","path":"utils/evaluation_utils.py","file_url":"https://github.com/FoundationAgents/SPO/blob/HEAD/utils/evaluation_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3c48f41137637146","mcp_get_code":{"code_sha256":"3c48f41137637146"}},{"arxiv_id":"2024.findings-emnlp.123","paper":null,"title":"arXiv:2024.findings-emnlp.123","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"HKUST-KnowComp/IntentionQA","path":"filter/simplifyName.py","file_url":"https://github.com/HKUST-KnowComp/IntentionQA/blob/HEAD/filter/simplifyName.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f54f0325dfa091c7","mcp_get_code":{"code_sha256":"f54f0325dfa091c7"}}]}