{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/load-tasks","entry":"load_tasks","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":11,"n_papers_ran":3,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":10,"n_samples_ran":3,"n_samples_fingerprinted":0,"n_places":11,"n_places_pointer_only":4,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":3,"unverified":7},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.00759","paper":"/paper/arxiv-2609-00759","title":"Compile, Don't Memorize: A Context Compilation Architecture (CCA) for In-Context Learning","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"TonyQJH/cca-emnlp2026","path":"code/run_cca.py","file_url":"https://github.com/TonyQJH/cca-emnlp2026/blob/HEAD/code/run_cca.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"5bacf5e653c53635","mcp_get_code":{"code_sha256":"5bacf5e653c53635"}},{"arxiv_id":"2608.00718","paper":"/paper/arxiv-2608-00718","title":"Adversarial Attacks in Multi-Agent LLM Pipelines: Unveiling Structural Vulnerabilities in Agentic AI Architectures","date":null,"month_inferred_from_arxiv_id":"2026-08","title_source":"syntology","repo":"SPaDeS-Lab/adversarial-llm-pipeline","path":"tasks/loader.py","file_url":"https://github.com/SPaDeS-Lab/adversarial-llm-pipeline/blob/HEAD/tasks/loader.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"8fe9b92168f524b5","mcp_get_code":{"code_sha256":"8fe9b92168f524b5"}},{"arxiv_id":"2607.25765","paper":"/paper/arxiv-2607-25765","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"haolpku/WorkSurface-Bench","path":"worksurface/derive_tasks.py","file_url":"https://github.com/haolpku/WorkSurface-Bench/blob/HEAD/worksurface/derive_tasks.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"1db4d10fb1273ea4","mcp_get_code":{"code_sha256":"1db4d10fb1273ea4"}},{"arxiv_id":"2607.23624","paper":"/paper/arxiv-2607-23624","title":"Where Is the Cost of Third-Party API Routers in Agentic Software Development?","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"Riyasushin/SIDEL","path":"manager/orchestrator.py","file_url":"https://github.com/Riyasushin/SIDEL/blob/HEAD/manager/orchestrator.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"bac9eb863a9cb9c7","mcp_get_code":{"code_sha256":"bac9eb863a9cb9c7"}},{"arxiv_id":"2607.18084","paper":"/paper/arxiv-2607-18084","title":"WorldCupArena: Fine-Grained Evaluation of Language Models and Deep-Research Agents on Football Forecasting","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"wzk1015/WorldCupArena","path":"src/graders/grade_match.py","file_url":"https://github.com/wzk1015/WorldCupArena/blob/HEAD/src/graders/grade_match.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8e7158f5b28feeac","mcp_get_code":{"code_sha256":"8e7158f5b28feeac"}},{"arxiv_id":"2604.15715","paper":"/paper/arxiv-2604-15715","title":"GTA-2: Benchmarking General Tool Agents from Atomic Tool-Use to Open-Ended Workflows","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"open-compass/GTA","path":"agent_app_eval/run_agents.py","file_url":"https://github.com/open-compass/GTA/blob/HEAD/agent_app_eval/run_agents.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"3ab02673548e9f4a","mcp_get_code":{"code_sha256":"3ab02673548e9f4a"}},{"arxiv_id":"2602.20156","paper":"/paper/arxiv-2602-20156","title":"SKILL-INJECT: Measuring Agent Vulnerability to Skill File Attacks","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"aisa-group/skill-inject","path":"judges/utility_baseline_judge.py","file_url":"https://github.com/aisa-group/skill-inject/blob/HEAD/judges/utility_baseline_judge.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"39d3d3516b652e6f","mcp_get_code":{"code_sha256":"39d3d3516b652e6f"}},{"arxiv_id":"2602.19672","paper":"/paper/arxiv-2602-19672","title":"SkillOrchestra: Learning to Route Agents via Skill Transfer","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"jiayuww/SkillOrchestra","path":"skillorchestra/converters/from_stage_router.py","file_url":"https://github.com/jiayuww/SkillOrchestra/blob/HEAD/skillorchestra/converters/from_stage_router.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"fbc7eabf139be7d4","mcp_get_code":{"code_sha256":"fbc7eabf139be7d4"}},{"arxiv_id":"2402.08178","paper":"/paper/lota-bench-benchmarking-language-oriented","title":"LoTa-Bench: Benchmarking Language-oriented Task Planners for Embodied Agents","date":"2024-02-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"lbaa2022/LLMTaskPlanning","path":"src/alfred/examine_alfred_data.py","file_url":"https://github.com/lbaa2022/LLMTaskPlanning/blob/HEAD/src/alfred/examine_alfred_data.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1cf7812afecd6dea","mcp_get_code":{"code_sha256":"1cf7812afecd6dea"}},{"arxiv_id":"2202.05983","paper":"/paper/uncalibrated-models-can-improve-human-ai","title":"Uncalibrated Models Can Improve Human-AI Collaboration","date":"2022-02-12","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kailas-v/human-ai-interactions","path":"haiid.py","file_url":"https://github.com/kailas-v/human-ai-interactions/blob/HEAD/haiid.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a6b4ca4905abac6d","mcp_get_code":{"code_sha256":"a6b4ca4905abac6d"}},{"arxiv_id":"2107.07015","paper":"/paper/do-humans-trust-advice-more-if-it-comes-from","title":"Do Humans Trust Advice More if it Comes from AI? An Analysis of Human-AI Interactions","date":"2021-07-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":null,"path":"","file_url":null,"status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":null,"inline_ok":false,"code_sha256_prefix":"a6b4ca4905abac6d","mcp_get_code":{"code_sha256":"a6b4ca4905abac6d"}}]}