{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/get-task","entry":"get_task","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":28,"n_papers_ran":1,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":14,"n_samples_ran":1,"n_samples_fingerprinted":0,"n_places":28,"n_places_pointer_only":8,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":1,"unverified":13},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2609.12265","paper":"/paper/arxiv-2609-12265","title":"GTA: Graph Theory Agent and Benchmark for Algorithmic Graph Reasoning with LLMs","date":null,"month_inferred_from_arxiv_id":"2026-09","title_source":"syntology","repo":"xzx34/GTA","path":"src/gtbench/tasks.py","file_url":"https://github.com/xzx34/GTA/blob/HEAD/src/gtbench/tasks.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"61d7b30722025b3d","mcp_get_code":{"code_sha256":"61d7b30722025b3d"}},{"arxiv_id":"2607.07847","paper":"/paper/arxiv-2607-07847","title":"When Does Continual Learning Require Learning","date":null,"month_inferred_from_arxiv_id":"2026-07","title_source":"syntology","repo":"anneharrington/studying-cl","path":"cl/tasks.py","file_url":"https://github.com/anneharrington/studying-cl/blob/HEAD/cl/tasks.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f387a906d3df00a6","mcp_get_code":{"code_sha256":"f387a906d3df00a6"}},{"arxiv_id":"2604.21454","paper":"/paper/arxiv-2604-21454","title":"Reasoning Primitives in Hybrid and Non-Hybrid LLMs: Do Architectural Differences Yield Advantages in State-Tracking and Recall?","date":null,"month_inferred_from_arxiv_id":"2026-04","title_source":"syntology","repo":"ultor1996/reasoning_primitives","path":"src/templates.py","file_url":"https://github.com/ultor1996/reasoning_primitives/blob/HEAD/src/templates.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"3a0a352c2bb0f352","mcp_get_code":{"code_sha256":"3a0a352c2bb0f352"}},{"arxiv_id":"2602.17155","paper":"/paper/arxiv-2602-17155","title":"Powering Up Zeroth-Order Training via Subspace Gradient Orthogonalization","date":null,"month_inferred_from_arxiv_id":"2026-02","title_source":"syntology","repo":"OPTML-Group/ZO-Muon","path":"llm/tasks.py","file_url":"https://github.com/OPTML-Group/ZO-Muon/blob/HEAD/llm/tasks.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f7c384f1e681ea7a","mcp_get_code":{"code_sha256":"f7c384f1e681ea7a"}},{"arxiv_id":"2505.13430","paper":"/paper/fine-tuning-quantized-neural-networks-with","title":"Fine-tuning Quantized Neural Networks with Zeroth-order Optimization","date":"2025-05-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"maifoundations/qzo","path":"large_language_models/tasks.py","file_url":"https://github.com/maifoundations/qzo/blob/HEAD/large_language_models/tasks.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f7c384f1e681ea7a","mcp_get_code":{"code_sha256":"f7c384f1e681ea7a"}},{"arxiv_id":"2412.11499","paper":"/paper/embodied-cot-distillation-from-llm-to-off-the","title":"Embodied CoT Distillation From LLM To Off-the-shelf Agents","date":"2024-12-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"osu-nlp-group/llm-planner","path":"e2e/alfred/env/tasks.py","file_url":"https://github.com/osu-nlp-group/llm-planner/blob/HEAD/e2e/alfred/env/tasks.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5ffac7b83a094577","mcp_get_code":{"code_sha256":"5ffac7b83a094577"}},{"arxiv_id":"2410.08989","paper":"/paper/subzero-random-subspace-zeroth-order","title":"Zeroth-Order Fine-Tuning of LLMs in Random Subspaces","date":"2024-10-11","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zimingyy/subzero","path":"large_models/tasks.py","file_url":"https://github.com/zimingyy/subzero/blob/HEAD/large_models/tasks.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"f7c384f1e681ea7a","mcp_get_code":{"code_sha256":"f7c384f1e681ea7a"}},{"arxiv_id":"2406.01382","paper":"/paper/do-large-language-models-perform-the-way","title":"Do Large Language Models Perform the Way People Expect? Measuring the Human Generalization Function","date":"2024-06-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"keyonvafa/human-generalization-llms","path":"utils.py","file_url":"https://github.com/keyonvafa/human-generalization-llms/blob/HEAD/utils.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0334516a40523434","mcp_get_code":{"code_sha256":"0334516a40523434"}},{"arxiv_id":"2406.01006","paper":"/paper/semcoder-training-code-language-models-with","title":"SemCoder: Training Code Language Models with Comprehensive Semantics Reasoning","date":"2024-06-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ARiSE-Lab/SemCoder","path":"experiments/cruxeval_utils.py","file_url":"https://github.com/ARiSE-Lab/SemCoder/blob/HEAD/experiments/cruxeval_utils.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"0ea1d7ce8d9d3e2a","mcp_get_code":{"code_sha256":"0ea1d7ce8d9d3e2a"}},{"arxiv_id":"2406.00132","paper":"/paper/quanta-efficient-high-rank-fine-tuning-of","title":"QuanTA: Efficient High-Rank Fine-Tuning of LLMs with Quantum-Informed Tensor Adaptation","date":"2024-05-31","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"quanta-fine-tuning/quanta","path":"run/utils/tasks.py","file_url":"https://github.com/quanta-fine-tuning/quanta/blob/HEAD/run/utils/tasks.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"f7c384f1e681ea7a","mcp_get_code":{"code_sha256":"f7c384f1e681ea7a"}},{"arxiv_id":"2402.17453","paper":"/paper/ds-agent-automated-data-science-by-empowering","title":"DS-Agent: Automated Data Science by Empowering Large Language Models with Case-Based Reasoning","date":"2024-02-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"guosyjlu/DS-Agent","path":"deployment/prompt.py","file_url":"https://github.com/guosyjlu/DS-Agent/blob/HEAD/deployment/prompt.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"a45eb6f9978ac8b0","mcp_get_code":{"code_sha256":"a45eb6f9978ac8b0"}},{"arxiv_id":"2402.11592","paper":"/paper/revisiting-zeroth-order-optimization-for","title":"Revisiting Zeroth-Order Optimization for Memory-Efficient LLM Fine-Tuning: A Benchmark","date":"2024-02-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"zo-bench/zo-llm","path":"zo-bench/tasks.py","file_url":"https://github.com/zo-bench/zo-llm/blob/HEAD/zo-bench/tasks.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"GPL-3.0","inline_ok":false,"code_sha256_prefix":"f7c384f1e681ea7a","mcp_get_code":{"code_sha256":"f7c384f1e681ea7a"}},{"arxiv_id":"2312.15184","paper":"/paper/zo-adamu-optimizer-adapting-perturbation-by","title":"ZO-AdaMU Optimizer: Adapting Perturbation by the Momentum and Uncertainty in Zeroth-order Optimization","date":"2023-12-23","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"mathisall/zo-adamu","path":"tasks.py","file_url":"https://github.com/mathisall/zo-adamu/blob/HEAD/tasks.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"f7c384f1e681ea7a","mcp_get_code":{"code_sha256":"f7c384f1e681ea7a"}},{"arxiv_id":"2310.12344","paper":"/paper/lacma-language-aligning-contrastive-learning","title":"LACMA: Language-Aligning Contrastive Learning with Meta-Actions for Embodied Instruction Following","date":"2023-10-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"joeyy5588/LACMA","path":"alfred/env/tasks.py","file_url":"https://github.com/joeyy5588/LACMA/blob/HEAD/alfred/env/tasks.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d11401259503326f","mcp_get_code":{"code_sha256":"d11401259503326f"}},{"arxiv_id":"2310.09639","paper":"/paper/dpzero-dimension-independent-and","title":"DPZero: Private Fine-Tuning of Language Models without Backpropagation","date":"2023-10-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Liang137/DPZero","path":"opt/src/tasks.py","file_url":"https://github.com/Liang137/DPZero/blob/HEAD/opt/src/tasks.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f7c384f1e681ea7a","mcp_get_code":{"code_sha256":"f7c384f1e681ea7a"}},{"arxiv_id":"2308.09387","paper":"/paper/multi-level-compositional-reasoning-for","title":"Multi-Level Compositional Reasoning for Interactive Instruction Following","date":"2023-08-18","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yonseivnl/mcr-agent","path":"env/tasks.py","file_url":"https://github.com/yonseivnl/mcr-agent/blob/HEAD/env/tasks.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"5ffac7b83a094577","mcp_get_code":{"code_sha256":"5ffac7b83a094577"}},{"arxiv_id":"2305.17333","paper":"/paper/fine-tuning-language-models-with-just-forward-1","title":"Fine-Tuning Language Models with Just Forward Passes","date":"2023-05-27","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"princeton-nlp/mezo","path":"large_models/tasks.py","file_url":"https://github.com/princeton-nlp/mezo/blob/HEAD/large_models/tasks.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"f7c384f1e681ea7a","mcp_get_code":{"code_sha256":"f7c384f1e681ea7a"}},{"arxiv_id":"2206.06614","paper":"/paper/transformers-are-meta-reinforcement-learners-1","title":"Transformers are Meta-Reinforcement Learners","date":"2022-06-14","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"luckeciano/transformers-metarl","path":"benchmarks/src/garage_benchmarks/benchmarks.py","file_url":"https://github.com/luckeciano/transformers-metarl/blob/HEAD/benchmarks/src/garage_benchmarks/benchmarks.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8574703c48d0d43d","mcp_get_code":{"code_sha256":"8574703c48d0d43d"}},{"arxiv_id":"2106.03427","paper":"/paper/hierarchical-task-learning-from-language","title":"Hierarchical Task Learning from Language Instructions with Unified Transformers and Self-Monitoring","date":"2021-06-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"594zyc/HiTUT","path":"env/tasks.py","file_url":"https://github.com/594zyc/HiTUT/blob/HEAD/env/tasks.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5ffac7b83a094577","mcp_get_code":{"code_sha256":"5ffac7b83a094577"}},{"arxiv_id":"2105.06453","paper":"/paper/episodic-transformer-for-vision-and-language","title":"Episodic Transformer for Vision-and-Language Navigation","date":"2021-05-13","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"alexpashevich/E.T.","path":"alfred/env/tasks.py","file_url":"https://github.com/alexpashevich/E.T./blob/HEAD/alfred/env/tasks.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"d11401259503326f","mcp_get_code":{"code_sha256":"d11401259503326f"}},{"arxiv_id":"2006.04061","paper":"/paper/dual-policy-distillation","title":"Dual Policy Distillation","date":"2020-06-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"kiminh/dual-policy-distillation","path":"baselines/bench/benchmarks.py","file_url":"https://github.com/kiminh/dual-policy-distillation/blob/HEAD/baselines/bench/benchmarks.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"3e82b9de30b1625f","mcp_get_code":{"code_sha256":"3e82b9de30b1625f"}},{"arxiv_id":"2001.03415","paper":"/paper/multi-agent-interactions-modeling-with-1","title":"Multi-Agent Interactions Modeling with Correlated Policies","date":"2020-01-04","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"apexrl/codail","path":"multi-agent-irl/rl/bench/benchmarks.py","file_url":"https://github.com/apexrl/codail/blob/HEAD/multi-agent-irl/rl/bench/benchmarks.py","status":"unverified","verification_level":0,"contract_check":"MISDECLARED","metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"7e1a88346f1d7ae7","mcp_get_code":{"code_sha256":"7e1a88346f1d7ae7"}},{"arxiv_id":"1912.01734","paper":"/paper/alfred-a-benchmark-for-interpreting-grounded","title":"ALFRED: A Benchmark for Interpreting Grounded Instructions for Everyday Tasks","date":"2019-12-03","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"askforalfred/alfred","path":"env/tasks.py","file_url":"https://github.com/askforalfred/alfred/blob/HEAD/env/tasks.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"5ffac7b83a094577","mcp_get_code":{"code_sha256":"5ffac7b83a094577"}},{"arxiv_id":"1905.02363","paper":"/paper/dimension-wise-importance-sampling-weight","title":"Dimension-Wise Importance Sampling Weight Clipping for Sample-Efficient Reinforcement Learning","date":"2019-05-07","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"seungyulhan/disc","path":"baselines/bench/benchmarks.py","file_url":"https://github.com/seungyulhan/disc/blob/HEAD/baselines/bench/benchmarks.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":false,"code_sha256_prefix":"3e82b9de30b1625f","mcp_get_code":{"code_sha256":"3e82b9de30b1625f"}},{"arxiv_id":"1606.03476","paper":"/paper/generative-adversarial-imitation-learning","title":"Generative Adversarial Imitation Learning","date":"2016-06-10","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"Techget/gail-tf-sc2","path":"gailtf/baselines/bench/benchmarks.py","file_url":"https://github.com/Techget/gail-tf-sc2/blob/HEAD/gailtf/baselines/bench/benchmarks.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"3e82b9de30b1625f","mcp_get_code":{"code_sha256":"3e82b9de30b1625f"}},{"arxiv_id":"1604.06778","paper":"/paper/benchmarking-deep-reinforcement-learning-for","title":"Benchmarking Deep Reinforcement Learning for Continuous Control","date":"2016-04-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"rlworkgroup/garage","path":"benchmarks/src/garage_benchmarks/benchmarks.py","file_url":"https://github.com/rlworkgroup/garage/blob/HEAD/benchmarks/src/garage_benchmarks/benchmarks.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"8574703c48d0d43d","mcp_get_code":{"code_sha256":"8574703c48d0d43d"}},{"arxiv_id":"1301.3781","paper":"/paper/efficient-estimation-of-word-representations","title":"Efficient Estimation of Word Representations in Vector Space","date":"2013-01-16","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ringbdstack/socialed","path":"SocialED/detector/hcrc.py","file_url":"https://github.com/ringbdstack/socialed/blob/HEAD/SocialED/detector/hcrc.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"BSD-2-Clause","inline_ok":true,"code_sha256_prefix":"410fc5e1b2944e14","mcp_get_code":{"code_sha256":"410fc5e1b2944e14"}},{"arxiv_id":"2025.findings-emnlp.176","paper":null,"title":"arXiv:2025.findings-emnlp.176","date":null,"month_inferred_from_arxiv_id":null,"title_source":null,"repo":"ContextualAI/LMUnit","path":"lmunit/tasks.py","file_url":"https://github.com/ContextualAI/LMUnit/blob/HEAD/lmunit/tasks.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"a35da997d266a35c","mcp_get_code":{"code_sha256":"a35da997d266a35c"}}]}