{"about":{"site":"https://codewithpapers.app","non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page"},"url":"/code/llm-2","entry":"llm","source":"Syntology graph, per-sample; not an archive number","read_at":"2026-09-24T18:15:14+00:00","claim":"Names are grouped by exact entry-name string. Same-named routines are NOT asserted to be equivalent; 'ran' means executed on a synthesized fixture, not correctness. n_samples_ran = sum of by_status over every status except 'unverified' (ran_draft_wrong and ran_fixture are failures of Syntology's instrument, not of the code); n_papers_ran = papers with at least one such sample.","status_vocabulary":{"ran_honours":"ran, honoured the contract we drafted","ran_violates":"ran, violated the contract we drafted","ran_draft_wrong":"ran; our contract draft was wrong, not the code","ran_fixture":"ran; our fixture could not drive it","ran":"ran on a synthesized input","unverified":"unverified (harvested, no recorded run)"},"n_papers":7,"n_papers_ran":3,"units":"n_samples, n_samples_ran, n_samples_fingerprinted and by_status count distinct code bodies (code_sha256); n_places and n_places_pointer_only count places, one per (paper, code body) pair, which is also the unit of the samples list","n_samples":9,"n_samples_ran":3,"n_samples_fingerprinted":0,"n_places":9,"n_places_pointer_only":7,"by_status":{"ran_honours":0,"ran_violates":0,"ran_draft_wrong":0,"ran_fixture":0,"ran":3,"unverified":6},"syntology":{"atlas_url":null,"mcp":null,"mcp_per_sample":{"tool":"get_code","arguments_in":"samples[].mcp_get_code"},"developers":"https://syntology.ai/developers"},"samples":[{"arxiv_id":"2410.02810","paper":"/paper/stateact-state-tracking-and-reasoning-for","title":"StateAct: State Tracking and Reasoning for Acting and Planning with Large Language Models","date":"2024-09-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ai-nikolai/stateact","path":"textcraft_runs/run_textcraft_adapt.py","file_url":"https://github.com/ai-nikolai/stateact/blob/HEAD/textcraft_runs/run_textcraft_adapt.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"0b9398a05fe8427e","mcp_get_code":{"code_sha256":"0b9398a05fe8427e"}},{"arxiv_id":"2410.02810","paper":"/paper/stateact-state-tracking-and-reasoning-for","title":"StateAct: State Tracking and Reasoning for Acting and Planning with Large Language Models","date":"2024-09-21","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"ai-nikolai/stateact","path":"webshop_runs/run_webshop_adapted.py","file_url":"https://github.com/ai-nikolai/stateact/blob/HEAD/webshop_runs/run_webshop_adapted.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"fa6aecfd93116153","mcp_get_code":{"code_sha256":"fa6aecfd93116153"}},{"arxiv_id":"2406.13862","paper":"/paper/knowledge-graph-enhanced-large-language","title":"Knowledge Graph-Enhanced Large Language Models via Path Selection","date":"2024-06-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"haochenliu2000/kelp","path":"factkg/make_training_set.py","file_url":"https://github.com/haochenliu2000/kelp/blob/HEAD/factkg/make_training_set.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"1495d07ab03e6a4a","mcp_get_code":{"code_sha256":"1495d07ab03e6a4a"}},{"arxiv_id":"2404.13874","paper":"/paper/valor-eval-holistic-coverage-and-faithfulness","title":"VALOR-EVAL: Holistic Coverage and Faithfulness Evaluation of Large Vision-Language Models","date":"2024-04-22","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"haoyiq114/valor","path":"evaluation/gpt_model.py","file_url":"https://github.com/haoyiq114/valor/blob/HEAD/evaluation/gpt_model.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"Apache-2.0","inline_ok":true,"code_sha256_prefix":"cd1a182553a1bf71","mcp_get_code":{"code_sha256":"cd1a182553a1bf71"}},{"arxiv_id":"2402.16181","paper":"/paper/how-can-llm-guide-rl-a-value-based-approach","title":"How Can LLM Guide RL? A Value-Based Approach","date":"2024-02-25","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"agentification/language-integrated-vi","path":"alfworld/alfworld_trial_wv.py","file_url":"https://github.com/agentification/language-integrated-vi/blob/HEAD/alfworld/alfworld_trial_wv.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"be74280fb8f684d9","mcp_get_code":{"code_sha256":"be74280fb8f684d9"}},{"arxiv_id":"2402.10178","paper":"/paper/tdag-a-multi-agent-framework-based-on-dynamic","title":"TDAG: A Multi-Agent Framework based on Dynamic Task Decomposition and Agent Generation","date":"2024-02-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yxwang8775/tdag","path":"run/run_textcraft.py","file_url":"https://github.com/yxwang8775/tdag/blob/HEAD/run/run_textcraft.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"6b6f5191859f7aef","mcp_get_code":{"code_sha256":"6b6f5191859f7aef"}},{"arxiv_id":"2402.10178","paper":"/paper/tdag-a-multi-agent-framework-based-on-dynamic","title":"TDAG: A Multi-Agent Framework based on Dynamic Task Decomposition and Agent Generation","date":"2024-02-15","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"yxwang8775/tdag","path":"run/run_webshop.py","file_url":"https://github.com/yxwang8775/tdag/blob/HEAD/run/run_webshop.py","status":"unverified","verification_level":0,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"02d74265eb1c412c","mcp_get_code":{"code_sha256":"02d74265eb1c412c"}},{"arxiv_id":"2401.09334","paper":"/paper/large-language-models-are-neurosymbolic","title":"Large Language Models Are Neurosymbolic Reasoners","date":"2024-01-17","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"hyintell/llmsymbolic","path":"play_game_by_LLM.py","file_url":"https://github.com/hyintell/llmsymbolic/blob/HEAD/play_game_by_LLM.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"NONE","inline_ok":false,"code_sha256_prefix":"349de02d458cce50","mcp_get_code":{"code_sha256":"349de02d458cce50"}},{"arxiv_id":"2305.11738","paper":"/paper/critic-large-language-models-can-self-correct","title":"CRITIC: Large Language Models Can Self-Correct with Tool-Interactive Critiquing","date":"2023-05-19","month_inferred_from_arxiv_id":null,"title_source":"archive","repo":"microsoft/ProphetNet","path":"CRITIC/src/program/critic.py","file_url":"https://github.com/microsoft/ProphetNet/blob/HEAD/CRITIC/src/program/critic.py","status":"ran","verification_level":1,"contract_check":null,"metamorphic_tier":null,"behaviour_fingerprint":false,"licence":"MIT","inline_ok":true,"code_sha256_prefix":"daebaf68fa6b443e","mcp_get_code":{"code_sha256":"daebaf68fa6b443e"}}]}