{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/decision-making/papers/ran/1","list_of":"/task/decision-making","task":"Decision Making","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":7,"rows_per_page":100,"rows":[1,100],"of":678,"counts":{"archive_papers_tagged":12311,"with_a_code_link":2946,"where_syntology_ran_a_sample":678,"not_listed_spam_title":0,"listed":12311,"listed_where_code_ran":678,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":560,"every_run_a_failure_of_syntologys_instrument":118,"listed_with_a_run_with_no_instrument_failure":560,"listed_every_run_a_failure_of_syntologys_instrument":118,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/decision-making/papers/ran/1","prev":null,"next":"/task/decision-making/papers/ran/2","papers":[{"url":"/paper/navmorph-a-self-evolving-world-model-for","slug":"navmorph-a-self-evolving-world-model-for","title":"NavMorph: A Self-Evolving World Model for Vision-and-Language Navigation in Continuous Environments","date":"2025-06-30","arxiv_id":"2506.23468","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":9,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/navmorph-a-self-evolving-world-model-for#ran","syntology_url":"https://syntology.ai/paper/2506.23468","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.23468"}},"official":{"repos":["feliciaxyao/navmorph"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/flow-based-single-step-completion-for","slug":"flow-based-single-step-completion-for","title":"Flow-Based Single-Step Completion for Efficient and Expressive Policy Learning","date":"2025-06-26","arxiv_id":"2506.21427","repositories_listed":0,"syntology":{"n":4,"n_ran":3,"n_constructed":2,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/flow-based-single-step-completion-for#ran","syntology_url":"https://syntology.ai/paper/2506.21427","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.21427"}},"official":null}},{"url":"/paper/causalpfn-amortized-causal-effect-estimation","slug":"causalpfn-amortized-causal-effect-estimation","title":"CausalPFN: Amortized Causal Effect Estimation via In-Context Learning","date":"2025-06-09","arxiv_id":"2506.07918","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/causalpfn-amortized-causal-effect-estimation#ran","syntology_url":"https://syntology.ai/paper/2506.07918","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.07918"}},"official":{"repos":["vdblm/CausalPFN"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/textatari-100k-frames-game-playing-with","slug":"textatari-100k-frames-game-playing-with","title":"TextAtari: 100K Frames Game Playing with Language Agents","date":"2025-06-04","arxiv_id":"2506.04098","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/textatari-100k-frames-game-playing-with#ran","syntology_url":"https://syntology.ai/paper/2506.04098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.04098"}},"official":{"repos":["Lww007/Text-Atari-Agents"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/causal-aware-large-language-models-enhancing","slug":"causal-aware-large-language-models-enhancing","title":"Causal-aware Large Language Models: Enhancing Decision-Making Through Learning, Adapting and Acting","date":"2025-05-30","arxiv_id":"2505.24710","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/causal-aware-large-language-models-enhancing#ran","syntology_url":"https://syntology.ai/paper/2505.24710","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.24710"}},"official":{"repos":["dmirlab-group/causal-aware_llms"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/k-2-vae-a-koopman-kalman-enhanced-variational","slug":"k-2-vae-a-koopman-kalman-enhanced-variational","title":"$K^2$VAE: A Koopman-Kalman Enhanced Variational AutoEncoder for Probabilistic Time Series Forecasting","date":"2025-05-29","arxiv_id":"2505.23017","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/k-2-vae-a-koopman-kalman-enhanced-variational#ran","syntology_url":"https://syntology.ai/paper/2505.23017","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.23017"}},"official":{"repos":["decisionintelligence/k2vae"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/attention-you-vision-language-model-could-be","slug":"attention-you-vision-language-model-could-be","title":"Attention! You Vision Language Model Could Be Maliciously Manipulated","date":"2025-05-26","arxiv_id":"2505.19911","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/attention-you-vision-language-model-could-be#ran","syntology_url":"https://syntology.ai/paper/2505.19911","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.19911"}},"official":null}},{"url":"/paper/sequential-monte-carlo-for-policy","slug":"sequential-monte-carlo-for-policy","title":"Sequential Monte Carlo for Policy Optimization in Continuous POMDPs","date":"2025-05-22","arxiv_id":"2505.16732","repositories_listed":0,"syntology":{"n":44,"n_ran":15,"n_constructed":8,"n_ran_checked":9,"n_instrument":6,"n_unverified":29,"n_honours":1,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"15 ran (of which 8 constructed an object rather than computing a result; 9 with no instrument failure: 1 honoured, 0 violated, 8 with no contract checked; 6 where Syntology's instrument failed) · 29 unverified","sample_list":"/paper/sequential-monte-carlo-for-policy#ran","syntology_url":"https://syntology.ai/paper/2505.16732","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.16732"}},"official":null}},{"url":"/paper/mf-llm-simulating-collective-decision","slug":"mf-llm-simulating-collective-decision","title":"MF-LLM: Simulating Population Decision Dynamics via a Mean-Field Large Language Model Framework","date":"2025-04-30","arxiv_id":"2504.21582","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mf-llm-simulating-collective-decision#ran","syntology_url":"https://syntology.ai/paper/2504.21582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.21582"}},"official":{"repos":["Miracle1207/Mean-Field-LLM"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/ragen-understanding-self-evolution-in-llm","slug":"ragen-understanding-self-evolution-in-llm","title":"RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning","date":"2025-04-24","arxiv_id":"2504.20073","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ragen-understanding-self-evolution-in-llm#ran","syntology_url":"https://syntology.ai/paper/2504.20073","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.20073"}},"official":{"repos":["ragen-ai/ragen"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/agentic-knowledgeable-self-awareness","slug":"agentic-knowledgeable-self-awareness","title":"Agentic Knowledgeable Self-awareness","date":"2025-04-04","arxiv_id":"2504.03553","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/agentic-knowledgeable-self-awareness#ran","syntology_url":"https://syntology.ai/paper/2504.03553","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.03553"}},"official":{"repos":["zjunlp/knowself"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/opendrivevla-towards-end-to-end-autonomous","slug":"opendrivevla-towards-end-to-end-autonomous","title":"OpenDriveVLA: Towards End-to-end Autonomous Driving with Large Vision Language Action Model","date":"2025-03-30","arxiv_id":"2503.23463","repositories_listed":1,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":2,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/opendrivevla-towards-end-to-end-autonomous#ran","syntology_url":"https://syntology.ai/paper/2503.23463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.23463"}},"official":{"repos":["DriveVLA/OpenDriveVLA"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/dissecting-and-mitigating-diffusion-bias-via","slug":"dissecting-and-mitigating-diffusion-bias-via","title":"Dissecting and Mitigating Diffusion Bias via Mechanistic Interpretability","date":"2025-03-26","arxiv_id":"2503.20483","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/dissecting-and-mitigating-diffusion-bias-via#ran","syntology_url":"https://syntology.ai/paper/2503.20483","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.20483"}},"official":null}},{"url":"/paper/llm-based-agent-simulation-for-maternal","slug":"llm-based-agent-simulation-for-maternal","title":"LLM-based Agent Simulation for Maternal Health Interventions: Uncertainty Estimation and Decision-focused Evaluation","date":"2025-03-25","arxiv_id":"2503.22719","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llm-based-agent-simulation-for-maternal#ran","syntology_url":"https://syntology.ai/paper/2503.22719","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.22719"}},"official":{"repos":["sarahmart/llm-abs-armman-prediction"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/when-words-outperform-vision-vlms-can-self","slug":"when-words-outperform-vision-vlms-can-self","title":"When Words Outperform Vision: VLMs Can Self-Improve Via Text-Only Training For Human-Centered Decision Making","date":"2025-03-21","arxiv_id":"2503.16965","repositories_listed":0,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/when-words-outperform-vision-vlms-can-self#ran","syntology_url":"https://syntology.ai/paper/2503.16965","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.16965"}},"official":null}},{"url":"/paper/txagent-an-ai-agent-for-therapeutic-reasoning","slug":"txagent-an-ai-agent-for-therapeutic-reasoning","title":"TxAgent: An AI Agent for Therapeutic Reasoning Across a Universe of Tools","date":"2025-03-14","arxiv_id":"2503.10970","repositories_listed":4,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/txagent-an-ai-agent-for-therapeutic-reasoning#ran","syntology_url":"https://syntology.ai/paper/2503.10970","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.10970"}},"official":{"repos":["mims-harvard/ToolUniverse","mims-harvard/TxAgent","huggingface.co/mims-harvard/ToolRAG-T1-GTE-Qwen2-1.5B","huggingface.co/mims-harvard/TxAgent-T1-Llama-3.1-8B"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/segagent-exploring-pixel-understanding-1","slug":"segagent-exploring-pixel-understanding-1","title":"SegAgent: Exploring Pixel Understanding Capabilities in MLLMs by Imitating Human Annotator Trajectories","date":"2025-03-11","arxiv_id":"2503.08625","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/segagent-exploring-pixel-understanding-1#ran","syntology_url":"https://syntology.ai/paper/2503.08625","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.08625"}},"official":{"repos":["aim-uofa/SegAgent"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/parallelized-planning-acting-for-efficient","slug":"parallelized-planning-acting-for-efficient","title":"Parallelized Planning-Acting for Efficient LLM-based Multi-Agent Systems","date":"2025-03-05","arxiv_id":"2503.03505","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/parallelized-planning-acting-for-efficient#ran","syntology_url":"https://syntology.ai/paper/2503.03505","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2503.03505"}},"official":{"repos":["zju-vipa/odyssey"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/digital-player-evaluating-large-language","slug":"digital-player-evaluating-large-language","title":"Digital Player: Evaluating Large Language Models based Human-like Agent in Games","date":"2025-02-28","arxiv_id":"2502.20807","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/digital-player-evaluating-large-language#ran","syntology_url":"https://syntology.ai/paper/2502.20807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.20807"}},"official":{"repos":["fuxiailab/civagent"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cirt-global-subseasonal-to-seasonal","slug":"cirt-global-subseasonal-to-seasonal","title":"CirT: Global Subseasonal-to-Seasonal Forecasting with Geometry-inspired Transformer","date":"2025-02-27","arxiv_id":"2502.19750","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cirt-global-subseasonal-to-seasonal#ran","syntology_url":"https://syntology.ai/paper/2502.19750","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.19750"}},"official":{"repos":["compasszzn/CirT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/citrus-leveraging-expert-cognitive-pathways","slug":"citrus-leveraging-expert-cognitive-pathways","title":"Citrus: Leveraging Expert Cognitive Pathways in a Medical Language Model for Advanced Medical Decision Support","date":"2025-02-25","arxiv_id":"2502.18274","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/citrus-leveraging-expert-cognitive-pathways#ran","syntology_url":"https://syntology.ai/paper/2502.18274","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.18274"}},"official":{"repos":["jdh-algo/Citrus"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/from-text-to-space-mapping-abstract-spatial","slug":"from-text-to-space-mapping-abstract-spatial","title":"From Text to Space: Mapping Abstract Spatial Models in LLMs during a Grid-World Navigation Task","date":"2025-02-23","arxiv_id":"2502.16690","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/from-text-to-space-mapping-abstract-spatial#ran","syntology_url":"https://syntology.ai/paper/2502.16690","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.16690"}},"official":{"repos":["mneuronico/griw-world-spatial-orientation-task"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/moving-beyond-medical-exam-questions-a","slug":"moving-beyond-medical-exam-questions-a","title":"Moving Beyond Medical Exam Questions: A Clinician-Annotated Dataset of Real-World Tasks and Ambiguity in Mental Healthcare","date":"2025-02-22","arxiv_id":"2502.16051","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/moving-beyond-medical-exam-questions-a#ran","syntology_url":"https://syntology.ai/paper/2502.16051","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.16051"}},"official":{"repos":["maxlampe/mentat"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptivestep-automatically-dividing-reasoning","slug":"adaptivestep-automatically-dividing-reasoning","title":"AdaptiveStep: Automatically Dividing Reasoning Step through Model Confidence","date":"2025-02-19","arxiv_id":"2502.13943","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaptivestep-automatically-dividing-reasoning#ran","syntology_url":"https://syntology.ai/paper/2502.13943","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.13943"}},"official":{"repos":["lux0926/asprm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-solve-the-min-max-mixed-shelves","slug":"learning-to-solve-the-min-max-mixed-shelves","title":"Learning to Solve the Min-Max Mixed-Shelves Picker-Routing Problem via Hierarchical and Parallel Decoding","date":"2025-02-14","arxiv_id":"2502.10233","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/learning-to-solve-the-min-max-mixed-shelves#ran","syntology_url":"https://syntology.ai/paper/2502.10233","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.10233"}},"official":{"repos":["ltluttmann/marl4msprp"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-conformal-abstention-policies-for","slug":"learning-conformal-abstention-policies-for","title":"Learning Conformal Abstention Policies for Adaptive Risk Management in Large Language and Vision-Language Models","date":"2025-02-08","arxiv_id":"2502.06884","repositories_listed":1,"syntology":{"n":16,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":8,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/learning-conformal-abstention-policies-for#ran","syntology_url":"https://syntology.ai/paper/2502.06884","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.06884"}},"official":{"repos":["sinatayebati/vlm-uncertainty"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":8,"ran_from_kinds":["official"]}}},{"url":"/paper/prism-a-robust-framework-for-skill-based-meta","slug":"prism-a-robust-framework-for-skill-based-meta","title":"PRISM: A Robust Framework for Skill-based Meta-Reinforcement Learning with Noisy Demonstrations","date":"2025-02-06","arxiv_id":"2502.03752","repositories_listed":0,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prism-a-robust-framework-for-skill-based-meta#ran","syntology_url":"https://syntology.ai/paper/2502.03752","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.03752"}},"official":null}},{"url":"/paper/on-the-guidance-of-flow-matching","slug":"on-the-guidance-of-flow-matching","title":"On the Guidance of Flow Matching","date":"2025-02-04","arxiv_id":"2502.02150","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 2 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/on-the-guidance-of-flow-matching#ran","syntology_url":"https://syntology.ai/paper/2502.02150","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2502.02150"}},"official":{"repos":["ai4science-westlakeu/flow_guidance"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/vintix-action-model-via-in-context","slug":"vintix-action-model-via-in-context","title":"Vintix: Action Model via In-Context Reinforcement Learning","date":"2025-01-31","arxiv_id":"2501.19400","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/vintix-action-model-via-in-context#ran","syntology_url":"https://syntology.ai/paper/2501.19400","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.19400"}},"official":{"repos":["dunnolab/vintix"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/leapvad-a-leap-in-autonomous-driving-via","slug":"leapvad-a-leap-in-autonomous-driving-via","title":"LeapVAD: A Leap in Autonomous Driving via Cognitive Perception and Dual-Process Thinking","date":"2025-01-14","arxiv_id":"2501.08168","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/leapvad-a-leap-in-autonomous-driving-via#ran","syntology_url":"https://syntology.ai/paper/2501.08168","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.08168"}},"official":null}},{"url":"/paper/mechanistic-understanding-and-validation-of","slug":"mechanistic-understanding-and-validation-of","title":"Mechanistic understanding and validation of large AI models with SemanticLens","date":"2025-01-09","arxiv_id":"2501.05398","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mechanistic-understanding-and-validation-of#ran","syntology_url":"https://syntology.ai/paper/2501.05398","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.05398"}},"official":{"repos":["jim-berend/semanticlens"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prmbench-a-fine-grained-and-challenging","slug":"prmbench-a-fine-grained-and-challenging","title":"PRMBench: A Fine-grained and Challenging Benchmark for Process-Level Reward Models","date":"2025-01-06","arxiv_id":"2501.03124","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prmbench-a-fine-grained-and-challenging#ran","syntology_url":"https://syntology.ai/paper/2501.03124","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.03124"}},"official":{"repos":["ssmisya/PRMBench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-cvar-leveraging-static-spectral-risk","slug":"beyond-cvar-leveraging-static-spectral-risk","title":"Beyond CVaR: Leveraging Static Spectral Risk Measures for Enhanced Decision-Making in Distributional Reinforcement Learning","date":"2025-01-03","arxiv_id":"2501.02087","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":1,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/beyond-cvar-leveraging-static-spectral-risk#ran","syntology_url":"https://syntology.ai/paper/2501.02087","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2501.02087"}},"official":{"repos":["mehrdadmoghimi/qrsrm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/constraint-adaptive-policy-switching-for","slug":"constraint-adaptive-policy-switching-for","title":"Constraint-Adaptive Policy Switching for Offline Safe Reinforcement Learning","date":"2024-12-25","arxiv_id":"2412.18946","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/constraint-adaptive-policy-switching-for#ran","syntology_url":"https://syntology.ai/paper/2412.18946","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.18946"}},"official":{"repos":["yassinech/caps"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/minsstudio-a-streamlined-package-for","slug":"minsstudio-a-streamlined-package-for","title":"MineStudio: A Streamlined Package for Minecraft AI Agent Development","date":"2024-12-24","arxiv_id":"2412.18293","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/minsstudio-a-streamlined-package-for#ran","syntology_url":"https://syntology.ai/paper/2412.18293","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.18293"}},"official":{"repos":["craftjarvis/minestudio"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/embodied-cot-distillation-from-llm-to-off-the","slug":"embodied-cot-distillation-from-llm-to-off-the","title":"Embodied CoT Distillation From LLM To Off-the-shelf Agents","date":"2024-12-16","arxiv_id":"2412.11499","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/embodied-cot-distillation-from-llm-to-off-the#ran","syntology_url":"https://syntology.ai/paper/2412.11499","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.11499"}},"official":{"repos":["osu-nlp-group/llm-planner"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/auctionnet-a-novel-benchmark-for-decision","slug":"auctionnet-a-novel-benchmark-for-decision","title":"AuctionNet: A Novel Benchmark for Decision-Making in Large-Scale Games","date":"2024-12-14","arxiv_id":"2412.10798","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/auctionnet-a-novel-benchmark-for-decision#ran","syntology_url":"https://syntology.ai/paper/2412.10798","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.10798"}},"official":{"repos":["alimama-tech/auctionnet"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/wisead-knowledge-augmented-end-to-end","slug":"wisead-knowledge-augmented-end-to-end","title":"WiseAD: Knowledge Augmented End-to-End Autonomous Driving with Vision-Language Model","date":"2024-12-13","arxiv_id":"2412.09951","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wisead-knowledge-augmented-end-to-end#ran","syntology_url":"https://syntology.ai/paper/2412.09951","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.09951"}},"official":{"repos":["wyddmw/WiseAD"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gaussianad-gaussian-centric-end-to-end","slug":"gaussianad-gaussian-centric-end-to-end","title":"GaussianAD: Gaussian-Centric End-to-End Autonomous Driving","date":"2024-12-13","arxiv_id":"2412.10371","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/gaussianad-gaussian-centric-end-to-end#ran","syntology_url":"https://syntology.ai/paper/2412.10371","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.10371"}},"official":{"repos":["wzzheng/gaussianad"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/forest-of-thought-scaling-test-time-compute","slug":"forest-of-thought-scaling-test-time-compute","title":"Forest-of-Thought: Scaling Test-Time Compute for Enhancing LLM Reasoning","date":"2024-12-12","arxiv_id":"2412.09078","repositories_listed":1,"syntology":{"n":11,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":11,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/forest-of-thought-scaling-test-time-compute#ran","syntology_url":"https://syntology.ai/paper/2412.09078","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.09078"}},"official":{"repos":["iamhankai/Forest-of-Thought"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/doe-1-closed-loop-autonomous-driving-with","slug":"doe-1-closed-loop-autonomous-driving-with","title":"Doe-1: Closed-Loop Autonomous Driving with Large World Model","date":"2024-12-12","arxiv_id":"2412.09627","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/doe-1-closed-loop-autonomous-driving-with#ran","syntology_url":"https://syntology.ai/paper/2412.09627","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2412.09627"}},"official":{"repos":["wzzheng/doe"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/auto-rag-autonomous-retrieval-augmented","slug":"auto-rag-autonomous-retrieval-augmented","title":"Auto-RAG: Autonomous Retrieval-Augmented Generation for Large Language Models","date":"2024-11-29","arxiv_id":"2411.19443","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/auto-rag-autonomous-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2411.19443","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.19443"}},"official":{"repos":["ictnlp/auto-rag"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/decision-making-under-the-exponential-family","slug":"decision-making-under-the-exponential-family","title":"Decision Making under the Exponential Family: Distributionally Robust Optimisation with Bayesian Ambiguity Sets","date":"2024-11-25","arxiv_id":"2411.16829","repositories_listed":0,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/decision-making-under-the-exponential-family#ran","syntology_url":"https://syntology.ai/paper/2411.16829","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.16829"}},"official":null}},{"url":"/paper/expert-elicitation-method-for-non-parametric","slug":"expert-elicitation-method-for-non-parametric","title":"Expert-elicitation method for non-parametric joint priors using normalizing flows","date":"2024-11-24","arxiv_id":"2411.15826","repositories_listed":1,"syntology":{"n":11,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/expert-elicitation-method-for-non-parametric#ran","syntology_url":"https://syntology.ai/paper/2411.15826","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.15826"}},"official":{"repos":["florence-bockting/prior_elicitation"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/natural-language-reinforcement-learning-1","slug":"natural-language-reinforcement-learning-1","title":"Natural Language Reinforcement Learning","date":"2024-11-21","arxiv_id":"2411.14251","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/natural-language-reinforcement-learning-1#ran","syntology_url":"https://syntology.ai/paper/2411.14251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.14251"}},"official":{"repos":["waterhorse1/natural-language-rl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gmai-vl-gmai-vl-5-5m-a-large-vision-language","slug":"gmai-vl-gmai-vl-5-5m-a-large-vision-language","title":"GMAI-VL & GMAI-VL-5.5M: A Large Vision-Language Model and A Comprehensive Multimodal Dataset Towards General Medical AI","date":"2024-11-21","arxiv_id":"2411.14522","repositories_listed":1,"syntology":{"n":10,"n_ran":10,"n_constructed":0,"n_ran_checked":9,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gmai-vl-gmai-vl-5-5m-a-large-vision-language#ran","syntology_url":"https://syntology.ai/paper/2411.14522","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.14522"}},"official":{"repos":["uni-medical/gmai-vl"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/disentangling-memory-and-reasoning-ability-in","slug":"disentangling-memory-and-reasoning-ability-in","title":"Disentangling Memory and Reasoning Ability in Large Language Models","date":"2024-11-20","arxiv_id":"2411.13504","repositories_listed":1,"syntology":{"n":8,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":8,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/disentangling-memory-and-reasoning-ability-in#ran","syntology_url":"https://syntology.ai/paper/2411.13504","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.13504"}},"official":{"repos":["mingyuj666/disentangling-memory-and-reasoning"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/large-scale-moral-machine-experiment-on-large","slug":"large-scale-moral-machine-experiment-on-large","title":"Large-scale moral machine experiment on large language models","date":"2024-11-11","arxiv_id":"2411.06790","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/large-scale-moral-machine-experiment-on-large#ran","syntology_url":"https://syntology.ai/paper/2411.06790","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.06790"}},"official":{"repos":["kztakemoto/mmllm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/assistrag-boosting-the-potential-of-large","slug":"assistrag-boosting-the-potential-of-large","title":"AssistRAG: Boosting the Potential of Large Language Models with an Intelligent Information Assistant","date":"2024-11-11","arxiv_id":"2411.06805","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":1,"n_ran_checked":8,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":13,"phrase":"10 ran (of which 1 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/assistrag-boosting-the-potential-of-large#ran","syntology_url":"https://syntology.ai/paper/2411.06805","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.06805"}},"official":{"repos":["smallporridge/assistrag"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":1,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-graph-neural-network-surrogates","slug":"towards-graph-neural-network-surrogates","title":"Graph Neural Network Surrogates to leverage Mechanistic Expert Knowledge towards Reliable and Immediate Pandemic Response","date":"2024-11-10","arxiv_id":"2411.06500","repositories_listed":1,"syntology":{"n":6,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/towards-graph-neural-network-surrogates#ran","syntology_url":"https://syntology.ai/paper/2411.06500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.06500"}},"official":null}},{"url":"/paper/concept-bottleneck-language-models-for","slug":"concept-bottleneck-language-models-for","title":"Concept Bottleneck Language Models For protein design","date":"2024-11-09","arxiv_id":"2411.06090","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/concept-bottleneck-language-models-for#ran","syntology_url":"https://syntology.ai/paper/2411.06090","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.06090"}},"official":{"repos":["prescient-design/lobster"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/game-theoretic-llm-agent-workflow-for","slug":"game-theoretic-llm-agent-workflow-for","title":"Game-theoretic LLM: Agent Workflow for Negotiation Games","date":"2024-11-08","arxiv_id":"2411.05990","repositories_listed":2,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/game-theoretic-llm-agent-workflow-for#ran","syntology_url":"https://syntology.ai/paper/2411.05990","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.05990"}},"official":{"repos":["wenyueh/game_theory"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/adasociety-an-adaptive-environment-with","slug":"adasociety-an-adaptive-environment-with","title":"AdaSociety: An Adaptive Environment with Social Structures for Multi-Agent Decision-Making","date":"2024-11-06","arxiv_id":"2411.03865","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/adasociety-an-adaptive-environment-with#ran","syntology_url":"https://syntology.ai/paper/2411.03865","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.03865"}},"official":{"repos":["bigai-ai/adasociety"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/pagerank-bandits-for-link-prediction","slug":"pagerank-bandits-for-link-prediction","title":"PageRank Bandits for Link Prediction","date":"2024-11-03","arxiv_id":"2411.01410","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pagerank-bandits-for-link-prediction#ran","syntology_url":"https://syntology.ai/paper/2411.01410","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.01410"}},"official":{"repos":["jiaruzouu/prb"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-expert-prompting-improves-reliability","slug":"multi-expert-prompting-improves-reliability","title":"Multi-expert Prompting Improves Reliability, Safety, and Usefulness of Large Language Models","date":"2024-11-01","arxiv_id":"2411.00492","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multi-expert-prompting-improves-reliability#ran","syntology_url":"https://syntology.ai/paper/2411.00492","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2411.00492"}},"official":{"repos":["dxlong2000/multi-expert-prompting"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/difflight-a-partial-rewards-conditioned","slug":"difflight-a-partial-rewards-conditioned","title":"DiffLight: A Partial Rewards Conditioned Diffusion Model for Traffic Signal Control with Missing Data","date":"2024-10-30","arxiv_id":"2410.22938","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/difflight-a-partial-rewards-conditioned#ran","syntology_url":"https://syntology.ai/paper/2410.22938","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.22938"}},"official":{"repos":["lokol5579/DiffLight-release"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/online-intrinsic-rewards-for-decision-making","slug":"online-intrinsic-rewards-for-decision-making","title":"Online Intrinsic Rewards for Decision Making Agents from Large Language Model Feedback","date":"2024-10-30","arxiv_id":"2410.23022","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":12,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/online-intrinsic-rewards-for-decision-making#ran","syntology_url":"https://syntology.ai/paper/2410.23022","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.23022"}},"official":{"repos":["facebookresearch/oni"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/activesplat-high-fidelity-scene","slug":"activesplat-high-fidelity-scene","title":"ActiveSplat: High-Fidelity Scene Reconstruction through Active Gaussian Splatting","date":"2024-10-29","arxiv_id":"2410.21955","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/activesplat-high-fidelity-scene#ran","syntology_url":"https://syntology.ai/paper/2410.21955","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.21955"}},"official":{"repos":["Li-Yuetao/ActiveSplat"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/toward-conditional-distribution-calibration","slug":"toward-conditional-distribution-calibration","title":"Toward Conditional Distribution Calibration in Survival Prediction","date":"2024-10-27","arxiv_id":"2410.20579","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/toward-conditional-distribution-calibration#ran","syntology_url":"https://syntology.ai/paper/2410.20579","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.20579"}},"official":{"repos":["shi-ang/makesurvivalcalibratedagain"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/context-is-key-a-benchmark-for-forecasting","slug":"context-is-key-a-benchmark-for-forecasting","title":"Context is Key: A Benchmark for Forecasting with Essential Textual Information","date":"2024-10-24","arxiv_id":"2410.18959","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/context-is-key-a-benchmark-for-forecasting#ran","syntology_url":"https://syntology.ai/paper/2410.18959","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.18959"}},"official":{"repos":["servicenow/context-is-key-forecasting"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-versatile-skills-with-curriculum","slug":"learning-versatile-skills-with-curriculum","title":"Learning Versatile Skills with Curriculum Masking","date":"2024-10-23","arxiv_id":"2410.17744","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-versatile-skills-with-curriculum#ran","syntology_url":"https://syntology.ai/paper/2410.17744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17744"}},"official":{"repos":["yaotang23/currmask"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/rocket-1-master-open-world-interaction-with","slug":"rocket-1-master-open-world-interaction-with","title":"ROCKET-1: Mastering Open-World Interaction with Visual-Temporal Context Prompting","date":"2024-10-23","arxiv_id":"2410.17856","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rocket-1-master-open-world-interaction-with#ran","syntology_url":"https://syntology.ai/paper/2410.17856","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17856"}},"official":{"repos":["CraftJarvis/ROCKET-1"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-causal-reasoning-in-large-language","slug":"improving-causal-reasoning-in-large-language","title":"Improving Causal Reasoning in Large Language Models: A Survey","date":"2024-10-22","arxiv_id":"2410.16676","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-causal-reasoning-in-large-language#ran","syntology_url":"https://syntology.ai/paper/2410.16676","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.16676"}},"official":{"repos":["chendl02/awesome-llm-causal-reasoning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/literature-meets-data-a-synergistic-approach","slug":"literature-meets-data-a-synergistic-approach","title":"Literature Meets Data: A Synergistic Approach to Hypothesis Generation","date":"2024-10-22","arxiv_id":"2410.17309","repositories_listed":3,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/literature-meets-data-a-synergistic-approach#ran","syntology_url":"https://syntology.ai/paper/2410.17309","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.17309"}},"official":{"repos":["chicagohai/hypothesis-generation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/a-comprehensive-evaluation-of-cognitive","slug":"a-comprehensive-evaluation-of-cognitive","title":"A Comprehensive Evaluation of Cognitive Biases in LLMs","date":"2024-10-20","arxiv_id":"2410.15413","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-comprehensive-evaluation-of-cognitive#ran","syntology_url":"https://syntology.ai/paper/2410.15413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.15413"}},"official":{"repos":["simonmalberg/cognitive-biases-in-llms"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/measuring-free-form-decision-making","slug":"measuring-free-form-decision-making","title":"Measuring Free-Form Decision-Making Inconsistency of Language Models in Military Crisis Simulations","date":"2024-10-17","arxiv_id":"2410.13204","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/measuring-free-form-decision-making#ran","syntology_url":"https://syntology.ai/paper/2410.13204","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13204"}},"official":{"repos":["aashrivastava/llmwargaminginconsistency"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/web-agents-with-world-models-learning-and","slug":"web-agents-with-world-models-learning-and","title":"Web Agents with World Models: Learning and Leveraging Environment Dynamics in Web Navigation","date":"2024-10-17","arxiv_id":"2410.13232","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/web-agents-with-world-models-learning-and#ran","syntology_url":"https://syntology.ai/paper/2410.13232","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13232"}},"official":{"repos":["kyle8581/wma-agents"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/fdf-flexible-decoupled-framework-for-time","slug":"fdf-flexible-decoupled-framework-for-time","title":"FDF: Flexible Decoupled Framework for Time Series Forecasting with Conditional Denoising and Polynomial Modeling","date":"2024-10-17","arxiv_id":"2410.13253","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fdf-flexible-decoupled-framework-for-time#ran","syntology_url":"https://syntology.ai/paper/2410.13253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.13253"}},"official":{"repos":["zjt-gpu/fdf"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sliding-puzzles-gym-a-scalable-benchmark-for","slug":"sliding-puzzles-gym-a-scalable-benchmark-for","title":"Sliding Puzzles Gym: A Scalable Benchmark for State Representation in Visual Reinforcement Learning","date":"2024-10-17","arxiv_id":"2410.14038","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sliding-puzzles-gym-a-scalable-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2410.14038","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.14038"}},"official":{"repos":["bryanoliveira/sliding-puzzles-gym"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/process-reward-model-with-q-value-rankings","slug":"process-reward-model-with-q-value-rankings","title":"Process Reward Model with Q-Value Rankings","date":"2024-10-15","arxiv_id":"2410.11287","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/process-reward-model-with-q-value-rankings#ran","syntology_url":"https://syntology.ai/paper/2410.11287","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11287"}},"official":{"repos":["WindyLee0822/Process_Q_Model"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/stable-hadamard-memory-revitalizing-memory","slug":"stable-hadamard-memory-revitalizing-memory","title":"Stable Hadamard Memory: Revitalizing Memory-Augmented Agents for Reinforcement Learning","date":"2024-10-14","arxiv_id":"2410.10132","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/stable-hadamard-memory-revitalizing-memory#ran","syntology_url":"https://syntology.ai/paper/2410.10132","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.10132"}},"official":null}},{"url":"/paper/persistent-topological-features-in-large","slug":"persistent-topological-features-in-large","title":"Persistent Topological Features in Large Language Models","date":"2024-10-14","arxiv_id":"2410.11042","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":5,"n_instrument":4,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/persistent-topological-features-in-large#ran","syntology_url":"https://syntology.ai/paper/2410.11042","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.11042"}},"official":{"repos":["RitAreaSciencePark/ZigZagLLMs"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ranking-over-regression-for-bayesian","slug":"ranking-over-regression-for-bayesian","title":"Ranking over Regression for Bayesian Optimization and Molecule Selection","date":"2024-10-11","arxiv_id":"2410.09290","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ranking-over-regression-for-bayesian#ran","syntology_url":"https://syntology.ai/paper/2410.09290","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.09290"}},"official":{"repos":["gkwt/rbo"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/identifying-and-addressing-delusions-for","slug":"identifying-and-addressing-delusions-for","title":"Rejecting Hallucinated State Targets during Planning","date":"2024-10-09","arxiv_id":"2410.07096","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/identifying-and-addressing-delusions-for#ran","syntology_url":"https://syntology.ai/paper/2410.07096","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07096"}},"official":{"repos":["mila-iqia/delusions"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/embodied-agent-interface-benchmarking-llms","slug":"embodied-agent-interface-benchmarking-llms","title":"Embodied Agent Interface: Benchmarking LLMs for Embodied Decision Making","date":"2024-10-09","arxiv_id":"2410.07166","repositories_listed":3,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/embodied-agent-interface-benchmarking-llms#ran","syntology_url":"https://syntology.ai/paper/2410.07166","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.07166"}},"official":{"repos":["embodied-agent-eval/embodied-agent-eval","embodied-agent-interface/embodied-agent-interface","embodied-agent-eval/embodied-agent-eval.github.io"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/riemann-sum-optimization-for-accurate","slug":"riemann-sum-optimization-for-accurate","title":"Riemann Sum Optimization for Accurate Integrated Gradients Computation","date":"2024-10-05","arxiv_id":"2410.04118","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/riemann-sum-optimization-for-accurate#ran","syntology_url":"https://syntology.ai/paper/2410.04118","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.04118"}},"official":{"repos":["ShreeSinghi/RiemannOpt"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/spatial-aware-decision-making-with-ring","slug":"spatial-aware-decision-making-with-ring","title":"Spatial-aware decision-making with ring attractors in reinforcement learning systems","date":"2024-10-04","arxiv_id":"2410.03119","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spatial-aware-decision-making-with-ring#ran","syntology_url":"https://syntology.ai/paper/2410.03119","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03119"}},"official":null}},{"url":"/paper/open-world-reinforcement-learning-over-long","slug":"open-world-reinforcement-learning-over-long","title":"Open-World Reinforcement Learning over Long Short-Term Imagination","date":"2024-10-04","arxiv_id":"2410.03618","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/open-world-reinforcement-learning-over-long#ran","syntology_url":"https://syntology.ai/paper/2410.03618","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.03618"}},"official":{"repos":["qiwang067/LS-Imagine"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-teachers-for-amortized-samplers","slug":"adaptive-teachers-for-amortized-samplers","title":"Adaptive teachers for amortized samplers","date":"2024-10-02","arxiv_id":"2410.01432","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":1,"n_ran_checked":2,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":5,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaptive-teachers-for-amortized-samplers#ran","syntology_url":"https://syntology.ai/paper/2410.01432","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2410.01432"}},"official":{"repos":["alstn12088/adaptive-teacher"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/forecastbench-a-dynamic-benchmark-of-ai","slug":"forecastbench-a-dynamic-benchmark-of-ai","title":"ForecastBench: A Dynamic Benchmark of AI Forecasting Capabilities","date":"2024-09-30","arxiv_id":"2409.19839","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/forecastbench-a-dynamic-benchmark-of-ai#ran","syntology_url":"https://syntology.ai/paper/2409.19839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.19839"}},"official":{"repos":["forecastingresearch/forecastbench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/end-to-end-conformal-calibration-for","slug":"end-to-end-conformal-calibration-for","title":"End-to-End Conformal Calibration for Optimization Under Uncertainty","date":"2024-09-30","arxiv_id":"2409.20534","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/end-to-end-conformal-calibration-for#ran","syntology_url":"https://syntology.ai/paper/2409.20534","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.20534"}},"official":{"repos":["chrisyeh96/e2e-conformal"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/maia-2-a-unified-model-for-human-ai-alignment","slug":"maia-2-a-unified-model-for-human-ai-alignment","title":"Maia-2: A Unified Model for Human-AI Alignment in Chess","date":"2024-09-30","arxiv_id":"2409.20553","repositories_listed":2,"syntology":{"n":18,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":10,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 10 unverified","sample_list":"/paper/maia-2-a-unified-model-for-human-ai-alignment#ran","syntology_url":"https://syntology.ai/paper/2409.20553","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.20553"}},"official":{"repos":["csslab/maia2"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/climate-adaptation-with-reinforcement","slug":"climate-adaptation-with-reinforcement","title":"Climate Adaptation with Reinforcement Learning: Experiments with Flooding and Transportation in Copenhagen","date":"2024-09-27","arxiv_id":"2409.18574","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/climate-adaptation-with-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2409.18574","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.18574"}},"official":{"repos":["mlsm-at-dtu/floods_transport_rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/glinsat-the-general-linear-satisfiability","slug":"glinsat-the-general-linear-satisfiability","title":"GLinSAT: The General Linear Satisfiability Neural Network Layer By Accelerated Gradient Descent","date":"2024-09-26","arxiv_id":"2409.17500","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/glinsat-the-general-linear-satisfiability#ran","syntology_url":"https://syntology.ai/paper/2409.17500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.17500"}},"official":{"repos":["huntertracer/glinsat"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/emit-event-based-masked-auto-encoding-for","slug":"emit-event-based-masked-auto-encoding-for","title":"EMIT- Event-Based Masked Auto Encoding for Irregular Time Series","date":"2024-09-25","arxiv_id":"2409.16554","repositories_listed":1,"syntology":{"n":15,"n_ran":14,"n_constructed":0,"n_ran_checked":14,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":0,"phrase":"14 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/emit-event-based-masked-auto-encoding-for#ran","syntology_url":"https://syntology.ai/paper/2409.16554","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.16554"}},"official":{"repos":["hrishi-ds/EMIT"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/parco-learning-parallel-autoregressive","slug":"parco-learning-parallel-autoregressive","title":"Parallel AutoRegressive Models for Multi-Agent Combinatorial Optimization","date":"2024-09-05","arxiv_id":"2409.03811","repositories_listed":2,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/parco-learning-parallel-autoregressive#ran","syntology_url":"https://syntology.ai/paper/2409.03811","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.03811"}},"official":{"repos":["ai4co/parco"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/inverse-decision-making-using-neural","slug":"inverse-decision-making-using-neural","title":"Inverse decision-making using neural amortized Bayesian actors","date":"2024-09-04","arxiv_id":"2409.03710","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/inverse-decision-making-using-neural#ran","syntology_url":"https://syntology.ai/paper/2409.03710","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.03710"}},"official":{"repos":["rothkopflab/naba"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/counterfactual-fairness-by-combining-factual","slug":"counterfactual-fairness-by-combining-factual","title":"Counterfactual Fairness by Combining Factual and Counterfactual Predictions","date":"2024-09-03","arxiv_id":"2409.01977","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/counterfactual-fairness-by-combining-factual#ran","syntology_url":"https://syntology.ai/paper/2409.01977","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.01977"}},"official":{"repos":["inouye-lab/pcf"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/epo-hierarchical-llm-agents-with-environment","slug":"epo-hierarchical-llm-agents-with-environment","title":"EPO: Hierarchical LLM Agents with Environment Preference Optimization","date":"2024-08-28","arxiv_id":"2408.16090","repositories_listed":1,"syntology":{"n":5,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/epo-hierarchical-llm-agents-with-environment#ran","syntology_url":"https://syntology.ai/paper/2408.16090","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.16090"}},"official":{"repos":["kevinz8866/epo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/wait-that-s-not-an-option-llms-robustness","slug":"wait-that-s-not-an-option-llms-robustness","title":"Wait, that's not an option: LLMs Robustness with Incorrect Multiple-Choice Options","date":"2024-08-27","arxiv_id":"2409.00113","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/wait-that-s-not-an-option-llms-robustness#ran","syntology_url":"https://syntology.ai/paper/2409.00113","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2409.00113"}},"official":{"repos":["gracjangoral/when-all-options-are-wrong"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/blade-benchmarking-language-model-agents-for","slug":"blade-benchmarking-language-model-agents-for","title":"BLADE: Benchmarking Language Model Agents for Data-Driven Science","date":"2024-08-19","arxiv_id":"2408.09667","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/blade-benchmarking-language-model-agents-for#ran","syntology_url":"https://syntology.ai/paper/2408.09667","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.09667"}},"official":{"repos":["behavioral-data/blade"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/measuring-visual-sycophancy-in-multimodal","slug":"measuring-visual-sycophancy-in-multimodal","title":"Measuring Agreeableness Bias in Multimodal Models","date":"2024-08-17","arxiv_id":"2408.09111","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/measuring-visual-sycophancy-in-multimodal#ran","syntology_url":"https://syntology.ai/paper/2408.09111","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.09111"}},"official":{"repos":["jasonlim131/looksRdeceiving"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/med-pmc-medical-personalized-multi-modal","slug":"med-pmc-medical-personalized-multi-modal","title":"Med-PMC: Medical Personalized Multi-modal Consultation with a Proactive Ask-First-Observe-Next Paradigm","date":"2024-08-16","arxiv_id":"2408.08693","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/med-pmc-medical-personalized-multi-modal#ran","syntology_url":"https://syntology.ai/paper/2408.08693","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08693"}},"official":{"repos":["liuhc0428/med-pmc"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/se-sgformer-a-self-explainable-signed-graph","slug":"se-sgformer-a-self-explainable-signed-graph","title":"Self-Explainable Graph Transformer for Link Sign Prediction","date":"2024-08-16","arxiv_id":"2408.08754","repositories_listed":1,"syntology":{"n":9,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":9,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/se-sgformer-a-self-explainable-signed-graph#ran","syntology_url":"https://syntology.ai/paper/2408.08754","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.08754"}},"official":{"repos":["liule66/SE-SGformer"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/the-data-addition-dilemma","slug":"the-data-addition-dilemma","title":"The Data Addition Dilemma","date":"2024-08-08","arxiv_id":"2408.04154","repositories_listed":1,"syntology":{"n":15,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":15,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/the-data-addition-dilemma#ran","syntology_url":"https://syntology.ai/paper/2408.04154","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2408.04154"}},"official":{"repos":["the-chen-lab/data-addition-dilemma"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/interpretable-concept-based-memory-reasoning","slug":"interpretable-concept-based-memory-reasoning","title":"Interpretable Concept-Based Memory Reasoning","date":"2024-07-22","arxiv_id":"2407.15527","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/interpretable-concept-based-memory-reasoning#ran","syntology_url":"https://syntology.ai/paper/2407.15527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15527"}},"official":{"repos":["daviddebot/CMR"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cocog-2-controllable-generation-of-visual","slug":"cocog-2-controllable-generation-of-visual","title":"CoCoG-2: Controllable generation of visual stimuli for understanding human concept representation","date":"2024-07-20","arxiv_id":"2407.14949","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cocog-2-controllable-generation-of-visual#ran","syntology_url":"https://syntology.ai/paper/2407.14949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.14949"}},"official":{"repos":["ncclab-sustech/cocog-2"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-foundation-models-for-online","slug":"adaptive-foundation-models-for-online","title":"Scalable Exploration via Ensemble++","date":"2024-07-18","arxiv_id":"2407.13195","repositories_listed":2,"syntology":{"n":8,"n_ran":5,"n_constructed":1,"n_ran_checked":3,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":8,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/adaptive-foundation-models-for-online#ran","syntology_url":"https://syntology.ai/paper/2407.13195","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.13195"}},"official":{"repos":["szrlee/GPT-HyperAgent","szrlee/ensemble_plus_plus"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/cod-towards-an-interpretable-medical-agent","slug":"cod-towards-an-interpretable-medical-agent","title":"CoD, Towards an Interpretable Medical Agent using Chain of Diagnosis","date":"2024-07-18","arxiv_id":"2407.13301","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cod-towards-an-interpretable-medical-agent#ran","syntology_url":"https://syntology.ai/paper/2407.13301","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.13301"}},"official":{"repos":["freedomintelligence/chain-of-diagnosis"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/last-iterate-global-convergence-of-policy","slug":"last-iterate-global-convergence-of-policy","title":"Last-Iterate Global Convergence of Policy Gradients for Constrained Reinforcement Learning","date":"2024-07-15","arxiv_id":"2407.10775","repositories_listed":0,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/last-iterate-global-convergence-of-policy#ran","syntology_url":"https://syntology.ai/paper/2407.10775","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.10775"}},"official":null}}],"record_sha256":"76d2a4441b723c3067b0599780ff4d4e34242d2bcf2c04f9002494fb90bfa6ae","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}