{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/decision-making/papers/ran/3","list_of":"/task/decision-making","task":"Decision Making","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":3,"pages_in_order":7,"rows_per_page":100,"rows":[201,300],"of":678,"counts":{"archive_papers_tagged":12311,"with_a_code_link":2946,"where_syntology_ran_a_sample":678,"not_listed_spam_title":0,"listed":12311,"listed_where_code_ran":678,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":560,"every_run_a_failure_of_syntologys_instrument":118,"listed_with_a_run_with_no_instrument_failure":560,"listed_every_run_a_failure_of_syntologys_instrument":118,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/decision-making/papers/ran/1","prev":"/task/decision-making/papers/ran/2","next":"/task/decision-making/papers/ran/4","papers":[{"url":"/paper/benchmarking-data-science-agents","slug":"benchmarking-data-science-agents","title":"Benchmarking Data Science Agents","date":"2024-02-27","arxiv_id":"2402.17168","repositories_listed":1,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/benchmarking-data-science-agents#ran","syntology_url":"https://syntology.ai/paper/2402.17168","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.17168"}},"official":{"repos":["metacopilot/dseval"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/ehrnoteqa-a-patient-specific-question","slug":"ehrnoteqa-a-patient-specific-question","title":"EHRNoteQA: An LLM Benchmark for Real-World Clinical Practice Using Discharge Summaries","date":"2024-02-25","arxiv_id":"2402.16040","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ehrnoteqa-a-patient-specific-question#ran","syntology_url":"https://syntology.ai/paper/2402.16040","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16040"}},"official":{"repos":["ji-youn-kim/ehrnoteqa"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/how-can-llm-guide-rl-a-value-based-approach","slug":"how-can-llm-guide-rl-a-value-based-approach","title":"How Can LLM Guide RL? A Value-Based Approach","date":"2024-02-25","arxiv_id":"2402.16181","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":5,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":4,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 1 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/how-can-llm-guide-rl-a-value-based-approach#ran","syntology_url":"https://syntology.ai/paper/2402.16181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.16181"}},"official":{"repos":["agentification/language-integrated-vi"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-agent-collaboration-framework-for","slug":"multi-agent-collaboration-framework-for","title":"MACRec: a Multi-Agent Collaboration Framework for Recommendation","date":"2024-02-23","arxiv_id":"2402.15235","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/multi-agent-collaboration-framework-for#ran","syntology_url":"https://syntology.ai/paper/2402.15235","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15235"}},"official":{"repos":["wzf2000/macrec"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sage-evaluating-moral-consistency-in-large","slug":"sage-evaluating-moral-consistency-in-large","title":"SaGE: Evaluating Moral Consistency in Large Language Models","date":"2024-02-21","arxiv_id":"2402.13709","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sage-evaluating-moral-consistency-in-large#ran","syntology_url":"https://syntology.ai/paper/2402.13709","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.13709"}},"official":{"repos":["vnnm404/SaGE"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-a-better-planning-with-transformers","slug":"beyond-a-better-planning-with-transformers","title":"Beyond A*: Better Planning with Transformers via Search Dynamics Bootstrapping","date":"2024-02-21","arxiv_id":"2402.14083","repositories_listed":1,"syntology":{"n":14,"n_ran":13,"n_constructed":0,"n_ran_checked":13,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":13,"n_pointer_only":14,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 13 with no instrument failure: 0 honoured, 0 violated, 13 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/beyond-a-better-planning-with-transformers#ran","syntology_url":"https://syntology.ai/paper/2402.14083","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14083"}},"official":{"repos":["facebookresearch/searchformer"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":13,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/social-environment-design","slug":"social-environment-design","title":"Social Environment Design","date":"2024-02-21","arxiv_id":"2402.14090","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/social-environment-design#ran","syntology_url":"https://syntology.ai/paper/2402.14090","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.14090"}},"official":{"repos":["ezhang7423/social-environment-design"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/pca-bench-evaluating-multimodal-large","slug":"pca-bench-evaluating-multimodal-large","title":"PCA-Bench: Evaluating Multimodal Large Language Models in Perception-Cognition-Action Chain","date":"2024-02-21","arxiv_id":"2402.15527","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pca-bench-evaluating-multimodal-large#ran","syntology_url":"https://syntology.ai/paper/2402.15527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.15527"}},"official":{"repos":["pkunlp-icler/pca-eval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reflect-rl-two-player-online-rl-fine-tuning","slug":"reflect-rl-two-player-online-rl-fine-tuning","title":"Reflect-RL: Two-Player Online RL Fine-Tuning for LMs","date":"2024-02-20","arxiv_id":"2402.12621","repositories_listed":1,"syntology":{"n":6,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/reflect-rl-two-player-online-rl-fine-tuning#ran","syntology_url":"https://syntology.ai/paper/2402.12621","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12621"}},"official":{"repos":["zhourunlong/reflect-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/unist-a-prompt-empowered-universal-model-for","slug":"unist-a-prompt-empowered-universal-model-for","title":"UniST: A Prompt-Empowered Universal Model for Urban Spatio-Temporal Prediction","date":"2024-02-19","arxiv_id":"2402.11838","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":11,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/unist-a-prompt-empowered-universal-model-for#ran","syntology_url":"https://syntology.ai/paper/2402.11838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.11838"}},"official":{"repos":["tsinghua-fib-lab/unist"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/artifacts-or-abduction-how-do-llms-answer","slug":"artifacts-or-abduction-how-do-llms-answer","title":"Artifacts or Abduction: How Do LLMs Answer Multiple-Choice Questions Without the Question?","date":"2024-02-19","arxiv_id":"2402.12483","repositories_listed":1,"syntology":{"n":10,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":10,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/artifacts-or-abduction-how-do-llms-answer#ran","syntology_url":"https://syntology.ai/paper/2402.12483","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.12483"}},"official":{"repos":["nbalepur/mcqa-artifacts"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/explaining-generative-diffusion-models-via","slug":"explaining-generative-diffusion-models-via","title":"Explaining generative diffusion models via visual analysis for interpretable decision-making process","date":"2024-02-16","arxiv_id":"2402.10404","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/explaining-generative-diffusion-models-via#ran","syntology_url":"https://syntology.ai/paper/2402.10404","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10404"}},"official":{"repos":["ian-jihoonpark/X-Diffusion"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/prise-learning-temporal-action-abstractions","slug":"prise-learning-temporal-action-abstractions","title":"PRISE: LLM-Style Sequence Compression for Learning Temporal Action Abstractions in Control","date":"2024-02-16","arxiv_id":"2402.10450","repositories_listed":1,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/prise-learning-temporal-action-abstractions#ran","syntology_url":"https://syntology.ai/paper/2402.10450","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10450"}},"official":{"repos":["frankzheng2022/prise"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/rag-driver-generalisable-driving-explanations","slug":"rag-driver-generalisable-driving-explanations","title":"RAG-Driver: Generalisable Driving Explanations with Retrieval-Augmented In-Context Learning in Multi-Modal Large Language Model","date":"2024-02-16","arxiv_id":"2402.10828","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rag-driver-generalisable-driving-explanations#ran","syntology_url":"https://syntology.ai/paper/2402.10828","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.10828"}},"official":null}},{"url":"/paper/jack-of-all-trades-master-of-some-a-multi","slug":"jack-of-all-trades-master-of-some-a-multi","title":"Jack of All Trades, Master of Some, a Multi-Purpose Transformer Agent","date":"2024-02-15","arxiv_id":"2402.09844","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/jack-of-all-trades-master-of-some-a-multi#ran","syntology_url":"https://syntology.ai/paper/2402.09844","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.09844"}},"official":{"repos":["huggingface/jat"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/uncertainty-quantification-for-forward-and","slug":"uncertainty-quantification-for-forward-and","title":"Uncertainty Quantification for Forward and Inverse Problems of PDEs via Latent Global Evolution","date":"2024-02-13","arxiv_id":"2402.08383","repositories_listed":2,"syntology":{"n":7,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":3,"n_honours":1,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/uncertainty-quantification-for-forward-and#ran","syntology_url":"https://syntology.ai/paper/2402.08383","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.08383"}},"official":{"repos":["ai4science-westlakeu/le-pde-uq"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/noise-adaptive-confidence-sets-for-linear","slug":"noise-adaptive-confidence-sets-for-linear","title":"Noise-Adaptive Confidence Sets for Linear Bandits and Application to Bayesian Optimization","date":"2024-02-12","arxiv_id":"2402.07341","repositories_listed":1,"syntology":{"n":15,"n_ran":15,"n_constructed":0,"n_ran_checked":12,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/noise-adaptive-confidence-sets-for-linear#ran","syntology_url":"https://syntology.ai/paper/2402.07341","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07341"}},"official":{"repos":["jungtaekkim/losan-lofav"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/smx-sequential-monte-carlo-planning-for","slug":"smx-sequential-monte-carlo-planning-for","title":"SPO: Sequential Monte Carlo Policy Optimisation","date":"2024-02-12","arxiv_id":"2402.07963","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/smx-sequential-monte-carlo-planning-for#ran","syntology_url":"https://syntology.ai/paper/2402.07963","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07963"}},"official":null}},{"url":"/paper/addressing-cognitive-bias-in-medical-language","slug":"addressing-cognitive-bias-in-medical-language","title":"Addressing cognitive bias in medical language models","date":"2024-02-12","arxiv_id":"2402.08113","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/addressing-cognitive-bias-in-medical-language#ran","syntology_url":"https://syntology.ai/paper/2402.08113","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.08113"}},"official":{"repos":["carlwharris/cog-bias-med-llms"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/self-consistent-conformal-prediction","slug":"self-consistent-conformal-prediction","title":"Self-Calibrating Conformal Prediction","date":"2024-02-11","arxiv_id":"2402.07307","repositories_listed":1,"syntology":{"n":25,"n_ran":23,"n_constructed":4,"n_ran_checked":9,"n_instrument":14,"n_unverified":2,"n_honours":2,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"23 ran (of which 4 constructed an object rather than computing a result; 9 with no instrument failure: 2 honoured, 0 violated, 7 with no contract checked; 14 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/self-consistent-conformal-prediction#ran","syntology_url":"https://syntology.ai/paper/2402.07307","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.07307"}},"official":{"repos":["larsvanderlaan/selfcalibratingconformal"],"state":"official (archive's flag): 23 ran","n_ran":23,"n_constructed":4,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/premier-taco-pretraining-multitask","slug":"premier-taco-pretraining-multitask","title":"Premier-TACO is a Few-Shot Policy Learner: Pretraining Multitask Representation via Temporal Action-Driven Contrastive Loss","date":"2024-02-09","arxiv_id":"2402.06187","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":4,"n_instrument":2,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/premier-taco-pretraining-multitask#ran","syntology_url":"https://syntology.ai/paper/2402.06187","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.06187"}},"official":{"repos":["premiertaco/premier-taco"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/entropy-regularized-token-level-policy","slug":"entropy-regularized-token-level-policy","title":"Entropy-Regularized Token-Level Policy Optimization for Language Agent Reinforcement","date":"2024-02-09","arxiv_id":"2402.06700","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/entropy-regularized-token-level-policy#ran","syntology_url":"https://syntology.ai/paper/2402.06700","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.06700"}},"official":{"repos":["morning9393/etpo"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/conformal-monte-carlo-meta-learners-for","slug":"conformal-monte-carlo-meta-learners-for","title":"Conformal Convolution and Monte Carlo Meta-learners for Predictive Inference of Individual Treatment Effects","date":"2024-02-07","arxiv_id":"2402.04906","repositories_listed":2,"syntology":{"n":9,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/conformal-monte-carlo-meta-learners-for#ran","syntology_url":"https://syntology.ai/paper/2402.04906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.04906"}},"official":{"repos":["predict-idlab/cct-cmc","predict-idlab/cmc-learner"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/measuring-implicit-bias-in-explicitly","slug":"measuring-implicit-bias-in-explicitly","title":"Measuring Implicit Bias in Explicitly Unbiased Large Language Models","date":"2024-02-06","arxiv_id":"2402.04105","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":2,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/measuring-implicit-bias-in-explicitly#ran","syntology_url":"https://syntology.ai/paper/2402.04105","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.04105"}},"official":{"repos":["baixuechunzi/llm-implicit-bias"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/skill-set-optimization-reinforcing-language","slug":"skill-set-optimization-reinforcing-language","title":"Skill Set Optimization: Reinforcing Language Model Behavior via Transferable Skills","date":"2024-02-05","arxiv_id":"2402.03244","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/skill-set-optimization-reinforcing-language#ran","syntology_url":"https://syntology.ai/paper/2402.03244","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03244"}},"official":{"repos":["allenai/sso"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/v-irl-grounding-virtual-intelligence-in-real","slug":"v-irl-grounding-virtual-intelligence-in-real","title":"V-IRL: Grounding Virtual Intelligence in Real Life","date":"2024-02-05","arxiv_id":"2402.03310","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/v-irl-grounding-virtual-intelligence-in-real#ran","syntology_url":"https://syntology.ai/paper/2402.03310","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03310"}},"official":{"repos":["VIRL-Platform/VIRL"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/climbing-the-ladder-of-interpretability-with","slug":"climbing-the-ladder-of-interpretability-with","title":"Counterfactual Concept Bottleneck Models","date":"2024-02-02","arxiv_id":"2402.01408","repositories_listed":1,"syntology":{"n":16,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":16,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/climbing-the-ladder-of-interpretability-with#ran","syntology_url":"https://syntology.ai/paper/2402.01408","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.01408"}},"official":{"repos":["gabriele-dominici/counterfactual-cbm"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/vertical-symbolic-regression-via-deep-policy","slug":"vertical-symbolic-regression-via-deep-policy","title":"Vertical Symbolic Regression via Deep Policy Gradient","date":"2024-02-01","arxiv_id":"2402.00254","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vertical-symbolic-regression-via-deep-policy#ran","syntology_url":"https://syntology.ai/paper/2402.00254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.00254"}},"official":{"repos":["jiangnanhugo/vsr-dpg"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hierarchical-bias-driven-stratification-for","slug":"hierarchical-bias-driven-stratification-for","title":"Hierarchical Bias-Driven Stratification for Interpretable Causal Effect Estimation","date":"2024-01-31","arxiv_id":"2401.17737","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hierarchical-bias-driven-stratification-for#ran","syntology_url":"https://syntology.ai/paper/2401.17737","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.17737"}},"official":{"repos":["ibm-hrl-mlhls/bicause-trees"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/zero-shot-reinforcement-learning-via-function","slug":"zero-shot-reinforcement-learning-via-function","title":"Zero-Shot Reinforcement Learning via Function Encoders","date":"2024-01-30","arxiv_id":"2401.17173","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/zero-shot-reinforcement-learning-via-function#ran","syntology_url":"https://syntology.ai/paper/2401.17173","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.17173"}},"official":{"repos":["anonymousresearcher5642/functionencoderrl","tyler-ingebrand/functionencoderrl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/goat-explaining-graph-neural-networks-via","slug":"goat-explaining-graph-neural-networks-via","title":"GOAt: Explaining Graph Neural Networks via Graph Output Attribution","date":"2024-01-26","arxiv_id":"2401.14578","repositories_listed":1,"syntology":{"n":13,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":6,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":13,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/goat-explaining-graph-neural-networks-via#ran","syntology_url":"https://syntology.ai/paper/2401.14578","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.14578"}},"official":{"repos":["sluxsr/goat"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/true-knowledge-comes-from-practice-aligning","slug":"true-knowledge-comes-from-practice-aligning","title":"True Knowledge Comes from Practice: Aligning LLMs with Embodied Environments via Reinforcement Learning","date":"2024-01-25","arxiv_id":"2401.14151","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/true-knowledge-comes-from-practice-aligning#ran","syntology_url":"https://syntology.ai/paper/2401.14151","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.14151"}},"official":{"repos":["weihaotan/twosome"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prompting-large-language-models-for-zero-shot-1","slug":"prompting-large-language-models-for-zero-shot-1","title":"Prompting Large Language Models for Zero-Shot Clinical Prediction with Structured Longitudinal Electronic Health Record Data","date":"2024-01-25","arxiv_id":"2402.01713","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/prompting-large-language-models-for-zero-shot-1#ran","syntology_url":"https://syntology.ai/paper/2402.01713","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.01713"}},"official":{"repos":["yhzhu99/llm4healthcare"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/respect-the-model-fine-grained-and-robust-1","slug":"respect-the-model-fine-grained-and-robust-1","title":"Respect the model: Fine-grained and Robust Explanation with Sharing Ratio Decomposition","date":"2024-01-25","arxiv_id":"2402.03348","repositories_listed":1,"syntology":{"n":11,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":8,"n_honours":1,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","sample_list":"/paper/respect-the-model-fine-grained-and-robust-1#ran","syntology_url":"https://syntology.ai/paper/2402.03348","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.03348"}},"official":null}},{"url":"/paper/conformal-prediction-sets-improve-human","slug":"conformal-prediction-sets-improve-human","title":"Conformal Prediction Sets Improve Human Decision Making","date":"2024-01-24","arxiv_id":"2401.13744","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conformal-prediction-sets-improve-human#ran","syntology_url":"https://syntology.ai/paper/2401.13744","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.13744"}},"official":{"repos":["layer6ai-labs/hitl-conformal-prediction"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/hazard-challenge-embodied-decision-making-in","slug":"hazard-challenge-embodied-decision-making-in","title":"HAZARD Challenge: Embodied Decision Making in Dynamically Changing Environments","date":"2024-01-23","arxiv_id":"2401.12975","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/hazard-challenge-embodied-decision-making-in#ran","syntology_url":"https://syntology.ai/paper/2401.12975","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.12975"}},"official":{"repos":["umass-foundation-model/hazard"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/organa-a-robotic-assistant-for-automated","slug":"organa-a-robotic-assistant-for-automated","title":"ORGANA: A Robotic Assistant for Automated Chemistry Experimentation and Characterization","date":"2024-01-13","arxiv_id":"2401.06949","repositories_listed":1,"syntology":{"n":18,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":18,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/organa-a-robotic-assistant-for-automated#ran","syntology_url":"https://syntology.ai/paper/2401.06949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.06949"}},"official":{"repos":["ac-rad/organa"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/manipulating-feature-visualizations-with","slug":"manipulating-feature-visualizations-with","title":"Manipulating Feature Visualizations with Gradient Slingshots","date":"2024-01-11","arxiv_id":"2401.06122","repositories_listed":1,"syntology":{"n":18,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":12,"n_pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/manipulating-feature-visualizations-with#ran","syntology_url":"https://syntology.ai/paper/2401.06122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.06122"}},"official":{"repos":["dilyabareeva/grad-slingshot"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/evaluating-language-model-agency-through","slug":"evaluating-language-model-agency-through","title":"Evaluating Language Model Agency through Negotiations","date":"2024-01-09","arxiv_id":"2401.04536","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evaluating-language-model-agency-through#ran","syntology_url":"https://syntology.ai/paper/2401.04536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.04536"}},"official":{"repos":["epfl-dlab/lamen"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/escalation-risks-from-language-models-in","slug":"escalation-risks-from-language-models-in","title":"Escalation Risks from Language Models in Military and Diplomatic Decision-Making","date":"2024-01-07","arxiv_id":"2401.03408","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/escalation-risks-from-language-models-in#ran","syntology_url":"https://syntology.ai/paper/2401.03408","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.03408"}},"official":{"repos":["jprivera44/EscalAItion"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/calibration-attack-a-framework-for","slug":"calibration-attack-a-framework-for","title":"Calibration Attacks: A Comprehensive Study of Adversarial Attacks on Model Confidence","date":"2024-01-05","arxiv_id":"2401.02718","repositories_listed":1,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":12,"n_instrument":0,"n_unverified":2,"n_honours":4,"n_violates":0,"n_no_contract":8,"n_pointer_only":14,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 4 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/calibration-attack-a-framework-for#ran","syntology_url":"https://syntology.ai/paper/2401.02718","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.02718"}},"official":{"repos":["phenetos/calibrationattack"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/chartassisstant-a-universal-chart-multimodal","slug":"chartassisstant-a-universal-chart-multimodal","title":"ChartAssisstant: A Universal Chart Multimodal Language Model via Chart-to-Table Pre-training and Multitask Instruction Tuning","date":"2024-01-04","arxiv_id":"2401.02384","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/chartassisstant-a-universal-chart-multimodal#ran","syntology_url":"https://syntology.ai/paper/2401.02384","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2401.02384"}},"official":{"repos":["opengvlab/chartast"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-as-traffic-signal","slug":"large-language-models-as-traffic-signal","title":"LLMLight: Large Language Models as Traffic Signal Control Agents","date":"2023-12-26","arxiv_id":"2312.16044","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/large-language-models-as-traffic-signal#ran","syntology_url":"https://syntology.ai/paper/2312.16044","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.16044"}},"official":{"repos":["usail-hkust/llmtscs"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/harnessing-the-power-of-federated-learning-in","slug":"harnessing-the-power-of-federated-learning-in","title":"Harnessing the Power of Federated Learning in Federated Contextual Bandits","date":"2023-12-26","arxiv_id":"2312.16341","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/harnessing-the-power-of-federated-learning-in#ran","syntology_url":"https://syntology.ai/paper/2312.16341","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.16341"}},"official":{"repos":["shengroup/fedigw"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/unlocking-the-potential-of-large-language","slug":"unlocking-the-potential-of-large-language","title":"Unlocking the Potential of Large Language Models for Explainable Recommendations","date":"2023-12-25","arxiv_id":"2312.15661","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/unlocking-the-potential-of-large-language#ran","syntology_url":"https://syntology.ai/paper/2312.15661","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.15661"}},"official":{"repos":["godfire66666/llm_rec_explanation"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/gencast-diffusion-based-ensemble-forecasting","slug":"gencast-diffusion-based-ensemble-forecasting","title":"GenCast: Diffusion-based ensemble forecasting for medium-range weather","date":"2023-12-25","arxiv_id":"2312.15796","repositories_listed":3,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gencast-diffusion-based-ensemble-forecasting#ran","syntology_url":"https://syntology.ai/paper/2312.15796","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.15796"}},"official":null}},{"url":"/paper/solving-long-run-average-reward-robust-mdps","slug":"solving-long-run-average-reward-robust-mdps","title":"Solving Long-run Average Reward Robust MDPs via Stochastic Games","date":"2023-12-21","arxiv_id":"2312.13912","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/solving-long-run-average-reward-robust-mdps#ran","syntology_url":"https://syntology.ai/paper/2312.13912","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.13912"}},"official":{"repos":["mehrdad76/rmdp-lra"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/risk-sensitive-stochastic-optimal-control-as","slug":"risk-sensitive-stochastic-optimal-control-as","title":"Risk-Sensitive Stochastic Optimal Control as Rao-Blackwellized Markovian Score Climbing","date":"2023-12-21","arxiv_id":"2312.14000","repositories_listed":1,"syntology":{"n":11,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":11,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified","sample_list":"/paper/risk-sensitive-stochastic-optimal-control-as#ran","syntology_url":"https://syntology.ai/paper/2312.14000","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14000"}},"official":{"repos":["hanyas/psoc"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":9,"ran_from_kinds":["official"]}}},{"url":"/paper/lingoqa-video-question-answering-for","slug":"lingoqa-video-question-answering-for","title":"LingoQA: Visual Question Answering for Autonomous Driving","date":"2023-12-21","arxiv_id":"2312.14115","repositories_listed":2,"syntology":{"n":10,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lingoqa-video-question-answering-for#ran","syntology_url":"https://syntology.ai/paper/2312.14115","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.14115"}},"official":{"repos":["wayveai/lingoqa"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/fifar-a-fraud-detection-dataset-for-learning","slug":"fifar-a-fraud-detection-dataset-for-learning","title":"FiFAR: A Fraud Detection Dataset for Learning to Defer","date":"2023-12-20","arxiv_id":"2312.13218","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fifar-a-fraud-detection-dataset-for-learning#ran","syntology_url":"https://syntology.ai/paper/2312.13218","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.13218"}},"official":{"repos":["feedzai/fifar-dataset"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/event-based-contrastive-learning-for-medical","slug":"event-based-contrastive-learning-for-medical","title":"Event-Based Contrastive Learning for Medical Time Series","date":"2023-12-16","arxiv_id":"2312.10308","repositories_listed":1,"syntology":{"n":14,"n_ran":8,"n_constructed":5,"n_ran_checked":6,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 5 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/event-based-contrastive-learning-for-medical#ran","syntology_url":"https://syntology.ai/paper/2312.10308","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.10308"}},"official":{"repos":["mit-ccrg/ebcl"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":5,"n_ran_no_instrument_failure":6,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/diff-history-for-long-context-language-agents","slug":"diff-history-for-long-context-language-agents","title":"diff History for Neural Language Agents","date":"2023-12-12","arxiv_id":"2312.07540","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/diff-history-for-long-context-language-agents#ran","syntology_url":"https://syntology.ai/paper/2312.07540","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.07540"}},"official":{"repos":["upiterbarg/diff_history"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/diffail-diffusion-adversarial-imitation","slug":"diffail-diffusion-adversarial-imitation","title":"DiffAIL: Diffusion Adversarial Imitation Learning","date":"2023-12-11","arxiv_id":"2312.06348","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diffail-diffusion-adversarial-imitation#ran","syntology_url":"https://syntology.ai/paper/2312.06348","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06348"}},"official":{"repos":["ml-group-sdu/diffail"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/bat-behavior-aware-human-like-trajectory","slug":"bat-behavior-aware-human-like-trajectory","title":"BAT: Behavior-Aware Human-Like Trajectory Prediction for Autonomous Driving","date":"2023-12-11","arxiv_id":"2312.06371","repositories_listed":1,"syntology":{"n":4,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":4,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/bat-behavior-aware-human-like-trajectory#ran","syntology_url":"https://syntology.ai/paper/2312.06371","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06371"}},"official":{"repos":["petrichor625/batraj-behavior-aware-model"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/can-reinforcement-learning-support-policy","slug":"can-reinforcement-learning-support-policy","title":"Can Reinforcement Learning support policy makers? A preliminary study with Integrated Assessment Models","date":"2023-12-11","arxiv_id":"2312.06527","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-reinforcement-learning-support-policy#ran","syntology_url":"https://syntology.ai/paper/2312.06527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.06527"}},"official":null}},{"url":"/paper/using-large-language-models-for","slug":"using-large-language-models-for","title":"Using Large Language Models for Hyperparameter Optimization","date":"2023-12-07","arxiv_id":"2312.04528","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/using-large-language-models-for#ran","syntology_url":"https://syntology.ai/paper/2312.04528","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.04528"}},"official":{"repos":["michaelrzhang/llm-hyperopt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/generalization-to-new-sequential-decision","slug":"generalization-to-new-sequential-decision","title":"Generalization to New Sequential Decision Making Tasks with In-Context Learning","date":"2023-12-06","arxiv_id":"2312.03801","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/generalization-to-new-sequential-decision#ran","syntology_url":"https://syntology.ai/paper/2312.03801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03801"}},"official":null}},{"url":"/paper/expert-guided-bayesian-optimisation-for-human","slug":"expert-guided-bayesian-optimisation-for-human","title":"Expert-guided Bayesian Optimisation for Human-in-the-loop Experimental Design of Known Systems","date":"2023-12-05","arxiv_id":"2312.02852","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/expert-guided-bayesian-optimisation-for-human#ran","syntology_url":"https://syntology.ai/paper/2312.02852","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.02852"}},"official":{"repos":["trsav/hitl-bo"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-curricula-in-open-ended-worlds","slug":"learning-curricula-in-open-ended-worlds","title":"Learning Curricula in Open-Ended Worlds","date":"2023-12-03","arxiv_id":"2312.03126","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":3,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-curricula-in-open-ended-worlds#ran","syntology_url":"https://syntology.ai/paper/2312.03126","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2312.03126"}},"official":{"repos":["facebookresearch/dcd"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/understanding-the-extra-ordinary-validating","slug":"understanding-the-extra-ordinary-validating","title":"Understanding the (Extra-)Ordinary: Validating Deep Model Decisions with Prototypical Concept-based Explanations","date":"2023-11-28","arxiv_id":"2311.16681","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/understanding-the-extra-ordinary-validating#ran","syntology_url":"https://syntology.ai/paper/2311.16681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.16681"}},"official":{"repos":["maxdreyer/pcx"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/utilizing-explainability-techniques-for","slug":"utilizing-explainability-techniques-for","title":"Utilizing Explainability Techniques for Reinforcement Learning Model Assurance","date":"2023-11-27","arxiv_id":"2311.15838","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/utilizing-explainability-techniques-for#ran","syntology_url":"https://syntology.ai/paper/2311.15838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.15838"}},"official":{"repos":["mitre/arlin"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/finme-a-performance-enhanced-large-language","slug":"finme-a-performance-enhanced-large-language","title":"FinMem: A Performance-Enhanced LLM Trading Agent with Layered Memory and Character Design","date":"2023-11-23","arxiv_id":"2311.13743","repositories_listed":2,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/finme-a-performance-enhanced-large-language#ran","syntology_url":"https://syntology.ai/paper/2311.13743","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.13743"}},"official":{"repos":["pipiku915/finmem-llm-stocktrading"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/2311-13594","slug":"2311-13594","title":"Labeling Neural Representations with Inverse Recognition","date":"2023-11-22","arxiv_id":"2311.13594","repositories_listed":2,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2311-13594#ran","syntology_url":"https://syntology.ai/paper/2311.13594","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.13594"}},"official":{"repos":["lapalap/invert"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/inherently-interpretable-time-series","slug":"inherently-interpretable-time-series","title":"Inherently Interpretable Time Series Classification via Multiple Instance Learning","date":"2023-11-16","arxiv_id":"2311.10049","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/inherently-interpretable-time-series#ran","syntology_url":"https://syntology.ai/paper/2311.10049","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10049"}},"official":{"repos":["jaearly/miltimeseriesclassification"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/navigating-the-ocean-of-biases-political-bias","slug":"navigating-the-ocean-of-biases-political-bias","title":"Exploring the Jungle of Bias: Political Bias Attribution in Language Models via Dependency Analysis","date":"2023-11-15","arxiv_id":"2311.08605","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/navigating-the-ocean-of-biases-political-bias#ran","syntology_url":"https://syntology.ai/paper/2311.08605","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.08605"}},"official":{"repos":["david-jenny/llm-political-study"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/tooltalk-evaluating-tool-usage-in-a","slug":"tooltalk-evaluating-tool-usage-in-a","title":"ToolTalk: Evaluating Tool-Usage in a Conversational Setting","date":"2023-11-15","arxiv_id":"2311.10775","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tooltalk-evaluating-tool-usage-in-a#ran","syntology_url":"https://syntology.ai/paper/2311.10775","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10775"}},"official":null}},{"url":"/paper/rethinking-and-benchmarking-predict-then","slug":"rethinking-and-benchmarking-predict-then","title":"Benchmarking PtO and PnO Methods in the Predictive Combinatorial Optimization Regime","date":"2023-11-13","arxiv_id":"2311.07633","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rethinking-and-benchmarking-predict-then#ran","syntology_url":"https://syntology.ai/paper/2311.07633","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07633"}},"official":{"repos":["thinklab-sjtu/predictiveco-benchmark"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/adapt-as-needed-decomposition-and-planning","slug":"adapt-as-needed-decomposition-and-planning","title":"ADaPT: As-Needed Decomposition and Planning with Language Models","date":"2023-11-08","arxiv_id":"2311.05772","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adapt-as-needed-decomposition-and-planning#ran","syntology_url":"https://syntology.ai/paper/2311.05772","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.05772"}},"official":null}},{"url":"/paper/everything-of-thoughts-defying-the-law-of","slug":"everything-of-thoughts-defying-the-law-of","title":"Everything of Thoughts: Defying the Law of Penrose Triangle for Thought Generation","date":"2023-11-07","arxiv_id":"2311.04254","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/everything-of-thoughts-defying-the-law-of#ran","syntology_url":"https://syntology.ai/paper/2311.04254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04254"}},"official":{"repos":["microsoft/everything-of-thoughts-xot-"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/proagent-from-robotic-process-automation-to","slug":"proagent-from-robotic-process-automation-to","title":"ProAgent: From Robotic Process Automation to Agentic Process Automation","date":"2023-11-02","arxiv_id":"2311.10751","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/proagent-from-robotic-process-automation-to#ran","syntology_url":"https://syntology.ai/paper/2311.10751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10751"}},"official":{"repos":["openbmb/proagent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-foundation-models-watch-talk-and-guide","slug":"can-foundation-models-watch-talk-and-guide","title":"Can Foundation Models Watch, Talk and Guide You Step by Step to Make a Cake?","date":"2023-11-01","arxiv_id":"2311.00738","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/can-foundation-models-watch-talk-and-guide#ran","syntology_url":"https://syntology.ai/paper/2311.00738","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.00738"}},"official":{"repos":["sled-group/watch-talk-and-guide"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/calibration-by-distribution-matching-1","slug":"calibration-by-distribution-matching-1","title":"Calibration by Distribution Matching: Trainable Kernel Calibration Metrics","date":"2023-10-31","arxiv_id":"2310.20211","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/calibration-by-distribution-matching-1#ran","syntology_url":"https://syntology.ai/paper/2310.20211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.20211"}},"official":{"repos":["kernel-calibration/kernel-calibration"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/interpretable-prototype-based-graph-1","slug":"interpretable-prototype-based-graph-1","title":"Interpretable Prototype-based Graph Information Bottleneck","date":"2023-10-30","arxiv_id":"2310.19906","repositories_listed":1,"syntology":{"n":49,"n_ran":27,"n_constructed":8,"n_ran_checked":12,"n_instrument":15,"n_unverified":22,"n_honours":3,"n_violates":0,"n_no_contract":9,"n_pointer_only":48,"phrase":"27 ran (of which 8 constructed an object rather than computing a result; 12 with no instrument failure: 3 honoured, 0 violated, 9 with no contract checked; 15 where Syntology's instrument failed) · 22 unverified","sample_list":"/paper/interpretable-prototype-based-graph-1#ran","syntology_url":"https://syntology.ai/paper/2310.19906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.19906"}},"official":{"repos":["sang-woo-seo/pgib"],"state":"official (archive's flag): 25 ran","n_ran":25,"n_constructed":8,"n_ran_no_instrument_failure":11,"n_unverified":22,"ran_from_kinds":["community","official","unlocated"]}}},{"url":"/paper/ehrxqa-a-multi-modal-question-answering-1","slug":"ehrxqa-a-multi-modal-question-answering-1","title":"EHRXQA: A Multi-Modal Question Answering Dataset for Electronic Health Records with Chest X-ray Images","date":"2023-10-28","arxiv_id":"2310.18652","repositories_listed":3,"syntology":{"n":20,"n_ran":20,"n_constructed":0,"n_ran_checked":17,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":17,"n_pointer_only":0,"phrase":"20 ran (of which 0 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 0 violated, 17 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/ehrxqa-a-multi-modal-question-answering-1#ran","syntology_url":"https://syntology.ai/paper/2310.18652","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.18652"}},"official":{"repos":["baeseongsu/ehrxqa","baeseongsu/mimic-cxr-vqa"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/how-well-do-feature-additive-explainers","slug":"how-well-do-feature-additive-explainers","title":"How Well Do Feature-Additive Explainers Explain Feature-Additive Predictors?","date":"2023-10-27","arxiv_id":"2310.18496","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/how-well-do-feature-additive-explainers#ran","syntology_url":"https://syntology.ai/paper/2310.18496","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.18496"}},"official":{"repos":["craymichael/PostHocExplainerEvaluation"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/sum-of-parts-models-faithful-attributions-for","slug":"sum-of-parts-models-faithful-attributions-for","title":"Sum-of-Parts: Faithful Attributions for Groups of Features","date":"2023-10-25","arxiv_id":"2310.16316","repositories_listed":2,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/sum-of-parts-models-faithful-attributions-for#ran","syntology_url":"https://syntology.ai/paper/2310.16316","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.16316"}},"official":{"repos":["brachiolab/sop","debugml/sop"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-successor-representations-with","slug":"learning-successor-representations-with","title":"Learning Successor Features with Distributed Hebbian Temporal Memory","date":"2023-10-20","arxiv_id":"2310.13391","repositories_listed":0,"syntology":{"n":15,"n_ran":8,"n_constructed":0,"n_ran_checked":1,"n_instrument":7,"n_unverified":7,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 7 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/learning-successor-representations-with#ran","syntology_url":"https://syntology.ai/paper/2310.13391","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.13391"}},"official":null}},{"url":"/paper/eureka-human-level-reward-design-via-coding","slug":"eureka-human-level-reward-design-via-coding","title":"Eureka: Human-Level Reward Design via Coding Large Language Models","date":"2023-10-19","arxiv_id":"2310.12931","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/eureka-human-level-reward-design-via-coding#ran","syntology_url":"https://syntology.ai/paper/2310.12931","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12931"}},"official":{"repos":["eureka-research/Eureka"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"url":"/paper/sensitivity-aware-amortized-bayesian","slug":"sensitivity-aware-amortized-bayesian","title":"Sensitivity-Aware Amortized Bayesian Inference","date":"2023-10-17","arxiv_id":"2310.11122","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sensitivity-aware-amortized-bayesian#ran","syntology_url":"https://syntology.ai/paper/2310.11122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.11122"}},"official":{"repos":["bayesflow-org/SA-ABI"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/agent-specific-effects","slug":"agent-specific-effects","title":"Agent-Specific Effects: A Causal Effect Propagation Analysis in Multi-Agent MDPs","date":"2023-10-17","arxiv_id":"2310.11334","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/agent-specific-effects#ran","syntology_url":"https://syntology.ai/paper/2310.11334","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.11334"}},"official":{"repos":["stelios30/agent-specific-effects"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-model-empowered-agents-for","slug":"large-language-model-empowered-agents-for","title":"EconAgent: Large Language Model-Empowered Agents for Simulating Macroeconomic Activities","date":"2023-10-16","arxiv_id":"2310.10436","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/large-language-model-empowered-agents-for#ran","syntology_url":"https://syntology.ai/paper/2310.10436","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.10436"}},"official":{"repos":["tsinghua-fib-lab/acl24-econagent"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/step-by-step-remediation-of-students","slug":"step-by-step-remediation-of-students","title":"Bridging the Novice-Expert Gap via Models of Decision-Making: A Case Study on Remediating Math Mistakes","date":"2023-10-16","arxiv_id":"2310.10648","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/step-by-step-remediation-of-students#ran","syntology_url":"https://syntology.ai/paper/2310.10648","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.10648"}},"official":{"repos":["rosewang2008/bridge"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/octopus-embodied-vision-language-programmer","slug":"octopus-embodied-vision-language-programmer","title":"Octopus: Embodied Vision-Language Programmer from Environmental Feedback","date":"2023-10-12","arxiv_id":"2310.08588","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":6,"n_instrument":6,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":4,"n_pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/octopus-embodied-vision-language-programmer#ran","syntology_url":"https://syntology.ai/paper/2310.08588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08588"}},"official":{"repos":["dongyh20/octopus"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/imitation-learning-from-purified","slug":"imitation-learning-from-purified","title":"Imitation Learning from Purified Demonstrations","date":"2023-10-11","arxiv_id":"2310.07143","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":2,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/imitation-learning-from-purified#ran","syntology_url":"https://syntology.ai/paper/2310.07143","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.07143"}},"official":{"repos":["yunke-wang/dp-il"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/linear-latent-world-models-in-simple","slug":"linear-latent-world-models-in-simple","title":"Linear Latent World Models in Simple Transformers: A Case Study on Othello-GPT","date":"2023-10-11","arxiv_id":"2310.07582","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/linear-latent-world-models-in-simple#ran","syntology_url":"https://syntology.ai/paper/2310.07582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.07582"}},"official":{"repos":["deanhazineh/emergent-world-representations-othello"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/qacheck-a-demonstration-system-for-question","slug":"qacheck-a-demonstration-system-for-question","title":"QACHECK: A Demonstration System for Question-Guided Multi-Hop Fact-Checking","date":"2023-10-11","arxiv_id":"2310.07609","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/qacheck-a-demonstration-system-for-question#ran","syntology_url":"https://syntology.ai/paper/2310.07609","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.07609"}},"official":{"repos":["xinyuanlu00/qacheck"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/dsac-t-distributional-soft-actor-critic-with","slug":"dsac-t-distributional-soft-actor-critic-with","title":"Distributional Soft Actor-Critic with Three Refinements","date":"2023-10-09","arxiv_id":"2310.05858","repositories_listed":2,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dsac-t-distributional-soft-actor-critic-with#ran","syntology_url":"https://syntology.ai/paper/2310.05858","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.05858"}},"official":{"repos":["jingliang-duan/dsac-t","jingliang-duan/dsac-v2"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/are-large-language-models-geospatially","slug":"are-large-language-models-geospatially","title":"Are Large Language Models Geospatially Knowledgeable?","date":"2023-10-09","arxiv_id":"2310.13002","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/are-large-language-models-geospatially#ran","syntology_url":"https://syntology.ai/paper/2310.13002","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.13002"}},"official":{"repos":["prabin525/spatial-llm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/language-agent-tree-search-unifies-reasoning","slug":"language-agent-tree-search-unifies-reasoning","title":"Language Agent Tree Search Unifies Reasoning Acting and Planning in Language Models","date":"2023-10-06","arxiv_id":"2310.04406","repositories_listed":2,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/language-agent-tree-search-unifies-reasoning#ran","syntology_url":"https://syntology.ai/paper/2310.04406","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04406"}},"official":{"repos":["lapisrocks/languageagenttreesearch","andyz245/LanguageAgentTreeSearch"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-uniform-sampling-offline-reinforcement-1","slug":"beyond-uniform-sampling-offline-reinforcement-1","title":"Beyond Uniform Sampling: Offline Reinforcement Learning with Imbalanced Datasets","date":"2023-10-06","arxiv_id":"2310.04413","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/beyond-uniform-sampling-offline-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2310.04413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04413"}},"official":{"repos":["Improbable-AI/dw-offline-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/benchmarking-large-language-models-as-ai","slug":"benchmarking-large-language-models-as-ai","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","date":"2023-10-05","arxiv_id":"2310.03302","repositories_listed":2,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/benchmarking-large-language-models-as-ai#ran","syntology_url":"https://syntology.ai/paper/2310.03302","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03302"}},"official":{"repos":["snap-stanford/mlagentbench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/metatool-benchmark-deciding-whether-to-use","slug":"metatool-benchmark-deciding-whether-to-use","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","date":"2023-10-04","arxiv_id":"2310.03128","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/metatool-benchmark-deciding-whether-to-use#ran","syntology_url":"https://syntology.ai/paper/2310.03128","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03128"}},"official":{"repos":["howiehwong/metatool"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-robust-fidelity-for-evaluating","slug":"towards-robust-fidelity-for-evaluating","title":"Towards Robust Fidelity for Evaluating Explainability of Graph Neural Networks","date":"2023-10-03","arxiv_id":"2310.01820","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/towards-robust-fidelity-for-evaluating#ran","syntology_url":"https://syntology.ai/paper/2310.01820","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01820"}},"official":{"repos":["AslanDing/Fidelity"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/mini-behavior-a-procedurally-generated","slug":"mini-behavior-a-procedurally-generated","title":"Mini-BEHAVIOR: A Procedurally Generated Benchmark for Long-horizon Decision-Making in Embodied AI","date":"2023-10-03","arxiv_id":"2310.01824","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/mini-behavior-a-procedurally-generated#ran","syntology_url":"https://syntology.ai/paper/2310.01824","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01824"}},"official":{"repos":["stanfordvl/mini_behavior"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/autocast-enhancing-world-event-prediction","slug":"autocast-enhancing-world-event-prediction","title":"AutoCast++: Enhancing World Event Prediction with Zero-shot Ranking-based Context Retrieval","date":"2023-10-03","arxiv_id":"2310.01880","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/autocast-enhancing-world-event-prediction#ran","syntology_url":"https://syntology.ai/paper/2310.01880","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01880"}},"official":{"repos":["BorealisAI/Autocast-plus-plus"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/talk2bev-language-enhanced-bird-s-eye-view","slug":"talk2bev-language-enhanced-bird-s-eye-view","title":"Talk2BEV: Language-enhanced Bird's-eye View Maps for Autonomous Driving","date":"2023-10-03","arxiv_id":"2310.02251","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/talk2bev-language-enhanced-bird-s-eye-view#ran","syntology_url":"https://syntology.ai/paper/2310.02251","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.02251"}},"official":null}},{"url":"/paper/autodan-generating-stealthy-jailbreak-prompts","slug":"autodan-generating-stealthy-jailbreak-prompts","title":"AutoDAN: Generating Stealthy Jailbreak Prompts on Aligned Large Language Models","date":"2023-10-03","arxiv_id":"2310.04451","repositories_listed":2,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":1,"n_instrument":5,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/autodan-generating-stealthy-jailbreak-prompts#ran","syntology_url":"https://syntology.ai/paper/2310.04451","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04451"}},"official":{"repos":["sheltonliu-n/autodan"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-neural-networks-tend-to-extrapolate","slug":"deep-neural-networks-tend-to-extrapolate","title":"Deep Neural Networks Tend To Extrapolate Predictably","date":"2023-10-02","arxiv_id":"2310.00873","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/deep-neural-networks-tend-to-extrapolate#ran","syntology_url":"https://syntology.ai/paper/2310.00873","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00873"}},"official":{"repos":["katiekang1998/cautious_extrapolation"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/gpt-driver-learning-to-drive-with-gpt","slug":"gpt-driver-learning-to-drive-with-gpt","title":"GPT-Driver: Learning to Drive with GPT","date":"2023-10-02","arxiv_id":"2310.01415","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/gpt-driver-learning-to-drive-with-gpt#ran","syntology_url":"https://syntology.ai/paper/2310.01415","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01415"}},"official":{"repos":["pointscoder/gpt-driver"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/combining-spatial-and-temporal-abstraction-in","slug":"combining-spatial-and-temporal-abstraction-in","title":"Consciousness-Inspired Spatio-Temporal Abstractions for Better Generalization in Reinforcement Learning","date":"2023-09-30","arxiv_id":"2310.00229","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/combining-spatial-and-temporal-abstraction-in#ran","syntology_url":"https://syntology.ai/paper/2310.00229","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.00229"}},"official":{"repos":["mila-iqia/skipper"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"7a190187cdbfb83780a7fa4736fde2cf676ae973bbb969ca96f1f12a4b75523e","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}