{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/decision-making/papers/14","list_of":"/task/decision-making","task":"Decision Making","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":14,"pages_in_order":124,"rows_per_page":100,"rows":[1301,1400],"of":12311,"counts":{"archive_papers_tagged":12311,"with_a_code_link":2946,"where_syntology_ran_a_sample":678,"not_listed_spam_title":0,"listed":12311,"listed_where_code_ran":678,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":560,"every_run_a_failure_of_syntologys_instrument":118,"listed_with_a_run_with_no_instrument_failure":560,"listed_every_run_a_failure_of_syntologys_instrument":118,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/decision-making","prev":"/task/decision-making/papers/13","next":"/task/decision-making/papers/15","papers":[{"url":"/paper/towards-clinical-prediction-with-transparency","slug":"towards-clinical-prediction-with-transparency","title":"Towards Clinical Prediction with Transparency: An Explainable AI Approach to Survival Modelling in Residential Aged Care","date":"2023-12-01","arxiv_id":"2312.00271","repositories_listed":1,"syntology":null},{"url":"/paper/detection-of-tonic-clonic-seizures-in","slug":"detection-of-tonic-clonic-seizures-in","title":"Detection of tonic-clonic seizures in children with epilepsy","date":"2023-11-30","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/language-model-agents-suffer-from","slug":"language-model-agents-suffer-from","title":"Exposing Limitations of Language Model Agents in Sequential-Task Compositions on the Web","date":"2023-11-30","arxiv_id":"2311.18751","repositories_listed":1,"syntology":null},{"url":"/paper/optimizing-zx-diagrams-with-deep","slug":"optimizing-zx-diagrams-with-deep","title":"Optimizing ZX-Diagrams with Deep Reinforcement Learning","date":"2023-11-30","arxiv_id":"2311.18588","repositories_listed":1,"syntology":null},{"url":"/paper/joint-network-for-specular-highlight","slug":"joint-network-for-specular-highlight","title":"Joint network for specular highlight detection and adversarial generation of specular-free images trained with polarimetric data","date":"2023-11-28","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/on-the-robustness-of-decision-focused","slug":"on-the-robustness-of-decision-focused","title":"On the Robustness of Decision-Focused Learning","date":"2023-11-28","arxiv_id":"2311.16487","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-the-extra-ordinary-validating","slug":"understanding-the-extra-ordinary-validating","title":"Understanding the (Extra-)Ordinary: Validating Deep Model Decisions with Prototypical Concept-based Explanations","date":"2023-11-28","arxiv_id":"2311.16681","repositories_listed":1,"syntology":{"n":6,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/understanding-the-extra-ordinary-validating#ran","syntology_url":"https://syntology.ai/paper/2311.16681","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.16681"}},"official":{"repos":["maxdreyer/pcx"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/utilizing-explainability-techniques-for","slug":"utilizing-explainability-techniques-for","title":"Utilizing Explainability Techniques for Reinforcement Learning Model Assurance","date":"2023-11-27","arxiv_id":"2311.15838","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/utilizing-explainability-techniques-for#ran","syntology_url":"https://syntology.ai/paper/2311.15838","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.15838"}},"official":{"repos":["mitre/arlin"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/video-bench-a-comprehensive-benchmark-and","slug":"video-bench-a-comprehensive-benchmark-and","title":"Video-Bench: A Comprehensive Benchmark and Toolkit for Evaluating Video-based Large Language Models","date":"2023-11-27","arxiv_id":"2311.16103","repositories_listed":1,"syntology":null},{"url":"/paper/token-recycling-for-efficient-sequential","slug":"token-recycling-for-efficient-sequential","title":"TORE: Token Recycling in Vision Transformers for Efficient Active Visual Exploration","date":"2023-11-26","arxiv_id":"2311.15335","repositories_listed":1,"syntology":null},{"url":"/paper/l-m-v-iql-multiple-intention-inverse","slug":"l-m-v-iql-multiple-intention-inverse","title":"Multi-intention Inverse Q-learning for Interpretable Behavior Representation","date":"2023-11-23","arxiv_id":"2311.13870","repositories_listed":1,"syntology":null},{"url":"/paper/learning-dynamic-selection-and-pricing-of-out","slug":"learning-dynamic-selection-and-pricing-of-out","title":"Learning Dynamic Selection and Pricing of Out-of-Home Deliveries","date":"2023-11-23","arxiv_id":"2311.13983","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-model-is-a-good-policy-teacher","slug":"large-language-model-is-a-good-policy-teacher","title":"Large Language Model as a Policy Teacher for Training Reinforcement Learning Agents","date":"2023-11-22","arxiv_id":"2311.13373","repositories_listed":1,"syntology":null},{"url":"/paper/physical-reasoning-and-object-planning-for","slug":"physical-reasoning-and-object-planning-for","title":"Physical Reasoning and Object Planning for Household Embodied Agents","date":"2023-11-22","arxiv_id":"2311.13577","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-text-unveiling-multimodal-proficiency","slug":"beyond-text-unveiling-multimodal-proficiency","title":"Beyond Text: Unveiling Multimodal Proficiency of Large Language Models with MultiAPI Benchmark","date":"2023-11-21","arxiv_id":"2311.13053","repositories_listed":1,"syntology":null},{"url":"/paper/from-classification-to-clinical-insights","slug":"from-classification-to-clinical-insights","title":"From Classification to Clinical Insights: Towards Analyzing and Reasoning About Mobile and Behavioral Health Data With Large Language Models","date":"2023-11-21","arxiv_id":"2311.13063","repositories_listed":1,"syntology":null},{"url":"/paper/minimax-efficient-baselines-for-autocurricula","slug":"minimax-efficient-baselines-for-autocurricula","title":"minimax: Efficient Baselines for Autocurricula in JAX","date":"2023-11-21","arxiv_id":"2311.12716","repositories_listed":1,"syntology":null},{"url":"/paper/smart-energy-network-digital-twins-findings","slug":"smart-energy-network-digital-twins-findings","title":"Smart Energy Network Digital Twins: Findings from a UK-Based Demonstrator Project","date":"2023-11-20","arxiv_id":"2311.11997","repositories_listed":1,"syntology":null},{"url":"/paper/cautious-decision-making-for-tree-ensembles","slug":"cautious-decision-making-for-tree-ensembles","title":"Cautious Decision-Making for Tree Ensembles","date":"2023-11-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/a-language-agent-for-autonomous-driving","slug":"a-language-agent-for-autonomous-driving","title":"A Language Agent for Autonomous Driving","date":"2023-11-17","arxiv_id":"2311.10813","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-pruning-of-deep-ensembles-with","slug":"hierarchical-pruning-of-deep-ensembles-with","title":"Hierarchical Pruning of Deep Ensembles with Focal Diversity","date":"2023-11-17","arxiv_id":"2311.10293","repositories_listed":1,"syntology":null},{"url":"/paper/inherently-interpretable-time-series","slug":"inherently-interpretable-time-series","title":"Inherently Interpretable Time Series Classification via Multiple Instance Learning","date":"2023-11-16","arxiv_id":"2311.10049","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":0,"n_instrument":6,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/inherently-interpretable-time-series#ran","syntology_url":"https://syntology.ai/paper/2311.10049","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10049"}},"official":{"repos":["jaearly/miltimeseriesclassification"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/ibgr-influence-based-group-recommendation","slug":"ibgr-influence-based-group-recommendation","title":"IBGR: Influence-Based Group Recommendation system","date":"2023-11-15","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/navigating-the-ocean-of-biases-political-bias","slug":"navigating-the-ocean-of-biases-political-bias","title":"Exploring the Jungle of Bias: Political Bias Attribution in Language Models via Dependency Analysis","date":"2023-11-15","arxiv_id":"2311.08605","repositories_listed":1,"syntology":{"n":15,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/navigating-the-ocean-of-biases-political-bias#ran","syntology_url":"https://syntology.ai/paper/2311.08605","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.08605"}},"official":{"repos":["david-jenny/llm-political-study"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/tooltalk-evaluating-tool-usage-in-a","slug":"tooltalk-evaluating-tool-usage-in-a","title":"ToolTalk: Evaluating Tool-Usage in a Conversational Setting","date":"2023-11-15","arxiv_id":"2311.10775","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tooltalk-evaluating-tool-usage-in-a#ran","syntology_url":"https://syntology.ai/paper/2311.10775","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10775"}},"official":null}},{"url":"/paper/xplainllm-a-qa-explanation-dataset-for","slug":"xplainllm-a-qa-explanation-dataset-for","title":"XplainLLM: A Knowledge-Augmented Dataset for Reliable Grounded Explanations in LLMs","date":"2023-11-15","arxiv_id":"2311.08614","repositories_listed":1,"syntology":null},{"url":"/paper/extrinsically-focused-evaluation-of-omissions","slug":"extrinsically-focused-evaluation-of-omissions","title":"Extrinsically-Focused Evaluation of Omissions in Medical Summarization","date":"2023-11-14","arxiv_id":"2311.08303","repositories_listed":1,"syntology":null},{"url":"/paper/a-comprehensive-evaluation-of-gpt-4v-on","slug":"a-comprehensive-evaluation-of-gpt-4v-on","title":"A Comprehensive Evaluation of GPT-4V on Knowledge-Intensive Visual Question Answering","date":"2023-11-13","arxiv_id":"2311.07536","repositories_listed":1,"syntology":null},{"url":"/paper/input-convex-lstm-a-convex-approach-for-fast","slug":"input-convex-lstm-a-convex-approach-for-fast","title":"Real-Time Machine-Learning-Based Optimization Using Input Convex Long Short-Term Memory Network","date":"2023-11-13","arxiv_id":"2311.07202","repositories_listed":1,"syntology":null},{"url":"/paper/rethinking-and-benchmarking-predict-then","slug":"rethinking-and-benchmarking-predict-then","title":"Benchmarking PtO and PnO Methods in the Predictive Combinatorial Optimization Regime","date":"2023-11-13","arxiv_id":"2311.07633","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rethinking-and-benchmarking-predict-then#ran","syntology_url":"https://syntology.ai/paper/2311.07633","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.07633"}},"official":{"repos":["thinklab-sjtu/predictiveco-benchmark"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/monoprob-self-supervised-monocular-depth","slug":"monoprob-self-supervised-monocular-depth","title":"MonoProb: Self-Supervised Monocular Depth Estimation with Interpretable Uncertainty","date":"2023-11-10","arxiv_id":"2311.06137","repositories_listed":1,"syntology":null},{"url":"/paper/green-resilience-of-cyber-physical-systems","slug":"green-resilience-of-cyber-physical-systems","title":"Green Resilience of Cyber-Physical Systems","date":"2023-11-09","arxiv_id":"2311.05201","repositories_listed":1,"syntology":null},{"url":"/paper/on-the-road-with-gpt-4v-ision-early","slug":"on-the-road-with-gpt-4v-ision-early","title":"On the Road with GPT-4V(ision): Early Explorations of Visual-Language Model on Autonomous Driving","date":"2023-11-09","arxiv_id":"2311.05332","repositories_listed":1,"syntology":null},{"url":"/paper/adapt-as-needed-decomposition-and-planning","slug":"adapt-as-needed-decomposition-and-planning","title":"ADaPT: As-Needed Decomposition and Planning with Language Models","date":"2023-11-08","arxiv_id":"2311.05772","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adapt-as-needed-decomposition-and-planning#ran","syntology_url":"https://syntology.ai/paper/2311.05772","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.05772"}},"official":null}},{"url":"/paper/cais-dma-a-decision-making-assistant-for","slug":"cais-dma-a-decision-making-assistant-for","title":"CAIS-DMA: A Decision-Making Assistant for Collaborative AI Systems","date":"2023-11-08","arxiv_id":"2311.04562","repositories_listed":1,"syntology":null},{"url":"/paper/everything-of-thoughts-defying-the-law-of","slug":"everything-of-thoughts-defying-the-law-of","title":"Everything of Thoughts: Defying the Law of Penrose Triangle for Thought Generation","date":"2023-11-07","arxiv_id":"2311.04254","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/everything-of-thoughts-defying-the-law-of#ran","syntology_url":"https://syntology.ai/paper/2311.04254","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.04254"}},"official":{"repos":["microsoft/everything-of-thoughts-xot-"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/alympics-language-agents-meet-game-theory","slug":"alympics-language-agents-meet-game-theory","title":"ALYMPICS: LLM Agents Meet Game Theory -- Exploring Strategic Decision-Making with AI Agents","date":"2023-11-06","arxiv_id":"2311.03220","repositories_listed":1,"syntology":null},{"url":"/paper/cal-detr-calibrated-detection-transformer-1","slug":"cal-detr-calibrated-detection-transformer-1","title":"Cal-DETR: Calibrated Detection Transformer","date":"2023-11-06","arxiv_id":"2311.03570","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/cal-detr-calibrated-detection-transformer-1#ran","syntology_url":"https://syntology.ai/paper/2311.03570","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.03570"}},"official":{"repos":["akhtarvision/cal-detr"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/procedural-fairness-through-decoupling","slug":"procedural-fairness-through-decoupling","title":"Procedural Fairness Through Decoupling Objectionable Data Generating Components","date":"2023-11-05","arxiv_id":"2311.14688","repositories_listed":1,"syntology":null},{"url":"/paper/uncertainty-quantification-in-multivariable","slug":"uncertainty-quantification-in-multivariable","title":"Uncertainty Quantification in Multivariable Regression for Material Property Prediction with Bayesian Neural Networks","date":"2023-11-04","arxiv_id":"2311.02495","repositories_listed":1,"syntology":null},{"url":"/paper/an-algorithmic-framework-for-synthetic-cost","slug":"an-algorithmic-framework-for-synthetic-cost","title":"An algorithmic framework for synthetic cost-aware decision making in molecular design","date":"2023-11-03","arxiv_id":"2311.02187","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-symbolic-policy-learning-with","slug":"efficient-symbolic-policy-learning-with","title":"Efficient Symbolic Policy Learning with Differentiable Symbolic Expression","date":"2023-11-02","arxiv_id":"2311.02104","repositories_listed":1,"syntology":null},{"url":"/paper/proagent-from-robotic-process-automation-to","slug":"proagent-from-robotic-process-automation-to","title":"ProAgent: From Robotic Process Automation to Agentic Process Automation","date":"2023-11-02","arxiv_id":"2311.10751","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/proagent-from-robotic-process-automation-to#ran","syntology_url":"https://syntology.ai/paper/2311.10751","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.10751"}},"official":{"repos":["openbmb/proagent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-foundation-models-watch-talk-and-guide","slug":"can-foundation-models-watch-talk-and-guide","title":"Can Foundation Models Watch, Talk and Guide You Step by Step to Make a Cake?","date":"2023-11-01","arxiv_id":"2311.00738","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/can-foundation-models-watch-talk-and-guide#ran","syntology_url":"https://syntology.ai/paper/2311.00738","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2311.00738"}},"official":{"repos":["sled-group/watch-talk-and-guide"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/costar-improved-temporal-counterfactual","slug":"costar-improved-temporal-counterfactual","title":"COSTAR: Improved Temporal Counterfactual Estimation with Self-Supervised Learning","date":"2023-11-01","arxiv_id":"2311.00886","repositories_listed":1,"syntology":null},{"url":"/paper/the-development-of-llms-for-embodied","slug":"the-development-of-llms-for-embodied","title":"Advances in Embodied Navigation Using Large Language Models: A Survey","date":"2023-11-01","arxiv_id":"2311.00530","repositories_listed":1,"syntology":null},{"url":"/paper/calibration-by-distribution-matching-1","slug":"calibration-by-distribution-matching-1","title":"Calibration by Distribution Matching: Trainable Kernel Calibration Metrics","date":"2023-10-31","arxiv_id":"2310.20211","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/calibration-by-distribution-matching-1#ran","syntology_url":"https://syntology.ai/paper/2310.20211","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.20211"}},"official":{"repos":["kernel-calibration/kernel-calibration"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/multi-user-multiwoz-task-oriented-dialogues","slug":"multi-user-multiwoz-task-oriented-dialogues","title":"Multi-User MultiWOZ: Task-Oriented Dialogues among Multiple Users","date":"2023-10-31","arxiv_id":"2310.20479","repositories_listed":1,"syntology":null},{"url":"/paper/free-from-bellman-completeness-trajectory","slug":"free-from-bellman-completeness-trajectory","title":"Free from Bellman Completeness: Trajectory Stitching via Model-based Return-conditioned Supervised Learning","date":"2023-10-30","arxiv_id":"2310.19308","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/free-from-bellman-completeness-trajectory#ran","syntology_url":"https://syntology.ai/paper/2310.19308","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.19308"}},"official":{"repos":["zhaoyizhou1123/mbrcsl"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/interpretable-prototype-based-graph-1","slug":"interpretable-prototype-based-graph-1","title":"Interpretable Prototype-based Graph Information Bottleneck","date":"2023-10-30","arxiv_id":"2310.19906","repositories_listed":1,"syntology":{"n":49,"n_ran":27,"n_constructed":8,"n_ran_checked":12,"n_instrument":15,"n_unverified":22,"n_honours":3,"n_violates":0,"n_no_contract":9,"n_pointer_only":48,"phrase":"27 ran (of which 8 constructed an object rather than computing a result; 12 with no instrument failure: 3 honoured, 0 violated, 9 with no contract checked; 15 where Syntology's instrument failed) · 22 unverified","sample_list":"/paper/interpretable-prototype-based-graph-1#ran","syntology_url":"https://syntology.ai/paper/2310.19906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.19906"}},"official":{"repos":["sang-woo-seo/pgib"],"state":"official (archive's flag): 25 ran","n_ran":25,"n_constructed":8,"n_ran_no_instrument_failure":11,"n_unverified":22,"ran_from_kinds":["community","official","unlocated"]}}},{"url":"/paper/modeling-the-telemarketing-process-using","slug":"modeling-the-telemarketing-process-using","title":"Modeling the Telemarketing Process using Genetic Algorithms and Extreme Boosting: Feature Selection and Cost-Sensitive Analytical Approach","date":"2023-10-30","arxiv_id":"2310.19843","repositories_listed":1,"syntology":null},{"url":"/paper/dense-retrieval-as-indirect-supervision-for","slug":"dense-retrieval-as-indirect-supervision-for","title":"Dense Retrieval as Indirect Supervision for Large-space Decision Making","date":"2023-10-28","arxiv_id":"2310.18619","repositories_listed":1,"syntology":null},{"url":"/paper/explaining-by-imitating-understanding-1","slug":"explaining-by-imitating-understanding-1","title":"Explaining by Imitating: Understanding Decisions by Interpretable Policy Learning","date":"2023-10-28","arxiv_id":"2310.19831","repositories_listed":1,"syntology":null},{"url":"/paper/hierarchical-framework-for-interpretable-and","slug":"hierarchical-framework-for-interpretable-and","title":"Hierarchical Framework for Interpretable and Probabilistic Model-Based Safe Reinforcement Learning","date":"2023-10-28","arxiv_id":"2310.18811","repositories_listed":1,"syntology":null},{"url":"/paper/a-comprehensive-and-reliable-feature","slug":"a-comprehensive-and-reliable-feature","title":"A Comprehensive and Reliable Feature Attribution Method: Double-sided Remove and Reconstruct (DoRaR)","date":"2023-10-27","arxiv_id":"2310.17945","repositories_listed":1,"syntology":null},{"url":"/paper/bayesian-optimization-with-hidden-constraints","slug":"bayesian-optimization-with-hidden-constraints","title":"Black-Box Optimization with Implicit Constraints for Public Policy","date":"2023-10-27","arxiv_id":"2310.18449","repositories_listed":1,"syntology":null},{"url":"/paper/how-well-do-feature-additive-explainers","slug":"how-well-do-feature-additive-explainers","title":"How Well Do Feature-Additive Explainers Explain Feature-Additive Predictors?","date":"2023-10-27","arxiv_id":"2310.18496","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/how-well-do-feature-additive-explainers#ran","syntology_url":"https://syntology.ai/paper/2310.18496","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.18496"}},"official":{"repos":["craymichael/PostHocExplainerEvaluation"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/bayesian-neural-controlled-differential","slug":"bayesian-neural-controlled-differential","title":"Bayesian Neural Controlled Differential Equations for Treatment Effect Estimation","date":"2023-10-26","arxiv_id":"2310.17463","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/bayesian-neural-controlled-differential#ran","syntology_url":"https://syntology.ai/paper/2310.17463","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.17463"}},"official":{"repos":["konstantinhess/bayesian-neural-cde"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/the-impact-of-using-an-ai-chatbot-to-respond","slug":"the-impact-of-using-an-ai-chatbot-to-respond","title":"The impact of responding to patient messages with large language model assistance","date":"2023-10-26","arxiv_id":"2310.17703","repositories_listed":1,"syntology":null},{"url":"/paper/utilizing-language-models-for-energy-load","slug":"utilizing-language-models-for-energy-load","title":"Utilizing Language Models for Energy Load Forecasting","date":"2023-10-26","arxiv_id":"2310.17788","repositories_listed":1,"syntology":null},{"url":"/paper/physician-detection-of-clinical-harm-in","slug":"physician-detection-of-clinical-harm-in","title":"Physician Detection of Clinical Harm in Machine Translation: Quality Estimation Aids in Reliance and Backtranslation Identifies Critical Errors","date":"2023-10-25","arxiv_id":"2310.16924","repositories_listed":1,"syntology":null},{"url":"/paper/cr-copec-causal-rationale-of-corporate","slug":"cr-copec-causal-rationale-of-corporate","title":"CR-COPEC: Causal Rationale of Corporate Performance Changes to Learn from Financial Reports","date":"2023-10-24","arxiv_id":"2310.16095","repositories_listed":1,"syntology":null},{"url":"/paper/elemantra-an-end-to-end-automated-framework","slug":"elemantra-an-end-to-end-automated-framework","title":"Elemantra: An End-to-End Automated Framework Empowered with AI and IoT for Tackling Human-Elephant Conflict in Elephant-Range Countries","date":"2023-10-23","arxiv_id":"2310.15012","repositories_listed":1,"syntology":null},{"url":"/paper/multimodal-graph-learning-for-modeling","slug":"multimodal-graph-learning-for-modeling","title":"Multimodal Graph Learning for Modeling Emerging Pandemics with Big Data","date":"2023-10-23","arxiv_id":"2310.14549","repositories_listed":1,"syntology":null},{"url":"/paper/text2topic-multi-label-text-classification","slug":"text2topic-multi-label-text-classification","title":"Text2Topic: Multi-Label Text Classification System for Efficient Topic Detection in User Generated Content with Zero-Shot Capabilities","date":"2023-10-23","arxiv_id":"2310.14817","repositories_listed":1,"syntology":null},{"url":"/paper/vision-language-models-in-autonomous-driving","slug":"vision-language-models-in-autonomous-driving","title":"Vision Language Models in Autonomous Driving: A Survey and Outlook","date":"2023-10-22","arxiv_id":"2310.14414","repositories_listed":1,"syntology":null},{"url":"/paper/handling-missing-values-in-local-post-hoc","slug":"handling-missing-values-in-local-post-hoc","title":"Handling Missing Values in Local Post-hoc Explainability","date":"2023-10-21","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/an-improved-artificial-fish-swarm-algorithm","slug":"an-improved-artificial-fish-swarm-algorithm","title":"An Improved Artificial Fish Swarm Algorithm for Solving the Problem of Investigation Path Planning","date":"2023-10-20","arxiv_id":"2310.13375","repositories_listed":1,"syntology":null},{"url":"/paper/explaining-interactions-between-text-spans","slug":"explaining-interactions-between-text-spans","title":"Explaining Interactions Between Text Spans","date":"2023-10-20","arxiv_id":"2310.13506","repositories_listed":1,"syntology":null},{"url":"/paper/eureka-human-level-reward-design-via-coding","slug":"eureka-human-level-reward-design-via-coding","title":"Eureka: Human-Level Reward Design via Coding Large Language Models","date":"2023-10-19","arxiv_id":"2310.12931","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":9,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":8,"n_pointer_only":1,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 1 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/eureka-human-level-reward-design-via-coding#ran","syntology_url":"https://syntology.ai/paper/2310.12931","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.12931"}},"official":{"repos":["eureka-research/Eureka"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["community","official"]}}},{"url":"/paper/generating-collective-counterfactual","slug":"generating-collective-counterfactual","title":"Generating collective counterfactual explanations in score-based classification via mathematical optimization","date":"2023-10-19","arxiv_id":"2310.12822","repositories_listed":1,"syntology":null},{"url":"/paper/agent-specific-effects","slug":"agent-specific-effects","title":"Agent-Specific Effects: A Causal Effect Propagation Analysis in Multi-Agent MDPs","date":"2023-10-17","arxiv_id":"2310.11334","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/agent-specific-effects#ran","syntology_url":"https://syntology.ai/paper/2310.11334","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.11334"}},"official":{"repos":["stelios30/agent-specific-effects"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":3,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sensitivity-aware-amortized-bayesian","slug":"sensitivity-aware-amortized-bayesian","title":"Sensitivity-Aware Amortized Bayesian Inference","date":"2023-10-17","arxiv_id":"2310.11122","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sensitivity-aware-amortized-bayesian#ran","syntology_url":"https://syntology.ai/paper/2310.11122","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.11122"}},"official":{"repos":["bayesflow-org/SA-ABI"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-model-empowered-agents-for","slug":"large-language-model-empowered-agents-for","title":"EconAgent: Large Language Model-Empowered Agents for Simulating Macroeconomic Activities","date":"2023-10-16","arxiv_id":"2310.10436","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/large-language-model-empowered-agents-for#ran","syntology_url":"https://syntology.ai/paper/2310.10436","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.10436"}},"official":{"repos":["tsinghua-fib-lab/acl24-econagent"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-knowledge-distillation-for","slug":"leveraging-knowledge-distillation-for","title":"Leveraging Knowledge Distillation for Efficient Deep Reinforcement Learning in Resource-Constrained Environments","date":"2023-10-16","arxiv_id":"2310.10170","repositories_listed":1,"syntology":null},{"url":"/paper/step-by-step-remediation-of-students","slug":"step-by-step-remediation-of-students","title":"Bridging the Novice-Expert Gap via Models of Decision-Making: A Case Study on Remediating Math Mistakes","date":"2023-10-16","arxiv_id":"2310.10648","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/step-by-step-remediation-of-students#ran","syntology_url":"https://syntology.ai/paper/2310.10648","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.10648"}},"official":{"repos":["rosewang2008/bridge"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-gpt-4v-ision-serve-medical-applications","slug":"can-gpt-4v-ision-serve-medical-applications","title":"Can GPT-4V(ision) Serve Medical Applications? Case Studies on GPT-4V for Multimodal Medical Diagnosis","date":"2023-10-15","arxiv_id":"2310.09909","repositories_listed":1,"syntology":null},{"url":"/paper/on-statistical-learning-of-branch-and-bound","slug":"on-statistical-learning-of-branch-and-bound","title":"On Statistical Learning of Branch and Bound for Vehicle Routing Optimization","date":"2023-10-15","arxiv_id":"2310.09986","repositories_listed":1,"syntology":null},{"url":"/paper/measuring-the-stability-of-process-outcome","slug":"measuring-the-stability-of-process-outcome","title":"Measuring the Stability of Process Outcome Predictions in Online Settings","date":"2023-10-13","arxiv_id":"2310.09000","repositories_listed":1,"syntology":null},{"url":"/paper/lightzero-a-unified-benchmark-for-monte-carlo-1","slug":"lightzero-a-unified-benchmark-for-monte-carlo-1","title":"LightZero: A Unified Benchmark for Monte Carlo Tree Search in General Sequential Decision Scenarios","date":"2023-10-12","arxiv_id":"2310.08348","repositories_listed":1,"syntology":null},{"url":"/paper/octopus-embodied-vision-language-programmer","slug":"octopus-embodied-vision-language-programmer","title":"Octopus: Embodied Vision-Language Programmer from Environmental Feedback","date":"2023-10-12","arxiv_id":"2310.08588","repositories_listed":1,"syntology":{"n":12,"n_ran":12,"n_constructed":0,"n_ran_checked":6,"n_instrument":6,"n_unverified":0,"n_honours":1,"n_violates":1,"n_no_contract":4,"n_pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 6 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/octopus-embodied-vision-language-programmer#ran","syntology_url":"https://syntology.ai/paper/2310.08588","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.08588"}},"official":{"repos":["dongyh20/octopus"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visual-attention-prompted-prediction-and","slug":"visual-attention-prompted-prediction-and","title":"Visual Attention Prompted Prediction and Learning","date":"2023-10-12","arxiv_id":"2310.08420","repositories_listed":1,"syntology":null},{"url":"/paper/explainable-image-similarity-integrating","slug":"explainable-image-similarity-integrating","title":"Explainable Image Similarity: Integrating Siamese Networks and Grad-CAM","date":"2023-10-11","arxiv_id":"2310.07678","repositories_listed":1,"syntology":null},{"url":"/paper/imitation-learning-from-purified","slug":"imitation-learning-from-purified","title":"Imitation Learning from Purified Demonstrations","date":"2023-10-11","arxiv_id":"2310.07143","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":2,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":2,"n_violates":0,"n_no_contract":2,"n_pointer_only":5,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 2 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/imitation-learning-from-purified#ran","syntology_url":"https://syntology.ai/paper/2310.07143","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.07143"}},"official":{"repos":["yunke-wang/dp-il"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/learning-a-reward-function-for-user-preferred","slug":"learning-a-reward-function-for-user-preferred","title":"Learning a Reward Function for User-Preferred Appliance Scheduling","date":"2023-10-11","arxiv_id":"2310.07389","repositories_listed":1,"syntology":null},{"url":"/paper/neuroinspect-interpretable-neuron-based","slug":"neuroinspect-interpretable-neuron-based","title":"NeuroInspect: Interpretable Neuron-based Debugging Framework through Class-conditional Visualizations","date":"2023-10-11","arxiv_id":"2310.07184","repositories_listed":1,"syntology":null},{"url":"/paper/qacheck-a-demonstration-system-for-question","slug":"qacheck-a-demonstration-system-for-question","title":"QACHECK: A Demonstration System for Question-Guided Multi-Hop Fact-Checking","date":"2023-10-11","arxiv_id":"2310.07609","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/qacheck-a-demonstration-system-for-question#ran","syntology_url":"https://syntology.ai/paper/2310.07609","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.07609"}},"official":{"repos":["xinyuanlu00/qacheck"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/what-if-the-tv-was-off-examining","slug":"what-if-the-tv-was-off-examining","title":"What If the TV Was Off? Examining Counterfactual Reasoning Abilities of Multi-modal Language Models","date":"2023-10-10","arxiv_id":"2310.06627","repositories_listed":1,"syntology":null},{"url":"/paper/are-large-language-models-geospatially","slug":"are-large-language-models-geospatially","title":"Are Large Language Models Geospatially Knowledgeable?","date":"2023-10-09","arxiv_id":"2310.13002","repositories_listed":1,"syntology":{"n":7,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/are-large-language-models-geospatially#ran","syntology_url":"https://syntology.ai/paper/2310.13002","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.13002"}},"official":{"repos":["prabin525/spatial-llm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/why-should-this-article-be-deleted","slug":"why-should-this-article-be-deleted","title":"Why Should This Article Be Deleted? Transparent Stance Detection in Multilingual Wikipedia Editor Discussions","date":"2023-10-09","arxiv_id":"2310.05779","repositories_listed":1,"syntology":null},{"url":"/paper/explainable-claim-verification-via-knowledge","slug":"explainable-claim-verification-via-knowledge","title":"Explainable Claim Verification via Knowledge-Grounded Reasoning with Large Language Models","date":"2023-10-08","arxiv_id":"2310.05253","repositories_listed":1,"syntology":null},{"url":"/paper/from-text-to-tactic-evaluating-llms-playing","slug":"from-text-to-tactic-evaluating-llms-playing","title":"AvalonBench: Evaluating LLMs Playing the Game of Avalon","date":"2023-10-08","arxiv_id":"2310.05036","repositories_listed":1,"syntology":null},{"url":"/paper/beyond-uniform-sampling-offline-reinforcement-1","slug":"beyond-uniform-sampling-offline-reinforcement-1","title":"Beyond Uniform Sampling: Offline Reinforcement Learning with Imbalanced Datasets","date":"2023-10-06","arxiv_id":"2310.04413","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/beyond-uniform-sampling-offline-reinforcement-1#ran","syntology_url":"https://syntology.ai/paper/2310.04413","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.04413"}},"official":{"repos":["Improbable-AI/dw-offline-rl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/neur2ro-neural-two-stage-robust-optimization","slug":"neur2ro-neural-two-stage-robust-optimization","title":"Deep Learning for Two-Stage Robust Integer Optimization","date":"2023-10-06","arxiv_id":"2310.04345","repositories_listed":1,"syntology":null},{"url":"/paper/consistency-regularization-improves-placenta","slug":"consistency-regularization-improves-placenta","title":"Consistency Regularization Improves Placenta Segmentation in Fetal EPI MRI Time Series","date":"2023-10-05","arxiv_id":"2310.03870","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-model-cascades-with-mixture-of","slug":"large-language-model-cascades-with-mixture-of","title":"Large Language Model Cascades with Mixture of Thoughts Representations for Cost-efficient Reasoning","date":"2023-10-04","arxiv_id":"2310.03094","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-reach-goals-via-diffusion","slug":"learning-to-reach-goals-via-diffusion","title":"Learning to Reach Goals via Diffusion","date":"2023-10-04","arxiv_id":"2310.02505","repositories_listed":1,"syntology":null},{"url":"/paper/metatool-benchmark-deciding-whether-to-use","slug":"metatool-benchmark-deciding-whether-to-use","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","date":"2023-10-04","arxiv_id":"2310.03128","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/metatool-benchmark-deciding-whether-to-use#ran","syntology_url":"https://syntology.ai/paper/2310.03128","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.03128"}},"official":{"repos":["howiehwong/metatool"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/autocast-enhancing-world-event-prediction","slug":"autocast-enhancing-world-event-prediction","title":"AutoCast++: Enhancing World Event Prediction with Zero-shot Ranking-based Context Retrieval","date":"2023-10-03","arxiv_id":"2310.01880","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/autocast-enhancing-world-event-prediction#ran","syntology_url":"https://syntology.ai/paper/2310.01880","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2310.01880"}},"official":{"repos":["BorealisAI/Autocast-plus-plus"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/driving-with-llms-fusing-object-level-vector","slug":"driving-with-llms-fusing-object-level-vector","title":"Driving with LLMs: Fusing Object-Level Vector Modality for Explainable Autonomous Driving","date":"2023-10-03","arxiv_id":"2310.01957","repositories_listed":1,"syntology":null}],"record_sha256":"4e221948cc5aa283971fa2988745b4e4ef543568e5ef7203f4d3f292451c2877","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}