{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/decision-making/papers/9","list_of":"/task/decision-making","task":"Decision Making","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":9,"pages_in_order":124,"rows_per_page":100,"rows":[801,900],"of":12311,"counts":{"archive_papers_tagged":12311,"with_a_code_link":2946,"where_syntology_ran_a_sample":678,"not_listed_spam_title":0,"listed":12311,"listed_where_code_ran":678,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":560,"every_run_a_failure_of_syntologys_instrument":118,"listed_with_a_run_with_no_instrument_failure":560,"listed_every_run_a_failure_of_syntologys_instrument":118,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/decision-making","prev":"/task/decision-making/papers/8","next":"/task/decision-making/papers/10","papers":[{"url":"/paper/enhancing-agent-learning-through-world","slug":"enhancing-agent-learning-through-world","title":"Enhancing Agent Learning through World Dynamics Modeling","date":"2024-07-25","arxiv_id":"2407.17695","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"0 ran · 1 unverified","sample_list":"/paper/enhancing-agent-learning-through-world#ran","syntology_url":"https://syntology.ai/paper/2407.17695","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.17695"}},"official":{"repos":["ZhiyuuanS/DiVE"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/toward-an-integrated-decision-making","slug":"toward-an-integrated-decision-making","title":"Toward an Integrated Decision Making Framework for Optimized Stroke Diagnosis with DSA and Treatment under Uncertainty","date":"2024-07-24","arxiv_id":"2407.16962","repositories_listed":1,"syntology":null},{"url":"/paper/towards-neural-network-based-cognitive-models","slug":"towards-neural-network-based-cognitive-models","title":"Towards Neural Network based Cognitive Models of Dynamic Decision-Making by Humans","date":"2024-07-24","arxiv_id":"2407.17622","repositories_listed":1,"syntology":null},{"url":"/paper/diffusion-models-as-optimizers-for-efficient","slug":"diffusion-models-as-optimizers-for-efficient","title":"Diffusion Models as Optimizers for Efficient Planning in Offline RL","date":"2024-07-23","arxiv_id":"2407.16142","repositories_listed":1,"syntology":null},{"url":"/paper/momaland-a-set-of-benchmarks-for-multi","slug":"momaland-a-set-of-benchmarks-for-multi","title":"MOMAland: A Set of Benchmarks for Multi-Objective Multi-Agent Reinforcement Learning","date":"2024-07-23","arxiv_id":"2407.16312","repositories_listed":1,"syntology":null},{"url":"/paper/pategail-a-privacy-preserving-mobility","slug":"pategail-a-privacy-preserving-mobility","title":"PateGail: A Privacy-Preserving Mobility Trajectory Generator with Imitation Learning","date":"2024-07-23","arxiv_id":"2407.16729","repositories_listed":1,"syntology":null},{"url":"/paper/reinforcement-learning-pair-trading-a-dynamic","slug":"reinforcement-learning-pair-trading-a-dynamic","title":"Reinforcement Learning Pair Trading: A Dynamic Scaling approach","date":"2024-07-23","arxiv_id":"2407.16103","repositories_listed":1,"syntology":null},{"url":"/paper/interpretable-concept-based-memory-reasoning","slug":"interpretable-concept-based-memory-reasoning","title":"Interpretable Concept-Based Memory Reasoning","date":"2024-07-22","arxiv_id":"2407.15527","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/interpretable-concept-based-memory-reasoning#ran","syntology_url":"https://syntology.ai/paper/2407.15527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.15527"}},"official":{"repos":["daviddebot/CMR"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforcement-learning-meets-visual-odometry","slug":"reinforcement-learning-meets-visual-odometry","title":"Reinforcement Learning Meets Visual Odometry","date":"2024-07-22","arxiv_id":"2407.15626","repositories_listed":1,"syntology":null},{"url":"/paper/cocog-2-controllable-generation-of-visual","slug":"cocog-2-controllable-generation-of-visual","title":"CoCoG-2: Controllable generation of visual stimuli for understanding human concept representation","date":"2024-07-20","arxiv_id":"2407.14949","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cocog-2-controllable-generation-of-visual#ran","syntology_url":"https://syntology.ai/paper/2407.14949","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.14949"}},"official":{"repos":["ncclab-sustech/cocog-2"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/achieving-well-informed-decision-making-in","slug":"achieving-well-informed-decision-making-in","title":"Achieving Well-Informed Decision-Making in Drug Discovery: A Comprehensive Calibration Study using Neural Network-Based Structure-Activity Models","date":"2024-07-19","arxiv_id":"2407.14185","repositories_listed":1,"syntology":null},{"url":"/paper/cod-towards-an-interpretable-medical-agent","slug":"cod-towards-an-interpretable-medical-agent","title":"CoD, Towards an Interpretable Medical Agent using Chain of Diagnosis","date":"2024-07-18","arxiv_id":"2407.13301","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cod-towards-an-interpretable-medical-agent#ran","syntology_url":"https://syntology.ai/paper/2407.13301","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.13301"}},"official":{"repos":["freedomintelligence/chain-of-diagnosis"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hyp2nav-hyperbolic-planning-and-curiosity-for","slug":"hyp2nav-hyperbolic-planning-and-curiosity-for","title":"Hyp2Nav: Hyperbolic Planning and Curiosity for Crowd Navigation","date":"2024-07-18","arxiv_id":"2407.13567","repositories_listed":1,"syntology":null},{"url":"/paper/semantic-prototypes-enhancing-transparency","slug":"semantic-prototypes-enhancing-transparency","title":"Semantic Prototypes: Enhancing Transparency Without Black Boxes","date":"2024-07-18","arxiv_id":"2407.15871","repositories_listed":1,"syntology":null},{"url":"/paper/unmasking-social-bots-how-confident-are-we","slug":"unmasking-social-bots-how-confident-are-we","title":"Unmasking Social Bots: How Confident Are We?","date":"2024-07-18","arxiv_id":"2407.13929","repositories_listed":1,"syntology":null},{"url":"/paper/actionswitch-class-agnostic-detection-of","slug":"actionswitch-class-agnostic-detection-of","title":"ActionSwitch: Class-agnostic Detection of Simultaneous Actions in Streaming Videos","date":"2024-07-17","arxiv_id":"2407.12987","repositories_listed":1,"syntology":null},{"url":"/paper/continuous-reasoning-for-adaptive-container","slug":"continuous-reasoning-for-adaptive-container","title":"Continuous reasoning for adaptive container image distribution in the cloud-edge continuum","date":"2024-07-17","arxiv_id":"2407.12605","repositories_listed":1,"syntology":null},{"url":"/paper/how-personality-traits-influence-negotiation","slug":"how-personality-traits-influence-negotiation","title":"How Personality Traits Influence Negotiation Outcomes? A Simulation based on Large Language Models","date":"2024-07-16","arxiv_id":"2407.11549","repositories_listed":1,"syntology":null},{"url":"/paper/invagent-a-large-language-model-based-multi","slug":"invagent-a-large-language-model-based-multi","title":"InvAgent: A Large Language Model based Multi-Agent System for Inventory Management in Supply Chains","date":"2024-07-16","arxiv_id":"2407.11384","repositories_listed":1,"syntology":null},{"url":"/paper/mask-free-neuron-concept-annotation-for","slug":"mask-free-neuron-concept-annotation-for","title":"Mask-Free Neuron Concept Annotation for Interpreting Neural Networks in Medical Domain","date":"2024-07-16","arxiv_id":"2407.11375","repositories_listed":1,"syntology":null},{"url":"/paper/predicting-emotion-intensity-in-polish","slug":"predicting-emotion-intensity-in-polish","title":"Predicting Emotion Intensity in Polish Political Texts: Comparing Supervised Models and Large Language Models in a Resource-Poor Language","date":"2024-07-16","arxiv_id":"2407.12141","repositories_listed":1,"syntology":null},{"url":"/paper/towards-consistency-of-rule-based-explainer","slug":"towards-consistency-of-rule-based-explainer","title":"Towards consistency of rule-based explainer and black box model -- fusion of rule induction and XAI-based feature importance","date":"2024-07-16","arxiv_id":"2407.14543","repositories_listed":1,"syntology":null},{"url":"/paper/urbanworld-an-urban-world-model-for-3d-city","slug":"urbanworld-an-urban-world-model-for-3d-city","title":"UrbanWorld: An Urban World Model for 3D City Generation","date":"2024-07-16","arxiv_id":"2407.11965","repositories_listed":1,"syntology":null},{"url":"/paper/communication-and-computation-efficient","slug":"communication-and-computation-efficient","title":"Communication- and Computation-Efficient Distributed Submodular Optimization in Robot Mesh Networks","date":"2024-07-15","arxiv_id":"2407.10382","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-the-dependence-of-perception","slug":"understanding-the-dependence-of-perception","title":"Understanding the Dependence of Perception Model Competency on Regions in an Image","date":"2024-07-15","arxiv_id":"2407.10543","repositories_listed":1,"syntology":null},{"url":"/paper/preserving-the-privacy-of-reward-functions-in","slug":"preserving-the-privacy-of-reward-functions-in","title":"Preserving the Privacy of Reward Functions in MDPs through Deception","date":"2024-07-13","arxiv_id":"2407.09809","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/preserving-the-privacy-of-reward-functions-in#ran","syntology_url":"https://syntology.ai/paper/2407.09809","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09809"}},"official":{"repos":["shshnkreddy/deceptiverl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/deep-attention-driven-reinforcement-learning","slug":"deep-attention-driven-reinforcement-learning","title":"Deep Attention Driven Reinforcement Learning (DAD-RL) for Autonomous Decision-Making in Dynamic Environment","date":"2024-07-12","arxiv_id":"2407.08932","repositories_listed":1,"syntology":null},{"url":"/paper/kunpeng-an-embodied-large-model-for","slug":"kunpeng-an-embodied-large-model-for","title":"KUNPENG: An Embodied Large Model for Intelligent Maritime","date":"2024-07-12","arxiv_id":"2407.09048","repositories_listed":1,"syntology":null},{"url":"/paper/latent-spaces-enable-transformer-based-dose","slug":"latent-spaces-enable-transformer-based-dose","title":"Latent Spaces Enable Transformer-Based Dose Prediction in Complex Radiotherapy Plans","date":"2024-07-11","arxiv_id":"2407.08650","repositories_listed":1,"syntology":null},{"url":"/paper/cm-dqn-a-value-based-deep-reinforcement","slug":"cm-dqn-a-value-based-deep-reinforcement","title":"CM-DQN: A Value-Based Deep Reinforcement Learning Model to Simulate Confirmation Bias","date":"2024-07-10","arxiv_id":"2407.07454","repositories_listed":1,"syntology":null},{"url":"/paper/long-term-fairness-in-sequential-multi-agent","slug":"long-term-fairness-in-sequential-multi-agent","title":"Long-Term Fairness in Sequential Multi-Agent Selection with Positive Reinforcement","date":"2024-07-10","arxiv_id":"2407.07350","repositories_listed":1,"syntology":null},{"url":"/paper/can-learned-optimization-make-reinforcement","slug":"can-learned-optimization-make-reinforcement","title":"Can Learned Optimization Make Reinforcement Learning Less Difficult?","date":"2024-07-09","arxiv_id":"2407.07082","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-learned-optimization-make-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2407.07082","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.07082"}},"official":{"repos":["alexgoldie/rl-learned-optimization"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/economic-span-selection-of-bridge-based-on","slug":"economic-span-selection-of-bridge-based-on","title":"Economic span selection of bridge based on deep reinforcement learning","date":"2024-07-09","arxiv_id":"2407.06507","repositories_listed":1,"syntology":null},{"url":"/paper/integrating-clinical-knowledge-into-concept","slug":"integrating-clinical-knowledge-into-concept","title":"Integrating Clinical Knowledge into Concept Bottleneck Models","date":"2024-07-09","arxiv_id":"2407.06600","repositories_listed":1,"syntology":null},{"url":"/paper/learning-to-complement-and-to-defer-to","slug":"learning-to-complement-and-to-defer-to","title":"Learning to Complement and to Defer to Multiple Users","date":"2024-07-09","arxiv_id":"2407.07003","repositories_listed":1,"syntology":{"n":21,"n_ran":19,"n_constructed":0,"n_ran_checked":14,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":21,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-to-complement-and-to-defer-to#ran","syntology_url":"https://syntology.ai/paper/2407.07003","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.07003"}},"official":{"repos":["zhengzhang37/lecodu"],"state":"official (archive's flag): 19 ran","n_ran":19,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/a-mamba-based-siamese-network-for-remote","slug":"a-mamba-based-siamese-network-for-remote","title":"A Mamba-based Siamese Network for Remote Sensing Change Detection","date":"2024-07-08","arxiv_id":"2407.06839","repositories_listed":1,"syntology":null},{"url":"/paper/object-oriented-material-classification-and","slug":"object-oriented-material-classification-and","title":"Object-Oriented Material Classification and 3D Clustering for Improved Semantic Perception and Mapping in Mobile Robots","date":"2024-07-08","arxiv_id":"2407.06077","repositories_listed":1,"syntology":null},{"url":"/paper/simulation-based-benchmarking-for-causal","slug":"simulation-based-benchmarking-for-causal","title":"Simulation-based Benchmarking for Causal Structure Learning in Gene Perturbation Experiments","date":"2024-07-08","arxiv_id":"2407.06015","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/simulation-based-benchmarking-for-causal#ran","syntology_url":"https://syntology.ai/paper/2407.06015","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.06015"}},"official":{"repos":["luka-kovacevic/causalregnet"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/climb-a-benchmark-of-clinical-bias-in-large","slug":"climb-a-benchmark-of-clinical-bias-in-large","title":"CLIMB: A Benchmark of Clinical Bias in Large Language Models","date":"2024-07-07","arxiv_id":"2407.05250","repositories_listed":1,"syntology":null},{"url":"/paper/prance-joint-token-optimization-and","slug":"prance-joint-token-optimization-and","title":"PRANCE: Joint Token-Optimization and Structural Channel-Pruning for Adaptive ViT Inference","date":"2024-07-06","arxiv_id":"2407.05010","repositories_listed":1,"syntology":null},{"url":"/paper/arigraph-learning-knowledge-graph-world","slug":"arigraph-learning-knowledge-graph-world","title":"AriGraph: Learning Knowledge Graph World Models with Episodic Memory for LLM Agents","date":"2024-07-05","arxiv_id":"2407.04363","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/arigraph-learning-knowledge-graph-world#ran","syntology_url":"https://syntology.ai/paper/2407.04363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04363"}},"official":{"repos":["airi-institute/arigraph"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/leveraging-graph-structures-to-detect","slug":"leveraging-graph-structures-to-detect","title":"Leveraging Graph Structures to Detect Hallucinations in Large Language Models","date":"2024-07-05","arxiv_id":"2407.04485","repositories_listed":1,"syntology":null},{"url":"/paper/chartgemma-visual-instruction-tuning-for","slug":"chartgemma-visual-instruction-tuning-for","title":"ChartGemma: Visual Instruction-tuning for Chart Reasoning in the Wild","date":"2024-07-04","arxiv_id":"2407.04172","repositories_listed":1,"syntology":null},{"url":"/paper/on-evaluating-explanation-utility-for-human","slug":"on-evaluating-explanation-utility-for-human","title":"On Evaluating Explanation Utility for Human-AI Decision Making in NLP","date":"2024-07-03","arxiv_id":"2407.03545","repositories_listed":1,"syntology":null},{"url":"/paper/on-large-language-models-in-national-security","slug":"on-large-language-models-in-national-security","title":"On Large Language Models in National Security Applications","date":"2024-07-03","arxiv_id":"2407.03453","repositories_listed":1,"syntology":null},{"url":"/paper/distributional-regression-u-nets-for-the","slug":"distributional-regression-u-nets-for-the","title":"Distributional Regression U-Nets for the Postprocessing of Precipitation Ensemble Forecasts","date":"2024-07-02","arxiv_id":"2407.02125","repositories_listed":1,"syntology":null},{"url":"/paper/econnli-evaluating-large-language-models-on","slug":"econnli-evaluating-large-language-models-on","title":"EconNLI: Evaluating Large Language Models on Economics Reasoning","date":"2024-07-01","arxiv_id":"2407.01212","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/econnli-evaluating-large-language-models-on#ran","syntology_url":"https://syntology.ai/paper/2407.01212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01212"}},"official":{"repos":["irenehere/econnli"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/the-house-always-wins-a-framework-for","slug":"the-house-always-wins-a-framework-for","title":"View From Above: A Framework for Evaluating Distribution Shifts in Model Behavior","date":"2024-07-01","arxiv_id":"2407.00948","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-house-always-wins-a-framework-for#ran","syntology_url":"https://syntology.ai/paper/2407.00948","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00948"}},"official":{"repos":["Bluefin-Tuna/ApartResearch"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/predict-optimize-revise-on-forecast-and","slug":"predict-optimize-revise-on-forecast-and","title":"Balancing Forecast Accuracy and Switching Costs in Online Optimization of Energy Management Systems","date":"2024-06-29","arxiv_id":"2407.03368","repositories_listed":1,"syntology":null},{"url":"/paper/puzzles-a-benchmark-for-neural-algorithmic","slug":"puzzles-a-benchmark-for-neural-algorithmic","title":"PUZZLES: A Benchmark for Neural Algorithmic Reasoning","date":"2024-06-29","arxiv_id":"2407.00401","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/puzzles-a-benchmark-for-neural-algorithmic#ran","syntology_url":"https://syntology.ai/paper/2407.00401","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00401"}},"official":{"repos":["eth-disco/rlp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/operator-world-models-for-reinforcement","slug":"operator-world-models-for-reinforcement","title":"Operator World Models for Reinforcement Learning","date":"2024-06-28","arxiv_id":"2406.19861","repositories_listed":1,"syntology":null},{"url":"/paper/boosting-cattle-face-recognition-under","slug":"boosting-cattle-face-recognition-under","title":"Boosting cattle face recognition under uncontrolled scenes by embedding enhancement and optimization","date":"2024-06-27","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/cello-causal-evaluation-of-large-vision","slug":"cello-causal-evaluation-of-large-vision","title":"CELLO: Causal Evaluation of Large Vision-Language Models","date":"2024-06-27","arxiv_id":"2406.19131","repositories_listed":1,"syntology":null},{"url":"/paper/evidential-concept-embedding-models-towards","slug":"evidential-concept-embedding-models-towards","title":"Evidential Concept Embedding Models: Towards Reliable Concept Explanations for Skin Disease Diagnosis","date":"2024-06-27","arxiv_id":"2406.19130","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evidential-concept-embedding-models-towards#ran","syntology_url":"https://syntology.ai/paper/2406.19130","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19130"}},"official":{"repos":["obiyoag/evi-cem"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/from-biased-selective-labels-to-pseudo-labels","slug":"from-biased-selective-labels-to-pseudo-labels","title":"From Biased Selective Labels to Pseudo-Labels: An Expectation-Maximization Framework for Learning from Biased Decisions","date":"2024-06-27","arxiv_id":"2406.18865","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":4,"n_honours":4,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 4 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/from-biased-selective-labels-to-pseudo-labels#ran","syntology_url":"https://syntology.ai/paper/2406.18865","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18865"}},"official":{"repos":["mld3/dcem"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["community","official"]}}},{"url":"/paper/instance-temperature-knowledge-distillation","slug":"instance-temperature-knowledge-distillation","title":"Instance Temperature Knowledge Distillation","date":"2024-06-27","arxiv_id":"2407.00115","repositories_listed":1,"syntology":null},{"url":"/paper/can-we-trust-the-performance-evaluation-of","slug":"can-we-trust-the-performance-evaluation-of","title":"Can We Trust the Performance Evaluation of Uncertainty Estimation Methods in Text Summarization?","date":"2024-06-25","arxiv_id":"2406.17274","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-we-trust-the-performance-evaluation-of#ran","syntology_url":"https://syntology.ai/paper/2406.17274","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17274"}},"official":{"repos":["he159ok/benchmark-of-uncertainty-estimation-methods-in-text-summarization"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/conditional-bayesian-quadrature","slug":"conditional-bayesian-quadrature","title":"Conditional Bayesian Quadrature","date":"2024-06-24","arxiv_id":"2406.16530","repositories_listed":1,"syntology":null},{"url":"/paper/differentiable-distributionally-robust","slug":"differentiable-distributionally-robust","title":"Differentiable Distributionally Robust Optimization Layers","date":"2024-06-24","arxiv_id":"2406.16571","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-models-assume-people-are-more","slug":"large-language-models-assume-people-are-more","title":"Large Language Models Assume People are More Rational than We Really are","date":"2024-06-24","arxiv_id":"2406.17055","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/large-language-models-assume-people-are-more#ran","syntology_url":"https://syntology.ai/paper/2406.17055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17055"}},"official":{"repos":["theryanl/llm-rationality"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-temporal-distances-contrastive","slug":"learning-temporal-distances-contrastive","title":"Learning Temporal Distances: Contrastive Successor Features Can Provide a Metric Structure for Decision-Making","date":"2024-06-24","arxiv_id":"2406.17098","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-temporal-distances-contrastive#ran","syntology_url":"https://syntology.ai/paper/2406.17098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17098"}},"official":{"repos":["vivekmyers/contrastive_metrics"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/what-do-vlms-notice-a-mechanistic","slug":"what-do-vlms-notice-a-mechanistic","title":"What Do VLMs NOTICE? A Mechanistic Interpretability Pipeline for Gaussian-Noise-free Text-Image Corruption and Evaluation","date":"2024-06-24","arxiv_id":"2406.16320","repositories_listed":1,"syntology":null},{"url":"/paper/hardware-aware-neural-dropout-search-for","slug":"hardware-aware-neural-dropout-search-for","title":"Hardware-Aware Neural Dropout Search for Reliable Uncertainty Prediction on FPGA","date":"2024-06-23","arxiv_id":"2406.16198","repositories_listed":1,"syntology":null},{"url":"/paper/catastrophic-risk-aware-reinforcement","slug":"catastrophic-risk-aware-reinforcement","title":"Catastrophic-risk-aware reinforcement learning with extreme-value-theory-based policy gradients","date":"2024-06-21","arxiv_id":"2406.15612","repositories_listed":1,"syntology":null},{"url":"/paper/pathowave-a-deep-learning-based-weight","slug":"pathowave-a-deep-learning-based-weight","title":"PathoWAve: A Deep Learning-based Weight Averaging Method for Improving Domain Generalization in Histopathology Images","date":"2024-06-21","arxiv_id":"2406.15685","repositories_listed":1,"syntology":null},{"url":"/paper/iwisdm-assessing-instruction-following-in","slug":"iwisdm-assessing-instruction-following-in","title":"IWISDM: Assessing instruction following in multimodal models at scale","date":"2024-06-20","arxiv_id":"2406.14343","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/iwisdm-assessing-instruction-following-in#ran","syntology_url":"https://syntology.ai/paper/2406.14343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14343"}},"official":{"repos":["bashivanlab/iwisdm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/macrohft-memory-augmented-context-aware","slug":"macrohft-memory-augmented-context-aware","title":"MacroHFT: Memory Augmented Context-aware Reinforcement Learning On High Frequency Trading","date":"2024-06-20","arxiv_id":"2406.14537","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/macrohft-memory-augmented-context-aware#ran","syntology_url":"https://syntology.ai/paper/2406.14537","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14537"}},"official":{"repos":["ZONG0004/MacroHFT"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/modeling-of-spatially-embedded-networks-via","slug":"modeling-of-spatially-embedded-networks-via","title":"Modeling of spatially embedded networks via regional spatial graph convolutional networks","date":"2024-06-20","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/reveal-it-reinforcement-learning-with","slug":"reveal-it-reinforcement-learning-with","title":"REVEAL-IT: REinforcement learning with Visibility of Evolving Agent poLicy for InTerpretability","date":"2024-06-20","arxiv_id":"2406.14214","repositories_listed":1,"syntology":null},{"url":"/paper/explaining-time-series-models-using-frequency","slug":"explaining-time-series-models-using-frequency","title":"FreqRISE: Explaining time series using frequency masking","date":"2024-06-19","arxiv_id":"2406.13584","repositories_listed":1,"syntology":null},{"url":"/paper/reasoning-with-trees-interpreting-cnns-using","slug":"reasoning-with-trees-interpreting-cnns-using","title":"Reasoning with trees: interpreting CNNs using hierarchies","date":"2024-06-19","arxiv_id":"2406.13257","repositories_listed":1,"syntology":null},{"url":"/paper/situational-instructions-database-task","slug":"situational-instructions-database-task","title":"SituationalLLM: Proactive language models with scene awareness for dynamic, contextual task guidance","date":"2024-06-19","arxiv_id":"2406.13302","repositories_listed":1,"syntology":null},{"url":"/paper/ask-before-plan-proactive-language-agents-for","slug":"ask-before-plan-proactive-language-agents-for","title":"Ask-before-Plan: Proactive Language Agents for Real-World Planning","date":"2024-06-18","arxiv_id":"2406.12639","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/ask-before-plan-proactive-language-agents-for#ran","syntology_url":"https://syntology.ai/paper/2406.12639","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12639"}},"official":{"repos":["magicgh/ask-before-plan"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/planrag-a-plan-then-retrieval-augmented","slug":"planrag-a-plan-then-retrieval-augmented","title":"PlanRAG: A Plan-then-Retrieval Augmented Generation for Generative Large Language Models as Decision Makers","date":"2024-06-18","arxiv_id":"2406.12430","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/planrag-a-plan-then-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2406.12430","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12430"}},"official":{"repos":["myeon9h/planrag"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/statistical-uncertainty-in-word-embeddings","slug":"statistical-uncertainty-in-word-embeddings","title":"Statistical Uncertainty in Word Embeddings: GloVe-V","date":"2024-06-18","arxiv_id":"2406.12165","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/statistical-uncertainty-in-word-embeddings#ran","syntology_url":"https://syntology.ai/paper/2406.12165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12165"}},"official":{"repos":["reglab/glove-v"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/computing-in-the-life-sciences-from-early","slug":"computing-in-the-life-sciences-from-early","title":"Computing in the Life Sciences: From Early Algorithms to Modern AI","date":"2024-06-17","arxiv_id":"2406.12108","repositories_listed":1,"syntology":null},{"url":"/paper/grade-score-quantifying-llm-performance-in","slug":"grade-score-quantifying-llm-performance-in","title":"Grade Score: Quantifying LLM Performance in Option Selection","date":"2024-06-17","arxiv_id":"2406.12043","repositories_listed":1,"syntology":null},{"url":"/paper/not-all-bias-is-bad-balancing-rational","slug":"not-all-bias-is-bad-balancing-rational","title":"Balancing Rigor and Utility: Mitigating Cognitive Biases in Large Language Models for Multiple-Choice Questions","date":"2024-06-16","arxiv_id":"2406.10999","repositories_listed":1,"syntology":null},{"url":"/paper/luma-a-benchmark-dataset-for-learning-from","slug":"luma-a-benchmark-dataset-for-learning-from","title":"LUMA: A Benchmark Dataset for Learning from Uncertain and Multimodal Data","date":"2024-06-14","arxiv_id":"2406.09864","repositories_listed":1,"syntology":null},{"url":"/paper/a-tutorial-on-fairness-in-machine-learning-in","slug":"a-tutorial-on-fairness-in-machine-learning-in","title":"What is Fair? Defining Fairness in Machine Learning for Health","date":"2024-06-13","arxiv_id":"2406.09307","repositories_listed":1,"syntology":null},{"url":"/paper/active-inference-meeting-energy-efficient","slug":"active-inference-meeting-energy-efficient","title":"Active Inference Meeting Energy-Efficient Control of Parallel and Identical Machines","date":"2024-06-13","arxiv_id":"2406.09322","repositories_listed":1,"syntology":null},{"url":"/paper/conceptual-learning-via-embedding","slug":"conceptual-learning-via-embedding","title":"Conceptual Learning via Embedding Approximations for Reinforcing Interpretability and Transparency","date":"2024-06-13","arxiv_id":"2406.08840","repositories_listed":1,"syntology":null},{"url":"/paper/mathematical-models-for-off-ball-scoring","slug":"mathematical-models-for-off-ball-scoring","title":"Mathematical models for off-ball scoring prediction in basketball","date":"2024-06-13","arxiv_id":"2406.08749","repositories_listed":1,"syntology":null},{"url":"/paper/opening-the-black-box-predicting-the","slug":"opening-the-black-box-predicting-the","title":"Opening the Black Box: predicting the trainability of deep neural networks with reconstruction entropy","date":"2024-06-13","arxiv_id":"2406.12916","repositories_listed":1,"syntology":null},{"url":"/paper/coxql-a-dataset-for-parsing-explanation","slug":"coxql-a-dataset-for-parsing-explanation","title":"CoXQL: A Dataset for Parsing Explanation Requests in Conversational XAI Systems","date":"2024-06-12","arxiv_id":"2406.08101","repositories_listed":1,"syntology":null},{"url":"/paper/lvbench-an-extreme-long-video-understanding","slug":"lvbench-an-extreme-long-video-understanding","title":"LVBench: An Extreme Long Video Understanding Benchmark","date":"2024-06-12","arxiv_id":"2406.08035","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lvbench-an-extreme-long-video-understanding#ran","syntology_url":"https://syntology.ai/paper/2406.08035","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08035"}},"official":{"repos":["THUDM/LVBench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/accessing-gpt-4-level-mathematical-olympiad","slug":"accessing-gpt-4-level-mathematical-olympiad","title":"Accessing GPT-4 level Mathematical Olympiad Solutions via Monte Carlo Tree Self-refine with LLaMa-3 8B","date":"2024-06-11","arxiv_id":"2406.07394","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/accessing-gpt-4-level-mathematical-olympiad#ran","syntology_url":"https://syntology.ai/paper/2406.07394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07394"}},"official":{"repos":["trotsky1997/mathblackbox"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-elbos-a-large-scale-evaluation-of","slug":"beyond-elbos-a-large-scale-evaluation-of","title":"Beyond ELBOs: A Large-Scale Evaluation of Variational Methods for Sampling","date":"2024-06-11","arxiv_id":"2406.07423","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/beyond-elbos-a-large-scale-evaluation-of#ran","syntology_url":"https://syntology.ai/paper/2406.07423","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07423"}},"official":{"repos":["denisbless/variational_sampling_methods"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/language-models-are-alignable-decision-makers","slug":"language-models-are-alignable-decision-makers","title":"Language Models are Alignable Decision-Makers: Dataset and Application to the Medical Triage Domain","date":"2024-06-10","arxiv_id":"2406.06435","repositories_listed":1,"syntology":null},{"url":"/paper/bosc-a-toolbox-for-aerial-imagery-mapping","slug":"bosc-a-toolbox-for-aerial-imagery-mapping","title":"BOSC: A toolbox for aerial imagery mapping","date":"2024-06-09","arxiv_id":"2406.05833","repositories_listed":1,"syntology":null},{"url":"/paper/numerical-solution-of-a-pde-arising-from","slug":"numerical-solution-of-a-pde-arising-from","title":"Numerical solution of a PDE arising from prediction with expert advice","date":"2024-06-09","arxiv_id":"2406.05754","repositories_listed":1,"syntology":null},{"url":"/paper/observation-denoising-in-cyrus-soccer-1","slug":"observation-denoising-in-cyrus-soccer-1","title":"Observation Denoising in CYRUS Soccer Simulation 2D Team For RoboCup 2024","date":"2024-06-09","arxiv_id":"2406.05623","repositories_listed":1,"syntology":null},{"url":"/paper/which-backbone-to-use-a-resource-efficient","slug":"which-backbone-to-use-a-resource-efficient","title":"Which Backbone to Use: A Resource-efficient Domain Specific Comparison for Computer Vision","date":"2024-06-09","arxiv_id":"2406.05612","repositories_listed":1,"syntology":null},{"url":"/paper/digital-twins-of-the-em-environment-benchmark","slug":"digital-twins-of-the-em-environment-benchmark","title":"Toward Real-Time Digital Twins of EM Environments: Computational Benchmark of Ray Launching Software","date":"2024-06-07","arxiv_id":"2406.05042","repositories_listed":1,"syntology":null},{"url":"/paper/predictive-dynamic-fusion","slug":"predictive-dynamic-fusion","title":"Predictive Dynamic Fusion","date":"2024-06-07","arxiv_id":"2406.04802","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/predictive-dynamic-fusion#ran","syntology_url":"https://syntology.ai/paper/2406.04802","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04802"}},"official":{"repos":["yinan-xia/pdf"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/contrastive-sparse-autoencoders-for","slug":"contrastive-sparse-autoencoders-for","title":"Contrastive Sparse Autoencoders for Interpreting Planning of Chess-Playing Agents","date":"2024-06-06","arxiv_id":"2406.04028","repositories_listed":1,"syntology":null},{"url":"/paper/explainability-and-hate-speech-structured","slug":"explainability-and-hate-speech-structured","title":"Explainability and Hate Speech: Structured Explanations Make Social Media Moderators Faster","date":"2024-06-06","arxiv_id":"2406.04106","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-automatic-strategy-discovery-to","slug":"leveraging-automatic-strategy-discovery-to","title":"Leveraging automatic strategy discovery to teach people how to select better projects","date":"2024-06-06","arxiv_id":"2406.04082","repositories_listed":1,"syntology":null},{"url":"/paper/efficient-exploration-of-the-rashomon-set-of","slug":"efficient-exploration-of-the-rashomon-set-of","title":"Efficient Exploration of the Rashomon Set of Rule Set Models","date":"2024-06-05","arxiv_id":"2406.03059","repositories_listed":1,"syntology":null},{"url":"/paper/open-grounded-planning-challenges-and","slug":"open-grounded-planning-challenges-and","title":"Open Grounded Planning: Challenges and Benchmark Construction","date":"2024-06-05","arxiv_id":"2406.02903","repositories_listed":1,"syntology":null}],"record_sha256":"e919dda305c2d23c735074446113c693eea14505277f16a139b1616220fb48a7","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}