{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/decision-making/papers/ran/2","list_of":"/task/decision-making","task":"Decision Making","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not run it on this task or check it against the task's benchmarks.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":2,"pages_in_order":7,"rows_per_page":100,"rows":[101,200],"of":678,"counts":{"archive_papers_tagged":12311,"with_a_code_link":2946,"where_syntology_ran_a_sample":678,"not_listed_spam_title":0,"listed":12311,"listed_where_code_ran":678,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":560,"every_run_a_failure_of_syntologys_instrument":118,"listed_with_a_run_with_no_instrument_failure":560,"listed_every_run_a_failure_of_syntologys_instrument":118,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/decision-making/papers/ran/1","prev":"/task/decision-making/papers/ran/1","next":"/task/decision-making/papers/ran/3","papers":[{"url":"/paper/preserving-the-privacy-of-reward-functions-in","slug":"preserving-the-privacy-of-reward-functions-in","title":"Preserving the Privacy of Reward Functions in MDPs through Deception","date":"2024-07-13","arxiv_id":"2407.09809","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/preserving-the-privacy-of-reward-functions-in#ran","syntology_url":"https://syntology.ai/paper/2407.09809","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.09809"}},"official":{"repos":["shshnkreddy/deceptiverl"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/fincon-a-synthesized-llm-multi-agent-system","slug":"fincon-a-synthesized-llm-multi-agent-system","title":"FinCon: A Synthesized LLM Multi-Agent System with Conceptual Verbal Reinforcement for Enhanced Financial Decision Making","date":"2024-07-09","arxiv_id":"2407.06567","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fincon-a-synthesized-llm-multi-agent-system#ran","syntology_url":"https://syntology.ai/paper/2407.06567","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.06567"}},"official":{"repos":["the-finai/fincon"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/learning-to-complement-and-to-defer-to","slug":"learning-to-complement-and-to-defer-to","title":"Learning to Complement and to Defer to Multiple Users","date":"2024-07-09","arxiv_id":"2407.07003","repositories_listed":1,"syntology":{"n":21,"n_ran":19,"n_constructed":0,"n_ran_checked":14,"n_instrument":5,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":14,"n_pointer_only":21,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 14 with no instrument failure: 0 honoured, 0 violated, 14 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/learning-to-complement-and-to-defer-to#ran","syntology_url":"https://syntology.ai/paper/2407.07003","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.07003"}},"official":{"repos":["zhengzhang37/lecodu"],"state":"official (archive's flag): 19 ran","n_ran":19,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/can-learned-optimization-make-reinforcement","slug":"can-learned-optimization-make-reinforcement","title":"Can Learned Optimization Make Reinforcement Learning Less Difficult?","date":"2024-07-09","arxiv_id":"2407.07082","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-learned-optimization-make-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2407.07082","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.07082"}},"official":{"repos":["alexgoldie/rl-learned-optimization"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/simulation-based-benchmarking-for-causal","slug":"simulation-based-benchmarking-for-causal","title":"Simulation-based Benchmarking for Causal Structure Learning in Gene Perturbation Experiments","date":"2024-07-08","arxiv_id":"2407.06015","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/simulation-based-benchmarking-for-causal#ran","syntology_url":"https://syntology.ai/paper/2407.06015","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.06015"}},"official":{"repos":["luka-kovacevic/causalregnet"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/arigraph-learning-knowledge-graph-world","slug":"arigraph-learning-knowledge-graph-world","title":"AriGraph: Learning Knowledge Graph World Models with Episodic Memory for LLM Agents","date":"2024-07-05","arxiv_id":"2407.04363","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":5,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/arigraph-learning-knowledge-graph-world#ran","syntology_url":"https://syntology.ai/paper/2407.04363","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04363"}},"official":{"repos":["airi-institute/arigraph"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/multilingual-trolley-problems-for-language","slug":"multilingual-trolley-problems-for-language","title":"Language Model Alignment in Multilingual Trolley Problems","date":"2024-07-02","arxiv_id":"2407.02273","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multilingual-trolley-problems-for-language#ran","syntology_url":"https://syntology.ai/paper/2407.02273","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.02273"}},"official":{"repos":["causalNLP/moralmachine","causalnlp/multitp"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/the-house-always-wins-a-framework-for","slug":"the-house-always-wins-a-framework-for","title":"View From Above: A Framework for Evaluating Distribution Shifts in Model Behavior","date":"2024-07-01","arxiv_id":"2407.00948","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-house-always-wins-a-framework-for#ran","syntology_url":"https://syntology.ai/paper/2407.00948","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00948"}},"official":{"repos":["Bluefin-Tuna/ApartResearch"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/econnli-evaluating-large-language-models-on","slug":"econnli-evaluating-large-language-models-on","title":"EconNLI: Evaluating Large Language Models on Economics Reasoning","date":"2024-07-01","arxiv_id":"2407.01212","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/econnli-evaluating-large-language-models-on#ran","syntology_url":"https://syntology.ai/paper/2407.01212","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01212"}},"official":{"repos":["irenehere/econnli"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusion-forcing-next-token-prediction-meets","slug":"diffusion-forcing-next-token-prediction-meets","title":"Diffusion Forcing: Next-token Prediction Meets Full-Sequence Diffusion","date":"2024-07-01","arxiv_id":"2407.01392","repositories_listed":3,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":2,"n_no_contract":2,"n_pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 2 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/diffusion-forcing-next-token-prediction-meets#ran","syntology_url":"https://syntology.ai/paper/2407.01392","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01392"}},"official":{"repos":["buoyancy99/diffusion-forcing"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/puzzles-a-benchmark-for-neural-algorithmic","slug":"puzzles-a-benchmark-for-neural-algorithmic","title":"PUZZLES: A Benchmark for Neural Algorithmic Reasoning","date":"2024-06-29","arxiv_id":"2407.00401","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/puzzles-a-benchmark-for-neural-algorithmic#ran","syntology_url":"https://syntology.ai/paper/2407.00401","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.00401"}},"official":{"repos":["eth-disco/rlp"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/from-biased-selective-labels-to-pseudo-labels","slug":"from-biased-selective-labels-to-pseudo-labels","title":"From Biased Selective Labels to Pseudo-Labels: An Expectation-Maximization Framework for Learning from Biased Decisions","date":"2024-06-27","arxiv_id":"2406.18865","repositories_listed":1,"syntology":{"n":9,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":4,"n_honours":4,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 4 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/from-biased-selective-labels-to-pseudo-labels#ran","syntology_url":"https://syntology.ai/paper/2406.18865","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18865"}},"official":{"repos":["mld3/dcem"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["community","official"]}}},{"url":"/paper/evidential-concept-embedding-models-towards","slug":"evidential-concept-embedding-models-towards","title":"Evidential Concept Embedding Models: Towards Reliable Concept Explanations for Skin Disease Diagnosis","date":"2024-06-27","arxiv_id":"2406.19130","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evidential-concept-embedding-models-towards#ran","syntology_url":"https://syntology.ai/paper/2406.19130","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19130"}},"official":{"repos":["obiyoag/evi-cem"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/can-we-trust-the-performance-evaluation-of","slug":"can-we-trust-the-performance-evaluation-of","title":"Can We Trust the Performance Evaluation of Uncertainty Estimation Methods in Text Summarization?","date":"2024-06-25","arxiv_id":"2406.17274","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/can-we-trust-the-performance-evaluation-of#ran","syntology_url":"https://syntology.ai/paper/2406.17274","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17274"}},"official":{"repos":["he159ok/benchmark-of-uncertainty-estimation-methods-in-text-summarization"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-assume-people-are-more","slug":"large-language-models-assume-people-are-more","title":"Large Language Models Assume People are More Rational than We Really are","date":"2024-06-24","arxiv_id":"2406.17055","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/large-language-models-assume-people-are-more#ran","syntology_url":"https://syntology.ai/paper/2406.17055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17055"}},"official":{"repos":["theryanl/llm-rationality"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-temporal-distances-contrastive","slug":"learning-temporal-distances-contrastive","title":"Learning Temporal Distances: Contrastive Successor Features Can Provide a Metric Structure for Decision-Making","date":"2024-06-24","arxiv_id":"2406.17098","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":11,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-temporal-distances-contrastive#ran","syntology_url":"https://syntology.ai/paper/2406.17098","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17098"}},"official":{"repos":["vivekmyers/contrastive_metrics"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/iwisdm-assessing-instruction-following-in","slug":"iwisdm-assessing-instruction-following-in","title":"IWISDM: Assessing instruction following in multimodal models at scale","date":"2024-06-20","arxiv_id":"2406.14343","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/iwisdm-assessing-instruction-following-in#ran","syntology_url":"https://syntology.ai/paper/2406.14343","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14343"}},"official":{"repos":["bashivanlab/iwisdm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/macrohft-memory-augmented-context-aware","slug":"macrohft-memory-augmented-context-aware","title":"MacroHFT: Memory Augmented Context-aware Reinforcement Learning On High Frequency Trading","date":"2024-06-20","arxiv_id":"2406.14537","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/macrohft-memory-augmented-context-aware#ran","syntology_url":"https://syntology.ai/paper/2406.14537","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14537"}},"official":{"repos":["ZONG0004/MacroHFT"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/imageflownet-forecasting-multiscale","slug":"imageflownet-forecasting-multiscale","title":"ImageFlowNet: Forecasting Multiscale Image-Level Trajectories of Disease Progression with Irregularly-Sampled Longitudinal Medical Images","date":"2024-06-20","arxiv_id":"2406.14794","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":5,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/imageflownet-forecasting-multiscale#ran","syntology_url":"https://syntology.ai/paper/2406.14794","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14794"}},"official":{"repos":["ChenLiu-1996/ImageFlowNet"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["named_in_paper","official"]}}},{"url":"/paper/statistical-uncertainty-in-word-embeddings","slug":"statistical-uncertainty-in-word-embeddings","title":"Statistical Uncertainty in Word Embeddings: GloVe-V","date":"2024-06-18","arxiv_id":"2406.12165","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/statistical-uncertainty-in-word-embeddings#ran","syntology_url":"https://syntology.ai/paper/2406.12165","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12165"}},"official":{"repos":["reglab/glove-v"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/planrag-a-plan-then-retrieval-augmented","slug":"planrag-a-plan-then-retrieval-augmented","title":"PlanRAG: A Plan-then-Retrieval Augmented Generation for Generative Large Language Models as Decision Makers","date":"2024-06-18","arxiv_id":"2406.12430","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/planrag-a-plan-then-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2406.12430","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12430"}},"official":{"repos":["myeon9h/planrag"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ask-before-plan-proactive-language-agents-for","slug":"ask-before-plan-proactive-language-agents-for","title":"Ask-before-Plan: Proactive Language Agents for Real-World Planning","date":"2024-06-18","arxiv_id":"2406.12639","repositories_listed":1,"syntology":{"n":12,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/ask-before-plan-proactive-language-agents-for#ran","syntology_url":"https://syntology.ai/paper/2406.12639","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12639"}},"official":{"repos":["magicgh/ask-before-plan"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/chg-shapley-efficient-data-valuation-and","slug":"chg-shapley-efficient-data-valuation-and","title":"CHG Shapley: Efficient Data Valuation and Selection towards Trustworthy Machine Learning","date":"2024-06-17","arxiv_id":"2406.11730","repositories_listed":2,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/chg-shapley-efficient-data-valuation-and#ran","syntology_url":"https://syntology.ai/paper/2406.11730","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11730"}},"official":{"repos":["caihuaiguang/CHG-Shapley-for-Data-Selection","caihuaiguang/CHG-Shapley-for-Data-Valuation"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/cleandiffuser-an-easy-to-use-modularized","slug":"cleandiffuser-an-easy-to-use-modularized","title":"CleanDiffuser: An Easy-to-use Modularized Library for Diffusion Models in Decision Making","date":"2024-06-13","arxiv_id":"2406.09509","repositories_listed":3,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/cleandiffuser-an-easy-to-use-modularized#ran","syntology_url":"https://syntology.ai/paper/2406.09509","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.09509"}},"official":{"repos":["cleandiffuserteam/cleandiffuser"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/lvbench-an-extreme-long-video-understanding","slug":"lvbench-an-extreme-long-video-understanding","title":"LVBench: An Extreme Long Video Understanding Benchmark","date":"2024-06-12","arxiv_id":"2406.08035","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lvbench-an-extreme-long-video-understanding#ran","syntology_url":"https://syntology.ai/paper/2406.08035","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08035"}},"official":{"repos":["THUDM/LVBench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/accessing-gpt-4-level-mathematical-olympiad","slug":"accessing-gpt-4-level-mathematical-olympiad","title":"Accessing GPT-4 level Mathematical Olympiad Solutions via Monte Carlo Tree Self-refine with LLaMa-3 8B","date":"2024-06-11","arxiv_id":"2406.07394","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/accessing-gpt-4-level-mathematical-olympiad#ran","syntology_url":"https://syntology.ai/paper/2406.07394","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07394"}},"official":{"repos":["trotsky1997/mathblackbox"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/beyond-elbos-a-large-scale-evaluation-of","slug":"beyond-elbos-a-large-scale-evaluation-of","title":"Beyond ELBOs: A Large-Scale Evaluation of Variational Methods for Sampling","date":"2024-06-11","arxiv_id":"2406.07423","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/beyond-elbos-a-large-scale-evaluation-of#ran","syntology_url":"https://syntology.ai/paper/2406.07423","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07423"}},"official":{"repos":["denisbless/variational_sampling_methods"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/predictive-dynamic-fusion","slug":"predictive-dynamic-fusion","title":"Predictive Dynamic Fusion","date":"2024-06-07","arxiv_id":"2406.04802","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/predictive-dynamic-fusion#ran","syntology_url":"https://syntology.ai/paper/2406.04802","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.04802"}},"official":{"repos":["yinan-xia/pdf"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/label-wise-aleatoric-and-epistemic","slug":"label-wise-aleatoric-and-epistemic","title":"Label-wise Aleatoric and Epistemic Uncertainty Quantification","date":"2024-06-04","arxiv_id":"2406.02354","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":14,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/label-wise-aleatoric-and-epistemic#ran","syntology_url":"https://syntology.ai/paper/2406.02354","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.02354"}},"official":{"repos":["YSale/label-uq"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/building-socially-equitable-public-models","slug":"building-socially-equitable-public-models","title":"Building Socially-Equitable Public Models","date":"2024-06-04","arxiv_id":"2406.02790","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":0,"n_instrument":7,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 7 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/building-socially-equitable-public-models#ran","syntology_url":"https://syntology.ai/paper/2406.02790","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.02790"}},"official":{"repos":["ren-research/socially-equitable-public-models"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/stochastic-bilevel-optimization-with-lower","slug":"stochastic-bilevel-optimization-with-lower","title":"Contextual Bilevel Reinforcement Learning for Incentive Alignment","date":"2024-06-03","arxiv_id":"2406.01575","repositories_listed":1,"syntology":{"n":13,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/stochastic-bilevel-optimization-with-lower#ran","syntology_url":"https://syntology.ai/paper/2406.01575","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.01575"}},"official":{"repos":["lasgroup/hpgd"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/in-context-decision-transformer-reinforcement","slug":"in-context-decision-transformer-reinforcement","title":"In-Context Decision Transformer: Reinforcement Learning via Hierarchical Chain-of-Thought","date":"2024-05-31","arxiv_id":"2405.20692","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/in-context-decision-transformer-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2405.20692","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.20692"}},"official":{"repos":["silihuang-ai/idt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/pursuing-overall-welfare-in-federated","slug":"pursuing-overall-welfare-in-federated","title":"Pursuing Overall Welfare in Federated Learning through Sequential Decision Making","date":"2024-05-31","arxiv_id":"2405.20821","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pursuing-overall-welfare-in-federated#ran","syntology_url":"https://syntology.ai/paper/2405.20821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.20821"}},"official":{"repos":["vaseline555/aaggff"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/g-transformer-for-conditional-average","slug":"g-transformer-for-conditional-average","title":"G-Transformer for Conditional Average Potential Outcome Estimation over Time","date":"2024-05-31","arxiv_id":"2405.21012","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/g-transformer-for-conditional-average#ran","syntology_url":"https://syntology.ai/paper/2405.21012","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.21012"}},"official":{"repos":["konstantinhess/G_transformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reward-machines-for-deep-rl-in-noisy-and","slug":"reward-machines-for-deep-rl-in-noisy-and","title":"Reward Machines for Deep RL in Noisy and Uncertain Environments","date":"2024-05-31","arxiv_id":"2406.00120","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/reward-machines-for-deep-rl-in-noisy-and#ran","syntology_url":"https://syntology.ai/paper/2406.00120","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.00120"}},"official":{"repos":["andrewli77/reward-machines-noisy-environments"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/streamflow-prediction-with-uncertainty","slug":"streamflow-prediction-with-uncertainty","title":"Streamflow Prediction with Uncertainty Quantification for Water Management: A Constrained Reasoning and Learning Approach","date":"2024-05-31","arxiv_id":"2406.00133","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/streamflow-prediction-with-uncertainty#ran","syntology_url":"https://syntology.ai/paper/2406.00133","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.00133"}},"official":{"repos":["aminegha/streampred"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/occsora-4d-occupancy-generation-models-as","slug":"occsora-4d-occupancy-generation-models-as","title":"OccSora: 4D Occupancy Generation Models as World Simulators for Autonomous Driving","date":"2024-05-30","arxiv_id":"2405.20337","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/occsora-4d-occupancy-generation-models-as#ran","syntology_url":"https://syntology.ai/paper/2405.20337","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.20337"}},"official":{"repos":["wzzheng/occsora"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/can-graph-learning-improve-task-planning","slug":"can-graph-learning-improve-task-planning","title":"Can Graph Learning Improve Planning in LLM-based Agents?","date":"2024-05-29","arxiv_id":"2405.19119","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/can-graph-learning-improve-task-planning#ran","syntology_url":"https://syntology.ai/paper/2405.19119","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.19119"}},"official":{"repos":["wxxshirley/gnn4taskplan"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/resisting-stochastic-risks-in-diffusion","slug":"resisting-stochastic-risks-in-diffusion","title":"Resisting Stochastic Risks in Diffusion Planners with the Trajectory Aggregation Tree","date":"2024-05-28","arxiv_id":"2405.17879","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":1,"n_instrument":3,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/resisting-stochastic-risks-in-diffusion#ran","syntology_url":"https://syntology.ai/paper/2405.17879","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17879"}},"official":{"repos":["langfengq/tree-diffusion-planner"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["community"]}}},{"url":"/paper/gta-generative-trajectory-augmentation-with","slug":"gta-generative-trajectory-augmentation-with","title":"GTA: Generative Trajectory Augmentation with Guidance for Offline Reinforcement Learning","date":"2024-05-27","arxiv_id":"2405.16907","repositories_listed":1,"syntology":{"n":37,"n_ran":34,"n_constructed":0,"n_ran_checked":29,"n_instrument":5,"n_unverified":3,"n_honours":3,"n_violates":4,"n_no_contract":22,"n_pointer_only":5,"phrase":"34 ran (of which 0 constructed an object rather than computing a result; 29 with no instrument failure: 3 honoured, 4 violated, 22 with no contract checked; 5 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/gta-generative-trajectory-augmentation-with#ran","syntology_url":"https://syntology.ai/paper/2405.16907","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.16907"}},"official":{"repos":["jaewoopudding/gta"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":14,"n_unverified":3,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/position-foundation-agents-as-the-paradigm","slug":"position-foundation-agents-as-the-paradigm","title":"Position: Foundation Agents as the Paradigm Shift for Decision Making","date":"2024-05-27","arxiv_id":"2405.17009","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":4,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/position-foundation-agents-as-the-paradigm#ran","syntology_url":"https://syntology.ai/paper/2405.17009","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17009"}},"official":{"repos":["microsoft/smart"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/exploring-and-steering-the-moral-compass-of","slug":"exploring-and-steering-the-moral-compass-of","title":"Exploring and steering the moral compass of Large Language Models","date":"2024-05-27","arxiv_id":"2405.17345","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/exploring-and-steering-the-moral-compass-of#ran","syntology_url":"https://syntology.ai/paper/2405.17345","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.17345"}},"official":{"repos":["atlaie/ethical-llms"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/on-the-algorithmic-bias-of-aligning-large","slug":"on-the-algorithmic-bias-of-aligning-large","title":"On the Algorithmic Bias of Aligning Large Language Models with RLHF: Preference Collapse and Matching Regularization","date":"2024-05-26","arxiv_id":"2405.16455","repositories_listed":1,"syntology":{"n":14,"n_ran":11,"n_constructed":0,"n_ran_checked":10,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":14,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/on-the-algorithmic-bias-of-aligning-large#ran","syntology_url":"https://syntology.ai/paper/2405.16455","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.16455"}},"official":{"repos":["JiancongXiao/PM_RLHF"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/stride-a-tool-assisted-llm-agent-framework","slug":"stride-a-tool-assisted-llm-agent-framework","title":"STRIDE: A Tool-Assisted LLM Agent Framework for Strategic and Interactive Decision-Making","date":"2024-05-25","arxiv_id":"2405.16376","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/stride-a-tool-assisted-llm-agent-framework#ran","syntology_url":"https://syntology.ai/paper/2405.16376","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.16376"}},"official":{"repos":["cyrilli/stride"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/diffusion-actor-critic-with-entropy-regulator","slug":"diffusion-actor-critic-with-entropy-regulator","title":"Diffusion Actor-Critic with Entropy Regulator","date":"2024-05-24","arxiv_id":"2405.15177","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/diffusion-actor-critic-with-entropy-regulator#ran","syntology_url":"https://syntology.ai/paper/2405.15177","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15177"}},"official":{"repos":["happy-yan/DACER-Diffusion-with-Online-RL"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/ivideogpt-interactive-videogpts-are-scalable","slug":"ivideogpt-interactive-videogpts-are-scalable","title":"iVideoGPT: Interactive VideoGPTs are Scalable World Models","date":"2024-05-24","arxiv_id":"2405.15223","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/ivideogpt-interactive-videogpts-are-scalable#ran","syntology_url":"https://syntology.ai/paper/2405.15223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15223"}},"official":{"repos":["thuml/iVideoGPT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-invariant-causal-mechanism-from","slug":"learning-invariant-causal-mechanism-from","title":"Learning Invariant Causal Mechanism from Vision-Language Models","date":"2024-05-24","arxiv_id":"2405.15289","repositories_listed":0,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/learning-invariant-causal-mechanism-from#ran","syntology_url":"https://syntology.ai/paper/2405.15289","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15289"}},"official":null}},{"url":"/paper/federated-behavioural-planes-explaining-the","slug":"federated-behavioural-planes-explaining-the","title":"Federated Behavioural Planes: Explaining the Evolution of Client Behaviour in Federated Learning","date":"2024-05-24","arxiv_id":"2405.15632","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/federated-behavioural-planes-explaining-the#ran","syntology_url":"https://syntology.ai/paper/2405.15632","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.15632"}},"official":{"repos":["dariofenoglio98/cf_fl"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/improving-single-domain-generalized-object","slug":"improving-single-domain-generalized-object","title":"Improving Single Domain-Generalized Object Detection: A Focus on Diversification and Alignment","date":"2024-05-23","arxiv_id":"2405.14497","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/improving-single-domain-generalized-object#ran","syntology_url":"https://syntology.ai/paper/2405.14497","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14497"}},"official":{"repos":["msohaildanish/divalign"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/didi-diffusion-guided-diversity-for-offline","slug":"didi-diffusion-guided-diversity-for-offline","title":"DIDI: Diffusion-Guided Diversity for Offline Behavioral Generation","date":"2024-05-23","arxiv_id":"2405.14790","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":8,"n_instrument":1,"n_unverified":4,"n_honours":1,"n_violates":0,"n_no_contract":7,"n_pointer_only":13,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/didi-diffusion-guided-diversity-for-offline#ran","syntology_url":"https://syntology.ai/paper/2405.14790","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.14790"}},"official":{"repos":["huey0528/icml24didi"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/configurable-mirror-descent-towards-a","slug":"configurable-mirror-descent-towards-a","title":"Configurable Mirror Descent: Towards a Unification of Decision Making","date":"2024-05-20","arxiv_id":"2405.11746","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/configurable-mirror-descent-towards-a#ran","syntology_url":"https://syntology.ai/paper/2405.11746","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.11746"}},"official":{"repos":["ipadli/cmd"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/conformal-alignment-knowing-when-to-trust","slug":"conformal-alignment-knowing-when-to-trust","title":"Conformal Alignment: Knowing When to Trust Foundation Models with Guarantees","date":"2024-05-16","arxiv_id":"2405.10301","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conformal-alignment-knowing-when-to-trust#ran","syntology_url":"https://syntology.ai/paper/2405.10301","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.10301"}},"official":{"repos":["yugjerry/conformal-alignment"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-decision-policies-with-instrumental","slug":"learning-decision-policies-with-instrumental","title":"Learning Decision Policies with Instrumental Variables through Double Machine Learning","date":"2024-05-14","arxiv_id":"2405.08498","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/learning-decision-policies-with-instrumental#ran","syntology_url":"https://syntology.ai/paper/2405.08498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.08498"}},"official":{"repos":["shaodaqian/DML-IV"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fedgcs-a-generative-framework-for-efficient","slug":"fedgcs-a-generative-framework-for-efficient","title":"FedGCS: A Generative Framework for Efficient Client Selection in Federated Learning via Gradient-based Optimization","date":"2024-05-10","arxiv_id":"2405.06312","repositories_listed":1,"syntology":{"n":9,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":9,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/fedgcs-a-generative-framework-for-efficient#ran","syntology_url":"https://syntology.ai/paper/2405.06312","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.06312"}},"official":{"repos":["zhiyuan-ning/GenerativeFL"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/weakly-supervised-residual-evidential","slug":"weakly-supervised-residual-evidential","title":"Weakly-Supervised Residual Evidential Learning for Multi-Instance Uncertainty Estimation","date":"2024-05-07","arxiv_id":"2405.04405","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/weakly-supervised-residual-evidential#ran","syntology_url":"https://syntology.ai/paper/2405.04405","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.04405"}},"official":{"repos":["liupei101/mirel"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/conformity-confabulation-and-impersonation","slug":"conformity-confabulation-and-impersonation","title":"Persona Inconstancy in Multi-Agent LLM Collaboration: Conformity, Confabulation, and Impersonation","date":"2024-05-06","arxiv_id":"2405.03862","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/conformity-confabulation-and-impersonation#ran","syntology_url":"https://syntology.ai/paper/2405.03862","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.03862"}},"official":{"repos":["baltaci-r/CulturedAgents"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hire-me-or-not-examining-language-model-s","slug":"hire-me-or-not-examining-language-model-s","title":"Hire Me or Not? Examining Language Model's Behavior with Occupation Attributes","date":"2024-05-06","arxiv_id":"2405.06687","repositories_listed":1,"syntology":{"n":13,"n_ran":10,"n_constructed":0,"n_ran_checked":10,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":10,"n_pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/hire-me-or-not-examining-language-model-s#ran","syntology_url":"https://syntology.ai/paper/2405.06687","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.06687"}},"official":{"repos":["daminz97/multi-step_gsv"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/omnidrive-a-holistic-llm-agent-framework-for","slug":"omnidrive-a-holistic-llm-agent-framework-for","title":"OmniDrive: A Holistic Vision-Language Dataset for Autonomous Driving with Counterfactual Reasoning","date":"2024-05-02","arxiv_id":"2405.01533","repositories_listed":1,"syntology":{"n":9,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/omnidrive-a-holistic-llm-agent-framework-for#ran","syntology_url":"https://syntology.ai/paper/2405.01533","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.01533"}},"official":{"repos":["nvlabs/omnidrive"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hard-thresholding-meets-evolution-strategies","slug":"hard-thresholding-meets-evolution-strategies","title":"Hard-Thresholding Meets Evolution Strategies in Reinforcement Learning","date":"2024-05-02","arxiv_id":"2405.01615","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/hard-thresholding-meets-evolution-strategies#ran","syntology_url":"https://syntology.ai/paper/2405.01615","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2405.01615"}},"official":{"repos":["cangcn/nes-ht"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/detecting-critical-treatment-effect-bias-in","slug":"detecting-critical-treatment-effect-bias-in","title":"Detecting critical treatment effect bias in small subgroups","date":"2024-04-29","arxiv_id":"2404.18905","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/detecting-critical-treatment-effect-bias-in#ran","syntology_url":"https://syntology.ai/paper/2404.18905","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.18905"}},"official":{"repos":["jaabmar/kernel-test-bias"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cocog-controllable-visual-stimuli-generation","slug":"cocog-controllable-visual-stimuli-generation","title":"CoCoG: Controllable Visual Stimuli Generation based on Human Concept Representations","date":"2024-04-25","arxiv_id":"2404.16482","repositories_listed":1,"syntology":{"n":5,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/cocog-controllable-visual-stimuli-generation#ran","syntology_url":"https://syntology.ai/paper/2404.16482","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.16482"}},"official":{"repos":["ncclab-sustech/cocog"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/cooperate-or-collapse-emergence-of","slug":"cooperate-or-collapse-emergence-of","title":"Cooperate or Collapse: Emergence of Sustainable Cooperation in a Society of LLM Agents","date":"2024-04-25","arxiv_id":"2404.16698","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cooperate-or-collapse-emergence-of#ran","syntology_url":"https://syntology.ai/paper/2404.16698","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.16698"}},"official":{"repos":["giorgiopiatti/govsim"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/hyperparameter-optimization-can-even-be","slug":"hyperparameter-optimization-can-even-be","title":"Hyperparameter Optimization Can Even be Harmful in Off-Policy Learning and How to Deal with It","date":"2024-04-23","arxiv_id":"2404.15084","repositories_listed":0,"syntology":{"n":9,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/hyperparameter-optimization-can-even-be#ran","syntology_url":"https://syntology.ai/paper/2404.15084","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.15084"}},"official":null}},{"url":"/paper/bias-patterns-in-the-application-of-llms-for","slug":"bias-patterns-in-the-application-of-llms-for","title":"Bias patterns in the application of LLMs for clinical decision support: A comprehensive study","date":"2024-04-23","arxiv_id":"2404.15149","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/bias-patterns-in-the-application-of-llms-for#ran","syntology_url":"https://syntology.ai/paper/2404.15149","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.15149"}},"official":{"repos":["healthylaife/faircdsllm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/battleagent-multi-modal-dynamic-emulation-on","slug":"battleagent-multi-modal-dynamic-emulation-on","title":"BattleAgent: Multi-modal Dynamic Emulation on Historical Battles to Complement Historical Analysis","date":"2024-04-23","arxiv_id":"2404.15532","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/battleagent-multi-modal-dynamic-emulation-on#ran","syntology_url":"https://syntology.ai/paper/2404.15532","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.15532"}},"official":{"repos":["agiresearch/battleagent"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/adaptive-collaboration-strategy-for-llms-in","slug":"adaptive-collaboration-strategy-for-llms-in","title":"MDAgents: An Adaptive Collaboration of LLMs for Medical Decision-Making","date":"2024-04-22","arxiv_id":"2404.15155","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaptive-collaboration-strategy-for-llms-in#ran","syntology_url":"https://syntology.ai/paper/2404.15155","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.15155"}},"official":{"repos":["mitmedialab/mdagents"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/procedural-dilemma-generation-for-evaluating","slug":"procedural-dilemma-generation-for-evaluating","title":"Procedural Dilemma Generation for Evaluating Moral Reasoning in Humans and Language Models","date":"2024-04-17","arxiv_id":"2404.10975","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/procedural-dilemma-generation-for-evaluating#ran","syntology_url":"https://syntology.ai/paper/2404.10975","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.10975"}},"official":{"repos":["cicl-stanford/moral-evals"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/what-hides-behind-unfairness-exploring","slug":"what-hides-behind-unfairness-exploring","title":"What Hides behind Unfairness? Exploring Dynamics Fairness in Reinforcement Learning","date":"2024-04-16","arxiv_id":"2404.10942","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/what-hides-behind-unfairness-exploring#ran","syntology_url":"https://syntology.ai/paper/2404.10942","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.10942"}},"official":{"repos":["familyld/insightfair"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/incremental-residual-concept-bottleneck","slug":"incremental-residual-concept-bottleneck","title":"Incremental Residual Concept Bottleneck Models","date":"2024-04-13","arxiv_id":"2404.08978","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/incremental-residual-concept-bottleneck#ran","syntology_url":"https://syntology.ai/paper/2404.08978","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.08978"}},"official":{"repos":["helloscm/res-cbm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/lm-transparency-tool-interactive-tool-for","slug":"lm-transparency-tool-interactive-tool-for","title":"LM Transparency Tool: Interactive Tool for Analyzing Transformer Language Models","date":"2024-04-10","arxiv_id":"2404.07004","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lm-transparency-tool-interactive-tool-for#ran","syntology_url":"https://syntology.ai/paper/2404.07004","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.07004"}},"official":null}},{"url":"/paper/percentile-criterion-optimization-in-offline-1","slug":"percentile-criterion-optimization-in-offline-1","title":"Percentile Criterion Optimization in Offline Reinforcement Learning","date":"2024-04-07","arxiv_id":"2404.05055","repositories_listed":1,"syntology":{"n":9,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/percentile-criterion-optimization-in-offline-1#ran","syntology_url":"https://syntology.ai/paper/2404.05055","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.05055"}},"official":{"repos":["elitalobo/varframework"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/autowebglm-bootstrap-and-reinforce-a-large","slug":"autowebglm-bootstrap-and-reinforce-a-large","title":"AutoWebGLM: A Large Language Model-based Web Navigating Agent","date":"2024-04-04","arxiv_id":"2404.03648","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/autowebglm-bootstrap-and-reinforce-a-large#ran","syntology_url":"https://syntology.ai/paper/2404.03648","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.03648"}},"official":{"repos":["thudm/autowebglm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cape-cam-as-a-probabilistic-ensemble-for","slug":"cape-cam-as-a-probabilistic-ensemble-for","title":"CAPE: CAM as a Probabilistic Ensemble for Enhanced DNN Interpretation","date":"2024-04-03","arxiv_id":"2404.02388","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/cape-cam-as-a-probabilistic-ensemble-for#ran","syntology_url":"https://syntology.ai/paper/2404.02388","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.02388"}},"official":{"repos":["aiml-med/cape"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/predictive-performance-comparison-of-decision","slug":"predictive-performance-comparison-of-decision","title":"Predictive Performance Comparison of Decision Policies Under Confounding","date":"2024-04-01","arxiv_id":"2404.00848","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/predictive-performance-comparison-of-decision#ran","syntology_url":"https://syntology.ai/paper/2404.00848","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2404.00848"}},"official":{"repos":["lguerdan/icml24_predictive_performance_comparison_dps"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/decision-mamba-reinforcement-learning-via","slug":"decision-mamba-reinforcement-learning-via","title":"Decision Mamba: Reinforcement Learning via Sequence Modeling with Selective State Spaces","date":"2024-03-29","arxiv_id":"2403.19925","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/decision-mamba-reinforcement-learning-via#ran","syntology_url":"https://syntology.ai/paper/2403.19925","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.19925"}},"official":{"repos":["toshihiro-ota/decision-mamba"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mitigating-hallucinations-in-large-vision","slug":"mitigating-hallucinations-in-large-vision","title":"Mitigating Hallucinations in Large Vision-Language Models with Instruction Contrastive Decoding","date":"2024-03-27","arxiv_id":"2403.18715","repositories_listed":2,"syntology":{"n":8,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/mitigating-hallucinations-in-large-vision#ran","syntology_url":"https://syntology.ai/paper/2403.18715","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.18715"}},"official":{"repos":["p1k0pan/ICD"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/optimization-based-prompt-injection-attack-to","slug":"optimization-based-prompt-injection-attack-to","title":"Optimization-based Prompt Injection Attack to LLM-as-a-Judge","date":"2024-03-26","arxiv_id":"2403.17710","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/optimization-based-prompt-injection-attack-to#ran","syntology_url":"https://syntology.ai/paper/2403.17710","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.17710"}},"official":{"repos":["shijiawenwen/judgedeceiver"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/an-image-computable-model-of-speeded-decision","slug":"an-image-computable-model-of-speeded-decision","title":"An image-computable model of speeded decision-making","date":"2024-03-25","arxiv_id":"2403.16382","repositories_listed":1,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/an-image-computable-model-of-speeded-decision#ran","syntology_url":"https://syntology.ai/paper/2403.16382","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.16382"}},"official":{"repos":["pauljaffe/vam"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/contrastive-balancing-representation-learning","slug":"contrastive-balancing-representation-learning","title":"Contrastive Balancing Representation Learning for Heterogeneous Dose-Response Curves Estimation","date":"2024-03-21","arxiv_id":"2403.14232","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":4,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":4,"n_pointer_only":7,"phrase":"7 ran (of which 4 constructed an object rather than computing a result; 7 with no instrument failure: 3 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/contrastive-balancing-representation-learning#ran","syntology_url":"https://syntology.ai/paper/2403.14232","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.14232"}},"official":{"repos":["euzmin/Contrastive-Balancing-Representation-Network-CRNet"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":4,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/uncertainty-quantification-for-data-driven-2","slug":"uncertainty-quantification-for-data-driven-2","title":"Uncertainty quantification for data-driven weather models","date":"2024-03-20","arxiv_id":"2403.13458","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/uncertainty-quantification-for-data-driven-2#ran","syntology_url":"https://syntology.ai/paper/2403.13458","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.13458"}},"official":{"repos":["cbuelt/dduq"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/chain-of-interaction-enhancing-large-language","slug":"chain-of-interaction-enhancing-large-language","title":"Chain-of-Interaction: Enhancing Large Language Models for Psychiatric Behavior Understanding by Dyadic Contexts","date":"2024-03-20","arxiv_id":"2403.13786","repositories_listed":1,"syntology":{"n":8,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/chain-of-interaction-enhancing-large-language#ran","syntology_url":"https://syntology.ai/paper/2403.13786","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.13786"}},"official":{"repos":["trust-nlp/coi-psychotherapy"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/insight-end-to-end-neuro-symbolic-visual","slug":"insight-end-to-end-neuro-symbolic-visual","title":"End-to-End Neuro-Symbolic Reinforcement Learning with Textual Explanations","date":"2024-03-19","arxiv_id":"2403.12451","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":1,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","sample_list":"/paper/insight-end-to-end-neuro-symbolic-visual#ran","syntology_url":"https://syntology.ai/paper/2403.12451","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12451"}},"official":{"repos":["liruiluo/nsrl-vision-pub"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/embodied-llm-agents-learn-to-cooperate-in","slug":"embodied-llm-agents-learn-to-cooperate-in","title":"Embodied LLM Agents Learn to Cooperate in Organized Teams","date":"2024-03-19","arxiv_id":"2403.12482","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/embodied-llm-agents-learn-to-cooperate-in#ran","syntology_url":"https://syntology.ai/paper/2403.12482","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12482"}},"official":{"repos":["tobeatraceur/Organized-LLM-Agents"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/a-practical-guide-to-statistical-distances","slug":"a-practical-guide-to-statistical-distances","title":"A Practical Guide to Sample-based Statistical Distances for Evaluating Generative Models in Science","date":"2024-03-19","arxiv_id":"2403.12636","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":3,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 3 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-practical-guide-to-statistical-distances#ran","syntology_url":"https://syntology.ai/paper/2403.12636","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12636"}},"official":{"repos":["mackelab/labproject"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-guided-evolution-the-automation-of-models","slug":"llm-guided-evolution-the-automation-of-models","title":"LLM Guided Evolution - The Automation of Models Advancing Models","date":"2024-03-18","arxiv_id":"2403.11446","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llm-guided-evolution-the-automation-of-models#ran","syntology_url":"https://syntology.ai/paper/2403.11446","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.11446"}},"official":{"repos":["clint-kristopher-morris/llm-guided-evolution"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/how-far-are-we-on-the-decision-making-of-llms","slug":"how-far-are-we-on-the-decision-making-of-llms","title":"How Far Are We on the Decision-Making of LLMs? Evaluating LLMs' Gaming Ability in Multi-Agent Environments","date":"2024-03-18","arxiv_id":"2403.11807","repositories_listed":1,"syntology":{"n":9,"n_ran":8,"n_constructed":0,"n_ran_checked":8,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":2,"n_no_contract":5,"n_pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 2 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/how-far-are-we-on-the-decision-making-of-llms#ran","syntology_url":"https://syntology.ai/paper/2403.11807","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.11807"}},"official":{"repos":["cuhk-arise/gamabench"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/expandable-subspace-ensemble-for-pre-trained","slug":"expandable-subspace-ensemble-for-pre-trained","title":"Expandable Subspace Ensemble for Pre-Trained Model-Based Class-Incremental Learning","date":"2024-03-18","arxiv_id":"2403.12030","repositories_listed":1,"syntology":{"n":7,"n_ran":2,"n_constructed":1,"n_ran_checked":1,"n_instrument":1,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":7,"phrase":"2 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/expandable-subspace-ensemble-for-pre-trained#ran","syntology_url":"https://syntology.ai/paper/2403.12030","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.12030"}},"official":{"repos":["sun-hailong/cvpr24-ease"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/auditing-fairness-under-unobserved","slug":"auditing-fairness-under-unobserved","title":"Auditing Fairness under Unobserved Confounding","date":"2024-03-18","arxiv_id":"2403.14713","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/auditing-fairness-under-unobserved#ran","syntology_url":"https://syntology.ai/paper/2403.14713","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.14713"}},"official":{"repos":["lasilab/inequity-bounds"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/reinforced-sequential-decision-making-for","slug":"reinforced-sequential-decision-making-for","title":"Reinforced Sequential Decision-Making for Sepsis Treatment: The POSNEGDM Framework with Mortality Classifier and Transformer","date":"2024-03-12","arxiv_id":"2403.07309","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":3,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":2,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/reinforced-sequential-decision-making-for#ran","syntology_url":"https://syntology.ai/paper/2403.07309","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07309"}},"official":{"repos":["dipeshtamboli/posnegdm-reinforced-sequential-decision-making-for-sepsis-treatment"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/better-than-classical-the-subtle-art-of","slug":"better-than-classical-the-subtle-art-of","title":"Better than classical? The subtle art of benchmarking quantum machine learning models","date":"2024-03-11","arxiv_id":"2403.07059","repositories_listed":3,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/better-than-classical-the-subtle-art-of#ran","syntology_url":"https://syntology.ai/paper/2403.07059","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.07059"}},"official":{"repos":["xanaduai/qml-benchmarks"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/tapilot-crossing-benchmarking-and-evolving","slug":"tapilot-crossing-benchmarking-and-evolving","title":"Tapilot-Crossing: Benchmarking and Evolving LLMs Towards Interactive Data Analysis Agents","date":"2024-03-08","arxiv_id":"2403.05307","repositories_listed":1,"syntology":{"n":16,"n_ran":16,"n_constructed":0,"n_ran_checked":16,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tapilot-crossing-benchmarking-and-evolving#ran","syntology_url":"https://syntology.ai/paper/2403.05307","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.05307"}},"official":{"repos":["tapilot-crossing/tapilot_code"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/human-vs-machine-language-models-and-wargames","slug":"human-vs-machine-language-models-and-wargames","title":"Human vs. Machine: Behavioral Differences Between Expert Humans and Language Models in Wargame Simulations","date":"2024-03-06","arxiv_id":"2403.03407","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/human-vs-machine-language-models-and-wargames#ran","syntology_url":"https://syntology.ai/paper/2403.03407","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.03407"}},"official":{"repos":["ancorso/llmwargaming"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/xai-based-detection-of-adversarial-attacks-on","slug":"xai-based-detection-of-adversarial-attacks-on","title":"XAI-Based Detection of Adversarial Attacks on Deepfake Detectors","date":"2024-03-05","arxiv_id":"2403.02955","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":10,"n_pointer_only":13,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 1 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/xai-based-detection-of-adversarial-attacks-on#ran","syntology_url":"https://syntology.ai/paper/2403.02955","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.02955"}},"official":{"repos":["razla/xai-based-detection-of-adversarial-attacks-on-deepfake-detectors"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/mikasa-multi-key-anchor-scene-aware","slug":"mikasa-multi-key-anchor-scene-aware","title":"MiKASA: Multi-Key-Anchor & Scene-Aware Transformer for 3D Visual Grounding","date":"2024-03-05","arxiv_id":"2403.03077","repositories_listed":1,"syntology":{"n":16,"n_ran":11,"n_constructed":10,"n_ran_checked":11,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":16,"phrase":"11 ran (of which 10 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mikasa-multi-key-anchor-scene-aware#ran","syntology_url":"https://syntology.ai/paper/2403.03077","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.03077"}},"official":{"repos":["dfki-av/mikasa-3dvg"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":10,"n_ran_no_instrument_failure":11,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/behavior-generation-with-latent-actions","slug":"behavior-generation-with-latent-actions","title":"Behavior Generation with Latent Actions","date":"2024-03-05","arxiv_id":"2403.03181","repositories_listed":2,"syntology":{"n":14,"n_ran":8,"n_constructed":0,"n_ran_checked":7,"n_instrument":1,"n_unverified":6,"n_honours":1,"n_violates":2,"n_no_contract":4,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 2 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/behavior-generation-with-latent-actions#ran","syntology_url":"https://syntology.ai/paper/2403.03181","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.03181"}},"official":{"repos":["jayLEE0301/vq_bet_official"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/playing-nethack-with-llms-potential","slug":"playing-nethack-with-llms-potential","title":"Playing NetHack with LLMs: Potential & Limitations as Zero-Shot Agents","date":"2024-03-01","arxiv_id":"2403.00690","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/playing-nethack-with-llms-potential#ran","syntology_url":"https://syntology.ai/paper/2403.00690","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.00690"}},"official":{"repos":["commandercero/netplay"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/enhancing-long-term-recommendation-with-bi","slug":"enhancing-long-term-recommendation-with-bi","title":"Large Language Models are Learnable Planners for Long-Term Recommendation","date":"2024-02-29","arxiv_id":"2403.00843","repositories_listed":1,"syntology":{"n":11,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/enhancing-long-term-recommendation-with-bi#ran","syntology_url":"https://syntology.ai/paper/2403.00843","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2403.00843"}},"official":{"repos":["jizhi-zhang/billp"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/decisionnce-embodied-multimodal","slug":"decisionnce-embodied-multimodal","title":"DecisionNCE: Embodied Multimodal Representations via Implicit Preference Learning","date":"2024-02-28","arxiv_id":"2402.18137","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/decisionnce-embodied-multimodal#ran","syntology_url":"https://syntology.ai/paper/2402.18137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18137"}},"official":{"repos":["2toinf/DecisionNCE"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unveiling-the-potential-of-robustness-in","slug":"unveiling-the-potential-of-robustness-in","title":"Unveiling the Potential of Robustness in Selecting Conditional Average Treatment Effect Estimators","date":"2024-02-28","arxiv_id":"2402.18392","repositories_listed":1,"syntology":{"n":10,"n_ran":7,"n_constructed":0,"n_ran_checked":6,"n_instrument":1,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":10,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/unveiling-the-potential-of-robustness-in#ran","syntology_url":"https://syntology.ai/paper/2402.18392","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18392"}},"official":{"repos":["yiyhuang3/cate_estimator_selection"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/rora-robust-free-text-rationale-evaluation","slug":"rora-robust-free-text-rationale-evaluation","title":"RORA: Robust Free-Text Rationale Evaluation","date":"2024-02-28","arxiv_id":"2402.18678","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/rora-robust-free-text-rationale-evaluation#ran","syntology_url":"https://syntology.ai/paper/2402.18678","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2402.18678"}},"official":{"repos":["zipjiang/rora"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"b58592b5e5eea50d3c86d61653477d15a7a624e28d4c0596584f22595fa943cf","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}