{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/set/papers/ran/6","list_of":"/method/set","method":"SET","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not isolate this method inside it.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":6,"pages_in_order":12,"rows_per_page":100,"rows":[501,600],"of":1186,"counts":{"archive_papers_tagged":13419,"with_a_code_link":4158,"where_syntology_ran_a_sample":1186,"not_listed_spam_title":0,"listed":13419,"listed_where_code_ran":1186,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":1038,"every_run_a_failure_of_syntologys_instrument":148,"listed_with_a_run_with_no_instrument_failure":1038,"listed_every_run_a_failure_of_syntologys_instrument":148,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/set/papers/ran/1","prev":"/method/set/papers/ran/5","next":"/method/set/papers/ran/7","papers":[{"paper":"/paper/one-prompt-is-not-enough-automated","slug":"one-prompt-is-not-enough-automated","title":"One Prompt is not Enough: Automated Construction of a Mixture-of-Expert Prompts","date":"2024-06-28","arxiv_id":"2407.00256","n_code_links":1,"syntology":{"ran":4,"of":7,"n_ran_checked":4,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["ruocwang/mixture-of-prompts"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/correspondence-free-non-rigid-point-set-1","slug":"correspondence-free-non-rigid-point-set-1","title":"Correspondence-Free Non-Rigid Point Set Registration Using Unsupervised Clustering Analysis","date":"2024-06-27","arxiv_id":"2406.18817","n_code_links":2,"syntology":{"ran":2,"of":5,"n_ran_checked":0,"n_instrument":2,"unverified":3,"pointer_only":5,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["zikai1/cvpr24_pointsetreg"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/compositional-image-decomposition-with","slug":"compositional-image-decomposition-with","title":"Compositional Image Decomposition with Diffusion Models","date":"2024-06-27","arxiv_id":"2406.19298","n_code_links":0,"syntology":{"ran":10,"of":11,"n_ran_checked":8,"n_instrument":2,"unverified":1,"pointer_only":4,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 1 honoured, 1 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/livebench-a-challenging-contamination-free","slug":"livebench-a-challenging-contamination-free","title":"LiveBench: A Challenging, Contamination-Limited LLM Benchmark","date":"2024-06-27","arxiv_id":"2406.19314","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["livebench/livebench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/dice-end-to-end-deformation-capture-of-hand","slug":"dice-end-to-end-deformation-capture-of-hand","title":"DICE: End-to-end Deformation Capture of Hand-Face Interactions from a Single Image","date":"2024-06-26","arxiv_id":"2406.17988","n_code_links":0,"syntology":{"ran":4,"of":5,"n_ran_checked":3,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/selective-prompting-tuning-for-personalized","slug":"selective-prompting-tuning-for-personalized","title":"Selective Prompting Tuning for Personalized Conversations with LLMs","date":"2024-06-26","arxiv_id":"2406.18187","n_code_links":1,"syntology":{"ran":16,"of":16,"n_ran_checked":16,"n_instrument":0,"unverified":0,"pointer_only":16,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hqsiswiliam/SPT"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/alphaforge-a-framework-to-mine-and","slug":"alphaforge-a-framework-to-mine-and","title":"AlphaForge: A Framework to Mine and Dynamically Combine Formulaic Alpha Factors","date":"2024-06-26","arxiv_id":"2406.18394","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["dulyhao/alphaforge"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/detecting-brittle-decisions-for-free","slug":"detecting-brittle-decisions-for-free","title":"Detecting Brittle Decisions for Free: Leveraging Margin Consistency in Deep Robust Classifiers","date":"2024-06-26","arxiv_id":"2406.18451","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ngnawejonas/margin-consistency"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/wildguard-open-one-stop-moderation-tools-for","slug":"wildguard-open-one-stop-moderation-tools-for","title":"WildGuard: Open One-Stop Moderation Tools for Safety Risks, Jailbreaks, and Refusals of LLMs","date":"2024-06-26","arxiv_id":"2406.18495","n_code_links":4,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["allenai/wildguard"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-stem-agnostic-single-decoder-system-for","slug":"a-stem-agnostic-single-decoder-system-for","title":"A Stem-Agnostic Single-Decoder System for Music Source Separation Beyond Four Stems","date":"2024-06-26","arxiv_id":"2406.18747","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["kwatcharasupat/query-bandit"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/varbench-robust-language-model-benchmarking","slug":"varbench-robust-language-model-benchmarking","title":"VarBench: Robust Language Model Benchmarking Through Dynamic Variable Perturbation","date":"2024-06-25","arxiv_id":"2406.17681","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["qbetterk/VarBench"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/arboretum-a-large-multimodal-dataset-enabling","slug":"arboretum-a-large-multimodal-dataset-enabling","title":"BioTrove: A Large Curated Image Dataset Enabling AI for Biodiversity","date":"2024-06-25","arxiv_id":"2406.17720","n_code_links":2,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["baskargroup/biotrove","baskargroup/Arboretum"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/recite-reconstruct-recollect-memorization-in","slug":"recite-reconstruct-recollect-memorization-in","title":"Recite, Reconstruct, Recollect: Memorization in LMs as a Multifaceted Phenomenon","date":"2024-06-25","arxiv_id":"2406.17746","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["eleutherai/semantic-memorization"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/confidence-regulation-neurons-in-language","slug":"confidence-regulation-neurons-in-language","title":"Confidence Regulation Neurons in Language Models","date":"2024-06-24","arxiv_id":"2406.16254","n_code_links":1,"syntology":{"ran":4,"of":5,"n_ran_checked":0,"n_instrument":4,"unverified":1,"pointer_only":5,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["bpwu1/confidence-regulation-neurons"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/token-based-decision-criteria-are-suboptimal","slug":"token-based-decision-criteria-are-suboptimal","title":"Token-based Decision Criteria Are Suboptimal in In-context Learning","date":"2024-06-24","arxiv_id":"2406.16535","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hc495/Hidden_Calibration"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/evalalign-evaluating-text-to-image-models","slug":"evalalign-evaluating-text-to-image-models","title":"EVALALIGN: Supervised Fine-Tuning Multimodal LLMs with Human-Aligned Data for Evaluating Text-to-Image Models","date":"2024-06-24","arxiv_id":"2406.16562","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":3,"n_instrument":3,"unverified":1,"pointer_only":2,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["sais-fuxi/evalalign"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/confidence-aware-inverse-constrained","slug":"confidence-aware-inverse-constrained","title":"Confidence Aware Inverse Constrained Reinforcement Learning","date":"2024-06-24","arxiv_id":"2406.16782","n_code_links":1,"syntology":{"ran":3,"of":6,"n_ran_checked":1,"n_instrument":2,"unverified":3,"pointer_only":0,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["sriram94/confidenceawareicrl"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/md-tree-a-model-diagnostic-tree-grown-on-loss","slug":"md-tree-a-model-diagnostic-tree-grown-on-loss","title":"MD tree: a model-diagnostic tree grown on loss landscape","date":"2024-06-24","arxiv_id":"2406.16988","n_code_links":1,"syntology":{"ran":15,"of":17,"n_ran_checked":6,"n_instrument":9,"unverified":2,"pointer_only":0,"phrase":"15 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 9 where Syntology's instrument failed) · 2 unverified","official":{"repos":["yefanzhou/modeldiagnosis"],"state":"official (archive's flag): 15 ran","n_ran":15,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/allenoise-large-scale-text-classification","slug":"allenoise-large-scale-text-classification","title":"AlleNoise: large-scale text classification benchmark dataset with real-world label noise","date":"2024-06-24","arxiv_id":"2407.10992","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["allegro/allenoise"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/blind-baselines-beat-membership-inference","slug":"blind-baselines-beat-membership-inference","title":"Blind Baselines Beat Membership Inference Attacks for Foundation Models","date":"2024-06-23","arxiv_id":"2406.16201","n_code_links":1,"syntology":{"ran":8,"of":11,"n_ran_checked":8,"n_instrument":0,"unverified":3,"pointer_only":11,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["ethz-spylab/Blind-MIA"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/semantic-entropy-probes-robust-and-cheap","slug":"semantic-entropy-probes-robust-and-cheap","title":"Semantic Entropy Probes: Robust and Cheap Hallucination Detection in LLMs","date":"2024-06-22","arxiv_id":"2406.15927","n_code_links":1,"syntology":{"ran":12,"of":13,"n_ran_checked":12,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/clip-decoder-zeroshot-multilabel","slug":"clip-decoder-zeroshot-multilabel","title":"CLIP-Decoder : ZeroShot Multilabel Classification using Multimodal CLIP Aligned Representation","date":"2024-06-21","arxiv_id":"2406.14830","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/dipex-dispersing-prompt-expansion-for-class","slug":"dipex-dispersing-prompt-expansion-for-class","title":"DiPEx: Dispersing Prompt Expansion for Class-Agnostic Object Detection","date":"2024-06-21","arxiv_id":"2406.14924","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":4,"n_instrument":1,"unverified":2,"pointer_only":2,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["jason-lim26/dipex"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/goal-a-generalist-combinatorial-optimization","slug":"goal-a-generalist-combinatorial-optimization","title":"GOAL: A Generalist Combinatorial Optimization Agent Learning","date":"2024-06-21","arxiv_id":"2406.15079","n_code_links":1,"syntology":{"ran":4,"of":7,"n_ran_checked":4,"n_instrument":0,"unverified":3,"pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["naver/goal-co"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/eclipse-expunging-clean-label-indiscriminate","slug":"eclipse-expunging-clean-label-indiscriminate","title":"ECLIPSE: Expunging Clean-label Indiscriminate Poisons via Sparse Diffusion Purification","date":"2024-06-21","arxiv_id":"2406.15093","n_code_links":1,"syntology":{"ran":18,"of":19,"n_ran_checked":15,"n_instrument":3,"unverified":1,"pointer_only":10,"phrase":"18 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 3 honoured, 0 violated, 12 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["cgcl-codes/eclipse"],"state":"official (archive's flag): 18 ran","n_ran":18,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/geolrm-geometry-aware-large-reconstruction","slug":"geolrm-geometry-aware-large-reconstruction","title":"GeoLRM: Geometry-Aware Large Reconstruction Model for High-Quality 3D Gaussian Generation","date":"2024-06-21","arxiv_id":"2406.15333","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":7,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["alibaba-yuanjing-aigclab/geolrm"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/multimodal-task-vectors-enable-many-shot","slug":"multimodal-task-vectors-enable-many-shot","title":"Multimodal Task Vectors Enable Many-Shot Multimodal In-Context Learning","date":"2024-06-21","arxiv_id":"2406.15334","n_code_links":1,"syntology":{"ran":16,"of":19,"n_ran_checked":15,"n_instrument":1,"unverified":3,"pointer_only":19,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 15 with no instrument failure: 4 honoured, 2 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["brandon3964/multimodal-task-vector"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":15,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/navsim-data-driven-non-reactive-autonomous","slug":"navsim-data-driven-non-reactive-autonomous","title":"NAVSIM: Data-Driven Non-Reactive Autonomous Vehicle Simulation and Benchmarking","date":"2024-06-21","arxiv_id":"2406.15349","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["autonomousvision/navsim"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/the-elusive-pursuit-of-replicating-pate-gan","slug":"the-elusive-pursuit-of-replicating-pate-gan","title":"The Elusive Pursuit of Reproducing PATE-GAN: Benchmarking, Auditing, Debugging","date":"2024-06-20","arxiv_id":"2406.13985","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["spalabucr/pategan-audit"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/the-reason-behind-good-or-bad-towards-a","slug":"the-reason-behind-good-or-bad-towards-a","title":"LLM Critics Help Catch Bugs in Mathematics: Towards a Better Mathematical Verifier with Natural Language Feedback","date":"2024-06-20","arxiv_id":"2406.14024","n_code_links":1,"syntology":{"ran":22,"of":23,"n_ran_checked":18,"n_instrument":4,"unverified":1,"pointer_only":23,"phrase":"22 ran (of which 0 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 1 violated, 17 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["kbsdjames/math-minos"],"state":"official (archive's flag): 22 ran","n_ran":22,"n_constructed":0,"n_ran_no_instrument_failure":18,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/timo-towards-better-temporal-reasoning-for","slug":"timo-towards-better-temporal-reasoning-for","title":"Timo: Towards Better Temporal Reasoning for Language Models","date":"2024-06-20","arxiv_id":"2406.14192","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":1,"n_instrument":3,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zhaochen0110/timo"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/postmark-a-robust-blackbox-watermark-for","slug":"postmark-a-robust-blackbox-watermark-for","title":"PostMark: A Robust Blackbox Watermark for Large Language Models","date":"2024-06-20","arxiv_id":"2406.14517","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["lilakk/postmark"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/patholm-identifying-pathogenicity-from-the","slug":"patholm-identifying-pathogenicity-from-the","title":"PathoLM: Identifying pathogenicity from the DNA sequence through the Genome Foundation Model","date":"2024-06-19","arxiv_id":"2406.13133","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Sajib-006/Patho-LM"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/biomedical-visual-instruction-tuning-with","slug":"biomedical-visual-instruction-tuning-with","title":"Biomedical Visual Instruction Tuning with Clinician Preference Alignment","date":"2024-06-19","arxiv_id":"2406.13173","n_code_links":1,"syntology":{"ran":10,"of":13,"n_ran_checked":8,"n_instrument":2,"unverified":3,"pointer_only":13,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["mao1207/BioMed-VITAL"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/adamoe-token-adaptive-routing-with-null","slug":"adamoe-token-adaptive-routing-with-null","title":"AdaMoE: Token-Adaptive Routing with Null Experts for Mixture-of-Experts Language Models","date":"2024-06-19","arxiv_id":"2406.13233","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":2,"n_instrument":1,"unverified":2,"pointer_only":5,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["cengzihao/adamoe"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/large-scale-dataset-pruning-in-adversarial","slug":"large-scale-dataset-pruning-in-adversarial","title":"Large-Scale Dataset Pruning in Adversarial Training through Data Importance Extrapolation","date":"2024-06-19","arxiv_id":"2406.13283","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":7,"n_instrument":0,"unverified":2,"pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["BjoernNieth/LS-Dataset-pruning-in-AT"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/sd-eval-a-benchmark-dataset-for-spoken","slug":"sd-eval-a-benchmark-dataset-for-spoken","title":"SD-Eval: A Benchmark Dataset for Spoken Dialogue Understanding Beyond Words","date":"2024-06-19","arxiv_id":"2406.13340","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["amphionspace/sd-eval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/jogging-the-memory-of-unlearned-model-through","slug":"jogging-the-memory-of-unlearned-model-through","title":"Jogging the Memory of Unlearned LLMs Through Targeted Relearning Attacks","date":"2024-06-19","arxiv_id":"2406.13356","n_code_links":1,"syntology":{"ran":13,"of":16,"n_ran_checked":11,"n_instrument":2,"unverified":3,"pointer_only":4,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["s-huu/jog_llm_memory"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/watt-weight-average-test-time-adaption-of","slug":"watt-weight-average-test-time-adaption-of","title":"WATT: Weight Average Test-Time Adaptation of CLIP","date":"2024-06-19","arxiv_id":"2406.13875","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":0,"n_instrument":5,"unverified":2,"pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 2 unverified","official":{"repos":["mehrdad-noori/watt"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/mathador-lm-a-dynamic-benchmark-for","slug":"mathador-lm-a-dynamic-benchmark-for","title":"Mathador-LM: A Dynamic Benchmark for Mathematical Reasoning on Large Language Models","date":"2024-06-18","arxiv_id":"2406.12572","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ist-daslab/mathador-lm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/magic-generating-self-correction-guideline","slug":"magic-generating-self-correction-guideline","title":"MAGIC: Generating Self-Correction Guideline for In-Context Text-to-SQL","date":"2024-06-18","arxiv_id":"2406.12692","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["microsoft/synqo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/olympicarena-benchmarking-multi-discipline","slug":"olympicarena-benchmarking-multi-discipline","title":"OlympicArena: Benchmarking Multi-discipline Cognitive Reasoning for Superintelligent AI","date":"2024-06-18","arxiv_id":"2406.12753","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["gair-nlp/olympicarena"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/chatglm-a-family-of-large-language-models","slug":"chatglm-a-family-of-large-language-models","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","date":"2024-06-18","arxiv_id":"2406.12793","n_code_links":7,"syntology":{"ran":21,"of":29,"n_ran_checked":20,"n_instrument":1,"unverified":8,"pointer_only":1,"phrase":"21 ran (of which 0 constructed an object rather than computing a result; 20 with no instrument failure: 0 honoured, 0 violated, 20 with no contract checked; 1 where Syntology's instrument failed) · 8 unverified","official":{"repos":["thudm/chatglm-6b"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/adversarial-attacks-on-multimodal-agents","slug":"adversarial-attacks-on-multimodal-agents","title":"Dissecting Adversarial Robustness of Multimodal LM Agents","date":"2024-06-18","arxiv_id":"2406.12814","n_code_links":1,"syntology":{"ran":16,"of":21,"n_ran_checked":11,"n_instrument":5,"unverified":5,"pointer_only":1,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 5 where Syntology's instrument failed) · 5 unverified","official":{"repos":["chenwu98/agent-attack"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/few-shot-recognition-via-stage-wise-augmented","slug":"few-shot-recognition-via-stage-wise-augmented","title":"Few-Shot Recognition via Stage-Wise Retrieval-Augmented Finetuning","date":"2024-06-17","arxiv_id":"2406.11148","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["tian1327/swat"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/multimodal-needle-in-a-haystack-benchmarking","slug":"multimodal-needle-in-a-haystack-benchmarking","title":"Multimodal Needle in a Haystack: Benchmarking Long-Context Capability of Multimodal Large Language Models","date":"2024-06-17","arxiv_id":"2406.11230","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wang-ml-lab/multimodal-needle-in-a-haystack"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/vocabulary-expansion-for-low-resource-cross","slug":"vocabulary-expansion-for-low-resource-cross","title":"How Can We Effectively Expand the Vocabulary of LLMs with 0.01GB of Target Language Text?","date":"2024-06-17","arxiv_id":"2406.11477","n_code_links":2,"syntology":{"ran":10,"of":12,"n_ran_checked":10,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["gucci-j/lowres-cva","gucci-j/lowres-cve"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/gigaspeech-2-an-evolving-large-scale-and","slug":"gigaspeech-2-an-evolving-large-scale-and","title":"GigaSpeech 2: An Evolving, Large-Scale and Multi-domain ASR Corpus for Low-Resource Languages with Automated Crawling, Transcription and Refinement","date":"2024-06-17","arxiv_id":"2406.11546","n_code_links":2,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["SpeechColab/GigaSpeech2"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/della-merging-reducing-interference-in-model","slug":"della-merging-reducing-interference-in-model","title":"DELLA-Merging: Reducing Interference in Model Merging through Magnitude-Based Sampling","date":"2024-06-17","arxiv_id":"2406.11617","n_code_links":1,"syntology":{"ran":12,"of":15,"n_ran_checked":12,"n_instrument":0,"unverified":3,"pointer_only":15,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["declare-lab/della"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/unveiling-encoder-free-vision-language-models","slug":"unveiling-encoder-free-vision-language-models","title":"Unveiling Encoder-Free Vision-Language Models","date":"2024-06-17","arxiv_id":"2406.11832","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["baaivision/eve"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["unlocated"]}}},{"paper":"/paper/medcalc-bench-evaluating-large-language","slug":"medcalc-bench-evaluating-large-language","title":"MedCalc-Bench: Evaluating Large Language Models for Medical Calculations","date":"2024-06-17","arxiv_id":"2406.12036","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ncbi-nlp/medcalc-bench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/scar-efficient-instruction-tuning-for-large","slug":"scar-efficient-instruction-tuning-for-large","title":"SCAR: Efficient Instruction-Tuning for Large Language Models via Style Consistency-Aware Response Ranking","date":"2024-06-16","arxiv_id":"2406.10882","n_code_links":1,"syntology":{"ran":5,"of":8,"n_ran_checked":2,"n_instrument":3,"unverified":3,"pointer_only":8,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 3 unverified","official":{"repos":["zhuang-li/scar"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/rwku-benchmarking-real-world-knowledge","slug":"rwku-benchmarking-real-world-knowledge","title":"RWKU: Benchmarking Real-World Knowledge Unlearning for Large Language Models","date":"2024-06-16","arxiv_id":"2406.10890","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":0,"n_instrument":1,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["jinzhuoran/rwku"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/haichart-human-and-ai-paired-visualization","slug":"haichart-human-and-ai-paired-visualization","title":"HAIChart: Human and AI Paired Visualization System","date":"2024-06-16","arxiv_id":"2406.11033","n_code_links":2,"syntology":{"ran":7,"of":9,"n_ran_checked":7,"n_instrument":0,"unverified":2,"pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["xypkent/haichart","hkustdial/haichart"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/active-anytime-valid-risk-controlling","slug":"active-anytime-valid-risk-controlling","title":"Active, anytime-valid risk controlling prediction sets","date":"2024-06-15","arxiv_id":"2406.10490","n_code_links":1,"syntology":{"ran":2,"of":6,"n_ran_checked":1,"n_instrument":1,"unverified":4,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["neilzxu/active-rcps"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/seeing-clearly-answering-incorrectly-a","slug":"seeing-clearly-answering-incorrectly-a","title":"Unveiling the Ignorance of MLLMs: Seeing Clearly, Answering Incorrectly","date":"2024-06-15","arxiv_id":"2406.10638","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":5,"n_instrument":3,"unverified":1,"pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","official":{"repos":["baai-dcai/multimodal-robustness-benchmark"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/evaluating-the-generalization-ability-of-1","slug":"evaluating-the-generalization-ability-of-1","title":"Evaluating the Generalization Ability of Quantized LLMs: Benchmark, Analysis, and Toolbox","date":"2024-06-15","arxiv_id":"2406.12928","n_code_links":1,"syntology":{"ran":4,"of":12,"n_ran_checked":4,"n_instrument":0,"unverified":8,"pointer_only":12,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 8 unverified","official":{"repos":["tsingmaoai/mi-optimize"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":8,"ran_from_kinds":["official"]}}},{"paper":"/paper/towards-scalable-and-versatile-weight-space","slug":"towards-scalable-and-versatile-weight-space","title":"Towards Scalable and Versatile Weight Space Learning","date":"2024-06-14","arxiv_id":"2406.09997","n_code_links":1,"syntology":{"ran":11,"of":15,"n_ran_checked":11,"n_instrument":0,"unverified":4,"pointer_only":15,"phrase":"11 ran (of which 7 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["hsg-aiml/sane"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":7,"n_ran_no_instrument_failure":11,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/protos-vit-visual-foundation-models-for","slug":"protos-vit-visual-foundation-models-for","title":"ProtoS-ViT: Visual foundation models for sparse self-explainable classifications","date":"2024-06-14","arxiv_id":"2406.10025","n_code_links":1,"syntology":{"ran":9,"of":13,"n_ran_checked":6,"n_instrument":3,"unverified":4,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","official":{"repos":["hturbe/protosvit"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/task-aligned-part-aware-panoptic-segmentation-1","slug":"task-aligned-part-aware-panoptic-segmentation-1","title":"Task-aligned Part-aware Panoptic Segmentation through Joint Object-Part Representations","date":"2024-06-14","arxiv_id":"2406.10114","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["tue-mps/tapps"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/babilong-testing-the-limits-of-llms-with-long","slug":"babilong-testing-the-limits-of-llms-with-long","title":"BABILong: Testing the Limits of LLMs with Long Context Reasoning-in-a-Haystack","date":"2024-06-14","arxiv_id":"2406.10149","n_code_links":4,"syntology":{"ran":7,"of":8,"n_ran_checked":7,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["booydar/babilong"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/devbench-a-multimodal-developmental-benchmark","slug":"devbench-a-multimodal-developmental-benchmark","title":"DevBench: A multimodal developmental benchmark for language learning","date":"2024-06-14","arxiv_id":"2406.10215","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":4,"n_instrument":1,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["alvinwmtan/dev-bench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/pareto-front-diverse-batch-multi-objective","slug":"pareto-front-diverse-batch-multi-objective","title":"Pareto Front-Diverse Batch Multi-Objective Bayesian Optimization","date":"2024-06-13","arxiv_id":"2406.08799","n_code_links":1,"syntology":{"ran":6,"of":13,"n_ran_checked":6,"n_instrument":0,"unverified":7,"pointer_only":13,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","official":{"repos":["alaleh/pdbo"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":"/paper/potion-towards-poison-unlearning","slug":"potion-towards-poison-unlearning","title":"Potion: Towards Poison Unlearning","date":"2024-06-13","arxiv_id":"2406.09173","n_code_links":1,"syntology":{"ran":7,"of":10,"n_ran_checked":7,"n_instrument":0,"unverified":3,"pointer_only":10,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["if-loops/towards_poison_unlearning"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/jailbreakeval-an-integrated-toolkit-for","slug":"jailbreakeval-an-integrated-toolkit-for","title":"JailbreakEval: An Integrated Toolkit for Evaluating Jailbreak Attempts Against Large Language Models","date":"2024-06-13","arxiv_id":"2406.09321","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thuccslab/jailbreakeval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/yo-llava-your-personalized-language-and","slug":"yo-llava-your-personalized-language-and","title":"Yo'LLaVA: Your Personalized Language and Vision Assistant","date":"2024-06-13","arxiv_id":"2406.09400","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":1,"n_instrument":3,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["WisconsinAIVision/YoLLaVA"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/interpreting-the-weight-space-of-customized","slug":"interpreting-the-weight-space-of-customized","title":"Interpreting the Weight Space of Customized Diffusion Models","date":"2024-06-13","arxiv_id":"2406.09413","n_code_links":1,"syntology":{"ran":4,"of":8,"n_ran_checked":4,"n_instrument":0,"unverified":4,"pointer_only":8,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["snap-research/weights2weights"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/videogpt-integrating-image-and-video-encoders","slug":"videogpt-integrating-image-and-video-encoders","title":"VideoGPT+: Integrating Image and Video Encoders for Enhanced Video Understanding","date":"2024-06-13","arxiv_id":"2406.09418","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["mbzuai-oryx/videogpt-plus"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/cleandiffuser-an-easy-to-use-modularized","slug":"cleandiffuser-an-easy-to-use-modularized","title":"CleanDiffuser: An Easy-to-use Modularized Library for Diffusion Models in Decision Making","date":"2024-06-13","arxiv_id":"2406.09509","n_code_links":3,"syntology":{"ran":8,"of":9,"n_ran_checked":6,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["cleandiffuserteam/cleandiffuser"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/drivaernet-a-large-scale-multimodal-car","slug":"drivaernet-a-large-scale-multimodal-car","title":"DrivAerNet++: A Large-Scale Multimodal Car Dataset with Computational Fluid Dynamics Simulations and Deep Learning Benchmarks","date":"2024-06-13","arxiv_id":"2406.09624","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["mohamedelrefaie/drivaernet"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/understanding-and-mitigating-compositional","slug":"understanding-and-mitigating-compositional","title":"Understanding and Mitigating Compositional Issues in Text-to-Image Generative Models","date":"2024-06-12","arxiv_id":"2406.07844","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ArmanZarei/Mitigating-T2I-Comp-Issues"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/lvbench-an-extreme-long-video-understanding","slug":"lvbench-an-extreme-long-video-understanding","title":"LVBench: An Extreme Long Video Understanding Benchmark","date":"2024-06-12","arxiv_id":"2406.08035","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["THUDM/LVBench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/rrls-robust-reinforcement-learning-suite","slug":"rrls-robust-reinforcement-learning-suite","title":"RRLS : Robust Reinforcement Learning Suite","date":"2024-06-12","arxiv_id":"2406.08406","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sureli/rrls"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/fine-tuned-small-llms-still-significantly","slug":"fine-tuned-small-llms-still-significantly","title":"Fine-Tuned 'Small' LLMs (Still) Significantly Outperform Zero-Shot Generative AI Models in Text Classification","date":"2024-06-12","arxiv_id":"2406.08660","n_code_links":1,"syntology":{"ran":5,"of":9,"n_ran_checked":5,"n_instrument":0,"unverified":4,"pointer_only":9,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["mnbucher/text-cls-llms"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/reconciling-kaplan-and-chinchilla-scaling","slug":"reconciling-kaplan-and-chinchilla-scaling","title":"Reconciling Kaplan and Chinchilla Scaling Laws","date":"2024-06-12","arxiv_id":"2406.12907","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["teapearce/reconciling_kaplan_chinchilla_scaling_laws"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-probabilistic-framework-for-llm","slug":"a-probabilistic-framework-for-llm","title":"A Probabilistic Framework for LLM Hallucination Detection via Belief Tree Propagation","date":"2024-06-11","arxiv_id":"2406.06950","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ucsb-nlp-chang/btprop"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/videollama-2-advancing-spatial-temporal","slug":"videollama-2-advancing-spatial-temporal","title":"VideoLLaMA 2: Advancing Spatial-Temporal Modeling and Audio Understanding in Video-LLMs","date":"2024-06-11","arxiv_id":"2406.07476","n_code_links":3,"syntology":{"ran":12,"of":17,"n_ran_checked":7,"n_instrument":5,"unverified":5,"pointer_only":8,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 1 violated, 6 with no contract checked; 5 where Syntology's instrument failed) · 5 unverified","official":{"repos":["damo-nlp-sg/videollama2"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/map-low-compute-model-merging-with-amortized","slug":"map-low-compute-model-merging-with-amortized","title":"MAP: Low-compute Model Merging with Amortized Pareto Fronts via Quadratic Approximation","date":"2024-06-11","arxiv_id":"2406.07529","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["luli-git/MAP"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/hearing-anything-anywhere","slug":"hearing-anything-anywhere","title":"Hearing Anything Anywhere","date":"2024-06-11","arxiv_id":"2406.07532","n_code_links":1,"syntology":{"ran":9,"of":12,"n_ran_checked":9,"n_instrument":0,"unverified":3,"pointer_only":12,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["maswang32/hearinganythinganywhere"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/open-llm-leaderboard-from-multi-choice-to","slug":"open-llm-leaderboard-from-multi-choice-to","title":"Open-LLM-Leaderboard: From Multi-choice to Open-style Questions for LLMs Evaluation, Benchmark, and Arena","date":"2024-06-11","arxiv_id":"2406.07545","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["vila-lab/open-llm-leaderboard"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/chain-of-scrutiny-detecting-backdoor-attacks","slug":"chain-of-scrutiny-detecting-backdoor-attacks","title":"Chain-of-Scrutiny: Detecting Backdoor Attacks for Large Language Models","date":"2024-06-10","arxiv_id":"2406.05948","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["lixi1994/CoS"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/get-rich-quick-exact-solutions-reveal-how","slug":"get-rich-quick-exact-solutions-reveal-how","title":"Get rich quick: exact solutions reveal how unbalanced initializations promote rapid feature learning","date":"2024-06-10","arxiv_id":"2406.06158","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["allanraventos/getrichquick"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/ears-an-anechoic-fullband-speech-dataset","slug":"ears-an-anechoic-fullband-speech-dataset","title":"EARS: An Anechoic Fullband Speech Dataset Benchmarked for Speech Enhancement and Dereverberation","date":"2024-06-10","arxiv_id":"2406.06185","n_code_links":2,"syntology":{"ran":8,"of":8,"n_ran_checked":8,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["facebookresearch/ears_dataset"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/self-tuning-instructing-llms-to-effectively","slug":"self-tuning-instructing-llms-to-effectively","title":"Self-Tuning: Instructing LLMs to Effectively Acquire New Knowledge through Self-Teaching","date":"2024-06-10","arxiv_id":"2406.06326","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":2,"n_instrument":1,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zhangxy-2019/Effective-Knowledge-Injection"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/husky-a-unified-open-source-language-agent","slug":"husky-a-unified-open-source-language-agent","title":"Husky: A Unified, Open-Source Language Agent for Multi-Step Reasoning","date":"2024-06-10","arxiv_id":"2406.06469","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":3,"n_instrument":3,"unverified":0,"pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 1 honoured, 2 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["agent-husky/husky-v1"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/when-is-multicalibration-post-processing","slug":"when-is-multicalibration-post-processing","title":"When is Multicalibration Post-Processing Necessary?","date":"2024-06-10","arxiv_id":"2406.06487","n_code_links":3,"syntology":{"ran":8,"of":8,"n_ran_checked":8,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["dutchhansen/multicalibration","dutchhansen/empirical-multicalibration","sid-devic/multicalibration"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/merlin-a-vision-language-foundation-model-for","slug":"merlin-a-vision-language-foundation-model-for","title":"Merlin: A Vision Language Foundation Model for 3D Computed Tomography","date":"2024-06-10","arxiv_id":"2406.06512","n_code_links":2,"syntology":{"ran":8,"of":9,"n_ran_checked":7,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/controlling-counterfactual-harm-in-decision","slug":"controlling-counterfactual-harm-in-decision","title":"Controlling Counterfactual Harm in Decision Support Systems Based on Prediction Sets","date":"2024-06-10","arxiv_id":"2406.06671","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":7,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["Networks-Learning/controlling-counterfactual-harm-prediction-sets"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/stable-neighbor-denoising-for-source-free","slug":"stable-neighbor-denoising-for-source-free","title":"Stable Neighbor Denoising for Source-free Domain Adaptive Segmentation","date":"2024-06-10","arxiv_id":"2406.06813","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["DZhaoXd/SND"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/conformal-prediction-for-class-wise-coverage","slug":"conformal-prediction-for-class-wise-coverage","title":"Conformal Prediction for Class-wise Coverage via Augmented Label Rank Calibration","date":"2024-06-10","arxiv_id":"2406.06818","n_code_links":1,"syntology":{"ran":21,"of":24,"n_ran_checked":20,"n_instrument":1,"unverified":3,"pointer_only":1,"phrase":"21 ran (of which 0 constructed an object rather than computing a result; 20 with no instrument failure: 0 honoured, 0 violated, 20 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["yuanjiesh/rc3p"],"state":"official (archive's flag): 21 ran","n_ran":21,"n_constructed":0,"n_ran_no_instrument_failure":20,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/an-llm-assisted-easy-to-trigger-backdoor","slug":"an-llm-assisted-easy-to-trigger-backdoor","title":"An LLM-Assisted Easy-to-Trigger Backdoor Attack on Code Completion Models: Injecting Disguised Vulnerabilities against Strong Detection","date":"2024-06-10","arxiv_id":"2406.06822","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["datasec-lab/codebreaker"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/what-can-we-learn-from-state-space-models-for","slug":"what-can-we-learn-from-state-space-models-for","title":"What Can We Learn from State Space Models for Machine Learning on Graphs?","date":"2024-06-09","arxiv_id":"2406.05815","n_code_links":1,"syntology":{"ran":10,"of":11,"n_ran_checked":6,"n_instrument":4,"unverified":1,"pointer_only":11,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["graph-com/gssc"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/winner-takes-all-learners-are-geometry-aware","slug":"winner-takes-all-learners-are-geometry-aware","title":"Winner-takes-all learners are geometry-aware conditional density estimators","date":"2024-06-07","arxiv_id":"2406.04706","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":4,"n_instrument":1,"unverified":0,"pointer_only":5,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Victorletzelter/VoronoiWTA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/probabilistic-perspectives-on-error","slug":"probabilistic-perspectives-on-error","title":"Probabilistic Perspectives on Error Minimization in Adversarial Reinforcement Learning","date":"2024-06-07","arxiv_id":"2406.04724","n_code_links":1,"syntology":{"ran":10,"of":12,"n_ran_checked":7,"n_instrument":3,"unverified":2,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["romanbelaire/acoe-robust-rl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"paper":"/paper/massively-multiagent-minigames-for-training","slug":"massively-multiagent-minigames-for-training","title":"Massively Multiagent Minigames for Training Generalist Agents","date":"2024-06-07","arxiv_id":"2406.05071","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["kywch/meta-mmo"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/split-and-fit-learning-b-reps-via-structure","slug":"split-and-fit-learning-b-reps-via-structure","title":"Split-and-Fit: Learning B-Reps via Structure-Aware Voronoi Partitioning","date":"2024-06-07","arxiv_id":"2406.05261","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":8,"n_instrument":0,"unverified":1,"pointer_only":9,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["yilinliu77/nvdnet"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/quality-diversity-with-limited-resources","slug":"quality-diversity-with-limited-resources","title":"Quality-Diversity with Limited Resources","date":"2024-06-06","arxiv_id":"2406.03731","n_code_links":1,"syntology":{"ran":3,"of":6,"n_ran_checked":3,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["lamda-bbo/refqd"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/rest-mcts-llm-self-training-via-process","slug":"rest-mcts-llm-self-training-via-process","title":"ReST-MCTS*: LLM Self-Training via Process Reward Guided Tree Search","date":"2024-06-06","arxiv_id":"2406.03816","n_code_links":2,"syntology":{"ran":9,"of":22,"n_ran_checked":2,"n_instrument":7,"unverified":13,"pointer_only":13,"phrase":"9 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 7 where Syntology's instrument failed) · 13 unverified","official":{"repos":["THUDM/ReST-MCTS"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":7,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/vectorized-conditional-neural-fields-a","slug":"vectorized-conditional-neural-fields-a","title":"Vectorized Conditional Neural Fields: A Framework for Solving Time-dependent Parametric Partial Differential Equations","date":"2024-06-06","arxiv_id":"2406.03919","n_code_links":1,"syntology":{"ran":9,"of":14,"n_ran_checked":9,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"9 ran (of which 9 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified; every one of the 9 samples that ran constructed an object rather than computing a result","official":{"repos":["jhagnberger/vcnef"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":9,"n_ran_no_instrument_failure":9,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/what-is-dataset-distillation-learning","slug":"what-is-dataset-distillation-learning","title":"What is Dataset Distillation Learning?","date":"2024-06-06","arxiv_id":"2406.04284","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":8,"n_instrument":0,"unverified":0,"pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["princetonvisualai/What-is-Dataset-Distillation-Learning"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}}],"record_sha256":"27a6f575dcdd9c958756ee7fdb1f501b01ba5f635ce3b2e2f4d4eebbdc1486d5","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}