{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt/papers/ran/1","list_of":"/method/gpt","method":"GPT","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not isolate this method inside it.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":1,"pages_in_order":2,"rows_per_page":100,"rows":[1,100],"of":152,"counts":{"archive_papers_tagged":1212,"with_a_code_link":453,"where_syntology_ran_a_sample":152,"not_listed_spam_title":0,"listed":1212,"listed_where_code_ran":152,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":130,"every_run_a_failure_of_syntologys_instrument":22,"listed_with_a_run_with_no_instrument_failure":130,"listed_every_run_a_failure_of_syntologys_instrument":22,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt/papers/ran/1","prev":null,"next":"/method/gpt/papers/ran/2","papers":[{"paper":"/paper/evaluating-llms-across-multi-cognitive-levels","slug":"evaluating-llms-across-multi-cognitive-levels","title":"Evaluating LLMs Across Multi-Cognitive Levels: From Medical Knowledge Mastery to Scenario-Based Problem Solving","date":"2025-06-10","arxiv_id":"2506.08349","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thumlp/multicogeval"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/combining-causal-models-for-more-accurate","slug":"combining-causal-models-for-more-accurate","title":"Combining Causal Models for More Accurate Abstractions of Neural Networks","date":"2025-03-14","arxiv_id":"2503.11429","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["marapislar/combining-causal-models-for-accurate-nn-abstractions"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/bp-gpt-auditory-neural-decoding-using-fmri","slug":"bp-gpt-auditory-neural-decoding-using-fmri","title":"BP-GPT: Auditory Neural Decoding Using fMRI-prompted LLM","date":"2025-02-21","arxiv_id":"2502.15172","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["1994cxy/bp-gpt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/wildchat-50m-a-deep-dive-into-the-role-of","slug":"wildchat-50m-a-deep-dive-into-the-role-of","title":"WILDCHAT-50M: A Deep Dive Into the Role of Synthetic Data in Post-Training","date":"2025-01-30","arxiv_id":"2501.18511","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":6,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["penfever/wildchat-50m"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/drivingworld-constructingworld-model-for","slug":"drivingworld-constructingworld-model-for","title":"DrivingWorld: Constructing World Model for Autonomous Driving via Video GPT","date":"2024-12-27","arxiv_id":"2412.19505","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":8,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 1 violated, 7 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yvanyin/drivingworld"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/causal-diffusion-transformers-for-generative","slug":"causal-diffusion-transformers-for-generative","title":"Causal Diffusion Transformers for Generative Modeling","date":"2024-12-16","arxiv_id":"2412.12095","n_code_links":1,"syntology":{"ran":8,"of":10,"n_ran_checked":6,"n_instrument":2,"unverified":2,"pointer_only":10,"phrase":"8 ran (of which 3 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["causalfusion/causalfusion"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":3,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/characterbox-evaluating-the-role-playing","slug":"characterbox-evaluating-the-role-playing","title":"CharacterBox: Evaluating the Role-Playing Capabilities of LLMs in Text-Based Virtual Worlds","date":"2024-12-07","arxiv_id":"2412.05631","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["paitesanshi/characterbox"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/marketgpt-developing-a-pre-trained","slug":"marketgpt-developing-a-pre-trained","title":"MarketGPT: Developing a Pre-trained transformer (GPT) for Modeling Financial Time Series","date":"2024-11-25","arxiv_id":"2411.16585","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["aaron-wheeler/marketgpt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/evaluating-the-robustness-of-analogical","slug":"evaluating-the-robustness-of-analogical","title":"Evaluating the Robustness of Analogical Reasoning in Large Language Models","date":"2024-11-21","arxiv_id":"2411.14215","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["marthaflinderslewis/robust-analogy"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/little-giants-synthesizing-high-quality","slug":"little-giants-synthesizing-high-quality","title":"Little Giants: Synthesizing High-Quality Embedding Data at Scale","date":"2024-10-24","arxiv_id":"2410.18634","n_code_links":1,"syntology":{"ran":11,"of":11,"n_ran_checked":11,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["haon-chen/SPEED"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/stabilize-the-latent-space-for-image","slug":"stabilize-the-latent-space-for-image","title":"Stabilize the Latent Space for Image Autoregressive Modeling: A Unified Perspective","date":"2024-10-16","arxiv_id":"2410.12490","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["DAMO-NLP-SG/DiGIT"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/deciphering-the-chaos-enhancing-jailbreak","slug":"deciphering-the-chaos-enhancing-jailbreak","title":"Deciphering the Chaos: Enhancing Jailbreak Attacks via Adversarial Prompt Translation","date":"2024-10-15","arxiv_id":"2410.11317","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["qizhangli/adversarial-prompt-translator"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/towards-better-multi-head-attention-via","slug":"towards-better-multi-head-attention-via","title":"Towards Better Multi-head Attention via Channel-wise Sample Permutation","date":"2024-10-14","arxiv_id":"2410.10914","n_code_links":1,"syntology":{"ran":4,"of":7,"n_ran_checked":4,"n_instrument":0,"unverified":3,"pointer_only":7,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["dashenzi721/csp"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/one-language-many-gaps-evaluating-dialect","slug":"one-language-many-gaps-evaluating-dialect","title":"One Language, Many Gaps: Evaluating Dialect Fairness and Robustness of Large Language Models in Reasoning Tasks","date":"2024-10-14","arxiv_id":"2410.11005","n_code_links":1,"syntology":{"ran":4,"of":7,"n_ran_checked":4,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["fangru-lin/redial_dialect_robustness_fairness"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/famma-a-benchmark-for-financial-domain","slug":"famma-a-benchmark-for-financial-domain","title":"FAMMA: A Benchmark for Financial Domain Multilingual Multimodal Question Answering","date":"2024-10-06","arxiv_id":"2410.04526","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["famma-bench/bench-script"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/seeing-eye-to-ai-human-alignment-via-gaze","slug":"seeing-eye-to-ai-human-alignment-via-gaze","title":"Seeing Eye to AI: Human Alignment via Gaze-Based Response Rewards for Large Language Models","date":"2024-10-02","arxiv_id":"2410.01532","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["telefonica-scientific-research/gaze_reward"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/quantifying-generalization-complexity-for","slug":"quantifying-generalization-complexity-for","title":"Quantifying Generalization Complexity for Large Language Models","date":"2024-10-02","arxiv_id":"2410.01769","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":4,"n_instrument":0,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zhentingqi/scylla"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/cottention-linear-transformers-with-cosine","slug":"cottention-linear-transformers-with-cosine","title":"Cottention: Linear Transformers With Cosine Attention","date":"2024-09-27","arxiv_id":"2409.18747","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["gmongaras/Cottention_Transformer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/2408-02946","slug":"2408-02946","title":"Data Poisoning in LLMs: Jailbreak-Tuning and Scaling Laws","date":"2024-08-06","arxiv_id":"2408.02946","n_code_links":2,"syntology":{"ran":1,"of":5,"n_ran_checked":1,"n_instrument":0,"unverified":4,"pointer_only":5,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["alignmentresearch/scaling-poisoning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/is-larger-always-better-evaluating-and","slug":"is-larger-always-better-evaluating-and","title":"ClinicRealm: Re-evaluating Large Language Models with Conventional Machine Learning for Non-Generative Clinical Prediction Tasks","date":"2024-07-26","arxiv_id":"2407.18525","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":6,"n_instrument":2,"unverified":0,"pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yhzhu99/ehr-llm-benchmark"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/adaptive-foundation-models-for-online","slug":"adaptive-foundation-models-for-online","title":"Scalable Exploration via Ensemble++","date":"2024-07-18","arxiv_id":"2407.13195","n_code_links":2,"syntology":{"ran":5,"of":8,"n_ran_checked":3,"n_instrument":2,"unverified":3,"pointer_only":8,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":{"repos":["szrlee/GPT-HyperAgent","szrlee/ensemble_plus_plus"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/werewolf-arena-a-case-study-in-llm-evaluation","slug":"werewolf-arena-a-case-study-in-llm-evaluation","title":"Werewolf Arena: A Case Study in LLM Evaluation via Social Deduction","date":"2024-07-18","arxiv_id":"2407.13943","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google/werewolf_arena"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/lami-detr-open-vocabulary-detection-with","slug":"lami-detr-open-vocabulary-detection-with","title":"LaMI-DETR: Open-Vocabulary Detection with Language Model Instruction","date":"2024-07-16","arxiv_id":"2407.11335","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["eternaldolphin/lami-detr"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/metallm-a-high-performant-and-cost-efficient","slug":"metallm-a-high-performant-and-cost-efficient","title":"MetaLLM: A High-performant and Cost-efficient Dynamic Framework for Wrapping LLMs","date":"2024-07-15","arxiv_id":"2407.10834","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["mail-research/metallm-wrapper"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/automatic-gradient-descent-with-generalized","slug":"automatic-gradient-descent-with-generalized","title":"Gradient descent with generalized Newton's method","date":"2024-07-03","arxiv_id":"2407.02772","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["shiyunxu/autogen"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/mathodyssey-benchmarking-mathematical-problem","slug":"mathodyssey-benchmarking-mathematical-problem","title":"MathOdyssey: Benchmarking Mathematical Problem-Solving Skills in Large Language Models Using Odyssey Math Data","date":"2024-06-26","arxiv_id":"2406.18321","n_code_links":3,"syntology":{"ran":5,"of":5,"n_ran_checked":4,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/dreambench-a-human-aligned-benchmark-for","slug":"dreambench-a-human-aligned-benchmark-for","title":"DreamBench++: A Human-Aligned Benchmark for Personalized Image Generation","date":"2024-06-24","arxiv_id":"2406.16855","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yuangpeng/dreambench_plus"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/how-to-compute-the-probability-of-a-word","slug":"how-to-compute-the-probability-of-a-word","title":"How to Compute the Probability of a Word","date":"2024-06-20","arxiv_id":"2406.14561","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["tpimentelms/probability-of-a-word"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/scaling-the-codebook-size-of-vqgan-to-100000","slug":"scaling-the-codebook-size-of-vqgan-to-100000","title":"Scaling the Codebook Size of VQGAN to 100,000 with a Utilization Rate of 99%","date":"2024-06-17","arxiv_id":"2406.11837","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["zh460045050/vqgan-lc"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/domainrag-a-chinese-benchmark-for-evaluating","slug":"domainrag-a-chinese-benchmark-for-evaluating","title":"DomainRAG: A Chinese Benchmark for Evaluating Domain-specific Retrieval-Augmented Generation","date":"2024-06-09","arxiv_id":"2406.05654","n_code_links":2,"syntology":{"ran":12,"of":12,"n_ran_checked":11,"n_instrument":1,"unverified":0,"pointer_only":12,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ShootingWong/DomainRAG"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/berts-are-generative-in-context-learners","slug":"berts-are-generative-in-context-learners","title":"BERTs are Generative In-Context Learners","date":"2024-06-07","arxiv_id":"2406.04823","n_code_links":1,"syntology":{"ran":13,"of":26,"n_ran_checked":12,"n_instrument":1,"unverified":13,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 1 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 13 unverified","official":{"repos":["ltgoslo/bert-in-context"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":13,"ran_from_kinds":["official"]}}},{"paper":"/paper/on-subjective-uncertainty-quantification-and","slug":"on-subjective-uncertainty-quantification-and","title":"On Subjective Uncertainty Quantification and Calibration in Natural Language Generation","date":"2024-06-07","arxiv_id":"2406.05213","n_code_links":1,"syntology":{"ran":12,"of":14,"n_ran_checked":12,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["meta-inf/suq-nlg"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/randomized-geometric-algebra-methods-for","slug":"randomized-geometric-algebra-methods-for","title":"Randomized Geometric Algebra Methods for Convex Neural Networks","date":"2024-06-04","arxiv_id":"2406.02806","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["pilancilab/Randomized-Geometric-Algebra-Methods-for-Convex-Neural-Networks"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/llamea-a-large-language-model-evolutionary","slug":"llamea-a-large-language-model-evolutionary","title":"LLaMEA: A Large Language Model Evolutionary Algorithm for Automatically Generating Metaheuristics","date":"2024-05-30","arxiv_id":"2405.20132","n_code_links":2,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["nikivanstein/LLaMEA"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/map-neo-highly-capable-and-transparent","slug":"map-neo-highly-capable-and-transparent","title":"MAP-Neo: Highly Capable and Transparent Bilingual Large Language Model Series","date":"2024-05-29","arxiv_id":"2405.19327","n_code_links":1,"syntology":{"ran":12,"of":14,"n_ran_checked":12,"n_instrument":0,"unverified":2,"pointer_only":14,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["multimodal-art-projection/map-neo"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/accelerating-transformers-with-spectrum-1","slug":"accelerating-transformers-with-spectrum-1","title":"Accelerating Transformers with Spectrum-Preserving Token Merging","date":"2024-05-25","arxiv_id":"2405.16148","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":2,"n_instrument":4,"unverified":1,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hchautran/PiToMe"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/learning-the-language-of-protein-structure","slug":"learning-the-language-of-protein-structure","title":"Learning the Language of Protein Structure","date":"2024-05-24","arxiv_id":"2405.15840","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["instadeepai/protein-structure-tokenizer"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/wise-rethinking-the-knowledge-memory-for","slug":"wise-rethinking-the-knowledge-memory-for","title":"WISE: Rethinking the Knowledge Memory for Lifelong Model Editing of Large Language Models","date":"2024-05-23","arxiv_id":"2405.14768","n_code_links":1,"syntology":{"ran":13,"of":16,"n_ran_checked":4,"n_instrument":9,"unverified":3,"pointer_only":0,"phrase":"13 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 0 violated, 3 with no contract checked; 9 where Syntology's instrument failed) · 3 unverified","official":{"repos":["zjunlp/easyedit"],"state":"official (archive's flag): 13 ran","n_ran":13,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/your-transformer-is-secretly-linear","slug":"your-transformer-is-secretly-linear","title":"Your Transformer is Secretly Linear","date":"2024-05-19","arxiv_id":"2405.12250","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["AIRI-Institute/LLM-Microscope"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/language-models-can-exploit-cross-task-in","slug":"language-models-can-exploit-cross-task-in","title":"Language Models can Exploit Cross-Task In-context Learning for Data-Scarce Novel Tasks","date":"2024-05-17","arxiv_id":"2405.10548","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":7,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["c-anwoy/cross-task-icl"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/freeva-offline-mllm-as-training-free-video","slug":"freeva-offline-mllm-as-training-free-video","title":"FreeVA: Offline MLLM as Training-Free Video Assistant","date":"2024-05-13","arxiv_id":"2405.07798","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":4,"n_instrument":4,"unverified":1,"pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","official":{"repos":["whwu95/freeva"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/on-training-data-influence-of-gpt-models","slug":"on-training-data-influence-of-gpt-models","title":"On Training Data Influence of GPT Models","date":"2024-04-11","arxiv_id":"2404.07840","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["eleutherai/pythia","ernie-research/gptfluence"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/exploring-the-potential-of-large-foundation","slug":"exploring-the-potential-of-large-foundation","title":"Exploring the Potential of Large Foundation Models for Open-Vocabulary HOI Detection","date":"2024-04-09","arxiv_id":"2404.06194","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ltttpku/cmd-se-release"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/visual-autoregressive-modeling-scalable-image","slug":"visual-autoregressive-modeling-scalable-image","title":"Visual Autoregressive Modeling: Scalable Image Generation via Next-Scale Prediction","date":"2024-04-03","arxiv_id":"2404.02905","n_code_links":3,"syntology":{"ran":7,"of":11,"n_ran_checked":2,"n_instrument":5,"unverified":4,"pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 5 where Syntology's instrument failed) · 4 unverified","official":{"repos":["FoundationVision/VAR"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/long-form-factuality-in-large-language-models","slug":"long-form-factuality-in-large-language-models","title":"Long-form factuality in large language models","date":"2024-03-27","arxiv_id":"2403.18802","n_code_links":3,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["google-deepmind/long-form-factuality"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/don-t-trust-verify-grounding-llm-quantitative","slug":"don-t-trust-verify-grounding-llm-quantitative","title":"Don't Trust: Verify -- Grounding LLM Quantitative Reasoning with Autoformalization","date":"2024-03-26","arxiv_id":"2403.18120","n_code_links":1,"syntology":{"ran":10,"of":12,"n_ran_checked":2,"n_instrument":8,"unverified":2,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 8 where Syntology's instrument failed) · 2 unverified","official":{"repos":["jinpz/dtv"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/psalm-pixelwise-segmentation-with-large-multi","slug":"psalm-pixelwise-segmentation-with-large-multi","title":"PSALM: Pixelwise SegmentAtion with Large Multi-Modal Model","date":"2024-03-21","arxiv_id":"2403.14598","n_code_links":1,"syntology":{"ran":3,"of":7,"n_ran_checked":3,"n_instrument":0,"unverified":4,"pointer_only":2,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["zamling/psalm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/emergent-world-models-and-latent-variable","slug":"emergent-world-models-and-latent-variable","title":"Emergent World Models and Latent Variable Estimation in Chess-Playing Language Models","date":"2024-03-21","arxiv_id":"2403.15498","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["adamkarvonen/chess_llm_interpretability"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/meta-prompting-for-automating-zero-shot","slug":"meta-prompting-for-automating-zero-shot","title":"Meta-Prompting for Automating Zero-shot Visual Recognition with LLMs","date":"2024-03-18","arxiv_id":"2403.11755","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":2,"n_instrument":3,"unverified":2,"pointer_only":3,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","official":{"repos":["jmiemirza/meta-prompting"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/rectifying-demonstration-shortcut-in-in","slug":"rectifying-demonstration-shortcut-in-in","title":"Rectifying Demonstration Shortcut in In-Context Learning","date":"2024-03-14","arxiv_id":"2403.09488","n_code_links":1,"syntology":{"ran":3,"of":9,"n_ran_checked":1,"n_instrument":2,"unverified":6,"pointer_only":9,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","official":{"repos":["lainshower/in-context-calibration"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/gpt-generated-text-detection-benchmark","slug":"gpt-generated-text-detection-benchmark","title":"GPT-generated Text Detection: Benchmark Dataset and Tensor-based Detection Method","date":"2024-03-12","arxiv_id":"2403.07321","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["madlab-ucr/grid"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/jmi-at-semeval-2024-task-3-two-step-approach","slug":"jmi-at-semeval-2024-task-3-two-step-approach","title":"JMI at SemEval 2024 Task 3: Two-step approach for multimodal ECAC using in-context learning with GPT and instruction-tuned Llama models","date":"2024-03-05","arxiv_id":"2403.04798","n_code_links":1,"syntology":{"ran":5,"of":12,"n_ran_checked":5,"n_instrument":0,"unverified":7,"pointer_only":12,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified","official":{"repos":["cmooncs/semeval-2024_multimodal_ecpe"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":7,"ran_from_kinds":["official"]}}},{"paper":"/paper/sciassess-benchmarking-llm-proficiency-in","slug":"sciassess-benchmarking-llm-proficiency-in","title":"SciAssess: Benchmarking LLM Proficiency in Scientific Literature Analysis","date":"2024-03-04","arxiv_id":"2403.01976","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sci-assess/sciassess"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/gradient-free-adaptive-global-pruning-for-pre","slug":"gradient-free-adaptive-global-pruning-for-pre","title":"SparseLLM: Towards Global Pruning for Pre-trained Language Models","date":"2024-02-28","arxiv_id":"2402.17946","n_code_links":2,"syntology":{"ran":3,"of":9,"n_ran_checked":2,"n_instrument":1,"unverified":6,"pointer_only":3,"phrase":"3 ran (of which 1 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["baithebest/adagp","baithebest/sparsellm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":1,"n_ran_no_instrument_failure":2,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/clustering-and-ranking-diversity-preserved","slug":"clustering-and-ranking-diversity-preserved","title":"Clustering and Ranking: Diversity-preserved Instruction Selection through Expert-aligned Quality Estimation","date":"2024-02-28","arxiv_id":"2402.18191","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":6,"n_instrument":0,"unverified":3,"pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["ironbeliever/car"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/analobench-benchmarking-the-identification-of","slug":"analobench-benchmarking-the-identification-of","title":"AnaloBench: Benchmarking the Identification of Abstract and Long-context Analogies","date":"2024-02-19","arxiv_id":"2402.12370","n_code_links":2,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["jhu-clsp/analogical-reasoning","JHU-CLSP/AnaloBench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/a-graph-is-worth-k-words-euclideanizing-graph","slug":"a-graph-is-worth-k-words-euclideanizing-graph","title":"A Graph is Worth $K$ Words: Euclideanizing Graph using Pure Transformer","date":"2024-02-04","arxiv_id":"2402.02464","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["A4Bio/GraphsGPT"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/improving-sequential-recommendations-with","slug":"improving-sequential-recommendations-with","title":"Improving Sequential Recommendations with LLMs","date":"2024-02-02","arxiv_id":"2402.01339","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":1,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["dh-r/llm-sequential-recommendation"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/e-eval-a-comprehensive-chinese-k-12-education","slug":"e-eval-a-comprehensive-chinese-k-12-education","title":"E-EVAL: A Comprehensive Chinese K-12 Education Evaluation Benchmark for Large Language Models","date":"2024-01-29","arxiv_id":"2401.15927","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ai-edu-lab/e-eval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/trove-inducing-verifiable-and-efficient","slug":"trove-inducing-verifiable-and-efficient","title":"TroVE: Inducing Verifiable and Efficient Toolboxes for Solving Programmatic Tasks","date":"2024-01-23","arxiv_id":"2401.12869","n_code_links":1,"syntology":{"ran":16,"of":19,"n_ran_checked":16,"n_instrument":0,"unverified":3,"pointer_only":19,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["zorazrw/trove"],"state":"official (archive's flag): 16 ran","n_ran":16,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/enhancing-in-context-learning-via-linear","slug":"enhancing-in-context-learning-via-linear","title":"Enhancing In-context Learning via Linear Probe Calibration","date":"2024-01-22","arxiv_id":"2401.12406","n_code_links":1,"syntology":{"ran":3,"of":6,"n_ran_checked":3,"n_instrument":0,"unverified":3,"pointer_only":6,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["mominabbass/linc"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/code-prompting-elicits-conditional-reasoning","slug":"code-prompting-elicits-conditional-reasoning","title":"Code Prompting Elicits Conditional Reasoning Abilities in Text+Code LLMs","date":"2024-01-18","arxiv_id":"2401.10065","n_code_links":1,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["ukplab/arxiv2024-conditional-reasoning-llms"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/rotbench-a-multi-level-benchmark-for","slug":"rotbench-a-multi-level-benchmark-for","title":"RoTBench: A Multi-Level Benchmark for Evaluating the Robustness of Large Language Models in Tool Learning","date":"2024-01-16","arxiv_id":"2401.08326","n_code_links":1,"syntology":{"ran":10,"of":15,"n_ran_checked":10,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["junjie-ye/rotbench"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/de-novo-drug-design-using-reinforcement-1","slug":"de-novo-drug-design-using-reinforcement-1","title":"De novo Drug Design using Reinforcement Learning with Multiple GPT Agents","date":"2023-12-21","arxiv_id":"2401.06155","n_code_links":2,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":4,"phrase":"3 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hxyfighter/molrl-mgpt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/an-in-depth-look-at-gemini-s-language","slug":"an-in-depth-look-at-gemini-s-language","title":"An In-depth Look at Gemini's Language Abilities","date":"2023-12-18","arxiv_id":"2312.11444","n_code_links":1,"syntology":{"ran":11,"of":12,"n_ran_checked":11,"n_instrument":0,"unverified":1,"pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["neulab/gemini-benchmark"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/causality-analysis-for-evaluating-the","slug":"causality-analysis-for-evaluating-the","title":"Causality Analysis for Evaluating the Security of Large Language Models","date":"2023-12-13","arxiv_id":"2312.07876","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":6,"n_instrument":1,"unverified":2,"pointer_only":9,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["casperllm/casper"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/gift-generative-interpretable-fine-tuning","slug":"gift-generative-interpretable-fine-tuning","title":"Generative Parameter-Efficient Fine-Tuning","date":"2023-12-01","arxiv_id":"2312.00700","n_code_links":1,"syntology":{"ran":12,"of":16,"n_ran_checked":12,"n_instrument":0,"unverified":4,"pointer_only":10,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 12 with no instrument failure: 0 honoured, 0 violated, 12 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["savadikarc/gift"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":12,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/characterglm-customizing-chinese","slug":"characterglm-customizing-chinese","title":"CharacterGLM: Customizing Chinese Conversational AI Characters with Large Language Models","date":"2023-11-28","arxiv_id":"2311.16832","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["thu-coai/characterglm-6b"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/uhgeval-benchmarking-the-hallucination-of","slug":"uhgeval-benchmarking-the-hallucination-of","title":"UHGEval: Benchmarking the Hallucination of Chinese Large Language Models via Unconstrained Generation","date":"2023-11-26","arxiv_id":"2311.15296","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["IAAR-Shanghai/UHGEval"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/event-causality-is-key-to-computational-story","slug":"event-causality-is-key-to-computational-story","title":"Event Causality Is Key to Computational Story Understanding","date":"2023-11-16","arxiv_id":"2311.09648","n_code_links":1,"syntology":{"ran":6,"of":9,"n_ran_checked":6,"n_instrument":0,"unverified":3,"pointer_only":9,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["insundaycathy/event-causality-extraction"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/smart-agent-based-modeling-on-the-use-of","slug":"smart-agent-based-modeling-on-the-use-of","title":"Smart Agent-Based Modeling: On the Use of Large Language Models in Computer Simulations","date":"2023-11-10","arxiv_id":"2311.06330","n_code_links":4,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["roihn/sabm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/lumos-learning-agents-with-unified-data","slug":"lumos-learning-agents-with-unified-data","title":"Agent Lumos: Unified and Modular Training for Open-Source Language Agents","date":"2023-11-09","arxiv_id":"2311.05657","n_code_links":2,"syntology":{"ran":2,"of":3,"n_ran_checked":0,"n_instrument":2,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["allenai/lumos"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/massive-editing-for-large-language-models-via","slug":"massive-editing-for-large-language-models-via","title":"Massive Editing for Large Language Models via Meta Learning","date":"2023-11-08","arxiv_id":"2311.04661","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["chenmientan/malmen"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/neuro-gpt-developing-a-foundation-model-for","slug":"neuro-gpt-developing-a-foundation-model-for","title":"Neuro-GPT: Towards A Foundation Model for EEG","date":"2023-11-07","arxiv_id":"2311.03764","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["wenhui0206/neurogpt"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/synthetic-imitation-edit-feedback-for-factual","slug":"synthetic-imitation-edit-feedback-for-factual","title":"Synthetic Imitation Edit Feedback for Factual Alignment in Clinical Summarization","date":"2023-10-30","arxiv_id":"2310.20033","n_code_links":1,"syntology":{"ran":7,"of":9,"n_ran_checked":7,"n_instrument":0,"unverified":2,"pointer_only":9,"phrase":"7 ran (of which 3 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["seasonyao/learnfromhumanedit"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":3,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/lightlm-a-lightweight-deep-and-narrow","slug":"lightlm-a-lightweight-deep-and-narrow","title":"LightLM: A Lightweight Deep and Narrow Language Model for Generative Recommendation","date":"2023-10-26","arxiv_id":"2310.17488","n_code_links":1,"syntology":{"ran":11,"of":12,"n_ran_checked":11,"n_instrument":0,"unverified":1,"pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 1 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["dongyuanjushi/lightlm"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/gemba-mqm-detecting-translation-quality-error","slug":"gemba-mqm-detecting-translation-quality-error","title":"GEMBA-MQM: Detecting Translation Quality Error Spans with GPT-4","date":"2023-10-21","arxiv_id":"2310.13988","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":8,"n_instrument":0,"unverified":0,"pointer_only":8,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/the-shifted-and-the-overlooked-a-task","slug":"the-shifted-and-the-overlooked-a-task","title":"The Shifted and The Overlooked: A Task-oriented Investigation of User-GPT Interactions","date":"2023-10-19","arxiv_id":"2310.12418","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ozyyshr/sharegpt_investigation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/data-contamination-through-the-lens-of-time","slug":"data-contamination-through-the-lens-of-time","title":"Data Contamination Through the Lens of Time","date":"2023-10-16","arxiv_id":"2310.10628","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["abacusai/to-the-cutoff"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/towards-end-to-end-4-bit-inference-on","slug":"towards-end-to-end-4-bit-inference-on","title":"QUIK: Towards End-to-End 4-Bit Inference on Generative Large Language Models","date":"2023-10-13","arxiv_id":"2310.09259","n_code_links":1,"syntology":{"ran":9,"of":13,"n_ran_checked":9,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["ist-daslab/quik"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/lauragpt-listen-attend-understand-and","slug":"lauragpt-listen-attend-understand-and","title":"LauraGPT: Listen, Attend, Understand, and Regenerate Audio with GPT","date":"2023-10-07","arxiv_id":"2310.04673","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/memoria-hebbian-memory-architecture-for-human","slug":"memoria-hebbian-memory-architecture-for-human","title":"Memoria: Resolving Fateful Forgetting Problem through Human-Inspired Memory Architecture","date":"2023-10-04","arxiv_id":"2310.03052","n_code_links":1,"syntology":{"ran":1,"of":10,"n_ran_checked":0,"n_instrument":1,"unverified":9,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 9 unverified","official":{"repos":["cosmoquester/memoria"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":9,"ran_from_kinds":["official"]}}},{"paper":"/paper/analyzing-and-mitigating-object-hallucination","slug":"analyzing-and-mitigating-object-hallucination","title":"Analyzing and Mitigating Object Hallucination in Large Vision-Language Models","date":"2023-10-01","arxiv_id":"2310.00754","n_code_links":1,"syntology":{"ran":7,"of":8,"n_ran_checked":2,"n_instrument":5,"unverified":1,"pointer_only":8,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 1 violated, 1 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","official":{"repos":["yiyangzhou/lure"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/mindgpt-interpreting-what-you-see-with-non","slug":"mindgpt-interpreting-what-you-see-with-non","title":"MindGPT: Interpreting What You See with Non-invasive Brain Recordings","date":"2023-09-27","arxiv_id":"2309.15729","n_code_links":1,"syntology":{"ran":3,"of":7,"n_ran_checked":3,"n_instrument":0,"unverified":4,"pointer_only":7,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["jxuanc/mindgpt"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/chatgpt-mt-competitive-for-high-but-not-low","slug":"chatgpt-mt-competitive-for-high-but-not-low","title":"ChatGPT MT: Competitive for High- (but not Low-) Resource Languages","date":"2023-09-14","arxiv_id":"2309.07423","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["cmu-llab/gpt_mt_benchmark"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/mllm-dataengine-an-iterative-refinement","slug":"mllm-dataengine-an-iterative-refinement","title":"MLLM-DataEngine: An Iterative Refinement Approach for MLLM","date":"2023-08-25","arxiv_id":"2308.13566","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":4,"n_instrument":4,"unverified":0,"pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["opendatalab/mllm-dataengine"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/evaluating-large-language-models-on-graphs","slug":"evaluating-large-language-models-on-graphs","title":"Evaluating Large Language Models on Graphs: Performance Insights and Comparative Analysis","date":"2023-08-22","arxiv_id":"2308.11224","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ayame1006/llmtograph"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/large-language-models-to-identify-social","slug":"large-language-models-to-identify-social","title":"Large Language Models to Identify Social Determinants of Health in Electronic Health Records","date":"2023-08-11","arxiv_id":"2308.06354","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":0,"n_instrument":2,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["aim-harvard/sdoh"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/new-interaction-paradigm-for-complex-eda","slug":"new-interaction-paradigm-for-complex-eda","title":"New Interaction Paradigm for Complex EDA Software Leveraging GPT","date":"2023-07-27","arxiv_id":"2307.14740","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["smarton-empower/smarton-ai"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/voicebox-text-guided-multilingual-universal","slug":"voicebox-text-guided-multilingual-universal","title":"Voicebox: Text-Guided Multilingual Universal Speech Generation at Scale","date":"2023-06-23","arxiv_id":"2306.15687","n_code_links":1,"syntology":{"ran":10,"of":10,"n_ran_checked":9,"n_instrument":1,"unverified":0,"pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 3 honoured, 3 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/event-stream-gpt-a-data-pre-processing-and-1","slug":"event-stream-gpt-a-data-pre-processing-and-1","title":"Event Stream GPT: A Data Pre-processing and Modeling Library for Generative, Pre-trained Transformers over Continuous-time Sequences of Complex Events","date":"2023-06-20","arxiv_id":"2306.11547","n_code_links":1,"syntology":{"ran":5,"of":10,"n_ran_checked":5,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["mmcdermott/eventstreamgpt"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/chessgpt-bridging-policy-learning-and-1","slug":"chessgpt-bridging-policy-learning-and-1","title":"ChessGPT: Bridging Policy Learning and Language Modeling","date":"2023-06-15","arxiv_id":"2306.09200","n_code_links":1,"syntology":{"ran":14,"of":20,"n_ran_checked":10,"n_instrument":4,"unverified":6,"pointer_only":0,"phrase":"14 ran (of which 8 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 4 where Syntology's instrument failed) · 6 unverified","official":{"repos":["waterhorse1/chessgpt"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":8,"n_ran_no_instrument_failure":10,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/explanations-as-features-llm-based-features","slug":"explanations-as-features-llm-based-features","title":"Harnessing Explanations: LLM-to-LM Interpreter for Enhanced Text-Attributed Graph Representation Learning","date":"2023-05-31","arxiv_id":"2305.19523","n_code_links":3,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["XiaoxinHe/TAPE","xiaoxinhe/tape_arxiv_2023"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/knowledge-augmented-reasoning-distillation-1","slug":"knowledge-augmented-reasoning-distillation-1","title":"Knowledge-Augmented Reasoning Distillation for Small Language Models in Knowledge-Intensive Tasks","date":"2023-05-28","arxiv_id":"2305.18395","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":3,"n_instrument":2,"unverified":1,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["nardien/kard"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/navgpt-explicit-reasoning-in-vision-and","slug":"navgpt-explicit-reasoning-in-vision-and","title":"NavGPT: Explicit Reasoning in Vision-and-Language Navigation with Large Language Models","date":"2023-05-26","arxiv_id":"2305.16986","n_code_links":2,"syntology":{"ran":4,"of":5,"n_ran_checked":2,"n_instrument":2,"unverified":1,"pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["gengzezhou/navgpt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/chain-of-thought-hub-a-continuous-effort-to","slug":"chain-of-thought-hub-a-continuous-effort-to","title":"Chain-of-Thought Hub: A Continuous Effort to Measure Large Language Models' Reasoning Performance","date":"2023-05-26","arxiv_id":"2305.17306","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":7,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["franxyao/chain-of-thought-hub"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/trusting-your-evidence-hallucinate-less-with","slug":"trusting-your-evidence-hallucinate-less-with","title":"Trusting Your Evidence: Hallucinate Less with Context-aware Decoding","date":"2023-05-24","arxiv_id":"2305.14739","n_code_links":3,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/editing-commonsense-knowledge-in-gpt","slug":"editing-commonsense-knowledge-in-gpt","title":"Editing Common Sense in Transformers","date":"2023-05-24","arxiv_id":"2305.14956","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":6,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["anshitag/memit_csk"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/tricking-llms-into-disobedience-understanding","slug":"tricking-llms-into-disobedience-understanding","title":"Tricking LLMs into Disobedience: Formalizing, Analyzing, and Detecting Jailbreaks","date":"2023-05-24","arxiv_id":"2305.14965","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["AetherPrior/TrickLLM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/have-llms-advanced-enough-a-challenging","slug":"have-llms-advanced-enough-a-challenging","title":"Have LLMs Advanced Enough? A Challenging Problem Solving Benchmark For Large Language Models","date":"2023-05-24","arxiv_id":"2305.15074","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hgaurav2k/jeebench"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}}],"record_sha256":"6e6e5bdb2788a9a06f9d0890abd49d69964c78cda698df3e74f2b65856def13d","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}