{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/linear-warmup-with-cosine-annealing/papers/ran/4","list_of":"/method/linear-warmup-with-cosine-annealing","method":"Linear Warmup With Cosine Annealing","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"ran","order_definition":"only papers where Syntology ran at least one harvested sample; date (newest first), ties by arXiv id","caption":"We ran code from the paper's repository; we did not isolate this method inside it.","absence":"A paper missing from this list is not a recorded non-run: it may have no arXiv id, no harvested code, or only samples that have not run yet.","page":4,"pages_in_order":7,"rows_per_page":100,"rows":[301,400],"of":602,"counts":{"archive_papers_tagged":3797,"with_a_code_link":1655,"where_syntology_ran_a_sample":602,"not_listed_spam_title":0,"listed":3797,"listed_where_code_ran":602,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":490,"every_run_a_failure_of_syntologys_instrument":112,"listed_with_a_run_with_no_instrument_failure":490,"listed_every_run_a_failure_of_syntologys_instrument":112,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/linear-warmup-with-cosine-annealing/papers/ran/1","prev":"/method/linear-warmup-with-cosine-annealing/papers/ran/3","next":"/method/linear-warmup-with-cosine-annealing/papers/ran/5","papers":[{"paper":"/paper/biocoder-a-benchmark-for-bioinformatics-code","slug":"biocoder-a-benchmark-for-bioinformatics-code","title":"BioCoder: A Benchmark for Bioinformatics Code Generation with Large Language Models","date":"2023-08-31","arxiv_id":"2308.16458","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["gersteinlab/biocoder"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/mllm-dataengine-an-iterative-refinement","slug":"mllm-dataengine-an-iterative-refinement","title":"MLLM-DataEngine: An Iterative Refinement Approach for MLLM","date":"2023-08-25","arxiv_id":"2308.13566","n_code_links":1,"syntology":{"ran":8,"of":8,"n_ran_checked":4,"n_instrument":4,"unverified":0,"pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["opendatalab/mllm-dataengine"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/evaluating-large-language-models-on-graphs","slug":"evaluating-large-language-models-on-graphs","title":"Evaluating Large Language Models on Graphs: Performance Insights and Comparative Analysis","date":"2023-08-22","arxiv_id":"2308.11224","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":2,"n_instrument":0,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["ayame1006/llmtograph"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/activation-addition-steering-language-models","slug":"activation-addition-steering-language-models","title":"Steering Language Models With Activation Engineering","date":"2023-08-20","arxiv_id":"2308.10248","n_code_links":2,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["montemac/activation_additions"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/how-good-are-large-language-models-at-out-of","slug":"how-good-are-large-language-models-at-out-of","title":"How Good Are LLMs at Out-of-Distribution Detection?","date":"2023-08-20","arxiv_id":"2308.10261","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["awenbocc/llm-ood"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/wizardmath-empowering-mathematical-reasoning","slug":"wizardmath-empowering-mathematical-reasoning","title":"WizardMath: Empowering Mathematical Reasoning for Large Language Models via Reinforced Evol-Instruct","date":"2023-08-18","arxiv_id":"2308.09583","n_code_links":1,"syntology":{"ran":10,"of":16,"n_ran_checked":8,"n_instrument":2,"unverified":6,"pointer_only":16,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","official":null}},{"paper":"/paper/beam-retrieval-general-end-to-end-retrieval","slug":"beam-retrieval-general-end-to-end-retrieval","title":"End-to-End Beam Retrieval for Multi-Hop Question Answering","date":"2023-08-17","arxiv_id":"2308.08973","n_code_links":3,"syntology":{"ran":10,"of":13,"n_ran_checked":9,"n_instrument":1,"unverified":3,"pointer_only":2,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["Alab-NII/2wikimultihop","canghongjian/beam_retriever"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/mindmap-knowledge-graph-prompting-sparks","slug":"mindmap-knowledge-graph-prompting-sparks","title":"MindMap: Knowledge Graph Prompting Sparks Graph of Thoughts in Large Language Models","date":"2023-08-17","arxiv_id":"2308.09729","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["wyl-willing/MindMap"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/large-language-models-to-identify-social","slug":"large-language-models-to-identify-social","title":"Large Language Models to Identify Social Determinants of Health in Electronic Health Records","date":"2023-08-11","arxiv_id":"2308.06354","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":0,"n_instrument":2,"unverified":2,"pointer_only":4,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","official":{"repos":["aim-harvard/sdoh"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/rtllm-an-open-source-benchmark-for-design-rtl","slug":"rtllm-an-open-source-benchmark-for-design-rtl","title":"RTLLM: An Open-Source Benchmark for Design RTL Generation with Large Language Model","date":"2023-08-10","arxiv_id":"2308.05345","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hkust-zhiyao/rtllm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/you-only-prompt-once-on-the-capabilities-of","slug":"you-only-prompt-once-on-the-capabilities-of","title":"You Only Prompt Once: On the Capabilities of Prompt Learning on Large Language Models to Tackle Toxic Content","date":"2023-08-10","arxiv_id":"2308.05596","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["xinleihe/toxic-prompt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/audioldm-2-learning-holistic-audio-generation","slug":"audioldm-2-learning-holistic-audio-generation","title":"AudioLDM 2: Learning Holistic Audio Generation with Self-supervised Pretraining","date":"2023-08-10","arxiv_id":"2308.05734","n_code_links":2,"syntology":{"ran":16,"of":27,"n_ran_checked":16,"n_instrument":0,"unverified":11,"pointer_only":19,"phrase":"16 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 2 honoured, 3 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 11 unverified","official":{"repos":["haoheliu/AudioLDM2"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":5,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/3d-vista-pre-trained-transformer-for-3d","slug":"3d-vista-pre-trained-transformer-for-3d","title":"3D-VisTA: Pre-trained Transformer for 3D Vision and Text Alignment","date":"2023-08-08","arxiv_id":"2308.04352","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":4,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 4 with no instrument failure: 1 honoured, 1 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":"/paper/from-sparse-to-soft-mixtures-of-experts","slug":"from-sparse-to-soft-mixtures-of-experts","title":"From Sparse to Soft Mixtures of Experts","date":"2023-08-02","arxiv_id":"2308.00951","n_code_links":5,"syntology":{"ran":29,"of":33,"n_ran_checked":18,"n_instrument":11,"unverified":4,"pointer_only":8,"phrase":"29 ran (of which 5 constructed an object rather than computing a result; 18 with no instrument failure: 3 honoured, 4 violated, 11 with no contract checked; 11 where Syntology's instrument failed) · 4 unverified","official":{"repos":["google-research/vmoe"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["listed","official","unlocated"]}}},{"paper":"/paper/instructed-to-bias-instruction-tuned-language","slug":"instructed-to-bias-instruction-tuned-language","title":"Instructed to Bias: Instruction-Tuned Language Models Exhibit Emergent Cognitive Bias","date":"2023-08-01","arxiv_id":"2308.00225","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["itay1itzhak/instructedtobias"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/new-interaction-paradigm-for-complex-eda","slug":"new-interaction-paradigm-for-complex-eda","title":"New Interaction Paradigm for Complex EDA Software Leveraging GPT","date":"2023-07-27","arxiv_id":"2307.14740","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["smarton-empower/smarton-ai"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/how-is-chatgpt-s-behavior-changing-over-time","slug":"how-is-chatgpt-s-behavior-changing-over-time","title":"How is ChatGPT's behavior changing over time?","date":"2023-07-18","arxiv_id":"2307.09009","n_code_links":4,"syntology":{"ran":6,"of":6,"n_ran_checked":4,"n_instrument":2,"unverified":0,"pointer_only":5,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lchen001/llmdrift"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/gear-augmenting-language-models-with","slug":"gear-augmenting-language-models-with","title":"GEAR: Augmenting Language Models with Generalizable and Efficient Tool Resolution","date":"2023-07-17","arxiv_id":"2307.08775","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yining610/gear"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/coupling-large-language-models-with-logic","slug":"coupling-large-language-models-with-logic","title":"Coupling Large Language Models with Logic Programming for Robust and General Reasoning from Text","date":"2023-07-15","arxiv_id":"2307.07696","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":9,"n_instrument":0,"unverified":0,"pointer_only":9,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["azreasoners/llm-asp"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/leveraging-large-language-models-to-generate","slug":"leveraging-large-language-models-to-generate","title":"Leveraging Large Language Models to Generate Answer Set Programs","date":"2023-07-15","arxiv_id":"2307.07699","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["azreasoners/gpt-asp-rules"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/improving-retrieval-augmented-large-language","slug":"improving-retrieval-augmented-large-language","title":"Improving Retrieval-Augmented Large Language Models via Data Importance Learning","date":"2023-07-06","arxiv_id":"2307.03027","n_code_links":1,"syntology":{"ran":2,"of":6,"n_ran_checked":2,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["amsterdata/ragbooster"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/came-confidence-guided-adaptive-memory","slug":"came-confidence-guided-adaptive-memory","title":"CAME: Confidence-guided Adaptive Memory Efficient Optimization","date":"2023-07-05","arxiv_id":"2307.02047","n_code_links":2,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["yangluo7/came","huawei-noah/Pretrained-Language-Model"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/embodied-task-planning-with-large-language","slug":"embodied-task-planning-with-large-language","title":"Embodied Task Planning with Large Language Models","date":"2023-07-04","arxiv_id":"2307.01848","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Gary3410/TaPA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/kdstm-neural-semi-supervised-topic-modeling","slug":"kdstm-neural-semi-supervised-topic-modeling","title":"KDSTM: Neural Semi-supervised Topic Modeling with Knowledge Distillation","date":"2023-07-04","arxiv_id":"2307.01878","n_code_links":0,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 1 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":null}},{"paper":"/paper/longcoder-a-long-range-pre-trained-language","slug":"longcoder-a-long-range-pre-trained-language","title":"LongCoder: A Long-Range Pre-trained Language Model for Code Completion","date":"2023-06-26","arxiv_id":"2306.14893","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["microsoft/CodeBERT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/system-level-natural-language-feedback","slug":"system-level-natural-language-feedback","title":"System-Level Natural Language Feedback","date":"2023-06-23","arxiv_id":"2306.13588","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yyy-apple/sys-nl-feedback"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/voicebox-text-guided-multilingual-universal","slug":"voicebox-text-guided-multilingual-universal","title":"Voicebox: Text-Guided Multilingual Universal Speech Generation at Scale","date":"2023-06-23","arxiv_id":"2306.15687","n_code_links":1,"syntology":{"ran":10,"of":10,"n_ran_checked":9,"n_instrument":1,"unverified":0,"pointer_only":3,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 3 honoured, 3 violated, 3 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/event-stream-gpt-a-data-pre-processing-and-1","slug":"event-stream-gpt-a-data-pre-processing-and-1","title":"Event Stream GPT: A Data Pre-processing and Modeling Library for Generative, Pre-trained Transformers over Continuous-time Sequences of Complex Events","date":"2023-06-20","arxiv_id":"2306.11547","n_code_links":1,"syntology":{"ran":5,"of":10,"n_ran_checked":5,"n_instrument":0,"unverified":5,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","official":{"repos":["mmcdermott/eventstreamgpt"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/chessgpt-bridging-policy-learning-and-1","slug":"chessgpt-bridging-policy-learning-and-1","title":"ChessGPT: Bridging Policy Learning and Language Modeling","date":"2023-06-15","arxiv_id":"2306.09200","n_code_links":1,"syntology":{"ran":14,"of":20,"n_ran_checked":10,"n_instrument":4,"unverified":6,"pointer_only":0,"phrase":"14 ran (of which 8 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 1 violated, 9 with no contract checked; 4 where Syntology's instrument failed) · 6 unverified","official":{"repos":["waterhorse1/chessgpt"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":8,"n_ran_no_instrument_failure":10,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/waffling-around-for-performance-visual","slug":"waffling-around-for-performance-visual","title":"Waffling around for Performance: Visual Classification with Random Words and Broad Concepts","date":"2023-06-12","arxiv_id":"2306.07282","n_code_links":2,"syntology":{"ran":5,"of":5,"n_ran_checked":0,"n_instrument":5,"unverified":0,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","official":{"repos":["explainableml/waffleclip"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/prefer-to-classify-improving-text-classifiers","slug":"prefer-to-classify-improving-text-classifiers","title":"Prefer to Classify: Improving Text Classifiers via Auxiliary Preference Learning","date":"2023-06-08","arxiv_id":"2306.04925","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["minnesotanlp/p2c"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/pandalm-an-automatic-evaluation-benchmark-for","slug":"pandalm-an-automatic-evaluation-benchmark-for","title":"PandaLM: An Automatic Evaluation Benchmark for LLM Instruction Tuning Optimization","date":"2023-06-08","arxiv_id":"2306.05087","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":1,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["weopenml/pandalm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/toolalpaca-generalized-tool-learning-for","slug":"toolalpaca-generalized-tool-learning-for","title":"ToolAlpaca: Generalized Tool Learning for Language Models with 3000 Simulated Cases","date":"2023-06-08","arxiv_id":"2306.05301","n_code_links":3,"syntology":{"ran":5,"of":8,"n_ran_checked":5,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["tangqiaoyu/ToolAlpaca"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/certified-reasoning-with-language-models","slug":"certified-reasoning-with-language-models","title":"Certified Deductive Reasoning with Language Models","date":"2023-06-06","arxiv_id":"2306.04031","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/auto-gpt-for-online-decision-making","slug":"auto-gpt-for-online-decision-making","title":"Auto-GPT for Online Decision Making: Benchmarks and Additional Opinions","date":"2023-06-04","arxiv_id":"2306.02224","n_code_links":1,"syntology":{"ran":6,"of":11,"n_ran_checked":5,"n_instrument":1,"unverified":5,"pointer_only":4,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["younghuman/llmagent"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"paper":"/paper/explanations-as-features-llm-based-features","slug":"explanations-as-features-llm-based-features","title":"Harnessing Explanations: LLM-to-LM Interpreter for Enhanced Text-Attributed Graph Representation Learning","date":"2023-05-31","arxiv_id":"2305.19523","n_code_links":3,"syntology":{"ran":6,"of":7,"n_ran_checked":6,"n_instrument":0,"unverified":1,"pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["XiaoxinHe/TAPE","xiaoxinhe/tape_arxiv_2023"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/do-large-language-models-know-what-they-don-t","slug":"do-large-language-models-know-what-they-don-t","title":"Do Large Language Models Know What They Don't Know?","date":"2023-05-29","arxiv_id":"2305.18153","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yinzhangyue/selfaware"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/marked-personas-using-natural-language","slug":"marked-personas-using-natural-language","title":"Marked Personas: Using Natural Language Prompts to Measure Stereotypes in Language Models","date":"2023-05-29","arxiv_id":"2305.18189","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["myracheng/markedpersonas"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/test-time-training-on-nearest-neighbors-for","slug":"test-time-training-on-nearest-neighbors-for","title":"Test-Time Training on Nearest Neighbors for Large Language Models","date":"2023-05-29","arxiv_id":"2305.18466","n_code_links":1,"syntology":{"ran":3,"of":5,"n_ran_checked":3,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["socialfoundations/tttlm"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/generating-edu-extracts-for-plan-guided","slug":"generating-edu-extracts-for-plan-guided","title":"Generating EDU Extracts for Plan-Guided Summary Re-Ranking","date":"2023-05-28","arxiv_id":"2305.17779","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":0,"n_instrument":6,"unverified":4,"pointer_only":10,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 6 where Syntology's instrument failed) · 4 unverified","official":{"repos":["griff4692/edu-sum"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/knowledge-augmented-reasoning-distillation-1","slug":"knowledge-augmented-reasoning-distillation-1","title":"Knowledge-Augmented Reasoning Distillation for Small Language Models in Knowledge-Intensive Tasks","date":"2023-05-28","arxiv_id":"2305.18395","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":3,"n_instrument":2,"unverified":1,"pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["nardien/kard"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/mitigating-label-biases-for-in-context","slug":"mitigating-label-biases-for-in-context","title":"Mitigating Label Biases for In-context Learning","date":"2023-05-28","arxiv_id":"2305.19148","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["fywalter/label-bias"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/dna-gpt-divergent-n-gram-analysis-for","slug":"dna-gpt-divergent-n-gram-analysis-for","title":"DNA-GPT: Divergent N-Gram Analysis for Training-Free Detection of GPT-Generated Text","date":"2023-05-27","arxiv_id":"2305.17359","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["xianjun-yang/dna-gpt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/backpack-language-models","slug":"backpack-language-models","title":"Backpack Language Models","date":"2023-05-26","arxiv_id":"2305.16765","n_code_links":1,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","official":null}},{"paper":"/paper/navgpt-explicit-reasoning-in-vision-and","slug":"navgpt-explicit-reasoning-in-vision-and","title":"NavGPT: Explicit Reasoning in Vision-and-Language Navigation with Large Language Models","date":"2023-05-26","arxiv_id":"2305.16986","n_code_links":2,"syntology":{"ran":4,"of":5,"n_ran_checked":2,"n_instrument":2,"unverified":1,"pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["gengzezhou/navgpt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/chain-of-thought-hub-a-continuous-effort-to","slug":"chain-of-thought-hub-a-continuous-effort-to","title":"Chain-of-Thought Hub: A Continuous Effort to Measure Large Language Models' Reasoning Performance","date":"2023-05-26","arxiv_id":"2305.17306","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":7,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["franxyao/chain-of-thought-hub"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/landmark-attention-random-access-infinite","slug":"landmark-attention-random-access-infinite","title":"Landmark Attention: Random-Access Infinite Context Length for Transformers","date":"2023-05-25","arxiv_id":"2305.16300","n_code_links":2,"syntology":{"ran":11,"of":13,"n_ran_checked":5,"n_instrument":6,"unverified":2,"pointer_only":0,"phrase":"11 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","official":{"repos":["epfml/landmark-attention"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["community","listed","official"]}}},{"paper":"/paper/self-checker-plug-and-play-modules-for-fact","slug":"self-checker-plug-and-play-modules-for-fact","title":"Self-Checker: Plug-and-Play Modules for Fact-Checking with Large Language Models","date":"2023-05-24","arxiv_id":"2305.14623","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Miaoranmmm/SelfChecker"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/expertprompting-instructing-large-language","slug":"expertprompting-instructing-large-language","title":"ExpertPrompting: Instructing Large Language Models to be Distinguished Experts","date":"2023-05-24","arxiv_id":"2305.14688","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ofa-sys/expertllama"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/a-causal-view-of-entity-bias-in-large","slug":"a-causal-view-of-entity-bias-in-large","title":"A Causal View of Entity Bias in (Large) Language Models","date":"2023-05-24","arxiv_id":"2305.14695","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["luka-group/causal-view-of-entity-bias"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/trusting-your-evidence-hallucinate-less-with","slug":"trusting-your-evidence-hallucinate-less-with","title":"Trusting Your Evidence: Hallucinate Less with Context-aware Decoding","date":"2023-05-24","arxiv_id":"2305.14739","n_code_links":3,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/editing-commonsense-knowledge-in-gpt","slug":"editing-commonsense-knowledge-in-gpt","title":"Editing Common Sense in Transformers","date":"2023-05-24","arxiv_id":"2305.14956","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":6,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["anshitag/memit_csk"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/tricking-llms-into-disobedience-understanding","slug":"tricking-llms-into-disobedience-understanding","title":"Tricking LLMs into Disobedience: Formalizing, Analyzing, and Detecting Jailbreaks","date":"2023-05-24","arxiv_id":"2305.14965","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["AetherPrior/TrickLLM"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/llmdet-a-large-language-models-detection-tool","slug":"llmdet-a-large-language-models-detection-tool","title":"LLMDet: A Third Party Large Language Models Generated Text Detection Tool","date":"2023-05-24","arxiv_id":"2305.15004","n_code_links":1,"syntology":{"ran":1,"of":3,"n_ran_checked":0,"n_instrument":1,"unverified":2,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["trustedllm/llmdet"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/inference-time-policy-adapters-ipa-tailoring","slug":"inference-time-policy-adapters-ipa-tailoring","title":"Inference-Time Policy Adapters (IPA): Tailoring Extreme-Scale LMs without Fine-tuning","date":"2023-05-24","arxiv_id":"2305.15065","n_code_links":1,"syntology":{"ran":9,"of":11,"n_ran_checked":5,"n_instrument":4,"unverified":2,"pointer_only":0,"phrase":"9 ran (of which 3 constructed an object rather than computing a result; 5 with no instrument failure: 2 honoured, 0 violated, 3 with no contract checked; 4 where Syntology's instrument failed) · 2 unverified","official":{"repos":["gximinglu/ipa"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":3,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/have-llms-advanced-enough-a-challenging","slug":"have-llms-advanced-enough-a-challenging","title":"Have LLMs Advanced Enough? A Challenging Problem Solving Benchmark For Large Language Models","date":"2023-05-24","arxiv_id":"2305.15074","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hgaurav2k/jeebench"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"paper":"/paper/peek-across-improving-multi-document-modeling","slug":"peek-across-improving-multi-document-modeling","title":"Peek Across: Improving Multi-Document Modeling via Cross-Document Question-Answering","date":"2023-05-24","arxiv_id":"2305.15387","n_code_links":1,"syntology":{"ran":12,"of":18,"n_ran_checked":11,"n_instrument":1,"unverified":6,"pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["aviclu/peekacross"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/harnessing-the-power-of-large-language-models","slug":"harnessing-the-power-of-large-language-models","title":"Harnessing the Power of Large Language Models for Natural Language to First-Order Logic Translation","date":"2023-05-24","arxiv_id":"2305.15541","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["gblackout/logicllama"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/wikichat-a-few-shot-llm-based-chatbot","slug":"wikichat-a-few-shot-llm-based-chatbot","title":"WikiChat: Stopping the Hallucination of Large Language Model Chatbots by Few-Shot Grounding on Wikipedia","date":"2023-05-23","arxiv_id":"2305.14292","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["stanford-oval/wikichat"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/dynosaur-a-dynamic-growth-paradigm-for","slug":"dynosaur-a-dynamic-growth-paradigm-for","title":"Dynosaur: A Dynamic Growth Paradigm for Instruction-Tuning Data Curation","date":"2023-05-23","arxiv_id":"2305.14327","n_code_links":1,"syntology":{"ran":10,"of":14,"n_ran_checked":10,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["wadeyin9712/dynosaur"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/sophia-a-scalable-stochastic-second-order","slug":"sophia-a-scalable-stochastic-second-order","title":"Sophia: A Scalable Stochastic Second-order Optimizer for Language Model Pre-training","date":"2023-05-23","arxiv_id":"2305.14342","n_code_links":7,"syntology":{"ran":13,"of":19,"n_ran_checked":12,"n_instrument":1,"unverified":6,"pointer_only":0,"phrase":"13 ran (of which 4 constructed an object rather than computing a result; 12 with no instrument failure: 1 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":null}},{"paper":"/paper/mathdial-a-dialogue-tutoring-dataset-with","slug":"mathdial-a-dialogue-tutoring-dataset-with","title":"MathDial: A Dialogue Tutoring Dataset with Rich Pedagogical Properties Grounded in Math Reasoning Problems","date":"2023-05-23","arxiv_id":"2305.14536","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":2,"n_instrument":2,"unverified":0,"pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["eth-nlped/mathdial"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/recurrentgpt-interactive-generation-of","slug":"recurrentgpt-interactive-generation-of","title":"RecurrentGPT: Interactive Generation of (Arbitrarily) Long Text","date":"2023-05-22","arxiv_id":"2305.13304","n_code_links":2,"syntology":{"ran":3,"of":4,"n_ran_checked":2,"n_instrument":1,"unverified":1,"pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["aiwaves-cn/recurrentgpt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/gene-set-summarization-using-large-language","slug":"gene-set-summarization-using-large-language","title":"Gene Set Summarization using Large Language Models","date":"2023-05-21","arxiv_id":"2305.13338","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["monarch-initiative/talisman"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/pointgpt-auto-regressively-generative-pre-1","slug":"pointgpt-auto-regressively-generative-pre-1","title":"PointGPT: Auto-regressively Generative Pre-training from Point Clouds","date":"2023-05-19","arxiv_id":"2305.11487","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":6,"n_instrument":1,"unverified":0,"pointer_only":4,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["CGuangyan-BIT/PointGPT"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/clinical-camel-an-open-source-expert-level","slug":"clinical-camel-an-open-source-expert-level","title":"Clinical Camel: An Open Expert-Level Medical Language Model with Dialogue-Based Knowledge Encoding","date":"2023-05-19","arxiv_id":"2305.12031","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":1,"n_instrument":1,"unverified":0,"pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["bowang-lab/clinical-camel"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/generalized-planning-in-pddl-domains-with","slug":"generalized-planning-in-pddl-domains-with","title":"Generalized Planning in PDDL Domains with Pretrained Large Language Models","date":"2023-05-18","arxiv_id":"2305.11014","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["tomsilver/llm-genplan"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/text-classification-via-large-language-models","slug":"text-classification-via-large-language-models","title":"Text Classification via Large Language Models","date":"2023-05-15","arxiv_id":"2305.08377","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["shannonai/gpt-cls-carp"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/rl4f-generating-natural-language-feedback","slug":"rl4f-generating-natural-language-feedback","title":"RL4F: Generating Natural Language Feedback with Reinforcement Learning for Repairing Model Outputs","date":"2023-05-15","arxiv_id":"2305.08844","n_code_links":1,"syntology":{"ran":7,"of":7,"n_ran_checked":7,"n_instrument":0,"unverified":0,"pointer_only":6,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["feyzaakyurek/rl4f"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"paper":"/paper/small-models-are-valuable-plug-ins-for-large","slug":"small-models-are-valuable-plug-ins-for-large","title":"Small Models are Valuable Plug-ins for Large Language Models","date":"2023-05-15","arxiv_id":"2305.08848","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["JetRunner/SuperICL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/mobile-env-a-universal-platform-for-training","slug":"mobile-env-a-universal-platform-for-training","title":"Mobile-Env: Building Qualified Evaluation Benchmarks for LLM-GUI Interaction","date":"2023-05-14","arxiv_id":"2305.08144","n_code_links":2,"syntology":{"ran":11,"of":17,"n_ran_checked":11,"n_instrument":0,"unverified":6,"pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 6 unverified","official":{"repos":["opendfm/mobile-env-expe","x-lance/mobile-env"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/investigating-emergent-goal-like-behaviour-in","slug":"investigating-emergent-goal-like-behaviour-in","title":"The Machine Psychology of Cooperation: Can GPT models operationalise prompts for altruism, cooperation, competitiveness and selfishness in economic games?","date":"2023-05-13","arxiv_id":"2305.07970","n_code_links":2,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["phelps-sg/llm-cooperation","gitlab.com/sphelps/llm-cooperation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/tinystories-how-small-can-language-models-be","slug":"tinystories-how-small-can-language-models-be","title":"TinyStories: How Small Can Language Models Be and Still Speak Coherent English?","date":"2023-05-12","arxiv_id":"2305.07759","n_code_links":8,"syntology":{"ran":10,"of":18,"n_ran_checked":8,"n_instrument":2,"unverified":8,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 8 unverified","official":null}},{"paper":"/paper/language-models-don-t-always-say-what-they-1","slug":"language-models-don-t-always-say-what-they-1","title":"Language Models Don't Always Say What They Think: Unfaithful Explanations in Chain-of-Thought Prompting","date":"2023-05-07","arxiv_id":"2305.04388","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["milesaturpin/cot-unfaithfulness"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/personallm-investigating-the-ability-of-gpt-3","slug":"personallm-investigating-the-ability-of-gpt-3","title":"PersonaLLM: Investigating the Ability of Large Language Models to Express Personality Traits","date":"2023-05-04","arxiv_id":"2305.02547","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hjian42/personallm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/towards-automated-circuit-discovery-for-1","slug":"towards-automated-circuit-discovery-for-1","title":"Towards Automated Circuit Discovery for Mechanistic Interpretability","date":"2023-04-28","arxiv_id":"2304.14997","n_code_links":4,"syntology":{"ran":2,"of":5,"n_ran_checked":2,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["arthurconmy/automatic-circuit-discovery","neelnanda-io/transformerlens"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/causal-reasoning-and-large-language-models","slug":"causal-reasoning-and-large-language-models","title":"Causal Reasoning and Large Language Models: Opening a New Frontier for Causality","date":"2023-04-28","arxiv_id":"2305.00050","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["py-why/pywhy-llm"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/large-language-models-are-state-of-the-art-1","slug":"large-language-models-are-state-of-the-art-1","title":"ICE-Score: Instructing Large Language Models to Evaluate Code","date":"2023-04-27","arxiv_id":"2304.14317","n_code_links":2,"syntology":{"ran":3,"of":4,"n_ran_checked":3,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 2 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["terryyz/llm-code-eval","terryyz/ice-score"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/exploring-the-curious-case-of-code-prompts","slug":"exploring-the-curious-case-of-code-prompts","title":"Exploring the Curious Case of Code Prompts","date":"2023-04-26","arxiv_id":"2304.13250","n_code_links":1,"syntology":{"ran":2,"of":6,"n_ran_checked":2,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["zharry29/codex_vs_gpt3"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/measuring-massive-multitask-chinese","slug":"measuring-massive-multitask-chinese","title":"Measuring Massive Multitask Chinese Understanding","date":"2023-04-25","arxiv_id":"2304.12986","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["Felixgithub2017/MMCU"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"paper":"/paper/genegpt-teaching-large-language-models-to-use","slug":"genegpt-teaching-large-language-models-to-use","title":"GeneGPT: Augmenting Large Language Models with Domain Tools for Improved Access to Biomedical Information","date":"2023-04-19","arxiv_id":"2304.09667","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["ncbi/GeneGPT"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/surgicalgpt-end-to-end-language-vision-gpt","slug":"surgicalgpt-end-to-end-language-vision-gpt","title":"SurgicalGPT: End-to-End Language-Vision GPT for Visual Question Answering in Surgery","date":"2023-04-19","arxiv_id":"2304.09974","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["lalithjets/surgicalgpt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/from-zero-to-hero-examining-the-power-of","slug":"from-zero-to-hero-examining-the-power-of","title":"From Zero to Hero: Examining the Power of Symbolic Tasks in Instruction Tuning","date":"2023-04-17","arxiv_id":"2304.07995","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sail-sg/symbolic-instruction-tuning"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/api-bank-a-benchmark-for-tool-augmented-llms","slug":"api-bank-a-benchmark-for-tool-augmented-llms","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","date":"2023-04-14","arxiv_id":"2304.08244","n_code_links":2,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","official":null}},{"paper":"/paper/medalpaca-an-open-source-collection-of","slug":"medalpaca-an-open-source-collection-of","title":"MedAlpaca -- An Open-Source Collection of Medical Conversational AI Models and Training Data","date":"2023-04-14","arxiv_id":"2304.08247","n_code_links":1,"syntology":{"ran":10,"of":13,"n_ran_checked":10,"n_instrument":0,"unverified":3,"pointer_only":1,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":"/paper/localizing-model-behavior-with-path-patching","slug":"localizing-model-behavior-with-path-patching","title":"Localizing Model Behavior with Path Patching","date":"2023-04-12","arxiv_id":"2304.05969","n_code_links":1,"syntology":{"ran":10,"of":13,"n_ran_checked":10,"n_instrument":0,"unverified":3,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","official":{"repos":["redwoodresearch/rust_circuit_public"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/multi-step-jailbreaking-privacy-attacks-on","slug":"multi-step-jailbreaking-privacy-attacks-on","title":"Multi-step Jailbreaking Privacy Attacks on ChatGPT","date":"2023-04-11","arxiv_id":"2304.05197","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":0,"n_instrument":5,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","official":{"repos":["hkust-knowcomp/llm-multistep-jailbreak"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/gpt-detectors-are-biased-against-non-native","slug":"gpt-detectors-are-biased-against-non-native","title":"GPT detectors are biased against non-native English writers","date":"2023-04-06","arxiv_id":"2304.02819","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["weixin-liang/chatgpt-detector-bias"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/zero-shot-next-item-recommendation-using","slug":"zero-shot-next-item-recommendation-using","title":"Zero-Shot Next-Item Recommendation using Large Pretrained Language Models","date":"2023-04-06","arxiv_id":"2304.03153","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":null}},{"paper":"/paper/making-ai-less-thirsty-uncovering-and","slug":"making-ai-less-thirsty-uncovering-and","title":"Making AI Less \"Thirsty\": Uncovering and Addressing the Secret Water Footprint of AI Models","date":"2023-04-06","arxiv_id":"2304.03271","n_code_links":1,"syntology":{"ran":6,"of":10,"n_ran_checked":6,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["ren-research/making-ai-less-thirsty"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/refiner-reasoning-feedback-on-intermediate","slug":"refiner-reasoning-feedback-on-intermediate","title":"REFINER: Reasoning Feedback on Intermediate Representations","date":"2023-04-04","arxiv_id":"2304.01904","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["debjitpaul/refiner"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/llm-adapters-an-adapter-family-for-parameter","slug":"llm-adapters-an-adapter-family-for-parameter","title":"LLM-Adapters: An Adapter Family for Parameter-Efficient Fine-Tuning of Large Language Models","date":"2023-04-04","arxiv_id":"2304.01933","n_code_links":2,"syntology":{"ran":4,"of":4,"n_ran_checked":0,"n_instrument":4,"unverified":0,"pointer_only":1,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","official":{"repos":["agi-edgerunners/llm-adapters"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/understanding-individual-and-team-based-human","slug":"understanding-individual-and-team-based-human","title":"Does Human Collaboration Enhance the Accuracy of Identifying LLM-Generated Deepfake Texts?","date":"2023-04-03","arxiv_id":"2304.01002","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["huashen218/llm-deepfake-human-study"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/annollm-making-large-language-models-to-be","slug":"annollm-making-large-language-models-to-be","title":"AnnoLLM: Making Large Language Models to Be Better Crowdsourced Annotators","date":"2023-03-29","arxiv_id":"2303.16854","n_code_links":2,"syntology":{"ran":3,"of":3,"n_ran_checked":3,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["nlpcode/annollm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"paper":"/paper/viewrefer-grasp-the-multi-view-knowledge-for","slug":"viewrefer-grasp-the-multi-view-knowledge-for","title":"ViewRefer: Grasp the Multi-view Knowledge for 3D Visual Grounding with GPT and Prototype Guidance","date":"2023-03-29","arxiv_id":"2303.16894","n_code_links":7,"syntology":{"ran":8,"of":12,"n_ran_checked":6,"n_instrument":2,"unverified":4,"pointer_only":12,"phrase":"8 ran (of which 2 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","official":{"repos":["ivan-tang-3d/viewrefer3d","ziyuguo99/viewrefer3d"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/autoad-movie-description-in-context","slug":"autoad-movie-description-in-context","title":"AutoAD: Movie Description in Context","date":"2023-03-29","arxiv_id":"2303.16899","n_code_links":1,"syntology":{"ran":5,"of":13,"n_ran_checked":3,"n_instrument":2,"unverified":8,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 2 where Syntology's instrument failed) · 8 unverified","official":{"repos":["Soldelli/MAD"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":8,"ran_from_kinds":["official"]}}},{"paper":"/paper/evaluating-gpt-3-5-and-gpt-4-models-on","slug":"evaluating-gpt-3-5-and-gpt-4-models-on","title":"Evaluating GPT-3.5 and GPT-4 Models on Brazilian University Admission Exams","date":"2023-03-29","arxiv_id":"2303.17003","n_code_links":1,"syntology":{"ran":8,"of":9,"n_ran_checked":6,"n_instrument":2,"unverified":1,"pointer_only":6,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 1 violated, 5 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["piresramon/gpt-4-enem"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/explicit-planning-helps-language-models-in","slug":"explicit-planning-helps-language-models-in","title":"Explicit Planning Helps Language Models in Logical Reasoning","date":"2023-03-28","arxiv_id":"2303.15714","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["cindermond/explicit-planning-for-reasoning","cindermond/leap"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/sift-sparse-iso-flop-transformations-for","slug":"sift-sparse-iso-flop-transformations-for","title":"Sparse-IFT: Sparse Iso-FLOP Transformations for Maximizing Training Efficiency","date":"2023-03-21","arxiv_id":"2303.11525","n_code_links":2,"syntology":{"ran":21,"of":24,"n_ran_checked":20,"n_instrument":1,"unverified":3,"pointer_only":2,"phrase":"21 ran (of which 0 constructed an object rather than computing a result; 20 with no instrument failure: 0 honoured, 0 violated, 20 with no contract checked; 1 where Syntology's instrument failed) · 3 unverified","official":{"repos":["cerebrasresearch/sift","cerebrasresearch/sparse-ift"],"state":"official (archive's flag): 21 ran","n_ran":21,"n_constructed":0,"n_ran_no_instrument_failure":20,"n_unverified":3,"ran_from_kinds":["official"]}}},{"paper":"/paper/learning-a-sparse-transformer-network-for","slug":"learning-a-sparse-transformer-network-for","title":"Learning A Sparse Transformer Network for Effective Image Deraining","date":"2023-03-21","arxiv_id":"2303.11950","n_code_links":1,"syntology":{"ran":8,"of":11,"n_ran_checked":8,"n_instrument":0,"unverified":3,"pointer_only":11,"phrase":"8 ran (of which 8 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified; every one of the 8 samples that ran constructed an object rather than computing a result","official":{"repos":["cschenxiang/drsformer"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":8,"n_ran_no_instrument_failure":8,"n_unverified":3,"ran_from_kinds":["official"]}}}],"record_sha256":"29036343b0f6b27bf2d935a27585987cd25aafb5717a02baac5a3e5ac2476d64","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}