{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/method/gpt-4/papers/28","list_of":"/method/gpt-4","method":"GPT-4","archive":{"snapshot":"2025-07-28"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"date (newest first), then slug","page":28,"pages_in_order":29,"rows_per_page":100,"rows":[2701,2800],"of":2870,"counts":{"archive_papers_tagged":2870,"with_a_code_link":1244,"where_syntology_ran_a_sample":526,"not_listed_spam_title":0,"listed":2870,"listed_where_code_ran":526,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":417,"every_run_a_failure_of_syntologys_instrument":109,"listed_with_a_run_with_no_instrument_failure":417,"listed_every_run_a_failure_of_syntologys_instrument":109,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/method/gpt-4","prev":"/method/gpt-4/papers/27","next":"/method/gpt-4/papers/29","papers":[{"paper":"/paper/the-rise-of-ai-language-pathologists","slug":"the-rise-of-ai-language-pathologists","title":"The Rise of AI Language Pathologists: Exploring Two-level Prompt Learning for Few-shot Weakly-supervised Whole Slide Image Classification","date":"2023-05-29","arxiv_id":"2305.17891","n_code_links":1,"syntology":{"ran":4,"of":6,"n_ran_checked":3,"n_instrument":1,"unverified":2,"pointer_only":6,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["miccaiif/top"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":2,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/large-language-models-scientific-knowledge","slug":"large-language-models-scientific-knowledge","title":"Large Language Models, scientific knowledge and factuality: A framework to streamline human expert evaluation","date":"2023-05-28","arxiv_id":"2305.17819","n_code_links":1,"syntology":null},{"paper":"/paper/reward-collapse-in-aligning-large-language","slug":"reward-collapse-in-aligning-large-language","title":"Reward Collapse in Aligning Large Language Models","date":"2023-05-28","arxiv_id":"2305.17608","n_code_links":1,"syntology":null},{"paper":"/paper/dna-gpt-divergent-n-gram-analysis-for","slug":"dna-gpt-divergent-n-gram-analysis-for","title":"DNA-GPT: Divergent N-Gram Analysis for Training-Free Detection of GPT-Generated Text","date":"2023-05-27","arxiv_id":"2305.17359","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["xianjun-yang/dna-gpt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/model-dementia-generated-data-makes-models","slug":"model-dementia-generated-data-makes-models","title":"The Curse of Recursion: Training on Generated Data Makes Models Forget","date":"2023-05-27","arxiv_id":"2305.17493","n_code_links":1,"syntology":null},{"paper":"/paper/swiftsage-a-generative-agent-with-fast-and-1","slug":"swiftsage-a-generative-agent-with-fast-and-1","title":"SwiftSage: A Generative Agent with Fast and Slow Thinking for Complex Interactive Tasks","date":"2023-05-27","arxiv_id":"2305.17390","n_code_links":2,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":null}},{"paper":"/paper/what-can-large-language-models-do-in","slug":"what-can-large-language-models-do-in","title":"What can Large Language Models do in chemistry? A comprehensive benchmark on eight tasks","date":"2023-05-27","arxiv_id":"2305.18365","n_code_links":1,"syntology":null},{"paper":"/paper/alignscore-evaluating-factual-consistency","slug":"alignscore-evaluating-factual-consistency","title":"AlignScore: Evaluating Factual Consistency with a Unified Alignment Function","date":"2023-05-26","arxiv_id":"2305.16739","n_code_links":2,"syntology":null},{"paper":"/paper/biomedgpt-a-unified-and-generalist-biomedical","slug":"biomedgpt-a-unified-and-generalist-biomedical","title":"BiomedGPT: A Generalist Vision-Language Foundation Model for Diverse Biomedical Tasks","date":"2023-05-26","arxiv_id":"2305.17100","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["taokz/biomedgpt"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/chain-of-thought-hub-a-continuous-effort-to","slug":"chain-of-thought-hub-a-continuous-effort-to","title":"Chain-of-Thought Hub: A Continuous Effort to Measure Large Language Models' Reasoning Performance","date":"2023-05-26","arxiv_id":"2305.17306","n_code_links":1,"syntology":{"ran":9,"of":9,"n_ran_checked":7,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["franxyao/chain-of-thought-hub"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"improving-accuracy-of-gpt-3-4-results-on","title":"Improving accuracy of GPT-3/4 results on biomedical data using a retrieval-augmented language model","date":"2023-05-26","arxiv_id":"2305.17116","n_code_links":0,"syntology":null},{"paper":"/paper/large-language-models-as-tool-makers","slug":"large-language-models-as-tool-makers","title":"Large Language Models as Tool Makers","date":"2023-05-26","arxiv_id":"2305.17126","n_code_links":1,"syntology":null},{"paper":"/paper/llms-and-the-abstraction-and-reasoning-corpus","slug":"llms-and-the-abstraction-and-reasoning-corpus","title":"LLMs and the Abstraction and Reasoning Corpus: Successes, Failures, and the Importance of Object-based Representations","date":"2023-05-26","arxiv_id":"2305.18354","n_code_links":1,"syntology":null},{"paper":"/paper/navgpt-explicit-reasoning-in-vision-and","slug":"navgpt-explicit-reasoning-in-vision-and","title":"NavGPT: Explicit Reasoning in Vision-and-Language Navigation with Large Language Models","date":"2023-05-26","arxiv_id":"2305.16986","n_code_links":2,"syntology":{"ran":4,"of":5,"n_ran_checked":2,"n_instrument":2,"unverified":1,"pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 2 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","official":{"repos":["gengzezhou/navgpt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["listed","official"]}}},{"paper":"/paper/neural-task-synthesis-for-visual-programming","slug":"neural-task-synthesis-for-visual-programming","title":"Neural Task Synthesis for Visual Programming","date":"2023-05-26","arxiv_id":"2305.18342","n_code_links":1,"syntology":null},{"paper":"/paper/on-evaluating-adversarial-robustness-of-large","slug":"on-evaluating-adversarial-robustness-of-large","title":"On Evaluating Adversarial Robustness of Large Vision-Language Models","date":"2023-05-26","arxiv_id":"2305.16934","n_code_links":1,"syntology":{"ran":4,"of":4,"n_ran_checked":2,"n_instrument":2,"unverified":0,"pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["yunqing-me/attackvlm"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"playing-repeated-games-with-large-language","title":"Playing repeated games with Large Language Models","date":"2023-05-26","arxiv_id":"2305.16867","n_code_links":0,"syntology":null},{"paper":null,"slug":"asking-before-action-gather-information-in","title":"Asking Before Acting: Gather Information in Embodied Decision Making with Language Models","date":"2023-05-25","arxiv_id":"2305.15695","n_code_links":0,"syntology":null},{"paper":null,"slug":"chatgpt-for-plc-dcs-control-logic-generation","title":"ChatGPT for PLC/DCS Control Logic Generation","date":"2023-05-25","arxiv_id":"2305.15809","n_code_links":0,"syntology":null},{"paper":"/paper/landmark-attention-random-access-infinite","slug":"landmark-attention-random-access-infinite","title":"Landmark Attention: Random-Access Infinite Context Length for Transformers","date":"2023-05-25","arxiv_id":"2305.16300","n_code_links":2,"syntology":{"ran":11,"of":13,"n_ran_checked":5,"n_instrument":6,"unverified":2,"pointer_only":0,"phrase":"11 ran (of which 5 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 6 where Syntology's instrument failed) · 2 unverified","official":{"repos":["epfml/landmark-attention"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["community","listed","official"]}}},{"paper":"/paper/on-the-tool-manipulation-capability-of-open","slug":"on-the-tool-manipulation-capability-of-open","title":"On the Tool Manipulation Capability of Open-source Large Language Models","date":"2023-05-25","arxiv_id":"2305.16504","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["sambanova/toolbench"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/self-contradictory-hallucinations-of-large","slug":"self-contradictory-hallucinations-of-large","title":"Self-contradictory Hallucinations of Large Language Models: Evaluation, Detection and Mitigation","date":"2023-05-25","arxiv_id":"2305.15852","n_code_links":1,"syntology":{"ran":5,"of":9,"n_ran_checked":5,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["eth-sri/chatprotect"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"undetectable-watermarks-for-language-models","title":"Undetectable Watermarks for Language Models","date":"2023-05-25","arxiv_id":"2306.09194","n_code_links":0,"syntology":null},{"paper":"/paper/voyager-an-open-ended-embodied-agent-with","slug":"voyager-an-open-ended-embodied-agent-with","title":"Voyager: An Open-Ended Embodied Agent with Large Language Models","date":"2023-05-25","arxiv_id":"2305.16291","n_code_links":1,"syntology":{"ran":5,"of":6,"n_ran_checked":5,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["MineDojo/Voyager"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"a-relentless-benchmark-for-modelling-graded","title":"A RelEntLess Benchmark for Modelling Graded Relations between Named Entities","date":"2023-05-24","arxiv_id":"2305.15002","n_code_links":0,"syntology":null},{"paper":null,"slug":"adversarial-demonstration-attacks-on-large","title":"Adversarial Demonstration Attacks on Large Language Models","date":"2023-05-24","arxiv_id":"2305.14950","n_code_links":0,"syntology":null},{"paper":null,"slug":"benchmarking-arabic-ai-with-large-language","title":"LAraBench: Benchmarking Arabic AI with Large Language Models","date":"2023-05-24","arxiv_id":"2305.14982","n_code_links":0,"syntology":null},{"paper":"/paper/bytesized32-a-corpus-and-challenge-task-for","slug":"bytesized32-a-corpus-and-challenge-task-for","title":"ByteSized32: A Corpus and Challenge Task for Generating Task-Specific World Models Expressed as Text Games","date":"2023-05-24","arxiv_id":"2305.14879","n_code_links":1,"syntology":{"ran":6,"of":8,"n_ran_checked":6,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["cognitiveailab/bytesized32"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"clever-hans-or-neural-theory-of-mind-stress","title":"Clever Hans or Neural Theory of Mind? Stress Testing Social Reasoning in Large Language Models","date":"2023-05-24","arxiv_id":"2305.14763","n_code_links":0,"syntology":null},{"paper":"/paper/from-words-to-wires-generating-functioning","slug":"from-words-to-wires-generating-functioning","title":"From Words to Wires: Generating Functioning Electronic Devices from Natural Language Descriptions","date":"2023-05-24","arxiv_id":"2305.14874","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["cognitiveailab/words2wires"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/gorilla-large-language-model-connected-with","slug":"gorilla-large-language-model-connected-with","title":"Gorilla: Large Language Model Connected with Massive APIs","date":"2023-05-24","arxiv_id":"2305.15334","n_code_links":1,"syntology":null},{"paper":null,"slug":"gptaraeval-a-comprehensive-evaluation-of","title":"GPTAraEval: A Comprehensive Evaluation of ChatGPT on Arabic NLP","date":"2023-05-24","arxiv_id":"2305.14976","n_code_links":0,"syntology":null},{"paper":"/paper/harnessing-the-power-of-large-language-models","slug":"harnessing-the-power-of-large-language-models","title":"Harnessing the Power of Large Language Models for Natural Language to First-Order Logic Translation","date":"2023-05-24","arxiv_id":"2305.15541","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["gblackout/logicllama"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/have-llms-advanced-enough-a-challenging","slug":"have-llms-advanced-enough-a-challenging","title":"Have LLMs Advanced Enough? A Challenging Problem Solving Benchmark For Large Language Models","date":"2023-05-24","arxiv_id":"2305.15074","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":2,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hgaurav2k/jeebench"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"paper":"/paper/huatuogpt-towards-taming-language-model-to-be","slug":"huatuogpt-towards-taming-language-model-to-be","title":"HuatuoGPT, towards Taming Language Model to Be a Doctor","date":"2023-05-24","arxiv_id":"2305.15075","n_code_links":2,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["freedomintelligence/huatuogpt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/is-gpt-4-a-good-data-analyst","slug":"is-gpt-4-a-good-data-analyst","title":"Is GPT-4 a Good Data Analyst?","date":"2023-05-24","arxiv_id":"2305.15038","n_code_links":1,"syntology":null},{"paper":"/paper/just-ask-for-calibration-strategies-for","slug":"just-ask-for-calibration-strategies-for","title":"Just Ask for Calibration: Strategies for Eliciting Calibrated Confidence Scores from Language Models Fine-Tuned with Human Feedback","date":"2023-05-24","arxiv_id":"2305.14975","n_code_links":1,"syntology":null},{"paper":"/paper/large-language-models-are-effective-table-to","slug":"large-language-models-are-effective-table-to","title":"Investigating Table-to-Text Generation Capabilities of LLMs in Real-World Information Seeking Scenarios","date":"2023-05-24","arxiv_id":"2305.14987","n_code_links":2,"syntology":null},{"paper":null,"slug":"leveraging-gpt-4-for-automatic-translation","title":"Leveraging GPT-4 for Automatic Translation Post-Editing","date":"2023-05-24","arxiv_id":"2305.14878","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-llms-for-kpis-retrieval-from","title":"Enabling and Analyzing How to Efficiently Extract Information from Hybrid Long Documents with LLMs","date":"2023-05-24","arxiv_id":"2305.16344","n_code_links":0,"syntology":null},{"paper":null,"slug":"leveraging-pre-trained-large-language-models","title":"Leveraging Pre-trained Large Language Models to Construct and Utilize World Models for Model-based Task Planning","date":"2023-05-24","arxiv_id":"2305.14909","n_code_links":0,"syntology":null},{"paper":"/paper/peek-across-improving-multi-document-modeling","slug":"peek-across-improving-multi-document-modeling","title":"Peek Across: Improving Multi-Document Modeling via Cross-Document Question-Answering","date":"2023-05-24","arxiv_id":"2305.15387","n_code_links":1,"syntology":{"ran":12,"of":18,"n_ran_checked":11,"n_instrument":1,"unverified":6,"pointer_only":0,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","official":{"repos":["aviclu/peekacross"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":6,"ran_from_kinds":["official"]}}},{"paper":"/paper/prompt-optimization-of-large-language-model","slug":"prompt-optimization-of-large-language-model","title":"AutoPlan: Automatic Planning of Interactive Decision-Making Tasks With Large Language Models","date":"2023-05-24","arxiv_id":"2305.15064","n_code_links":1,"syntology":{"ran":2,"of":4,"n_ran_checked":1,"n_instrument":1,"unverified":2,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","official":{"repos":["owaski/autoplan"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/reasoning-with-language-model-is-planning","slug":"reasoning-with-language-model-is-planning","title":"Reasoning with Language Model is Planning with World Model","date":"2023-05-24","arxiv_id":"2305.14992","n_code_links":3,"syntology":{"ran":4,"of":7,"n_ran_checked":2,"n_instrument":2,"unverified":3,"pointer_only":0,"phrase":"4 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 2 where Syntology's instrument failed) · 3 unverified","official":null}},{"paper":"/paper/refgpt-reference-truthful-customized","slug":"refgpt-reference-truthful-customized","title":"RefGPT: Dialogue Generation of GPT, by GPT, and for GPT","date":"2023-05-24","arxiv_id":"2305.14994","n_code_links":1,"syntology":null},{"paper":"/paper/spring-gpt-4-out-performs-rl-algorithms-by","slug":"spring-gpt-4-out-performs-rl-algorithms-by","title":"SPRING: Studying the Paper and Reasoning to Play Games","date":"2023-05-24","arxiv_id":"2305.15486","n_code_links":1,"syntology":null},{"paper":"/paper/testing-causal-models-of-word-meaning-in-gpt","slug":"testing-causal-models-of-word-meaning-in-gpt","title":"Testing Causal Models of Word Meaning in GPT-3 and -4","date":"2023-05-24","arxiv_id":"2305.14630","n_code_links":1,"syntology":null},{"paper":"/paper/towards-reliable-misinformation-mitigation","slug":"towards-reliable-misinformation-mitigation","title":"Towards Reliable Misinformation Mitigation: Generalization, Uncertainty, and GPT-4","date":"2023-05-24","arxiv_id":"2305.14928","n_code_links":1,"syntology":{"ran":6,"of":6,"n_ran_checked":6,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["complexdata-mila/mitigatemisinfo"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/aligning-large-language-models-through","slug":"aligning-large-language-models-through","title":"Aligning Large Language Models through Synthetic Feedback","date":"2023-05-23","arxiv_id":"2305.13735","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":1,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"1 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified; the one sample that ran constructed an object rather than computing a result","official":{"repos":["naver-ai/almost"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/automatic-model-selection-with-large-language","slug":"automatic-model-selection-with-large-language","title":"Automatic Model Selection with Large Language Models for Reasoning","date":"2023-05-23","arxiv_id":"2305.14333","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["xuzhao0/model-selection-reasoning"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/cgce-a-chinese-generative-chat-evaluation","slug":"cgce-a-chinese-generative-chat-evaluation","title":"CGCE: A Chinese Generative Chat Evaluation Benchmark for General and Financial Domains","date":"2023-05-23","arxiv_id":"2305.14471","n_code_links":1,"syntology":null},{"paper":"/paper/dynosaur-a-dynamic-growth-paradigm-for","slug":"dynosaur-a-dynamic-growth-paradigm-for","title":"Dynosaur: A Dynamic Growth Paradigm for Instruction-Tuning Data Curation","date":"2023-05-23","arxiv_id":"2305.14327","n_code_links":1,"syntology":{"ran":10,"of":14,"n_ran_checked":10,"n_instrument":0,"unverified":4,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","official":{"repos":["wadeyin9712/dynosaur"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":10,"n_unverified":4,"ran_from_kinds":["official"]}}},{"paper":"/paper/empowering-llm-based-machine-translation-with","slug":"empowering-llm-based-machine-translation-with","title":"Benchmarking Machine Translation with Cultural Awareness","date":"2023-05-23","arxiv_id":"2305.14328","n_code_links":1,"syntology":null},{"paper":"/paper/factscore-fine-grained-atomic-evaluation-of","slug":"factscore-fine-grained-atomic-evaluation-of","title":"FActScore: Fine-grained Atomic Evaluation of Factual Precision in Long Form Text Generation","date":"2023-05-23","arxiv_id":"2305.14251","n_code_links":4,"syntology":{"ran":11,"of":15,"n_ran_checked":10,"n_instrument":1,"unverified":4,"pointer_only":6,"phrase":"11 ran (of which 1 constructed an object rather than computing a result; 10 with no instrument failure: 0 honoured, 0 violated, 10 with no contract checked; 1 where Syntology's instrument failed) · 4 unverified","official":{"repos":["shmsw25/factscore"],"state":"official (archive's flag): 10 ran","n_ran":10,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":2,"ran_from_kinds":["listed","official"]}}},{"paper":null,"slug":"genspectrum-chat-data-exploration-in-public","title":"GenSpectrum Chat: Data Exploration in Public Health Using Large Language Models","date":"2023-05-23","arxiv_id":"2305.13821","n_code_links":0,"syntology":null},{"paper":"/paper/goat-fine-tuned-llama-outperforms-gpt-4-on","slug":"goat-fine-tuned-llama-outperforms-gpt-4-on","title":"Goat: Fine-tuned LLaMA Outperforms GPT-4 on Arithmetic Tasks","date":"2023-05-23","arxiv_id":"2305.14201","n_code_links":1,"syntology":null},{"paper":"/paper/instructscore-towards-explainable-text","slug":"instructscore-towards-explainable-text","title":"INSTRUCTSCORE: Explainable Text Generation Evaluation with Finegrained Feedback","date":"2023-05-23","arxiv_id":"2305.14282","n_code_links":2,"syntology":null},{"paper":"/paper/learning-to-generate-novel-scientific","slug":"learning-to-generate-novel-scientific","title":"SciMON: Scientific Inspiration Machines Optimized for Novelty","date":"2023-05-23","arxiv_id":"2305.14259","n_code_links":1,"syntology":null},{"paper":"/paper/let-s-think-frame-by-frame-evaluating-video","slug":"let-s-think-frame-by-frame-evaluating-video","title":"Let's Think Frame by Frame with VIP: A Video Infilling and Prediction Dataset for Evaluating Video Chain-of-Thought","date":"2023-05-23","arxiv_id":"2305.13903","n_code_links":1,"syntology":null},{"paper":"/paper/llm-powered-data-augmentation-for-enhanced","slug":"llm-powered-data-augmentation-for-enhanced","title":"LLM-powered Data Augmentation for Enhanced Cross-lingual Performance","date":"2023-05-23","arxiv_id":"2305.14288","n_code_links":1,"syntology":null},{"paper":"/paper/llms-as-factual-reasoners-insights-from","slug":"llms-as-factual-reasoners-insights-from","title":"LLMs as Factual Reasoners: Insights from Existing Benchmarks and Beyond","date":"2023-05-23","arxiv_id":"2305.14540","n_code_links":1,"syntology":null},{"paper":"/paper/qlora-efficient-finetuning-of-quantized-llms","slug":"qlora-efficient-finetuning-of-quantized-llms","title":"QLoRA: Efficient Finetuning of Quantized LLMs","date":"2023-05-23","arxiv_id":"2305.14314","n_code_links":20,"syntology":{"ran":18,"of":26,"n_ran_checked":6,"n_instrument":12,"unverified":8,"pointer_only":17,"phrase":"18 ran (of which 1 constructed an object rather than computing a result; 6 with no instrument failure: 2 honoured, 2 violated, 2 with no contract checked; 12 where Syntology's instrument failed) · 8 unverified","official":{"repos":["artidoro/qlora","timdettmers/bitsandbytes"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["community","listed","official","unlocated"]}}},{"paper":"/paper/towards-massively-multi-domain-multilingual","slug":"towards-massively-multi-domain-multilingual","title":"ReadMe++: Benchmarking Multilingual Language Models for Multi-Domain Readability Assessment","date":"2023-05-23","arxiv_id":"2305.14463","n_code_links":1,"syntology":null},{"paper":"/paper/wikichat-a-few-shot-llm-based-chatbot","slug":"wikichat-a-few-shot-llm-based-chatbot","title":"WikiChat: Stopping the Hallucination of Large Language Model Chatbots by Few-Shot Grounding on Wikipedia","date":"2023-05-23","arxiv_id":"2305.14292","n_code_links":1,"syntology":{"ran":5,"of":7,"n_ran_checked":5,"n_instrument":0,"unverified":2,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","official":{"repos":["stanford-oval/wikichat"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"paper":"/paper/zeroscrolls-a-zero-shot-benchmark-for-long","slug":"zeroscrolls-a-zero-shot-benchmark-for-long","title":"ZeroSCROLLS: A Zero-Shot Benchmark for Long Text Understanding","date":"2023-05-23","arxiv_id":"2305.14196","n_code_links":1,"syntology":{"ran":2,"of":3,"n_ran_checked":1,"n_instrument":1,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","official":{"repos":["tau-nlp/zero_scrolls"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/are-large-language-models-good-evaluators-for","slug":"are-large-language-models-good-evaluators-for","title":"Large Language Models are Not Yet Human-Level Evaluators for Abstractive Summarization","date":"2023-05-22","arxiv_id":"2305.13091","n_code_links":1,"syntology":{"ran":3,"of":3,"n_ran_checked":0,"n_instrument":3,"unverified":0,"pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","official":{"repos":["damo-nlp-sg/llm_summeval"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/beneath-surface-similarity-large-language","slug":"beneath-surface-similarity-large-language","title":"Beneath Surface Similarity: Large Language Models Make Reasonable Scientific Analogies after Structure Abduction","date":"2023-05-22","arxiv_id":"2305.12660","n_code_links":1,"syntology":null},{"paper":null,"slug":"can-chatgpt-defend-the-truth-automatic","title":"Can ChatGPT Defend its Belief in Truth? Evaluating LLM Reasoning via Debate","date":"2023-05-22","arxiv_id":"2305.13160","n_code_links":0,"syntology":null},{"paper":null,"slug":"cognitive-network-science-reveals-bias-in-gpt","title":"Cognitive network science reveals bias in GPT-3, ChatGPT, and GPT-4 mirroring math anxiety in high-school students","date":"2023-05-22","arxiv_id":"2305.18320","n_code_links":0,"syntology":null},{"paper":"/paper/evaluating-and-enhancing-structural","slug":"evaluating-and-enhancing-structural","title":"Table Meets LLM: Can Large Language Models Understand Structured Table Data? A Benchmark and Empirical Study","date":"2023-05-22","arxiv_id":"2305.13062","n_code_links":1,"syntology":null},{"paper":"/paper/explaincpe-a-free-text-explanation-benchmark","slug":"explaincpe-a-free-text-explanation-benchmark","title":"ExplainCPE: A Free-text Explanation Benchmark of Chinese Pharmacist Examination","date":"2023-05-22","arxiv_id":"2305.12945","n_code_links":1,"syntology":null},{"paper":null,"slug":"g3detector-general-gpt-generated-text","title":"G3Detector: General GPT-Generated Text Detector","date":"2023-05-22","arxiv_id":"2305.12680","n_code_links":0,"syntology":null},{"paper":"/paper/how-language-model-hallucinations-can","slug":"how-language-model-hallucinations-can","title":"How Language Model Hallucinations Can Snowball","date":"2023-05-22","arxiv_id":"2305.13534","n_code_links":1,"syntology":null},{"paper":"/paper/llms-for-knowledge-graph-construction-and","slug":"llms-for-knowledge-graph-construction-and","title":"LLMs for Knowledge Graph Construction and Reasoning: Recent Capabilities and Future Opportunities","date":"2023-05-22","arxiv_id":"2305.13168","n_code_links":1,"syntology":{"ran":0,"of":1,"n_ran_checked":0,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"0 ran · 1 unverified","official":{"repos":["zjunlp/autokg"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"paper":null,"slug":"multi-task-instruction-tuning-of-llama-for","title":"Multi-Task Instruction Tuning of LLaMa for Specific Scenarios: A Preliminary Study on Writing Assistance","date":"2023-05-22","arxiv_id":"2305.13225","n_code_links":0,"syntology":null},{"paper":"/paper/scitab-a-challenging-benchmark-for","slug":"scitab-a-challenging-benchmark-for","title":"SCITAB: A Challenging Benchmark for Compositional Reasoning and Claim Verification on Scientific Tables","date":"2023-05-22","arxiv_id":"2305.13186","n_code_links":1,"syntology":null},{"paper":null,"slug":"a-symbolic-framework-for-systematic","title":"A Symbolic Framework for Evaluating Mathematical Reasoning and Generalisation with Transformers","date":"2023-05-21","arxiv_id":"2305.12563","n_code_links":0,"syntology":null},{"paper":"/paper/contrastive-learning-with-logic-driven-data","slug":"contrastive-learning-with-logic-driven-data","title":"Abstract Meaning Representation-Based Logic-Driven Data Augmentation for Logical Reasoning","date":"2023-05-21","arxiv_id":"2305.12599","n_code_links":1,"syntology":null},{"paper":"/paper/evaluating-the-performance-of-large-language","slug":"evaluating-the-performance-of-large-language","title":"Evaluating the Performance of Large Language Models on GAOKAO Benchmark","date":"2023-05-21","arxiv_id":"2305.12474","n_code_links":1,"syntology":{"ran":5,"of":5,"n_ran_checked":5,"n_instrument":0,"unverified":0,"pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","official":{"repos":["openlmlab/gaokao-bench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"gpt-3-5-vs-gpt-4-evaluating-chatgpt-s","title":"GPT-3.5, GPT-4, or BARD? Evaluating LLMs Reasoning Ability in Zero-Shot Setting and Performance Boosting Through Prompts","date":"2023-05-21","arxiv_id":"2305.12477","n_code_links":0,"syntology":null},{"paper":"/paper/theoremqa-a-theorem-driven-question-answering","slug":"theoremqa-a-theorem-driven-question-answering","title":"TheoremQA: A Theorem-driven Question Answering dataset","date":"2023-05-21","arxiv_id":"2305.12524","n_code_links":1,"syntology":{"ran":10,"of":10,"n_ran_checked":8,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["wenhuchen/theoremqa"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text","official"]}}},{"paper":null,"slug":"experimental-results-from-applying-gpt-4-to","title":"Experimental results from applying GPT-4 to an unpublished formal language","date":"2023-05-20","arxiv_id":"2305.12196","n_code_links":0,"syntology":null},{"paper":"/paper/logicot-logical-chain-of-thought-instruction","slug":"logicot-logical-chain-of-thought-instruction","title":"LogiCoT: Logical Chain-of-Thought Instruction-Tuning","date":"2023-05-20","arxiv_id":"2305.12147","n_code_links":1,"syntology":null},{"paper":"/paper/diving-into-the-inter-consistency-of-large","slug":"diving-into-the-inter-consistency-of-large","title":"Examining Inter-Consistency of Large Language Models Collaboration: An In-depth Analysis via Debate","date":"2023-05-19","arxiv_id":"2305.11595","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["waste-wood/ford"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/self-qa-unsupervised-knowledge-guided","slug":"self-qa-unsupervised-knowledge-guided","title":"Self-QA: Unsupervised Knowledge Guided Language Model Alignment","date":"2023-05-19","arxiv_id":"2305.11952","n_code_links":1,"syntology":null},{"paper":"/paper/generalized-planning-in-pddl-domains-with","slug":"generalized-planning-in-pddl-domains-with","title":"Generalized Planning in PDDL Domains with Pretrained Large Language Models","date":"2023-05-18","arxiv_id":"2305.11014","n_code_links":1,"syntology":{"ran":1,"of":2,"n_ran_checked":1,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["tomsilver/llm-genplan"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/lima-less-is-more-for-alignment","slug":"lima-less-is-more-for-alignment","title":"LIMA: Less Is More for Alignment","date":"2023-05-18","arxiv_id":"2305.11206","n_code_links":5,"syntology":null},{"paper":null,"slug":"large-scale-text-analysis-using-generative","title":"Large-Scale Text Analysis Using Generative Language Models: A Case Study in Discovering Public Value Expressions in AI Patents","date":"2023-05-17","arxiv_id":"2305.10383","n_code_links":0,"syntology":null},{"paper":null,"slug":"qualifying-chinese-medical-licensing","title":"Large Language Models Leverage External Knowledge to Extend Clinical Insight Beyond Language Boundaries","date":"2023-05-17","arxiv_id":"2305.10163","n_code_links":0,"syntology":null},{"paper":"/paper/tree-of-thoughts-deliberate-problem-solving-1","slug":"tree-of-thoughts-deliberate-problem-solving-1","title":"Tree of Thoughts: Deliberate Problem Solving with Large Language Models","date":"2023-05-17","arxiv_id":"2305.10601","n_code_links":6,"syntology":{"ran":19,"of":24,"n_ran_checked":18,"n_instrument":1,"unverified":5,"pointer_only":1,"phrase":"19 ran (of which 6 constructed an object rather than computing a result; 18 with no instrument failure: 0 honoured, 0 violated, 18 with no contract checked; 1 where Syntology's instrument failed) · 5 unverified","official":{"repos":["princeton-nlp/tree-of-thought-llm","ysymyth/tree-of-thought-llm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["community","listed","official"]}}},{"paper":"/paper/c-eval-a-multi-level-multi-discipline-chinese-1","slug":"c-eval-a-multi-level-multi-discipline-chinese-1","title":"C-Eval: A Multi-Level Multi-Discipline Chinese Evaluation Suite for Foundation Models","date":"2023-05-15","arxiv_id":"2305.08322","n_code_links":1,"syntology":{"ran":1,"of":1,"n_ran_checked":0,"n_instrument":1,"unverified":0,"pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","official":{"repos":["hkust-nlp/ceval"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":null,"slug":"sensitivity-and-robustness-of-large-language","title":"Sensitivity and Robustness of Large Language Models to Prompt Template in Japanese Text Classification Tasks","date":"2023-05-15","arxiv_id":"2305.08714","n_code_links":0,"syntology":null},{"paper":"/paper/small-models-are-valuable-plug-ins-for-large","slug":"small-models-are-valuable-plug-ins-for-large","title":"Small Models are Valuable Plug-ins for Large Language Models","date":"2023-05-15","arxiv_id":"2305.08848","n_code_links":1,"syntology":{"ran":2,"of":2,"n_ran_checked":0,"n_instrument":2,"unverified":0,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","official":{"repos":["JetRunner/SuperICL"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"paper":"/paper/investigating-emergent-goal-like-behaviour-in","slug":"investigating-emergent-goal-like-behaviour-in","title":"The Machine Psychology of Cooperation: Can GPT models operationalise prompts for altruism, cooperation, competitiveness and selfishness in economic games?","date":"2023-05-13","arxiv_id":"2305.07970","n_code_links":2,"syntology":{"ran":2,"of":3,"n_ran_checked":2,"n_instrument":0,"unverified":1,"pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","official":{"repos":["phelps-sg/llm-cooperation","gitlab.com/sphelps/llm-cooperation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"paper":"/paper/artgpt-4-artistic-vision-language","slug":"artgpt-4-artistic-vision-language","title":"ArtGPT-4: Towards Artistic-understanding Large Vision-Language Models with Enhanced Adapter","date":"2023-05-12","arxiv_id":"2305.07490","n_code_links":1,"syntology":null},{"paper":null,"slug":"dr-llama-improving-small-language-models-in","title":"Improving Small Language Models on PubMedQA via Generative Data Augmentation","date":"2023-05-12","arxiv_id":"2305.07804","n_code_links":0,"syntology":null},{"paper":"/paper/tinystories-how-small-can-language-models-be","slug":"tinystories-how-small-can-language-models-be","title":"TinyStories: How Small Can Language Models Be and Still Speak Coherent English?","date":"2023-05-12","arxiv_id":"2305.07759","n_code_links":8,"syntology":{"ran":10,"of":18,"n_ran_checked":8,"n_instrument":2,"unverified":8,"pointer_only":0,"phrase":"10 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 8 unverified","official":null}},{"paper":null,"slug":"large-language-models-can-be-used-to","title":"Spear Phishing With Large Language Models","date":"2023-05-11","arxiv_id":"2305.06972","n_code_links":0,"syntology":null},{"paper":"/paper/the-conceptarc-benchmark-evaluating","slug":"the-conceptarc-benchmark-evaluating","title":"The ConceptARC Benchmark: Evaluating Understanding and Generalization in the ARC Domain","date":"2023-05-11","arxiv_id":"2305.07141","n_code_links":1,"syntology":null},{"paper":"/paper/are-chatgpt-and-gpt-4-general-purpose-solvers","slug":"are-chatgpt-and-gpt-4-general-purpose-solvers","title":"Are ChatGPT and GPT-4 General-Purpose Solvers for Financial Text Analytics? A Study on Several Typical Tasks","date":"2023-05-10","arxiv_id":"2305.05862","n_code_links":0,"syntology":null}],"record_sha256":"90eebbcc67f6aeef335032d87725efaa5eb8e1330d21b9f2677341025b7e226c","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}