{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/large-language-model/papers/3","list_of":"/task/large-language-model","task":"Large Language Model","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":3,"pages_in_order":61,"rows_per_page":100,"rows":[201,300],"of":6097,"counts":{"archive_papers_tagged":6097,"with_a_code_link":2250,"where_syntology_ran_a_sample":801,"not_listed_spam_title":0,"listed":6097,"listed_where_code_ran":801,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":683,"every_run_a_failure_of_syntologys_instrument":118,"listed_with_a_run_with_no_instrument_failure":683,"listed_every_run_a_failure_of_syntologys_instrument":118,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/large-language-model","prev":"/task/large-language-model/papers/2","next":"/task/large-language-model/papers/4","papers":[{"url":"/paper/omniquant-omnidirectionally-calibrated","slug":"omniquant-omnidirectionally-calibrated","title":"OmniQuant: Omnidirectionally Calibrated Quantization for Large Language Models","date":"2023-08-25","arxiv_id":"2308.13137","repositories_listed":2,"syntology":{"n":16,"n_ran":10,"n_constructed":1,"n_ran_checked":8,"n_instrument":2,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":9,"phrase":"10 ran (of which 1 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 2 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/omniquant-omnidirectionally-calibrated#ran","syntology_url":"https://syntology.ai/paper/2308.13137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.13137"}},"official":{"repos":["opengvlab/omniquant"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":6,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/scieval-a-multi-level-large-language-model","slug":"scieval-a-multi-level-large-language-model","title":"SciEval: A Multi-Level Large Language Model Evaluation Benchmark for Scientific Research","date":"2023-08-25","arxiv_id":"2308.13149","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/scieval-a-multi-level-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2308.13149","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.13149"}},"official":{"repos":["opendfm/bai-scieval","opendfm/scieval"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/large-multilingual-models-pivot-zero-shot","slug":"large-multilingual-models-pivot-zero-shot","title":"Large Multilingual Models Pivot Zero-Shot Multimodal Learning across Languages","date":"2023-08-23","arxiv_id":"2308.12038","repositories_listed":2,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/large-multilingual-models-pivot-zero-shot#ran","syntology_url":"https://syntology.ai/paper/2308.12038","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.12038"}},"official":{"repos":["openbmb/viscpm"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/a-survey-on-large-language-model-based","slug":"a-survey-on-large-language-model-based","title":"A Survey on Large Language Model based Autonomous Agents","date":"2023-08-22","arxiv_id":"2308.11432","repositories_listed":2,"syntology":null},{"url":"/paper/chat-3d-data-efficiently-tuning-large","slug":"chat-3d-data-efficiently-tuning-large","title":"Chat-3D: Data-efficiently Tuning Large Language Model for Universal Dialogue of 3D Scenes","date":"2023-08-17","arxiv_id":"2308.08769","repositories_listed":2,"syntology":null},{"url":"/paper/mt4crossoie-multi-stage-tuning-for-cross","slug":"mt4crossoie-multi-stage-tuning-for-cross","title":"MT4CrossOIE: Multi-stage Tuning for Cross-lingual Open Information Extraction","date":"2023-08-12","arxiv_id":"2308.06552","repositories_listed":2,"syntology":null},{"url":"/paper/esrl-efficient-sampling-based-reinforcement","slug":"esrl-efficient-sampling-based-reinforcement","title":"ESRL: Efficient Sampling-based Reinforcement Learning for Sequence Generation","date":"2023-08-04","arxiv_id":"2308.02223","repositories_listed":2,"syntology":{"n":14,"n_ran":12,"n_constructed":0,"n_ran_checked":11,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":3,"phrase":"12 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/esrl-efficient-sampling-based-reinforcement#ran","syntology_url":"https://syntology.ai/paper/2308.02223","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2308.02223"}},"official":{"repos":["wangclnlp/DeepSpeed-Chat-Extension"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/lisa-reasoning-segmentation-via-large","slug":"lisa-reasoning-segmentation-via-large","title":"LISA: Reasoning Segmentation via Large Language Model","date":"2023-08-01","arxiv_id":"2308.00692","repositories_listed":2,"syntology":null},{"url":"/paper/lp-musiccaps-llm-based-pseudo-music","slug":"lp-musiccaps-llm-based-pseudo-music","title":"LP-MusicCaps: LLM-Based Pseudo Music Captioning","date":"2023-07-31","arxiv_id":"2307.16372","repositories_listed":2,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":6,"n_pointer_only":4,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 1 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/lp-musiccaps-llm-based-pseudo-music#ran","syntology_url":"https://syntology.ai/paper/2307.16372","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.16372"}},"official":{"repos":["seungheondoh/lp-music-caps"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/scaling-transnormer-to-175-billion-parameters","slug":"scaling-transnormer-to-175-billion-parameters","title":"TransNormerLLM: A Faster and Better Large Language Model with Improved TransNormer","date":"2023-07-27","arxiv_id":"2307.14995","repositories_listed":2,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":2,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/scaling-transnormer-to-175-billion-parameters#ran","syntology_url":"https://syntology.ai/paper/2307.14995","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.14995"}},"official":{"repos":["opennlplab/transnormerllm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/in-context-autoencoder-for-context","slug":"in-context-autoencoder-for-context","title":"In-context Autoencoder for Context Compression in a Large Language Model","date":"2023-07-13","arxiv_id":"2307.06945","repositories_listed":2,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/in-context-autoencoder-for-context#ran","syntology_url":"https://syntology.ai/paper/2307.06945","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2307.06945"}},"official":{"repos":["getao/icae"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/composing-parameter-efficient-modules-with","slug":"composing-parameter-efficient-modules-with","title":"Composing Parameter-Efficient Modules with Arithmetic Operations","date":"2023-06-26","arxiv_id":"2306.14870","repositories_listed":2,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":4,"n_pointer_only":4,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 1 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/composing-parameter-efficient-modules-with#ran","syntology_url":"https://syntology.ai/paper/2306.14870","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.14870"}},"official":{"repos":["hkust-nlp/pem_composition","sjtu-lit/pem_composition"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/kosmos-2-grounding-multimodal-large-language","slug":"kosmos-2-grounding-multimodal-large-language","title":"Kosmos-2: Grounding Multimodal Large Language Models to the World","date":"2023-06-26","arxiv_id":"2306.14824","repositories_listed":2,"syntology":null},{"url":"/paper/webglm-towards-an-efficient-web-enhanced","slug":"webglm-towards-an-efficient-web-enhanced","title":"WebGLM: Towards An Efficient Web-Enhanced Question Answering System with Human Preferences","date":"2023-06-13","arxiv_id":"2306.07906","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/webglm-towards-an-efficient-web-enhanced#ran","syntology_url":"https://syntology.ai/paper/2306.07906","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.07906"}},"official":{"repos":["thudm/webglm"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}},{"url":"/paper/14-examples-of-how-llms-can-transform","slug":"14-examples-of-how-llms-can-transform","title":"14 Examples of How LLMs Can Transform Materials Science and Chemistry: A Reflection on a Large Language Model Hackathon","date":"2023-06-09","arxiv_id":"2306.06283","repositories_listed":2,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/14-examples-of-how-llms-can-transform#ran","syntology_url":"https://syntology.ai/paper/2306.06283","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.06283"}},"official":{"repos":["qai222/llm_organic_synthesis","doncamilom/bollama"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/fingpt-open-source-financial-large-language","slug":"fingpt-open-source-financial-large-language","title":"FinGPT: Open-Source Financial Large Language Models","date":"2023-06-09","arxiv_id":"2306.06031","repositories_listed":2,"syntology":{"n":11,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/fingpt-open-source-financial-large-language#ran","syntology_url":"https://syntology.ai/paper/2306.06031","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.06031"}},"official":{"repos":["ai4finance-foundation/fingpt","ai4finance-foundation/finnlp"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/pandalm-an-automatic-evaluation-benchmark-for","slug":"pandalm-an-automatic-evaluation-benchmark-for","title":"PandaLM: An Automatic Evaluation Benchmark for LLM Instruction Tuning Optimization","date":"2023-06-08","arxiv_id":"2306.05087","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pandalm-an-automatic-evaluation-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2306.05087","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.05087"}},"official":{"repos":["weopenml/pandalm"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/pixiu-a-large-language-model-instruction-data","slug":"pixiu-a-large-language-model-instruction-data","title":"PIXIU: A Large Language Model, Instruction Data and Evaluation Benchmark for Finance","date":"2023-06-08","arxiv_id":"2306.05443","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/pixiu-a-large-language-model-instruction-data#ran","syntology_url":"https://syntology.ai/paper/2306.05443","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2306.05443"}},"official":{"repos":["chancefocus/pixiu"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/llmzip-lossless-text-compression-using-large","slug":"llmzip-lossless-text-compression-using-large","title":"LLMZip: Lossless Text Compression using Large Language Models","date":"2023-06-06","arxiv_id":"2306.04050","repositories_listed":2,"syntology":null},{"url":"/paper/kl-divergence-guided-temperature-sampling","slug":"kl-divergence-guided-temperature-sampling","title":"KL-Divergence Guided Temperature Sampling","date":"2023-06-02","arxiv_id":"2306.01286","repositories_listed":2,"syntology":null},{"url":"/paper/vast-a-vision-audio-subtitle-text-omni-1","slug":"vast-a-vision-audio-subtitle-text-omni-1","title":"VAST: A Vision-Audio-Subtitle-Text Omni-Modality Foundation Model and Dataset","date":"2023-05-29","arxiv_id":"2305.18500","repositories_listed":2,"syntology":{"n":42,"n_ran":35,"n_constructed":4,"n_ran_checked":29,"n_instrument":6,"n_unverified":7,"n_honours":2,"n_violates":1,"n_no_contract":26,"n_pointer_only":8,"phrase":"35 ran (of which 4 constructed an object rather than computing a result; 29 with no instrument failure: 2 honoured, 1 violated, 26 with no contract checked; 6 where Syntology's instrument failed) · 7 unverified","sample_list":"/paper/vast-a-vision-audio-subtitle-text-omni-1#ran","syntology_url":"https://syntology.ai/paper/2305.18500","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.18500"}},"official":{"repos":["txh-mercury/vast"],"state":"official (archive's flag): 12 ran","n_ran":12,"n_constructed":4,"n_ran_no_instrument_failure":12,"n_unverified":7,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/language-models-can-improve-event-prediction-1","slug":"language-models-can-improve-event-prediction-1","title":"Language Models Can Improve Event Prediction by Few-Shot Abductive Reasoning","date":"2023-05-26","arxiv_id":"2305.16646","repositories_listed":2,"syntology":{"n":24,"n_ran":19,"n_constructed":4,"n_ran_checked":17,"n_instrument":2,"n_unverified":5,"n_honours":0,"n_violates":1,"n_no_contract":16,"n_pointer_only":4,"phrase":"19 ran (of which 4 constructed an object rather than computing a result; 17 with no instrument failure: 0 honoured, 1 violated, 16 with no contract checked; 2 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/language-models-can-improve-event-prediction-1#ran","syntology_url":"https://syntology.ai/paper/2305.16646","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.16646"}},"official":{"repos":["ant-research/easytemporalpointprocess","ilampard/lamp"],"state":"official (archive's flag): 14 ran","n_ran":14,"n_constructed":4,"n_ran_no_instrument_failure":12,"n_unverified":3,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/expertprompting-instructing-large-language","slug":"expertprompting-instructing-large-language","title":"ExpertPrompting: Instructing Large Language Models to be Distinguished Experts","date":"2023-05-24","arxiv_id":"2305.14688","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/expertprompting-instructing-large-language#ran","syntology_url":"https://syntology.ai/paper/2305.14688","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.14688"}},"official":{"repos":["ofa-sys/expertllama"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/huatuogpt-towards-taming-language-model-to-be","slug":"huatuogpt-towards-taming-language-model-to-be","title":"HuatuoGPT, towards Taming Language Model to Be a Doctor","date":"2023-05-24","arxiv_id":"2305.15075","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/huatuogpt-towards-taming-language-model-to-be#ran","syntology_url":"https://syntology.ai/paper/2305.15075","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.15075"}},"official":{"repos":["freedomintelligence/huatuogpt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-grounded-diffusion-enhancing-prompt","slug":"llm-grounded-diffusion-enhancing-prompt","title":"LLM-grounded Diffusion: Enhancing Prompt Understanding of Text-to-Image Diffusion Models with Large Language Models","date":"2023-05-23","arxiv_id":"2305.13655","repositories_listed":2,"syntology":null},{"url":"/paper/reticl-sequential-retrieval-of-in-context","slug":"reticl-sequential-retrieval-of-in-context","title":"RetICL: Sequential Retrieval of In-Context Examples with Reinforcement Learning","date":"2023-05-23","arxiv_id":"2305.14502","repositories_listed":2,"syntology":null},{"url":"/paper/recurrentgpt-interactive-generation-of","slug":"recurrentgpt-interactive-generation-of","title":"RecurrentGPT: Interactive Generation of (Arbitrarily) Long Text","date":"2023-05-22","arxiv_id":"2305.13304","repositories_listed":2,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":2,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/recurrentgpt-interactive-generation-of#ran","syntology_url":"https://syntology.ai/paper/2305.13304","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.13304"}},"official":{"repos":["aiwaves-cn/recurrentgpt"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed","official"]}}},{"url":"/paper/clinical-camel-an-open-source-expert-level","slug":"clinical-camel-an-open-source-expert-level","title":"Clinical Camel: An Open Expert-Level Medical Language Model with Dialogue-Based Knowledge Encoding","date":"2023-05-19","arxiv_id":"2305.12031","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/clinical-camel-an-open-source-expert-level#ran","syntology_url":"https://syntology.ai/paper/2305.12031","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.12031"}},"official":{"repos":["bowang-lab/clinical-camel"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/visionllm-large-language-model-is-also-an","slug":"visionllm-large-language-model-is-also-an","title":"VisionLLM: Large Language Model is also an Open-Ended Decoder for Vision-Centric Tasks","date":"2023-05-18","arxiv_id":"2305.11175","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 1 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/visionllm-large-language-model-is-also-an#ran","syntology_url":"https://syntology.ai/paper/2305.11175","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.11175"}},"official":{"repos":["opengvlab/visionllm","opengvlab/interngpt"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/pmc-vqa-visual-instruction-tuning-for-medical","slug":"pmc-vqa-visual-instruction-tuning-for-medical","title":"PMC-VQA: Visual Instruction Tuning for Medical Visual Question Answering","date":"2023-05-17","arxiv_id":"2305.10415","repositories_listed":2,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":3,"n_pointer_only":2,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 1 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/pmc-vqa-visual-instruction-tuning-for-medical#ran","syntology_url":"https://syntology.ai/paper/2305.10415","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.10415"}},"official":{"repos":["xiaoman-zhang/PMC-VQA"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/investigating-emergent-goal-like-behaviour-in","slug":"investigating-emergent-goal-like-behaviour-in","title":"The Machine Psychology of Cooperation: Can GPT models operationalise prompts for altruism, cooperation, competitiveness and selfishness in economic games?","date":"2023-05-13","arxiv_id":"2305.07970","repositories_listed":2,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/investigating-emergent-goal-like-behaviour-in#ran","syntology_url":"https://syntology.ai/paper/2305.07970","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.07970"}},"official":{"repos":["phelps-sg/llm-cooperation","gitlab.com/sphelps/llm-cooperation"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/x-llm-bootstrapping-advanced-large-language","slug":"x-llm-bootstrapping-advanced-large-language","title":"X-LLM: Bootstrapping Advanced Large Language Models by Treating Multi-Modalities as Foreign Languages","date":"2023-05-07","arxiv_id":"2305.04160","repositories_listed":2,"syntology":{"n":11,"n_ran":8,"n_constructed":0,"n_ran_checked":4,"n_instrument":4,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":2,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 4 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/x-llm-bootstrapping-advanced-large-language#ran","syntology_url":"https://syntology.ai/paper/2305.04160","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2305.04160"}},"official":null}},{"url":"/paper/how-to-unleash-the-power-of-large-language","slug":"how-to-unleash-the-power-of-large-language","title":"How to Unleash the Power of Large Language Models for Few-shot Relation Extraction?","date":"2023-05-02","arxiv_id":"2305.01555","repositories_listed":2,"syntology":null},{"url":"/paper/assessing-working-memory-capacity-of-chatgpt","slug":"assessing-working-memory-capacity-of-chatgpt","title":"Working Memory Capacity of ChatGPT: An Empirical Study","date":"2023-04-30","arxiv_id":"2305.03731","repositories_listed":2,"syntology":null},{"url":"/paper/teaching-large-language-models-to-self-debug","slug":"teaching-large-language-models-to-self-debug","title":"Teaching Large Language Models to Self-Debug","date":"2023-04-11","arxiv_id":"2304.05128","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/teaching-large-language-models-to-self-debug#ran","syntology_url":"https://syntology.ai/paper/2304.05128","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2304.05128"}},"official":null}},{"url":"/paper/inference-with-reference-lossless","slug":"inference-with-reference-lossless","title":"Inference with Reference: Lossless Acceleration of Large Language Models","date":"2023-04-10","arxiv_id":"2304.04487","repositories_listed":2,"syntology":null},{"url":"/paper/bloomberggpt-a-large-language-model-for","slug":"bloomberggpt-a-large-language-model-for","title":"BloombergGPT: A Large Language Model for Finance","date":"2023-03-30","arxiv_id":"2303.17564","repositories_listed":2,"syntology":null},{"url":"/paper/evaluation-of-chatgpt-as-a-question-answering","slug":"evaluation-of-chatgpt-as-a-question-answering","title":"Can ChatGPT Replace Traditional KBQA Models? An In-depth Analysis of the Question Answering Performance of the GPT LLM Family","date":"2023-03-14","arxiv_id":"2303.07992","repositories_listed":2,"syntology":null},{"url":"/paper/palm-e-an-embodied-multimodal-language-model","slug":"palm-e-an-embodied-multimodal-language-model","title":"PaLM-E: An Embodied Multimodal Language Model","date":"2023-03-06","arxiv_id":"2303.03378","repositories_listed":2,"syntology":null},{"url":"/paper/adaptive-test-generation-using-a-large","slug":"adaptive-test-generation-using-a-large","title":"An Empirical Evaluation of Using Large Language Models for Automated Unit Test Generation","date":"2023-02-13","arxiv_id":"2302.06527","repositories_listed":2,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/adaptive-test-generation-using-a-large#ran","syntology_url":"https://syntology.ai/paper/2302.06527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2302.06527"}},"official":{"repos":["githubnext/testpilot"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/vicarious-offense-and-noise-audit-of","slug":"vicarious-offense-and-noise-audit-of","title":"Vicarious Offense and Noise Audit of Offensive Speech Classifiers: Unifying Human and Machine Disagreement on What is Offensive","date":"2023-01-29","arxiv_id":"2301.12534","repositories_listed":2,"syntology":null},{"url":"/paper/batch-prompting-efficient-inference-with","slug":"batch-prompting-efficient-inference-with","title":"Batch Prompting: Efficient Inference with Large Language Model APIs","date":"2023-01-19","arxiv_id":"2301.08721","repositories_listed":2,"syntology":null},{"url":"/paper/emergent-analogical-reasoning-in-large","slug":"emergent-analogical-reasoning-in-large","title":"Emergent Analogical Reasoning in Large Language Models","date":"2022-12-19","arxiv_id":"2212.09196","repositories_listed":2,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/emergent-analogical-reasoning-in-large#ran","syntology_url":"https://syntology.ai/paper/2212.09196","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2212.09196"}},"official":{"repos":["taylorwwebb/emergent_analogies_llm"],"state":"community repositories only","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["listed"]}}},{"url":"/paper/elixir-train-a-large-language-model-on-a","slug":"elixir-train-a-large-language-model-on-a","title":"Elixir: Train a Large Language Model on a Small GPU Cluster","date":"2022-12-10","arxiv_id":"2212.05339","repositories_listed":2,"syntology":null},{"url":"/paper/attributed-text-generation-via-post-hoc","slug":"attributed-text-generation-via-post-hoc","title":"RARR: Researching and Revising What Language Models Say, Using Language Models","date":"2022-10-17","arxiv_id":"2210.08726","repositories_listed":2,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/attributed-text-generation-via-post-hoc#ran","syntology_url":"https://syntology.ai/paper/2210.08726","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2210.08726"}},"official":{"repos":["anthonywchen/rarr"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/generate-rather-than-retrieve-large-language","slug":"generate-rather-than-retrieve-large-language","title":"Generate rather than Retrieve: Large Language Models are Strong Context Generators","date":"2022-09-21","arxiv_id":"2209.10063","repositories_listed":2,"syntology":{"n":6,"n_ran":5,"n_constructed":1,"n_ran_checked":1,"n_instrument":4,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":6,"phrase":"5 ran (of which 1 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 4 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/generate-rather-than-retrieve-large-language#ran","syntology_url":"https://syntology.ai/paper/2209.10063","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2209.10063"}},"official":{"repos":["wyu97/GenRead"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":1,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/fast-r2d2-a-pretrained-recursive-neural","slug":"fast-r2d2-a-pretrained-recursive-neural","title":"Fast-R2D2: A Pretrained Recursive Neural Network based on Pruned CKY for Grammar Induction and Text Representation","date":"2022-03-01","arxiv_id":"2203.00281","repositories_listed":2,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/fast-r2d2-a-pretrained-recursive-neural#ran","syntology_url":"https://syntology.ai/paper/2203.00281","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2203.00281"}},"official":{"repos":["alipay/StructuredLM_RTDT"],"state":"official: harvested for another paper","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":[]}}},{"url":"/paper/assay2mol-large-language-model-based-drug","slug":"assay2mol-large-language-model-based-drug","title":"Assay2Mol: large language model-based drug design using BioAssay context","date":"2025-07-16","arxiv_id":"2507.12574","repositories_listed":1,"syntology":null},{"url":"/paper/drafterbench-benchmarking-large-language","slug":"drafterbench-benchmarking-large-language","title":"DrafterBench: Benchmarking Large Language Models for Tasks Automation in Civil Engineering","date":"2025-07-15","arxiv_id":"2507.11527","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/drafterbench-benchmarking-large-language#ran","syntology_url":"https://syntology.ai/paper/2507.11527","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.11527"}},"official":{"repos":["eason-li-ais/drafterbench"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/learning-to-tune-like-an-expert-interpretable","slug":"learning-to-tune-like-an-expert-interpretable","title":"Learning to Tune Like an Expert: Interpretable and Scene-Aware Navigation via MLLM Reasoning and CVAE-Based Adaptation","date":"2025-07-15","arxiv_id":"2507.11001","repositories_listed":1,"syntology":null},{"url":"/paper/mfgdiffusion-mask-guided-smoke-synthesis-for","slug":"mfgdiffusion-mask-guided-smoke-synthesis-for","title":"MFGDiffusion: Mask-Guided Smoke Synthesis for Enhanced Forest Fire Detection","date":"2025-07-15","arxiv_id":"2507.11252","repositories_listed":1,"syntology":null},{"url":"/paper/seq-vs-seq-an-open-suite-of-paired-encoders","slug":"seq-vs-seq-an-open-suite-of-paired-encoders","title":"Seq vs Seq: An Open Suite of Paired Encoders and Decoders","date":"2025-07-15","arxiv_id":"2507.11412","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/seq-vs-seq-an-open-suite-of-paired-encoders#ran","syntology_url":"https://syntology.ai/paper/2507.11412","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.11412"}},"official":{"repos":["jhu-clsp/ettin-encoder-vs-decoder"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dynamic-parameter-memory-temporary-lora","slug":"dynamic-parameter-memory-temporary-lora","title":"Dynamic Parameter Memory: Temporary LoRA-Enhanced LLM for Long-Sequence Emotion Recognition in Conversation","date":"2025-07-11","arxiv_id":"2507.09076","repositories_listed":1,"syntology":null},{"url":"/paper/bilateral-collaboration-with-large-vision","slug":"bilateral-collaboration-with-large-vision","title":"Bilateral Collaboration with Large Vision-Language Models for Open Vocabulary Human-Object Interaction Detection","date":"2025-07-09","arxiv_id":"2507.06510","repositories_listed":1,"syntology":null},{"url":"/paper/open-source-planning-control-system-with","slug":"open-source-planning-control-system-with","title":"Open Source Planning & Control System with Language Agents for Autonomous Scientific Discovery","date":"2025-07-09","arxiv_id":"2507.07257","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/open-source-planning-control-system-with#ran","syntology_url":"https://syntology.ai/paper/2507.07257","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.07257"}},"official":{"repos":["cmbagents/cmbagent"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["community","official"]}}},{"url":"/paper/early-signs-of-steganographic-capabilities-in","slug":"early-signs-of-steganographic-capabilities-in","title":"Early Signs of Steganographic Capabilities in Frontier LLMs","date":"2025-07-03","arxiv_id":"2507.02737","repositories_listed":1,"syntology":null},{"url":"/paper/llava-sp-enhancing-visual-representation-with","slug":"llava-sp-enhancing-visual-representation-with","title":"LLaVA-SP: Enhancing Visual Representation with Visual Spatial Tokens for MLLMs","date":"2025-07-01","arxiv_id":"2507.00505","repositories_listed":1,"syntology":{"n":8,"n_ran":8,"n_constructed":0,"n_ran_checked":3,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":1,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llava-sp-enhancing-visual-representation-with#ran","syntology_url":"https://syntology.ai/paper/2507.00505","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2507.00505"}},"official":{"repos":["cnfaker/llava-sp"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/thought-augmented-planning-for-llm-powered","slug":"thought-augmented-planning-for-llm-powered","title":"Thought-Augmented Planning for LLM-Powered Interactive Recommender Agent","date":"2025-06-30","arxiv_id":"2506.23485","repositories_listed":1,"syntology":null},{"url":"/paper/where-what-why-towards-explainable-driver","slug":"where-what-why-towards-explainable-driver","title":"Where, What, Why: Towards Explainable Driver Attention Prediction","date":"2025-06-29","arxiv_id":"2506.23088","repositories_listed":1,"syntology":null},{"url":"/paper/decoupled-seg-tokens-make-stronger-reasoning","slug":"decoupled-seg-tokens-make-stronger-reasoning","title":"Decoupled Seg Tokens Make Stronger Reasoning Video Segmenter and Grounder","date":"2025-06-28","arxiv_id":"2506.22880","repositories_listed":1,"syntology":null},{"url":"/paper/agentstealth-reinforcing-large-language-model","slug":"agentstealth-reinforcing-large-language-model","title":"AgentStealth: Reinforcing Large Language Model for Anonymizing User-generated Text","date":"2025-06-26","arxiv_id":"2506.22508","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":7,"n_pointer_only":0,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 0 honoured, 0 violated, 7 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/agentstealth-reinforcing-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2506.22508","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.22508"}},"official":{"repos":["tsinghua-fib-lab/agentstealth"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/detecting-referring-expressions-in-visually","slug":"detecting-referring-expressions-in-visually","title":"Detecting Referring Expressions in Visually Grounded Dialogue with Autoregressive Language Models","date":"2025-06-26","arxiv_id":"2506.21294","repositories_listed":1,"syntology":null},{"url":"/paper/humanomniv2-from-understanding-to-omni-modal","slug":"humanomniv2-from-understanding-to-omni-modal","title":"HumanOmniV2: From Understanding to Omni-Modal Reasoning with Context","date":"2025-06-26","arxiv_id":"2506.21277","repositories_listed":1,"syntology":null},{"url":"/paper/mtsbench-benchmarking-multivariate-time","slug":"mtsbench-benchmarking-multivariate-time","title":"mTSBench: Benchmarking Multivariate Time Series Anomaly Detection and Model Selection at Scale","date":"2025-06-26","arxiv_id":"2506.21550","repositories_listed":1,"syntology":null},{"url":"/paper/oraclefusion-assisting-the-decipherment-of","slug":"oraclefusion-assisting-the-decipherment-of","title":"OracleFusion: Assisting the Decipherment of Oracle Bone Script with Structurally Constrained Semantic Typography","date":"2025-06-26","arxiv_id":"2506.21101","repositories_listed":1,"syntology":null},{"url":"/paper/psylite-technical-report","slug":"psylite-technical-report","title":"PsyLite Technical Report","date":"2025-06-26","arxiv_id":"2506.21536","repositories_listed":1,"syntology":null},{"url":"/paper/thinksound-chain-of-thought-reasoning-in","slug":"thinksound-chain-of-thought-reasoning-in","title":"ThinkSound: Chain-of-Thought Reasoning in Multimodal Large Language Models for Audio Generation and Editing","date":"2025-06-26","arxiv_id":"2506.21448","repositories_listed":1,"syntology":null},{"url":"/paper/a-multi-pass-large-language-model-framework","slug":"a-multi-pass-large-language-model-framework","title":"A Multi-Pass Large Language Model Framework for Precise and Efficient Radiology Report Error Detection","date":"2025-06-25","arxiv_id":"2506.20112","repositories_listed":1,"syntology":null},{"url":"/paper/aalc-large-language-model-efficient-reasoning","slug":"aalc-large-language-model-efficient-reasoning","title":"AALC: Large Language Model Efficient Reasoning via Adaptive Accuracy-Length Control","date":"2025-06-25","arxiv_id":"2506.20160","repositories_listed":1,"syntology":null},{"url":"/paper/gptailor-large-language-model-pruning-through","slug":"gptailor-large-language-model-pruning-through","title":"GPTailor: Large Language Model Pruning Through Layer Cutting and Stitching","date":"2025-06-25","arxiv_id":"2506.20480","repositories_listed":1,"syntology":null},{"url":"/paper/narrative-shift-detection-a-hybrid-approach","slug":"narrative-shift-detection-a-hybrid-approach","title":"Narrative Shift Detection: A Hybrid Approach of Dynamic Topic Models and Large Language Models","date":"2025-06-25","arxiv_id":"2506.20269","repositories_listed":1,"syntology":null},{"url":"/paper/towards-community-driven-agents-for-machine","slug":"towards-community-driven-agents-for-machine","title":"Towards Community-Driven Agents for Machine Learning Engineering","date":"2025-06-25","arxiv_id":"2506.20640","repositories_listed":1,"syntology":{"n":4,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/towards-community-driven-agents-for-machine#ran","syntology_url":"https://syntology.ai/paper/2506.20640","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.20640"}},"official":{"repos":["comind-ml/comind"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/confucius3-math-a-lightweight-high","slug":"confucius3-math-a-lightweight-high","title":"Confucius3-Math: A Lightweight High-Performance Reasoning LLM for Chinese K-12 Mathematics Learning","date":"2025-06-23","arxiv_id":"2506.18330","repositories_listed":1,"syntology":null},{"url":"/paper/medtvt-r1-a-multimodal-llm-empowering-medical","slug":"medtvt-r1-a-multimodal-llm-empowering-medical","title":"MedTVT-R1: A Multimodal LLM Empowering Medical Reasoning and Diagnosis","date":"2025-06-23","arxiv_id":"2506.18512","repositories_listed":1,"syntology":null},{"url":"/paper/evolving-prompts-in-context-an-open-ended","slug":"evolving-prompts-in-context-an-open-ended","title":"Evolving Prompts In-Context: An Open-ended, Self-replicating Perspective","date":"2025-06-22","arxiv_id":"2506.17930","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/evolving-prompts-in-context-an-open-ended#ran","syntology_url":"https://syntology.ai/paper/2506.17930","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.17930"}},"official":{"repos":["jianyu-cs/promptquine"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mechanistic-interpretability-in-the-presence","slug":"mechanistic-interpretability-in-the-presence","title":"Mechanistic Interpretability in the Presence of Architectural Obfuscation","date":"2025-06-22","arxiv_id":"2506.18053","repositories_listed":1,"syntology":null},{"url":"/paper/pre-trained-llm-is-a-semantic-aware-and","slug":"pre-trained-llm-is-a-semantic-aware-and","title":"Pre-Trained LLM is a Semantic-Aware and Generalizable Segmentation Booster","date":"2025-06-22","arxiv_id":"2506.18034","repositories_listed":1,"syntology":null},{"url":"/paper/sharegpt-4o-image-aligning-multimodal-models","slug":"sharegpt-4o-image-aligning-multimodal-models","title":"ShareGPT-4o-Image: Aligning Multimodal Models with GPT-4o-Level Image Generation","date":"2025-06-22","arxiv_id":"2506.18095","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/sharegpt-4o-image-aligning-multimodal-models#ran","syntology_url":"https://syntology.ai/paper/2506.18095","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.18095"}},"official":{"repos":["freedomintelligence/sharegpt-4o-image"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/drama-x-a-fine-grained-intent-prediction-and","slug":"drama-x-a-fine-grained-intent-prediction-and","title":"DRAMA-X: A Fine-grained Intent Prediction and Risk Reasoning Benchmark For Driving","date":"2025-06-21","arxiv_id":"2506.17590","repositories_listed":1,"syntology":null},{"url":"/paper/lmr-bench-evaluating-llm-agent-s-ability-on","slug":"lmr-bench-evaluating-llm-agent-s-ability-on","title":"LMR-BENCH: Evaluating LLM Agent's Ability on Reproducing Language Modeling Research","date":"2025-06-19","arxiv_id":"2506.17335","repositories_listed":1,"syntology":{"n":6,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":6,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/lmr-bench-evaluating-llm-agent-s-ability-on#ran","syntology_url":"https://syntology.ai/paper/2506.17335","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.17335"}},"official":{"repos":["du-nlp-lab/lmr-bench"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/probe-before-you-talk-towards-black-box","slug":"probe-before-you-talk-towards-black-box","title":"Probe before You Talk: Towards Black-box Defense against Backdoor Unalignment for Large Language Models","date":"2025-06-19","arxiv_id":"2506.16447","repositories_listed":1,"syntology":null},{"url":"/paper/the-condition-number-as-a-scale-invariant","slug":"the-condition-number-as-a-scale-invariant","title":"The Condition Number as a Scale-Invariant Proxy for Information Encoding in Neural Units","date":"2025-06-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/agentgroupchat-v2-divide-and-conquer-is-what","slug":"agentgroupchat-v2-divide-and-conquer-is-what","title":"AgentGroupChat-V2: Divide-and-Conquer Is What LLM-Based Multi-Agent System Need","date":"2025-06-18","arxiv_id":"2506.15451","repositories_listed":1,"syntology":null},{"url":"/paper/ras-eval-a-comprehensive-benchmark-for","slug":"ras-eval-a-comprehensive-benchmark-for","title":"RAS-Eval: A Comprehensive Benchmark for Security Evaluation of LLM Agents in Real-World Environments","date":"2025-06-18","arxiv_id":"2506.15253","repositories_listed":1,"syntology":{"n":4,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":4,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/ras-eval-a-comprehensive-benchmark-for#ran","syntology_url":"https://syntology.ai/paper/2506.15253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.15253"}},"official":{"repos":["lanzer-tree/ras-eval"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/sonicverse-multi-task-learning-for-music","slug":"sonicverse-multi-task-learning-for-music","title":"SonicVerse: Multi-Task Learning for Music Feature-Informed Captioning","date":"2025-06-18","arxiv_id":"2506.15154","repositories_listed":1,"syntology":null},{"url":"/paper/video-salmonn-2-captioning-enhanced-audio","slug":"video-salmonn-2-captioning-enhanced-audio","title":"video-SALMONN 2: Captioning-Enhanced Audio-Visual Large Language Models","date":"2025-06-18","arxiv_id":"2506.15220","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/video-salmonn-2-captioning-enhanced-audio#ran","syntology_url":"https://syntology.ai/paper/2506.15220","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.15220"}},"official":{"repos":["bytedance/video-salmonn-2"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/from-what-to-respond-to-when-to-respond","slug":"from-what-to-respond-to-when-to-respond","title":"From What to Respond to When to Respond: Timely Response Generation for Open-domain Dialogue Agents","date":"2025-06-17","arxiv_id":"2506.14285","repositories_listed":1,"syntology":null},{"url":"/paper/rmit-adm-s-at-the-sigir-2025-liverag","slug":"rmit-adm-s-at-the-sigir-2025-liverag","title":"RMIT-ADM+S at the SIGIR 2025 LiveRAG Challenge","date":"2025-06-17","arxiv_id":"2506.14516","repositories_listed":1,"syntology":null},{"url":"/paper/emonews-a-spoken-dialogue-system-for","slug":"emonews-a-spoken-dialogue-system-for","title":"EmoNews: A Spoken Dialogue System for Expressive News Conversations","date":"2025-06-16","arxiv_id":"2506.13894","repositories_listed":1,"syntology":null},{"url":"/paper/vis-shepherd-constructing-critic-for-llm","slug":"vis-shepherd-constructing-critic-for-llm","title":"VIS-Shepherd: Constructing Critic for LLM-based Data Visualization Generation","date":"2025-06-16","arxiv_id":"2506.13326","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":6,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/vis-shepherd-constructing-critic-for-llm#ran","syntology_url":"https://syntology.ai/paper/2506.13326","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2506.13326"}},"official":{"repos":["bopan3/vis-shepherd"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/flexrag-a-flexible-and-comprehensive","slug":"flexrag-a-flexible-and-comprehensive","title":"FlexRAG: A Flexible and Comprehensive Framework for Retrieval-Augmented Generation","date":"2025-06-14","arxiv_id":"2506.12494","repositories_listed":1,"syntology":null},{"url":"/paper/tagrouter-learning-route-to-llms-through-tags","slug":"tagrouter-learning-route-to-llms-through-tags","title":"TagRouter: Learning Route to LLMs through Tags for Open-Domain Text Generation Tasks","date":"2025-06-14","arxiv_id":"2506.12473","repositories_listed":1,"syntology":null},{"url":"/paper/from-emergence-to-control-probing-and","slug":"from-emergence-to-control-probing-and","title":"From Emergence to Control: Probing and Modulating Self-Reflection in Language Models","date":"2025-06-13","arxiv_id":"2506.12217","repositories_listed":1,"syntology":null},{"url":"/paper/improving-large-language-model-safety-with","slug":"improving-large-language-model-safety-with","title":"Improving Large Language Model Safety with Contrastive Representation Learning","date":"2025-06-13","arxiv_id":"2506.11938","repositories_listed":1,"syntology":null},{"url":"/paper/sec-bench-automated-benchmarking-of-llm","slug":"sec-bench-automated-benchmarking-of-llm","title":"SEC-bench: Automated Benchmarking of LLM Agents on Real-World Software Security Tasks","date":"2025-06-13","arxiv_id":"2506.11791","repositories_listed":1,"syntology":null},{"url":"/paper/2506-10326","slug":"2506-10326","title":"A Benchmark for Generalizing Across Diverse Team Strategies in Competitive Pokémon","date":"2025-06-12","arxiv_id":"2506.10326","repositories_listed":1,"syntology":null},{"url":"/paper/2506-10678","slug":"2506-10678","title":"Automated Validation of Textual Constraints Against AutomationML via LLMs and SHACL","date":"2025-06-12","arxiv_id":"2506.10678","repositories_listed":1,"syntology":null},{"url":"/paper/2506-10974","slug":"2506-10974","title":"AutoMind: Adaptive Knowledgeable Agent for Automated Data Science","date":"2025-06-12","arxiv_id":"2506.10974","repositories_listed":1,"syntology":null},{"url":"/paper/llm-as-a-fuzzy-judge-fine-tuning-large","slug":"llm-as-a-fuzzy-judge-fine-tuning-large","title":"LLM-as-a-Fuzzy-Judge: Fine-Tuning Large Language Models as a Clinical Evaluation Judge with Fuzzy Logic","date":"2025-06-12","arxiv_id":"2506.11221","repositories_listed":1,"syntology":null},{"url":"/paper/neuralnexus-at-bea-2025-shared-task-retrieval","slug":"neuralnexus-at-bea-2025-shared-task-retrieval","title":"NeuralNexus at BEA 2025 Shared Task: Retrieval-Augmented Prompting for Mistake Identification in AI Tutors","date":"2025-06-12","arxiv_id":"2506.10627","repositories_listed":1,"syntology":null}],"record_sha256":"985bcb8f14f17db89932dcb9f0f3bbe0a6908134c06723cbcc9093e9bb6b5169","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}