{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/large-language-model/papers/13","list_of":"/task/large-language-model","task":"Large Language Model","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":13,"pages_in_order":61,"rows_per_page":100,"rows":[1201,1300],"of":6097,"counts":{"archive_papers_tagged":6097,"with_a_code_link":2250,"where_syntology_ran_a_sample":801,"not_listed_spam_title":0,"listed":6097,"listed_where_code_ran":801,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":683,"every_run_a_failure_of_syntologys_instrument":118,"listed_with_a_run_with_no_instrument_failure":683,"listed_every_run_a_failure_of_syntologys_instrument":118,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/large-language-model","prev":"/task/large-language-model/papers/12","next":"/task/large-language-model/papers/14","papers":[{"url":"/paper/me-myself-and-ai-the-situational-awareness","slug":"me-myself-and-ai-the-situational-awareness","title":"Me, Myself, and AI: The Situational Awareness Dataset (SAD) for LLMs","date":"2024-07-05","arxiv_id":"2407.04694","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/me-myself-and-ai-the-situational-awareness#ran","syntology_url":"https://syntology.ai/paper/2407.04694","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04694"}},"official":{"repos":["lrudl/sad"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/poprero-a-new-dataset-for-popularity","slug":"poprero-a-new-dataset-for-popularity","title":"PoPreRo: A New Dataset for Popularity Prediction of Romanian Reddit Posts","date":"2024-07-05","arxiv_id":"2407.04541","repositories_listed":1,"syntology":null},{"url":"/paper/spikellm-scaling-up-spiking-neural-network-to","slug":"spikellm-scaling-up-spiking-neural-network-to","title":"SpikeLLM: Scaling up Spiking Neural Network to Large Language Models via Saliency-based Spiking","date":"2024-07-05","arxiv_id":"2407.04752","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/spikellm-scaling-up-spiking-neural-network-to#ran","syntology_url":"https://syntology.ai/paper/2407.04752","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04752"}},"official":null}},{"url":"/paper/historical-ink-19th-century-latin-american","slug":"historical-ink-19th-century-latin-american","title":"Historical Ink: 19th Century Latin American Spanish Newspaper Corpus with LLM OCR Correction","date":"2024-07-04","arxiv_id":"2407.12838","repositories_listed":1,"syntology":null},{"url":"/paper/minigpt-med-large-language-model-as-a-general","slug":"minigpt-med-large-language-model-as-a-general","title":"MiniGPT-Med: Large Language Model as a General Interface for Radiology Diagnosis","date":"2024-07-04","arxiv_id":"2407.04106","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":1,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":1,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 1 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/minigpt-med-large-language-model-as-a-general#ran","syntology_url":"https://syntology.ai/paper/2407.04106","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.04106"}},"official":{"repos":["vision-cair/minigpt-med"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/uncertainty-guided-optimization-on-large","slug":"uncertainty-guided-optimization-on-large","title":"Uncertainty-Guided Optimization on Large Language Model Search Trees","date":"2024-07-04","arxiv_id":"2407.03951","repositories_listed":1,"syntology":null},{"url":"/paper/wilddesed-an-llm-powered-dataset-for-wild","slug":"wilddesed-an-llm-powered-dataset-for-wild","title":"WildDESED: An LLM-Powered Dataset for Wild Domestic Environment Sound Event Detection System","date":"2024-07-04","arxiv_id":"2407.03656","repositories_listed":1,"syntology":null},{"url":"/paper/fine-tuning-with-divergent-chains-of-thought","slug":"fine-tuning-with-divergent-chains-of-thought","title":"Fine-Tuning with Divergent Chains of Thought Boosts Reasoning Through Self-Correction in Language Models","date":"2024-07-03","arxiv_id":"2407.03181","repositories_listed":1,"syntology":null},{"url":"/paper/a-bounding-box-is-worth-one-token","slug":"a-bounding-box-is-worth-one-token","title":"A Bounding Box is Worth One Token: Interleaving Layout and Text in a Large Language Model for Document Understanding","date":"2024-07-02","arxiv_id":"2407.01976","repositories_listed":1,"syntology":{"n":7,"n_ran":7,"n_constructed":0,"n_ran_checked":7,"n_instrument":0,"n_unverified":0,"n_honours":2,"n_violates":1,"n_no_contract":4,"n_pointer_only":7,"phrase":"7 ran (of which 0 constructed an object rather than computing a result; 7 with no instrument failure: 2 honoured, 1 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/a-bounding-box-is-worth-one-token#ran","syntology_url":"https://syntology.ai/paper/2407.01976","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01976"}},"official":{"repos":["laytextllm/laytextllm"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":0,"n_ran_no_instrument_failure":7,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/is-your-large-language-model-knowledgeable-or","slug":"is-your-large-language-model-knowledgeable-or","title":"Is Your Large Language Model Knowledgeable or a Choices-Only Cheater?","date":"2024-07-02","arxiv_id":"2407.01992","repositories_listed":1,"syntology":null},{"url":"/paper/tokenpacker-efficient-visual-projector-for","slug":"tokenpacker-efficient-visual-projector-for","title":"TokenPacker: Efficient Visual Projector for Multimodal LLM","date":"2024-07-02","arxiv_id":"2407.02392","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/tokenpacker-efficient-visual-projector-for#ran","syntology_url":"https://syntology.ai/paper/2407.02392","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.02392"}},"official":{"repos":["circleradon/tokenpacker"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/autoflow-automated-workflow-generation-for","slug":"autoflow-automated-workflow-generation-for","title":"AutoFlow: Automated Workflow Generation for Large Language Model Agents","date":"2024-07-01","arxiv_id":"2407.12821","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/autoflow-automated-workflow-generation-for#ran","syntology_url":"https://syntology.ai/paper/2407.12821","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.12821"}},"official":{"repos":["agiresearch/autoflow"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/empathic-grounding-explorations-using","slug":"empathic-grounding-explorations-using","title":"Empathic Grounding: Explorations using Multimodal Interaction and Large Language Models with Conversational Agents","date":"2024-07-01","arxiv_id":"2407.01824","repositories_listed":1,"syntology":null},{"url":"/paper/meerkat-audio-visual-large-language-model-for","slug":"meerkat-audio-visual-large-language-model-for","title":"Meerkat: Audio-Visual Large Language Model for Grounding in Space and Time","date":"2024-07-01","arxiv_id":"2407.01851","repositories_listed":1,"syntology":{"n":12,"n_ran":8,"n_constructed":0,"n_ran_checked":6,"n_instrument":2,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"8 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 2 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/meerkat-audio-visual-large-language-model-for#ran","syntology_url":"https://syntology.ai/paper/2407.01851","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01851"}},"official":{"repos":["schowdhury671/meerkat"],"state":"official (archive's flag): 8 ran","n_ran":8,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/regmix-data-mixture-as-regression-for","slug":"regmix-data-mixture-as-regression-for","title":"RegMix: Data Mixture as Regression for Language Model Pre-training","date":"2024-07-01","arxiv_id":"2407.01492","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/regmix-data-mixture-as-regression-for#ran","syntology_url":"https://syntology.ai/paper/2407.01492","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.01492"}},"official":{"repos":["sail-sg/regmix"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/sinkt-a-structure-aware-inductive-knowledge","slug":"sinkt-a-structure-aware-inductive-knowledge","title":"SINKT: A Structure-Aware Inductive Knowledge Tracing Model with Large Language Model","date":"2024-07-01","arxiv_id":"2407.01245","repositories_listed":1,"syntology":null},{"url":"/paper/teola-towards-end-to-end-optimization-of-llm","slug":"teola-towards-end-to-end-optimization-of-llm","title":"Teola: Towards End-to-End Optimization of LLM-based Applications","date":"2024-06-29","arxiv_id":"2407.00326","repositories_listed":1,"syntology":null},{"url":"/paper/the-factuality-tax-of-diversity-intervened","slug":"the-factuality-tax-of-diversity-intervened","title":"The Factuality Tax of Diversity-Intervened Text-to-Image Generation: Benchmark and Fact-Augmented Intervention","date":"2024-06-29","arxiv_id":"2407.00377","repositories_listed":1,"syntology":null},{"url":"/paper/into-the-unknown-generating-geospatial","slug":"into-the-unknown-generating-geospatial","title":"Into the Unknown: Generating Geospatial Descriptions for New Environments","date":"2024-06-28","arxiv_id":"2406.19967","repositories_listed":1,"syntology":null},{"url":"/paper/molecular-facts-desiderata-for","slug":"molecular-facts-desiderata-for","title":"Molecular Facts: Desiderata for Decontextualization in LLM Fact Verification","date":"2024-06-28","arxiv_id":"2406.20079","repositories_listed":1,"syntology":null},{"url":"/paper/yulan-an-open-source-large-language-model","slug":"yulan-an-open-source-large-language-model","title":"YuLan: An Open-source Large Language Model","date":"2024-06-28","arxiv_id":"2406.19853","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/yulan-an-open-source-large-language-model#ran","syntology_url":"https://syntology.ai/paper/2406.19853","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.19853"}},"official":{"repos":["ruc-gsai/yulan-chat"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/length-optimization-in-conformal-prediction","slug":"length-optimization-in-conformal-prediction","title":"Length Optimization in Conformal Prediction","date":"2024-06-27","arxiv_id":"2406.18814","repositories_listed":1,"syntology":null},{"url":"/paper/a-refer-and-ground-multimodal-large-language","slug":"a-refer-and-ground-multimodal-large-language","title":"A Refer-and-Ground Multimodal Large Language Model for Biomedicine","date":"2024-06-26","arxiv_id":"2406.18146","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/a-refer-and-ground-multimodal-large-language#ran","syntology_url":"https://syntology.ai/paper/2406.18146","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.18146"}},"official":{"repos":["shawnhuang497/bird"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/badge-badminton-report-generation-and","slug":"badge-badminton-report-generation-and","title":"BADGE: BADminton report Generation and Evaluation with LLM","date":"2024-06-26","arxiv_id":"2406.18116","repositories_listed":1,"syntology":null},{"url":"/paper/cascading-large-language-models-for-salient","slug":"cascading-large-language-models-for-salient","title":"Cascading Large Language Models for Salient Event Graph Generation","date":"2024-06-26","arxiv_id":"2406.18449","repositories_listed":1,"syntology":null},{"url":"/paper/s3-a-simple-strong-sample-effective","slug":"s3-a-simple-strong-sample-effective","title":"S3: A Simple Strong Sample-effective Multimodal Dialog System","date":"2024-06-26","arxiv_id":"2406.18305","repositories_listed":1,"syntology":null},{"url":"/paper/cogmg-collaborative-augmentation-between","slug":"cogmg-collaborative-augmentation-between","title":"CogMG: Collaborative Augmentation Between Large Language Model and Knowledge Graph","date":"2024-06-25","arxiv_id":"2406.17231","repositories_listed":1,"syntology":null},{"url":"/paper/cosafe-evaluating-large-language-model-safety","slug":"cosafe-evaluating-large-language-model-safety","title":"CoSafe: Evaluating Large Language Model Safety in Multi-Turn Dialogue Coreference","date":"2024-06-25","arxiv_id":"2406.17626","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-tool-retrieval-with-iterative","slug":"enhancing-tool-retrieval-with-iterative","title":"Enhancing Tool Retrieval with Iterative Feedback from Large Language Models","date":"2024-06-25","arxiv_id":"2406.17465","repositories_listed":1,"syntology":null},{"url":"/paper/from-distributional-to-overton-pluralism","slug":"from-distributional-to-overton-pluralism","title":"From Distributional to Overton Pluralism: Investigating Large Language Model Alignment","date":"2024-06-25","arxiv_id":"2406.17692","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/from-distributional-to-overton-pluralism#ran","syntology_url":"https://syntology.ai/paper/2406.17692","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17692"}},"official":{"repos":["thomlake/investigating-alignment"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/grass-compute-efficient-low-memory-llm","slug":"grass-compute-efficient-low-memory-llm","title":"Grass: Compute Efficient Low-Memory LLM Training with Structured Sparse Gradients","date":"2024-06-25","arxiv_id":"2406.17660","repositories_listed":1,"syntology":null},{"url":"/paper/make-some-noise-unlocking-language-model","slug":"make-some-noise-unlocking-language-model","title":"Make Some Noise: Unlocking Language Model Parallel Inference Capability through Noisy Training","date":"2024-06-25","arxiv_id":"2406.17404","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/make-some-noise-unlocking-language-model#ran","syntology_url":"https://syntology.ai/paper/2406.17404","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.17404"}},"official":{"repos":["wyxstriker/MakeSomeNoiseInference"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/the-alchemist-automated-labeling-500x-cheaper","slug":"the-alchemist-automated-labeling-500x-cheaper","title":"The ALCHEmist: Automated Labeling 500x CHEaper Than LLM Data Annotators","date":"2024-06-25","arxiv_id":"2407.11004","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":1,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-alchemist-automated-labeling-500x-cheaper#ran","syntology_url":"https://syntology.ai/paper/2407.11004","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.11004"}},"official":{"repos":["sprocketlab/alchemist"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-fineweb-datasets-decanting-the-web-for","slug":"the-fineweb-datasets-decanting-the-web-for","title":"The FineWeb Datasets: Decanting the Web for the Finest Text Data at Scale","date":"2024-06-25","arxiv_id":"2406.17557","repositories_listed":1,"syntology":null},{"url":"/paper/trawl-tensor-reduced-and-approximated-weights","slug":"trawl-tensor-reduced-and-approximated-weights","title":"TRAWL: Tensor Reduced and Approximated Weights for Large Language Models","date":"2024-06-25","arxiv_id":"2406.17261","repositories_listed":1,"syntology":null},{"url":"/paper/variable-layer-wise-quantization-a-simple-and","slug":"variable-layer-wise-quantization-a-simple-and","title":"Layer-Wise Quantization: A Pragmatic and Effective Method for Quantizing LLMs Beyond Integer Bit-Levels","date":"2024-06-25","arxiv_id":"2406.17415","repositories_listed":1,"syntology":null},{"url":"/paper/a-large-language-model-for-predicting-t-cell","slug":"a-large-language-model-for-predicting-t-cell","title":"tcrLM: a lightweight protein language model for predicting T cell receptor and epitope binding specificity","date":"2024-06-24","arxiv_id":"2406.16995","repositories_listed":1,"syntology":null},{"url":"/paper/c-llm-learn-to-check-chinese-spelling-errors","slug":"c-llm-learn-to-check-chinese-spelling-errors","title":"C-LLM: Learn to Check Chinese Spelling Errors Character by Character","date":"2024-06-24","arxiv_id":"2406.16536","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/c-llm-learn-to-check-chinese-spelling-errors#ran","syntology_url":"https://syntology.ai/paper/2406.16536","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16536"}},"official":{"repos":["ktlktl/c-llm"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dalpsr-leverage-degradation-aligned-language","slug":"dalpsr-leverage-degradation-aligned-language","title":"DaLPSR: Leverage Degradation-Aligned Language Prompt for Real-World Image Super-Resolution","date":"2024-06-24","arxiv_id":"2406.16477","repositories_listed":1,"syntology":null},{"url":"/paper/res-q-evaluating-code-editing-large-language","slug":"res-q-evaluating-code-editing-large-language","title":"RES-Q: Evaluating Code-Editing Large Language Model Systems at the Repository Scale","date":"2024-06-24","arxiv_id":"2406.16801","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/res-q-evaluating-code-editing-large-language#ran","syntology_url":"https://syntology.ai/paper/2406.16801","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.16801"}},"official":{"repos":["qurrent-ai/res-q"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/edge-llm-enabling-efficient-large-language","slug":"edge-llm-enabling-efficient-large-language","title":"EDGE-LLM: Enabling Efficient Large Language Model Adaptation on Edge Devices via Layerwise Unified Compression and Adaptive Layer Tuning and Voting","date":"2024-06-22","arxiv_id":"2406.15758","repositories_listed":1,"syntology":null},{"url":"/paper/genotex-a-benchmark-for-evaluating-llm-based","slug":"genotex-a-benchmark-for-evaluating-llm-based","title":"GenoTEX: An LLM Agent Benchmark for Automated Gene Expression Data Analysis","date":"2024-06-21","arxiv_id":"2406.15341","repositories_listed":1,"syntology":null},{"url":"/paper/internlm-law-an-open-source-chinese-legal","slug":"internlm-law-an-open-source-chinese-legal","title":"InternLM-Law: An Open Source Chinese Legal Large Language Model","date":"2024-06-21","arxiv_id":"2406.14887","repositories_listed":1,"syntology":null},{"url":"/paper/moa-mixture-of-sparse-attention-for-automatic","slug":"moa-mixture-of-sparse-attention-for-automatic","title":"MoA: Mixture of Sparse Attention for Automatic Large Language Model Compression","date":"2024-06-21","arxiv_id":"2406.14909","repositories_listed":1,"syntology":{"n":22,"n_ran":19,"n_constructed":0,"n_ran_checked":19,"n_instrument":0,"n_unverified":3,"n_honours":0,"n_violates":0,"n_no_contract":19,"n_pointer_only":0,"phrase":"19 ran (of which 0 constructed an object rather than computing a result; 19 with no instrument failure: 0 honoured, 0 violated, 19 with no contract checked; 0 where Syntology's instrument failed) · 3 unverified","sample_list":"/paper/moa-mixture-of-sparse-attention-for-automatic#ran","syntology_url":"https://syntology.ai/paper/2406.14909","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14909"}},"official":{"repos":["thu-nics/moa"],"state":"official (archive's flag): 19 ran","n_ran":19,"n_constructed":0,"n_ran_no_instrument_failure":19,"n_unverified":3,"ran_from_kinds":["official"]}}},{"url":"/paper/safely-learning-with-private-data-a-federated","slug":"safely-learning-with-private-data-a-federated","title":"Safely Learning with Private Data: A Federated Learning Framework for Large Language Model","date":"2024-06-21","arxiv_id":"2406.14898","repositories_listed":1,"syntology":{"n":10,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":3,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 3 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/safely-learning-with-private-data-a-federated#ran","syntology_url":"https://syntology.ai/paper/2406.14898","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14898"}},"official":{"repos":["TAP-LLM/SplitFedLLM"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":3,"ran_from_kinds":["found_in_text","official"]}}},{"url":"/paper/asynchronous-large-language-model-enhanced","slug":"asynchronous-large-language-model-enhanced","title":"Asynchronous Large Language Model Enhanced Planner for Autonomous Driving","date":"2024-06-20","arxiv_id":"2406.14556","repositories_listed":1,"syntology":{"n":13,"n_ran":11,"n_constructed":0,"n_ran_checked":8,"n_instrument":3,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":8,"n_pointer_only":0,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 8 with no instrument failure: 0 honoured, 0 violated, 8 with no contract checked; 3 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/asynchronous-large-language-model-enhanced#ran","syntology_url":"https://syntology.ai/paper/2406.14556","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14556"}},"official":{"repos":["memberre/asyncdriver"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":8,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/citybench-evaluating-the-capabilities-of","slug":"citybench-evaluating-the-capabilities-of","title":"CityBench: Evaluating the Capabilities of Large Language Models for Urban Tasks","date":"2024-06-20","arxiv_id":"2406.13945","repositories_listed":1,"syntology":{"n":23,"n_ran":17,"n_constructed":0,"n_ran_checked":16,"n_instrument":1,"n_unverified":6,"n_honours":0,"n_violates":0,"n_no_contract":16,"n_pointer_only":0,"phrase":"17 ran (of which 0 constructed an object rather than computing a result; 16 with no instrument failure: 0 honoured, 0 violated, 16 with no contract checked; 1 where Syntology's instrument failed) · 6 unverified","sample_list":"/paper/citybench-evaluating-the-capabilities-of#ran","syntology_url":"https://syntology.ai/paper/2406.13945","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.13945"}},"official":{"repos":["tsinghua-fib-lab/citybench"],"state":"official (archive's flag): 17 ran","n_ran":17,"n_constructed":0,"n_ran_no_instrument_failure":16,"n_unverified":6,"ran_from_kinds":["official"]}}},{"url":"/paper/inference-time-decontamination-reusing-leaked","slug":"inference-time-decontamination-reusing-leaked","title":"Inference-Time Decontamination: Reusing Leaked Benchmarks for Large Language Model Evaluation","date":"2024-06-20","arxiv_id":"2406.13990","repositories_listed":1,"syntology":null},{"url":"/paper/livemind-low-latency-large-language-models","slug":"livemind-low-latency-large-language-models","title":"LiveMind: Low-latency Large Language Models with Simultaneous Inference","date":"2024-06-20","arxiv_id":"2406.14319","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/livemind-low-latency-large-language-models#ran","syntology_url":"https://syntology.ai/paper/2406.14319","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14319"}},"official":{"repos":["chuangtaochen-tum/livemind"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/llasa-large-multimodal-agent-for-human","slug":"llasa-large-multimodal-agent-for-human","title":"LLaSA: A Multimodal LLM for Human Activity Analysis Through Wearable and Smartphone Sensors","date":"2024-06-20","arxiv_id":"2406.14498","repositories_listed":1,"syntology":{"n":7,"n_ran":6,"n_constructed":0,"n_ran_checked":3,"n_instrument":3,"n_unverified":1,"n_honours":0,"n_violates":1,"n_no_contract":2,"n_pointer_only":7,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 1 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llasa-large-multimodal-agent-for-human#ran","syntology_url":"https://syntology.ai/paper/2406.14498","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14498"}},"official":{"repos":["bashlab/llasa"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-a-large-language-model-enhanced","slug":"llm-a-large-language-model-enhanced","title":"LLM-A*: Large Language Model Enhanced Incremental Heuristic Search on Path Planning","date":"2024-06-20","arxiv_id":"2407.02511","repositories_listed":1,"syntology":{"n":6,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/llm-a-large-language-model-enhanced#ran","syntology_url":"https://syntology.ai/paper/2407.02511","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2407.02511"}},"official":{"repos":["SilinMeng0510/llm-astar"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/prism-a-framework-for-decoupling-and","slug":"prism-a-framework-for-decoupling-and","title":"Prism: A Framework for Decoupling and Assessing the Capabilities of VLMs","date":"2024-06-20","arxiv_id":"2406.14544","repositories_listed":1,"syntology":{"n":13,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":4,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 4 unverified","sample_list":"/paper/prism-a-framework-for-decoupling-and#ran","syntology_url":"https://syntology.ai/paper/2406.14544","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.14544"}},"official":{"repos":["sparksjoe/prism"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":4,"ran_from_kinds":["official"]}}},{"url":"/paper/realhf-optimized-rlhf-training-for-large","slug":"realhf-optimized-rlhf-training-for-large","title":"ReaL: Efficient RLHF Training of Large Language Models with Parameter Reallocation","date":"2024-06-20","arxiv_id":"2406.14088","repositories_listed":1,"syntology":null},{"url":"/paper/sorry-bench-systematically-evaluating-large","slug":"sorry-bench-systematically-evaluating-large","title":"SORRY-Bench: Systematically Evaluating Large Language Model Safety Refusal Behaviors","date":"2024-06-20","arxiv_id":"2406.14598","repositories_listed":1,"syntology":null},{"url":"/paper/appl-a-prompt-programming-language-for","slug":"appl-a-prompt-programming-language-for","title":"APPL: A Prompt Programming Language for Harmonious Integration of Programs and Large Language Model Prompts","date":"2024-06-19","arxiv_id":"2406.13161","repositories_listed":1,"syntology":null},{"url":"/paper/bild-bi-directional-logits-difference-loss","slug":"bild-bi-directional-logits-difference-loss","title":"BiLD: Bi-directional Logits Difference Loss for Large Language Model Distillation","date":"2024-06-19","arxiv_id":"2406.13555","repositories_listed":1,"syntology":null},{"url":"/paper/detecting-hallucinations-in-large-language-1","slug":"detecting-hallucinations-in-large-language-1","title":"Detecting hallucinations in large language models using semantic entropy","date":"2024-06-19","arxiv_id":null,"repositories_listed":1,"syntology":null},{"url":"/paper/on-ai-inspired-ui-design","slug":"on-ai-inspired-ui-design","title":"On AI-Inspired UI-Design","date":"2024-06-19","arxiv_id":"2406.13631","repositories_listed":1,"syntology":null},{"url":"/paper/situational-instructions-database-task","slug":"situational-instructions-database-task","title":"SituationalLLM: Proactive language models with scene awareness for dynamic, contextual task guidance","date":"2024-06-19","arxiv_id":"2406.13302","repositories_listed":1,"syntology":null},{"url":"/paper/agentreview-exploring-peer-review-dynamics","slug":"agentreview-exploring-peer-review-dynamics","title":"AgentReview: Exploring Peer Review Dynamics with LLM Agents","date":"2024-06-18","arxiv_id":"2406.12708","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/agentreview-exploring-peer-review-dynamics#ran","syntology_url":"https://syntology.ai/paper/2406.12708","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12708"}},"official":{"repos":["ahren09/agentreview"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/automatic-benchmarking-of-large-multimodal","slug":"automatic-benchmarking-of-large-multimodal","title":"Automatic benchmarking of large multimodal models via iterative experiment programming","date":"2024-06-18","arxiv_id":"2406.12321","repositories_listed":1,"syntology":null},{"url":"/paper/breaking-the-ceiling-of-the-llm-community-by","slug":"breaking-the-ceiling-of-the-llm-community-by","title":"Breaking the Ceiling of the LLM Community by Treating Token Generation as a Classification for Ensembling","date":"2024-06-18","arxiv_id":"2406.12585","repositories_listed":1,"syntology":{"n":7,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/breaking-the-ceiling-of-the-llm-community-by#ran","syntology_url":"https://syntology.ai/paper/2406.12585","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12585"}},"official":{"repos":["yaoching0/gac"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/detectbench-can-large-language-model-detect","slug":"detectbench-can-large-language-model-detect","title":"DetectBench: Can Large Language Model Detect and Piece Together Implicit Evidence?","date":"2024-06-18","arxiv_id":"2406.12641","repositories_listed":1,"syntology":null},{"url":"/paper/detecting-errors-through-ensembling-prompts","slug":"detecting-errors-through-ensembling-prompts","title":"Detecting Errors through Ensembling Prompts (DEEP): An End-to-End LLM Framework for Detecting Factual Errors","date":"2024-06-18","arxiv_id":"2406.13009","repositories_listed":1,"syntology":null},{"url":"/paper/holmes-vad-towards-unbiased-and-explainable","slug":"holmes-vad-towards-unbiased-and-explainable","title":"Holmes-VAD: Towards Unbiased and Explainable Video Anomaly Detection via Multi-modal LLM","date":"2024-06-18","arxiv_id":"2406.12235","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":5,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":1,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/holmes-vad-towards-unbiased-and-explainable#ran","syntology_url":"https://syntology.ai/paper/2406.12235","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12235"}},"official":{"repos":["pipixin321/holmesvad"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/low-redundant-optimization-for-large-language","slug":"low-redundant-optimization-for-large-language","title":"Not Everything is All You Need: Toward Low-Redundant Optimization for Large Language Model Alignment","date":"2024-06-18","arxiv_id":"2406.12606","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/low-redundant-optimization-for-large-language#ran","syntology_url":"https://syntology.ai/paper/2406.12606","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12606"}},"official":{"repos":["rucaibox/allo"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/magic-generating-self-correction-guideline","slug":"magic-generating-self-correction-guideline","title":"MAGIC: Generating Self-Correction Guideline for In-Context Text-to-SQL","date":"2024-06-18","arxiv_id":"2406.12692","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/magic-generating-self-correction-guideline#ran","syntology_url":"https://syntology.ai/paper/2406.12692","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12692"}},"official":{"repos":["microsoft/synqo"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/moleculargpt-open-large-language-model-llm","slug":"moleculargpt-open-large-language-model-llm","title":"MolecularGPT: Open Large Language Model (LLM) for Few-Shot Molecular Property Prediction","date":"2024-06-18","arxiv_id":"2406.12950","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/moleculargpt-open-large-language-model-llm#ran","syntology_url":"https://syntology.ai/paper/2406.12950","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12950"}},"official":{"repos":["nyushcs/moleculargpt"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":9,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/rs-gpt4v-a-unified-multimodal-instruction","slug":"rs-gpt4v-a-unified-multimodal-instruction","title":"RS-GPT4V: A Unified Multimodal Instruction-Following Dataset for Remote Sensing Image Understanding","date":"2024-06-18","arxiv_id":"2406.12479","repositories_listed":1,"syntology":null},{"url":"/paper/adversarial-style-augmentation-via-large","slug":"adversarial-style-augmentation-via-large","title":"Adversarial Style Augmentation via Large Language Model for Robust Fake News Detection","date":"2024-06-17","arxiv_id":"2406.11260","repositories_listed":1,"syntology":null},{"url":"/paper/avatar-optimizing-llm-agents-for-tool","slug":"avatar-optimizing-llm-agents-for-tool","title":"AvaTaR: Optimizing LLM Agents for Tool Usage via Contrastive Reasoning","date":"2024-06-17","arxiv_id":"2406.11200","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/avatar-optimizing-llm-agents-for-tool#ran","syntology_url":"https://syntology.ai/paper/2406.11200","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11200"}},"official":{"repos":["zou-group/avatar"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/cosqa-enhancing-code-search-dataset-with","slug":"cosqa-enhancing-code-search-dataset-with","title":"CoSQA+: Pioneering the Multi-Choice Code Search Benchmark with Test-Driven Agents","date":"2024-06-17","arxiv_id":"2406.11589","repositories_listed":1,"syntology":null},{"url":"/paper/fintruthqa-a-benchmark-dataset-for-evaluating","slug":"fintruthqa-a-benchmark-dataset-for-evaluating","title":"FinTruthQA: A Benchmark Dataset for Evaluating the Quality of Financial Information Disclosure","date":"2024-06-17","arxiv_id":"2406.12009","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/fintruthqa-a-benchmark-dataset-for-evaluating#ran","syntology_url":"https://syntology.ai/paper/2406.12009","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.12009"}},"official":{"repos":["bethxx99/FinTruthQA"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/global-data-constraints-ethical-and","slug":"global-data-constraints-ethical-and","title":"Problematic Tokens: Tokenizer Bias in Large Language Models","date":"2024-06-17","arxiv_id":"2406.11214","repositories_listed":1,"syntology":null},{"url":"/paper/knowledge-to-jailbreak-one-knowledge-point","slug":"knowledge-to-jailbreak-one-knowledge-point","title":"Knowledge-to-Jailbreak: Investigating Knowledge-driven Jailbreaking Attacks for Large Language Models","date":"2024-06-17","arxiv_id":"2406.11682","repositories_listed":1,"syntology":null},{"url":"/paper/mdpo-conditional-preference-optimization-for","slug":"mdpo-conditional-preference-optimization-for","title":"mDPO: Conditional Preference Optimization for Multimodal Large Language Models","date":"2024-06-17","arxiv_id":"2406.11839","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":0,"n_ran_checked":2,"n_instrument":3,"n_unverified":5,"n_honours":1,"n_violates":0,"n_no_contract":1,"n_pointer_only":10,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 1 honoured, 0 violated, 1 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/mdpo-conditional-preference-optimization-for#ran","syntology_url":"https://syntology.ai/paper/2406.11839","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11839"}},"official":{"repos":["luka-group/mDPO"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/mmneuron-discovering-neuron-level-domain","slug":"mmneuron-discovering-neuron-level-domain","title":"MMNeuron: Discovering Neuron-Level Domain-Specific Interpretation in Multimodal Large Language Model","date":"2024-06-17","arxiv_id":"2406.11193","repositories_listed":1,"syntology":null},{"url":"/paper/prefixing-attention-sinks-can-mitigate","slug":"prefixing-attention-sinks-can-mitigate","title":"Prefixing Attention Sinks can Mitigate Activation Outliers for Large Language Model Quantization","date":"2024-06-17","arxiv_id":"2406.12016","repositories_listed":1,"syntology":null},{"url":"/paper/watch-every-step-llm-agent-learning-via","slug":"watch-every-step-llm-agent-learning-via","title":"Watch Every Step! LLM Agent Learning via Iterative Step-Level Process Refinement","date":"2024-06-17","arxiv_id":"2406.11176","repositories_listed":1,"syntology":{"n":8,"n_ran":6,"n_constructed":0,"n_ran_checked":6,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":8,"phrase":"6 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/watch-every-step-llm-agent-learning-via#ran","syntology_url":"https://syntology.ai/paper/2406.11176","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.11176"}},"official":{"repos":["weiminxiong/ipr"],"state":"official (archive's flag): 6 ran","n_ran":6,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/avoiding-copyright-infringement-via-machine","slug":"avoiding-copyright-infringement-via-machine","title":"Avoiding Copyright Infringement via Large Language Model Unlearning","date":"2024-06-16","arxiv_id":"2406.10952","repositories_listed":1,"syntology":{"n":4,"n_ran":4,"n_constructed":0,"n_ran_checked":0,"n_instrument":4,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":4,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 4 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/avoiding-copyright-infringement-via-machine#ran","syntology_url":"https://syntology.ai/paper/2406.10952","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10952"}},"official":{"repos":["guangyaodou/SSU_Unlearn"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/crisissense-llm-instruction-fine-tuned-large","slug":"crisissense-llm-instruction-fine-tuned-large","title":"CrisisSense-LLM: Instruction Fine-Tuned Large Language Model for Multi-label Social Media Text Classification in Disaster Informatics","date":"2024-06-16","arxiv_id":"2406.15477","repositories_listed":1,"syntology":null},{"url":"/paper/micl-improving-in-context-learning-through","slug":"micl-improving-in-context-learning-through","title":"Logit Separability-Driven Samples and Multiple Class-Related Words Selection for Advancing In-Context Learning","date":"2024-06-16","arxiv_id":"2406.10908","repositories_listed":1,"syntology":null},{"url":"/paper/not-all-bias-is-bad-balancing-rational","slug":"not-all-bias-is-bad-balancing-rational","title":"Balancing Rigor and Utility: Mitigating Cognitive Biases in Large Language Models for Multiple-Choice Questions","date":"2024-06-16","arxiv_id":"2406.10999","repositories_listed":1,"syntology":null},{"url":"/paper/optimization-of-armv9-architecture-general","slug":"optimization-of-armv9-architecture-general","title":"Optimization of Armv9 architecture general large language model inference performance based on Llama.cpp","date":"2024-06-16","arxiv_id":"2406.10816","repositories_listed":1,"syntology":null},{"url":"/paper/sharelora-parameter-efficient-and-robust","slug":"sharelora-parameter-efficient-and-robust","title":"ShareLoRA: Parameter Efficient and Robust Large Language Model Fine-tuning via Shared Low-Rank Adaptation","date":"2024-06-16","arxiv_id":"2406.10785","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":0,"n_instrument":5,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":6,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 5 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/sharelora-parameter-efficient-and-robust#ran","syntology_url":"https://syntology.ai/paper/2406.10785","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.10785"}},"official":{"repos":["Rain9876/ShareLoRA"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/luma-a-benchmark-dataset-for-learning-from","slug":"luma-a-benchmark-dataset-for-learning-from","title":"LUMA: A Benchmark Dataset for Learning from Uncertain and Multimodal Data","date":"2024-06-14","arxiv_id":"2406.09864","repositories_listed":1,"syntology":null},{"url":"/paper/rapport-driven-virtual-agent-rapport-building","slug":"rapport-driven-virtual-agent-rapport-building","title":"Rapport-Driven Virtual Agent: Rapport Building Dialogue Strategy for Improving User Experience at First Meeting","date":"2024-06-14","arxiv_id":"2406.09839","repositories_listed":1,"syntology":null},{"url":"/paper/sycophancy-to-subterfuge-investigating-reward","slug":"sycophancy-to-subterfuge-investigating-reward","title":"Sycophancy to Subterfuge: Investigating Reward-Tampering in Large Language Models","date":"2024-06-14","arxiv_id":"2406.10162","repositories_listed":1,"syntology":null},{"url":"/paper/explore-the-limits-of-omni-modal-pretraining","slug":"explore-the-limits-of-omni-modal-pretraining","title":"Explore the Limits of Omni-modal Pretraining at Scale","date":"2024-06-13","arxiv_id":"2406.09412","repositories_listed":1,"syntology":null},{"url":"/paper/streambench-towards-benchmarking-continuous","slug":"streambench-towards-benchmarking-continuous","title":"StreamBench: Towards Benchmarking Continuous Improvement of Language Agents","date":"2024-06-13","arxiv_id":"2406.08747","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/streambench-towards-benchmarking-continuous#ran","syntology_url":"https://syntology.ai/paper/2406.08747","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08747"}},"official":{"repos":["stream-bench/stream-bench"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/conme-rethinking-evaluation-of-compositional","slug":"conme-rethinking-evaluation-of-compositional","title":"ConMe: Rethinking Evaluation of Compositional Reasoning for Modern VLMs","date":"2024-06-12","arxiv_id":"2406.08164","repositories_listed":1,"syntology":{"n":12,"n_ran":11,"n_constructed":0,"n_ran_checked":11,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":11,"n_pointer_only":12,"phrase":"11 ran (of which 0 constructed an object rather than computing a result; 11 with no instrument failure: 0 honoured, 0 violated, 11 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/conme-rethinking-evaluation-of-compositional#ran","syntology_url":"https://syntology.ai/paper/2406.08164","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08164"}},"official":{"repos":["jmiemirza/conme"],"state":"official (archive's flag): 11 ran","n_ran":11,"n_constructed":0,"n_ran_no_instrument_failure":11,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/dataset-and-lessons-learned-from-the-2024","slug":"dataset-and-lessons-learned-from-the-2024","title":"Dataset and Lessons Learned from the 2024 SaTML LLM Capture-the-Flag Competition","date":"2024-06-12","arxiv_id":"2406.07954","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/dataset-and-lessons-learned-from-the-2024#ran","syntology_url":"https://syntology.ai/paper/2406.07954","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07954"}},"official":{"repos":["ethz-spylab/ctf-satml24-data-analysis"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-model-unlearning-via-embedding","slug":"large-language-model-unlearning-via-embedding","title":"Large Language Model Unlearning via Embedding-Corrupted Prompts","date":"2024-06-12","arxiv_id":"2406.07933","repositories_listed":1,"syntology":{"n":3,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/large-language-model-unlearning-via-embedding#ran","syntology_url":"https://syntology.ai/paper/2406.07933","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07933"}},"official":{"repos":["chrisliu298/llm-unlearn-eco"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":["unlocated"]}}},{"url":"/paper/multimodal-table-understanding","slug":"multimodal-table-understanding","title":"Multimodal Table Understanding","date":"2024-06-12","arxiv_id":"2406.08100","repositories_listed":1,"syntology":{"n":10,"n_ran":9,"n_constructed":0,"n_ran_checked":6,"n_instrument":3,"n_unverified":1,"n_honours":1,"n_violates":1,"n_no_contract":4,"n_pointer_only":1,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 6 with no instrument failure: 1 honoured, 1 violated, 4 with no contract checked; 3 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/multimodal-table-understanding#ran","syntology_url":"https://syntology.ai/paper/2406.08100","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.08100"}},"official":{"repos":["spursgozmy/table-llava"],"state":"official (archive's flag): 9 ran","n_ran":9,"n_constructed":0,"n_ran_no_instrument_failure":6,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/visionllm-v2-an-end-to-end-generalist","slug":"visionllm-v2-an-end-to-end-generalist","title":"VisionLLM v2: An End-to-End Generalist Multimodal Large Language Model for Hundreds of Vision-Language Tasks","date":"2024-06-12","arxiv_id":"2406.08394","repositories_listed":1,"syntology":null},{"url":"/paper/paying-more-attention-to-source-context","slug":"paying-more-attention-to-source-context","title":"Paying More Attention to Source Context: Mitigating Unfaithful Translations from Large Language Model","date":"2024-06-11","arxiv_id":"2406.07036","repositories_listed":1,"syntology":null},{"url":"/paper/rs-agent-automating-remote-sensing-tasks","slug":"rs-agent-automating-remote-sensing-tasks","title":"RS-Agent: Automating Remote Sensing Tasks through Intelligent Agent","date":"2024-06-11","arxiv_id":"2406.07089","repositories_listed":1,"syntology":{"n":5,"n_ran":4,"n_constructed":0,"n_ran_checked":4,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":4,"n_pointer_only":0,"phrase":"4 ran (of which 0 constructed an object rather than computing a result; 4 with no instrument failure: 0 honoured, 0 violated, 4 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/rs-agent-automating-remote-sensing-tasks#ran","syntology_url":"https://syntology.ai/paper/2406.07089","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2406.07089"}},"official":{"repos":["intellisensing/rs-agent"],"state":"official (archive's flag): 4 ran","n_ran":4,"n_constructed":0,"n_ran_no_instrument_failure":4,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/scaling-large-language-model-based-multi","slug":"scaling-large-language-model-based-multi","title":"Scaling Large Language Model-based Multi-Agent Collaboration","date":"2024-06-11","arxiv_id":"2406.07155","repositories_listed":1,"syntology":null},{"url":"/paper/scholarly-question-answering-using-large","slug":"scholarly-question-answering-using-large","title":"Scholarly Question Answering using Large Language Models in the NFDI4DataScience Gateway","date":"2024-06-11","arxiv_id":"2406.07257","repositories_listed":1,"syntology":null},{"url":"/paper/versicode-towards-version-controllable-code","slug":"versicode-towards-version-controllable-code","title":"VersiCode: Towards Version-controllable Code Generation","date":"2024-06-11","arxiv_id":"2406.07411","repositories_listed":1,"syntology":null}],"record_sha256":"565709341436f2c3951d3c0596b0ba2f6dea6bd7bf39263b3bdb0d29c7128f27","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}