{"about":{"non_affiliation":"Code with Papers and Syntology are not affiliated with, endorsed by, or sponsored by Papers with Code, Meta, or the pwc-archive mirror.","site":"https://codewithpapers.app","licence":"CC BY-SA 4.0","licence_url":"https://creativecommons.org/licenses/by-sa/4.0/legalcode","attribution":"https://codewithpapers.app/attribution","modified":"archive material modified by Syntology; see the attribution page","syntology":{"site":"https://syntology.ai","developers":"https://syntology.ai/developers","mcp":{"server":"https://syntology.ai/mcp","transport":"streamable-http","server_card":"https://syntology.ai/.well-known/mcp/server-card.json","auth":{"type":"trial token, no account","trial_token":"https://syntology.ai/api/oauth/trial/token","method":"POST","docs":"https://syntology.ai/developers"}},"have":"https://syntology.ai/api/graph/have?x=<method, arXiv id or title> (free, answers coverage only)","paper_base":"https://syntology.ai/paper/","atlas_base":"https://app.syntology.ai/?focus="},"machine_readable":[{"url":"https://codewithpapers.app/llms.txt","what":"the machine catalog: every machine-readable file, counted"},{"url":"https://codewithpapers.app/index/manifest.json","what":"paper-to-code index by arXiv id, with Syntology's counts"},{"url":"https://codewithpapers.app/search/manifest.json","what":"site search index (titles, authors) and its files"},{"url":"https://codewithpapers.app/download","what":"bulk files: Syntology's layer, described there"},{"url":"https://codewithpapers.app/build_manifest.json","what":"the build record: inputs, counts, exclusions, probes"}]},"url":"/task/large-language-model/papers/5","list_of":"/task/large-language-model","task":"Large Language Model","archive":{"snapshot":"2025-07-28"},"key_notes":{"n_ran_checked":"legacy name, kept unchanged so existing readers do not break: it counts the samples that ran with no instrument failure (honoured, violated, and ran with no contract checked); it does not mean a contract was checked, and the pages print it as 'K with no instrument failure', not 'K checked'","n_constructed":"a sub-count of the samples that ran, never subtracted from them and never a failure: an executed sample whose run returned an instance of its own class (fixture_out_type equals the entry name): the run built an object and did not compute a result (Syntology's RAN record, counts.constructed)"},"syntology_read_at":"2026-09-28T10:30:06+00:00","order":"archive","order_definition":"repositories listed in the archive (most first), then date (newest first), then slug","page":5,"pages_in_order":61,"rows_per_page":100,"rows":[401,500],"of":6097,"counts":{"archive_papers_tagged":6097,"with_a_code_link":2250,"where_syntology_ran_a_sample":801,"not_listed_spam_title":0,"listed":6097,"listed_where_code_ran":801,"where_syntology_ran_a_sample_split":{"with_a_run_with_no_instrument_failure":683,"every_run_a_failure_of_syntologys_instrument":118,"listed_with_a_run_with_no_instrument_failure":683,"listed_every_run_a_failure_of_syntologys_instrument":118,"filter":{"states":["a run with no instrument failure","any run, instrument failures included"],"default":"a run with no instrument failure","note":"on the 'only where code ran' pages the default hides, in the browser, the rows where every run was a failure of Syntology's instrument; the second state shows them again. Rows are hidden, never re-ordered; these twins list every row"}},"definition":"distinct papers the archive tags; 'where Syntology ran a sample' counts papers with at least one harvested sample that ran, which is not a correctness claim"},"first_page":"/task/large-language-model","prev":"/task/large-language-model/papers/4","next":"/task/large-language-model/papers/6","papers":[{"url":"/paper/u-sam-an-audio-language-model-for-unified","slug":"u-sam-an-audio-language-model-for-unified","title":"U-SAM: An audio language Model for Unified Speech, Audio, and Music Understanding","date":"2025-05-20","arxiv_id":"2505.13880","repositories_listed":1,"syntology":null},{"url":"/paper/busterx-mllm-powered-ai-generated-video","slug":"busterx-mllm-powered-ai-generated-video","title":"BusterX: MLLM-Powered AI-Generated Video Forgery Detection and Explanation","date":"2025-05-19","arxiv_id":"2505.12620","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":2,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/busterx-mllm-powered-ai-generated-video#ran","syntology_url":"https://syntology.ai/paper/2505.12620","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12620"}},"official":{"repos":["l8cv/busterx"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":2,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/cpret-a-dataset-benchmark-and-model-for","slug":"cpret-a-dataset-benchmark-and-model-for","title":"CPRet: A Dataset, Benchmark, and Model for Retrieval in Competitive Programming","date":"2025-05-19","arxiv_id":"2505.12925","repositories_listed":1,"syntology":{"n":11,"n_ran":9,"n_constructed":0,"n_ran_checked":9,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":9,"n_pointer_only":0,"phrase":"9 ran (of which 0 constructed an object rather than computing a result; 9 with no instrument failure: 0 honoured, 0 violated, 9 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/cpret-a-dataset-benchmark-and-model-for#ran","syntology_url":"https://syntology.ai/paper/2505.12925","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12925"}},"official":{"repos":["coldchair/cpret"],"state":"official: no sample here; runs from other or unrecorded repositories","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["found_in_text"]}}},{"url":"/paper/mindomni-unleashing-reasoning-generation-in","slug":"mindomni-unleashing-reasoning-generation-in","title":"MindOmni: Unleashing Reasoning Generation in Vision Language Models with RGPO","date":"2025-05-19","arxiv_id":"2505.13031","repositories_listed":1,"syntology":null},{"url":"/paper/temporal-oriented-recipe-for-transferring","slug":"temporal-oriented-recipe-for-transferring","title":"Temporal-Oriented Recipe for Transferring Large Vision-Language Model to Video Understanding","date":"2025-05-19","arxiv_id":"2505.12605","repositories_listed":1,"syntology":null},{"url":"/paper/the-traitors-deception-and-trust-in-multi","slug":"the-traitors-deception-and-trust-in-multi","title":"The Traitors: Deception and Trust in Multi-Agent Language Model Simulations","date":"2025-05-19","arxiv_id":"2505.12923","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/the-traitors-deception-and-trust-in-multi#ran","syntology_url":"https://syntology.ai/paper/2505.12923","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.12923"}},"official":{"repos":["pedrocurvo/thetraitors"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/towards-ds-ner-unveiling-and-addressing","slug":"towards-ds-ner-unveiling-and-addressing","title":"Towards DS-NER: Unveiling and Addressing Latent Noise in Distant Annotations","date":"2025-05-18","arxiv_id":"2505.12454","repositories_listed":1,"syntology":null},{"url":"/paper/demystifying-and-enhancing-the-efficiency-of","slug":"demystifying-and-enhancing-the-efficiency-of","title":"Demystifying and Enhancing the Efficiency of Large Language Model Based Search Agents","date":"2025-05-17","arxiv_id":"2505.12065","repositories_listed":1,"syntology":null},{"url":"/paper/lifelongagentbench-evaluating-llm-agents-as","slug":"lifelongagentbench-evaluating-llm-agents-as","title":"LifelongAgentBench: Evaluating LLM Agents as Lifelong Learners","date":"2025-05-17","arxiv_id":"2505.11942","repositories_listed":1,"syntology":null},{"url":"/paper/reasoning-large-language-model-errors-arise","slug":"reasoning-large-language-model-errors-arise","title":"Reasoning Large Language Model Errors Arise from Hallucinating Critical Problem Features","date":"2025-05-17","arxiv_id":"2505.12151","repositories_listed":1,"syntology":null},{"url":"/paper/tiny-qa-benchmark-ultra-lightweight-synthetic","slug":"tiny-qa-benchmark-ultra-lightweight-synthetic","title":"Tiny QA Benchmark++: Ultra-Lightweight, Synthetic Multilingual Dataset Generation & Smoke-Tests for Continuous LLM Evaluation","date":"2025-05-17","arxiv_id":"2505.12058","repositories_listed":1,"syntology":null},{"url":"/paper/2505-10769","slug":"2505-10769","title":"Unifying Segment Anything in Microscopy with Multimodal Large Language Model","date":"2025-05-16","arxiv_id":"2505.10769","repositories_listed":1,"syntology":null},{"url":"/paper/2505-10861","slug":"2505-10861","title":"Improving the Data-efficiency of Reinforcement Learning by Warm-starting with LLM","date":"2025-05-16","arxiv_id":"2505.10861","repositories_listed":1,"syntology":{"n":5,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/2505-10861#ran","syntology_url":"https://syntology.ai/paper/2505.10861","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.10861"}},"official":{"repos":["duongnhatthang/llamagym"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/2505-11140","slug":"2505-11140","title":"Scaling Reasoning can Improve Factuality in Large Language Models","date":"2025-05-16","arxiv_id":"2505.11140","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11177","slug":"2505-11177","title":"Low-Resource Language Processing: An OCR-Driven Summarization and Translation Pipeline","date":"2025-05-16","arxiv_id":"2505.11177","repositories_listed":1,"syntology":null},{"url":"/paper/2505-11417","slug":"2505-11417","title":"EdgeWisePersona: A Dataset for On-Device User Profiling from Natural Language Interactions","date":"2025-05-16","arxiv_id":"2505.11417","repositories_listed":1,"syntology":null},{"url":"/paper/an-agentic-system-with-reinforcement-learned","slug":"an-agentic-system-with-reinforcement-learned","title":"An agentic system with reinforcement-learned subsystem improvements for parsing form-like documents","date":"2025-05-16","arxiv_id":"2505.13504","repositories_listed":1,"syntology":null},{"url":"/paper/does-feasibility-matter-understanding-the","slug":"does-feasibility-matter-understanding-the","title":"Does Feasibility Matter? Understanding the Impact of Feasibility on Synthetic Training Data","date":"2025-05-15","arxiv_id":"2505.10551","repositories_listed":1,"syntology":null},{"url":"/paper/imaginebench-evaluating-reinforcement","slug":"imaginebench-evaluating-reinforcement","title":"ImagineBench: Evaluating Reinforcement Learning with Large Language Model Rollouts","date":"2025-05-15","arxiv_id":"2505.10010","repositories_listed":1,"syntology":null},{"url":"/paper/towards-a-deeper-understanding-of-reasoning","slug":"towards-a-deeper-understanding-of-reasoning","title":"Towards a Deeper Understanding of Reasoning Capabilities in Large Language Models","date":"2025-05-15","arxiv_id":"2505.10543","repositories_listed":1,"syntology":null},{"url":"/paper/contrastive-cross-course-knowledge-tracing","slug":"contrastive-cross-course-knowledge-tracing","title":"Contrastive Cross-Course Knowledge Tracing via Concept Graph Guided Knowledge Transfer","date":"2025-05-14","arxiv_id":"2505.13489","repositories_listed":1,"syntology":{"n":9,"n_ran":7,"n_constructed":5,"n_ran_checked":6,"n_instrument":1,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":6,"n_pointer_only":0,"phrase":"7 ran (of which 5 constructed an object rather than computing a result; 6 with no instrument failure: 0 honoured, 0 violated, 6 with no contract checked; 1 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/contrastive-cross-course-knowledge-tracing#ran","syntology_url":"https://syntology.ai/paper/2505.13489","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.13489"}},"official":{"repos":["dqyzhwk/transkt"],"state":"official (archive's flag): 7 ran","n_ran":7,"n_constructed":5,"n_ran_no_instrument_failure":6,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/large-language-models-are-more-persuasive","slug":"large-language-models-are-more-persuasive","title":"Large Language Models Are More Persuasive Than Incentivized Human Persuaders","date":"2025-05-14","arxiv_id":"2505.09662","repositories_listed":1,"syntology":null},{"url":"/paper/layered-unlearning-for-adversarial-relearning","slug":"layered-unlearning-for-adversarial-relearning","title":"Layered Unlearning for Adversarial Relearning","date":"2025-05-14","arxiv_id":"2505.09500","repositories_listed":1,"syntology":null},{"url":"/paper/zero-shot-multi-modal-large-language-model-v","slug":"zero-shot-multi-modal-large-language-model-v","title":"Zero-Shot Multi-modal Large Language Model v.s. Supervised Deep Learning: A Comparative Study on CT-Based Intracranial Hemorrhage Subtyping","date":"2025-05-14","arxiv_id":"2505.09252","repositories_listed":1,"syntology":null},{"url":"/paper/celltypeagent-trustworthy-cell-type","slug":"celltypeagent-trustworthy-cell-type","title":"CellTypeAgent: Trustworthy cell type annotation with Large Language Models","date":"2025-05-13","arxiv_id":"2505.08844","repositories_listed":1,"syntology":null},{"url":"/paper/large-language-model-psychometrics-a","slug":"large-language-model-psychometrics-a","title":"Large Language Model Psychometrics: A Systematic Review of Evaluation, Validation, and Enhancement","date":"2025-05-13","arxiv_id":"2505.08245","repositories_listed":1,"syntology":null},{"url":"/paper/prioritizing-image-related-tokens-enhances","slug":"prioritizing-image-related-tokens-enhances","title":"Prioritizing Image-Related Tokens Enhances Vision-Language Pre-Training","date":"2025-05-13","arxiv_id":"2505.08971","repositories_listed":1,"syntology":null},{"url":"/paper/dynamicrag-leveraging-outputs-of-large","slug":"dynamicrag-leveraging-outputs-of-large","title":"DynamicRAG: Leveraging Outputs of Large Language Model as Feedback for Dynamic Reranking in Retrieval-Augmented Generation","date":"2025-05-12","arxiv_id":"2505.07233","repositories_listed":1,"syntology":null},{"url":"/paper/measuring-general-intelligence-with-generated","slug":"measuring-general-intelligence-with-generated","title":"Measuring General Intelligence with Generated Games","date":"2025-05-12","arxiv_id":"2505.07215","repositories_listed":1,"syntology":{"n":6,"n_ran":5,"n_constructed":0,"n_ran_checked":5,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":5,"n_pointer_only":0,"phrase":"5 ran (of which 0 constructed an object rather than computing a result; 5 with no instrument failure: 0 honoured, 0 violated, 5 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/measuring-general-intelligence-with-generated#ran","syntology_url":"https://syntology.ai/paper/2505.07215","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07215"}},"official":{"repos":["vivek3141/gg-bench"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":0,"n_ran_no_instrument_failure":5,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mimo-unlocking-the-reasoning-potential-of","slug":"mimo-unlocking-the-reasoning-potential-of","title":"MiMo: Unlocking the Reasoning Potential of Language Model -- From Pretraining to Posttraining","date":"2025-05-12","arxiv_id":"2505.07608","repositories_listed":1,"syntology":null},{"url":"/paper/mle-dojo-interactive-environments-for","slug":"mle-dojo-interactive-environments-for","title":"MLE-Dojo: Interactive Environments for Empowering LLM Agents in Machine Learning Engineering","date":"2025-05-12","arxiv_id":"2505.07782","repositories_listed":1,"syntology":null},{"url":"/paper/yulan-onesim-towards-the-next-generation-of","slug":"yulan-onesim-towards-the-next-generation-of","title":"YuLan-OneSim: Towards the Next Generation of Social Simulator with Large Language Models","date":"2025-05-12","arxiv_id":"2505.07581","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/yulan-onesim-towards-the-next-generation-of#ran","syntology_url":"https://syntology.ai/paper/2505.07581","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07581"}},"official":{"repos":["RUC-GSAI/YuLan-OneSim"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":1,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/guidedquant-large-language-model-quantization","slug":"guidedquant-large-language-model-quantization","title":"GuidedQuant: Large Language Model Quantization via Exploiting End Loss Guidance","date":"2025-05-11","arxiv_id":"2505.07004","repositories_listed":1,"syntology":{"n":2,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/guidedquant-large-language-model-quantization#ran","syntology_url":"https://syntology.ai/paper/2505.07004","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.07004"}},"official":{"repos":["snu-mllab/guidedquant"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/mellm-exploring-llm-powered-micro-expression","slug":"mellm-exploring-llm-powered-micro-expression","title":"MELLM: Exploring LLM-Powered Micro-Expression Understanding Enhanced by Subtle Motion Perception","date":"2025-05-11","arxiv_id":"2505.07007","repositories_listed":1,"syntology":null},{"url":"/paper/web-page-classification-using-llms-for","slug":"web-page-classification-using-llms-for","title":"Web Page Classification using LLMs for Crawling Support","date":"2025-05-11","arxiv_id":"2505.06972","repositories_listed":1,"syntology":null},{"url":"/paper/batch-augmentation-with-unimodal-fine-tuning","slug":"batch-augmentation-with-unimodal-fine-tuning","title":"Batch Augmentation with Unimodal Fine-tuning for Multimodal Learning","date":"2025-05-10","arxiv_id":"2505.06592","repositories_listed":1,"syntology":null},{"url":"/paper/kcluster-an-llm-based-clustering-approach-to","slug":"kcluster-an-llm-based-clustering-approach-to","title":"KCluster: An LLM-based Clustering Approach to Knowledge Component Discovery","date":"2025-05-09","arxiv_id":"2505.06469","repositories_listed":1,"syntology":null},{"url":"/paper/summarisation-of-german-judgments-in","slug":"summarisation-of-german-judgments-in","title":"Summarisation of German Judgments in conjunction with a Class-based Evaluation","date":"2025-05-09","arxiv_id":"2505.05947","repositories_listed":1,"syntology":null},{"url":"/paper/understanding-stragglers-in-large-model","slug":"understanding-stragglers-in-large-model","title":"Understanding Stragglers in Large Model Training Using What-if Analysis","date":"2025-05-09","arxiv_id":"2505.05713","repositories_listed":1,"syntology":null},{"url":"/paper/biomed-dpt-dual-modality-prompt-tuning-for","slug":"biomed-dpt-dual-modality-prompt-tuning-for","title":"Biomed-DPT: Dual Modality Prompt Tuning for Biomedical Vision-Language Models","date":"2025-05-08","arxiv_id":"2505.05189","repositories_listed":1,"syntology":null},{"url":"/paper/citynavagent-aerial-vision-and-language","slug":"citynavagent-aerial-vision-and-language","title":"CityNavAgent: Aerial Vision-and-Language Navigation with Hierarchical Semantic Planning and Global Memory","date":"2025-05-08","arxiv_id":"2505.05622","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-large-language-models-with-faster","slug":"enhancing-large-language-models-with-faster","title":"Enhancing Large Language Models with Faster Code Preprocessing for Vulnerability Detection","date":"2025-05-08","arxiv_id":"2505.05600","repositories_listed":1,"syntology":null},{"url":"/paper/generating-physically-stable-and-buildable","slug":"generating-physically-stable-and-buildable","title":"Generating Physically Stable and Buildable LEGO Designs from Text","date":"2025-05-08","arxiv_id":"2505.05469","repositories_listed":1,"syntology":null},{"url":"/paper/revealing-weaknesses-in-text-watermarking","slug":"revealing-weaknesses-in-text-watermarking","title":"Revealing Weaknesses in Text Watermarking Through Self-Information Rewrite Attacks","date":"2025-05-08","arxiv_id":"2505.05190","repositories_listed":1,"syntology":{"n":10,"n_ran":5,"n_constructed":2,"n_ran_checked":2,"n_instrument":3,"n_unverified":5,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":10,"phrase":"5 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 3 where Syntology's instrument failed) · 5 unverified","sample_list":"/paper/revealing-weaknesses-in-text-watermarking#ran","syntology_url":"https://syntology.ai/paper/2505.05190","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.05190"}},"official":{"repos":["allencheng97/self-information-rewrite-attack"],"state":"official (archive's flag): 5 ran","n_ran":5,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":5,"ran_from_kinds":["official"]}}},{"url":"/paper/waterdrum-watermarking-for-data-centric","slug":"waterdrum-watermarking-for-data-centric","title":"WaterDrum: Watermarking for Data-centric Unlearning Metric","date":"2025-05-08","arxiv_id":"2505.05064","repositories_listed":1,"syntology":null},{"url":"/paper/apply-hierarchical-chain-of-generation-to","slug":"apply-hierarchical-chain-of-generation-to","title":"Apply Hierarchical-Chain-of-Generation to Complex Attributes Text-to-3D Generation","date":"2025-05-07","arxiv_id":"2505.05505","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/apply-hierarchical-chain-of-generation-to#ran","syntology_url":"https://syntology.ai/paper/2505.05505","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2505.05505"}},"official":{"repos":["wakals/gascol"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":["official"]}}},{"url":"/paper/llm-e-guess-can-llms-capabilities-advance","slug":"llm-e-guess-can-llms-capabilities-advance","title":"LLM-e Guess: Can LLMs Capabilities Advance Without Hardware Progress?","date":"2025-05-07","arxiv_id":"2505.04075","repositories_listed":1,"syntology":null},{"url":"/paper/on-device-llm-for-context-aware-wi-fi-roaming","slug":"on-device-llm-for-context-aware-wi-fi-roaming","title":"On-Device LLM for Context-Aware Wi-Fi Roaming","date":"2025-05-07","arxiv_id":"2505.04174","repositories_listed":1,"syntology":null},{"url":"/paper/vita-audio-fast-interleaved-cross-modal-token","slug":"vita-audio-fast-interleaved-cross-modal-token","title":"VITA-Audio: Fast Interleaved Cross-Modal Token Generation for Efficient Large Speech-Language Model","date":"2025-05-06","arxiv_id":"2505.03739","repositories_listed":1,"syntology":null},{"url":"/paper/emorl-ensemble-multi-objective-reinforcement","slug":"emorl-ensemble-multi-objective-reinforcement","title":"EMORL: Ensemble Multi-Objective Reinforcement Learning for Efficient and Flexible LLM Fine-Tuning","date":"2025-05-05","arxiv_id":"2505.02579","repositories_listed":1,"syntology":null},{"url":"/paper/leceval-an-automated-metric-for-multimodal","slug":"leceval-an-automated-metric-for-multimodal","title":"LecEval: An Automated Metric for Multimodal Knowledge Acquisition in Multimedia Learning","date":"2025-05-04","arxiv_id":"2505.02078","repositories_listed":1,"syntology":null},{"url":"/paper/leveraging-llm-agents-and-digital-twins-for","slug":"leveraging-llm-agents-and-digital-twins-for","title":"Leveraging LLM Agents and Digital Twins for Fault Handling in Process Plants","date":"2025-05-04","arxiv_id":"2505.02076","repositories_listed":1,"syntology":null},{"url":"/paper/memengine-a-unified-and-modular-library-for","slug":"memengine-a-unified-and-modular-library-for","title":"MemEngine: A Unified and Modular Library for Developing Advanced Memory of LLM-based Agents","date":"2025-05-04","arxiv_id":"2505.02099","repositories_listed":1,"syntology":null},{"url":"/paper/tutorgym-a-testbed-for-evaluating-ai-agents","slug":"tutorgym-a-testbed-for-evaluating-ai-agents","title":"TutorGym: A Testbed for Evaluating AI Agents as Tutors and Students","date":"2025-05-02","arxiv_id":"2505.01563","repositories_listed":1,"syntology":null},{"url":"/paper/a-survey-on-large-language-model-based-human","slug":"a-survey-on-large-language-model-based-human","title":"A Survey on Large Language Model based Human-Agent Systems","date":"2025-05-01","arxiv_id":"2505.00753","repositories_listed":1,"syntology":null},{"url":"/paper/sentient-agent-as-a-judge-evaluating-higher","slug":"sentient-agent-as-a-judge-evaluating-higher","title":"Sentient Agent as a Judge: Evaluating Higher-Order Social Cognition in Large Language Models","date":"2025-05-01","arxiv_id":"2505.02847","repositories_listed":1,"syntology":null},{"url":"/paper/consistency-aware-fake-videos-detection-on","slug":"consistency-aware-fake-videos-detection-on","title":"Consistency-aware Fake Videos Detection on Short Video Platforms","date":"2025-04-30","arxiv_id":"2504.21495","repositories_listed":1,"syntology":null},{"url":"/paper/deepseek-prover-v2-advancing-formal","slug":"deepseek-prover-v2-advancing-formal","title":"DeepSeek-Prover-V2: Advancing Formal Mathematical Reasoning via Reinforcement Learning for Subgoal Decomposition","date":"2025-04-30","arxiv_id":"2504.21801","repositories_listed":1,"syntology":null},{"url":"/paper/mf-llm-simulating-collective-decision","slug":"mf-llm-simulating-collective-decision","title":"MF-LLM: Simulating Population Decision Dynamics via a Mean-Field Large Language Model Framework","date":"2025-04-30","arxiv_id":"2504.21582","repositories_listed":1,"syntology":{"n":2,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":2,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/mf-llm-simulating-collective-decision#ran","syntology_url":"https://syntology.ai/paper/2504.21582","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.21582"}},"official":{"repos":["Miracle1207/Mean-Field-LLM"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/unibiomed-a-universal-foundation-model-for","slug":"unibiomed-a-universal-foundation-model-for","title":"UniBiomed: A Universal Foundation Model for Grounded Biomedical Image Interpretation","date":"2025-04-30","arxiv_id":"2504.21336","repositories_listed":1,"syntology":null},{"url":"/paper/detecting-manipulated-contents-using","slug":"detecting-manipulated-contents-using","title":"Detecting Manipulated Contents Using Knowledge-Grounded Inference","date":"2025-04-29","arxiv_id":"2504.21165","repositories_listed":1,"syntology":null},{"url":"/paper/turing-machine-evaluation-for-large-language","slug":"turing-machine-evaluation-for-large-language","title":"Computational Reasoning of Large Language Models","date":"2025-04-29","arxiv_id":"2504.20771","repositories_listed":1,"syntology":null},{"url":"/paper/codebc-a-more-secure-large-language-model-for","slug":"codebc-a-more-secure-large-language-model-for","title":"CodeBC: A More Secure Large Language Model for Smart Contract Code Generation in Blockchain","date":"2025-04-28","arxiv_id":"2504.21043","repositories_listed":1,"syntology":null},{"url":"/paper/phenoassistant-a-conversational-multi-agent","slug":"phenoassistant-a-conversational-multi-agent","title":"PhenoAssistant: A Conversational Multi-Agent AI System for Automated Plant Phenotyping","date":"2025-04-28","arxiv_id":"2504.19818","repositories_listed":1,"syntology":null},{"url":"/paper/an-empirical-study-of-evaluating-long-form","slug":"an-empirical-study-of-evaluating-long-form","title":"An Empirical Study of Evaluating Long-form Question Answering","date":"2025-04-25","arxiv_id":"2504.18413","repositories_listed":1,"syntology":null},{"url":"/paper/expressing-stigma-and-inappropriate-responses","slug":"expressing-stigma-and-inappropriate-responses","title":"Expressing stigma and inappropriate responses prevents LLMs from safely replacing mental health providers","date":"2025-04-25","arxiv_id":"2504.18412","repositories_listed":1,"syntology":null},{"url":"/paper/leam-a-prompt-only-large-language-model","slug":"leam-a-prompt-only-large-language-model","title":"LEAM: A Prompt-only Large Language Model-enabled Antenna Modeling Method","date":"2025-04-25","arxiv_id":"2504.18271","repositories_listed":1,"syntology":null},{"url":"/paper/automated-bug-report-prioritization-in-large","slug":"automated-bug-report-prioritization-in-large","title":"Automated Bug Report Prioritization in Large Open-Source Projects","date":"2025-04-22","arxiv_id":"2504.15912","repositories_listed":1,"syntology":null},{"url":"/paper/datetime-a-new-benchmark-to-measure-llm","slug":"datetime-a-new-benchmark-to-measure-llm","title":"DATETIME: A new benchmark to measure LLM translation and reasoning capabilities","date":"2025-04-22","arxiv_id":"2504.16155","repositories_listed":1,"syntology":null},{"url":"/paper/easyedit2-an-easy-to-use-steering-framework","slug":"easyedit2-an-easy-to-use-steering-framework","title":"EasyEdit2: An Easy-to-use Steering Framework for Editing Large Language Models","date":"2025-04-21","arxiv_id":"2504.15133","repositories_listed":1,"syntology":null},{"url":"/paper/evaluating-judges-as-evaluators-the-jetts","slug":"evaluating-judges-as-evaluators-the-jetts","title":"Evaluating Judges as Evaluators: The JETTS Benchmark of LLM-as-Judges as Test-Time Scaling Evaluators","date":"2025-04-21","arxiv_id":"2504.15253","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":0,"n_instrument":1,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 1 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/evaluating-judges-as-evaluators-the-jetts#ran","syntology_url":"https://syntology.ai/paper/2504.15253","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.15253"}},"official":{"repos":["salesforceairesearch/jetts-benchmark"],"state":"official (archive's flag): 1 ran","n_ran":1,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/virology-capabilities-test-vct-a-multimodal","slug":"virology-capabilities-test-vct-a-multimodal","title":"Virology Capabilities Test (VCT): A Multimodal Virology Q&A Benchmark","date":"2025-04-21","arxiv_id":"2504.16137","repositories_listed":1,"syntology":{"n":1,"n_ran":1,"n_constructed":0,"n_ran_checked":1,"n_instrument":0,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":1,"n_pointer_only":0,"phrase":"1 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 0 honoured, 0 violated, 1 with no contract checked; 0 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/virology-capabilities-test-vct-a-multimodal#ran","syntology_url":"https://syntology.ai/paper/2504.16137","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.16137"}},"official":null}},{"url":"/paper/walk-the-talk-measuring-the-faithfulness-of","slug":"walk-the-talk-measuring-the-faithfulness-of","title":"Walk the Talk? Measuring the Faithfulness of Large Language Model Explanations","date":"2025-04-19","arxiv_id":"2504.14150","repositories_listed":1,"syntology":null},{"url":"/paper/towards-a-multi-agent-vision-language-system","slug":"towards-a-multi-agent-vision-language-system","title":"Towards a Multi-Agent Vision-Language System for Zero-Shot Novel Hazardous Object Detection for Autonomous Driving Safety","date":"2025-04-18","arxiv_id":"2504.13399","repositories_listed":1,"syntology":{"n":2,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 2 unverified","sample_list":"/paper/towards-a-multi-agent-vision-language-system#ran","syntology_url":"https://syntology.ai/paper/2504.13399","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.13399"}},"official":{"repos":["mi3labucm/coooler"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":2,"ran_from_kinds":[]}}},{"url":"/paper/zero-shot-industrial-anomaly-segmentation","slug":"zero-shot-industrial-anomaly-segmentation","title":"Zero-Shot Industrial Anomaly Segmentation with Image-Aware Prompt Generation","date":"2025-04-18","arxiv_id":"2504.13560","repositories_listed":1,"syntology":null},{"url":"/paper/can-llms-reason-over-extended-multilingual","slug":"can-llms-reason-over-extended-multilingual","title":"Can LLMs reason over extended multilingual contexts? Towards long-context evaluation beyond retrieval and haystacks","date":"2025-04-17","arxiv_id":"2504.12845","repositories_listed":1,"syntology":null},{"url":"/paper/enhancing-the-geometric-problem-solving","slug":"enhancing-the-geometric-problem-solving","title":"Enhancing the Geometric Problem-Solving Ability of Multimodal LLMs via Symbolic-Neural Integration","date":"2025-04-17","arxiv_id":"2504.12773","repositories_listed":1,"syntology":null},{"url":"/paper/retrieval-augmented-generation-with-3","slug":"retrieval-augmented-generation-with-3","title":"Retrieval-Augmented Generation with Conflicting Evidence","date":"2025-04-17","arxiv_id":"2504.13079","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":0,"n_instrument":3,"n_unverified":0,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 3 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/retrieval-augmented-generation-with-3#ran","syntology_url":"https://syntology.ai/paper/2504.13079","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.13079"}},"official":{"repos":["hannight/ramdocs"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/skyreels-v2-infinite-length-film-generative","slug":"skyreels-v2-infinite-length-film-generative","title":"SkyReels-V2: Infinite-length Film Generative Model","date":"2025-04-17","arxiv_id":"2504.13074","repositories_listed":1,"syntology":null},{"url":"/paper/smartfreeedit-mask-free-spatial-aware-image","slug":"smartfreeedit-mask-free-spatial-aware-image","title":"SmartFreeEdit: Mask-Free Spatial-Aware Image Editing with Complex Instruction Understanding","date":"2025-04-17","arxiv_id":"2504.12704","repositories_listed":1,"syntology":null},{"url":"/paper/uncertainty-aware-trajectory-prediction-via","slug":"uncertainty-aware-trajectory-prediction-via","title":"Uncertainty-Aware Trajectory Prediction via Rule-Regularized Heteroscedastic Deep Classification","date":"2025-04-17","arxiv_id":"2504.13111","repositories_listed":1,"syntology":null},{"url":"/paper/anomalyr1-a-grpo-based-end-to-end-mllm-for","slug":"anomalyr1-a-grpo-based-end-to-end-mllm-for","title":"AnomalyR1: A GRPO-based End-to-end MLLM for Industrial Anomaly Detection","date":"2025-04-16","arxiv_id":"2504.11914","repositories_listed":1,"syntology":null},{"url":"/paper/hls-eval-a-benchmark-and-framework-for","slug":"hls-eval-a-benchmark-and-framework-for","title":"HLS-Eval: A Benchmark and Framework for Evaluating LLMs on High-Level Synthesis Design Tasks","date":"2025-04-16","arxiv_id":"2504.12268","repositories_listed":1,"syntology":null},{"url":"/paper/internvl3-exploring-advanced-training-and","slug":"internvl3-exploring-advanced-training-and","title":"InternVL3: Exploring Advanced Training and Test-Time Recipes for Open-Source Multimodal Models","date":"2025-04-14","arxiv_id":"2504.10479","repositories_listed":1,"syntology":{"n":3,"n_ran":3,"n_constructed":0,"n_ran_checked":1,"n_instrument":2,"n_unverified":0,"n_honours":1,"n_violates":0,"n_no_contract":0,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 1 with no instrument failure: 1 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 0 unverified","sample_list":"/paper/internvl3-exploring-advanced-training-and#ran","syntology_url":"https://syntology.ai/paper/2504.10479","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.10479"}},"official":{"repos":["opengvlab/internvl"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official","unlocated"]}}},{"url":"/paper/llm-unlearning-reveals-a-stronger-than","slug":"llm-unlearning-reveals-a-stronger-than","title":"LLM Unlearning Reveals a Stronger-Than-Expected Coreset Effect in Current Benchmarks","date":"2025-04-14","arxiv_id":"2504.10185","repositories_listed":1,"syntology":{"n":3,"n_ran":2,"n_constructed":0,"n_ran_checked":0,"n_instrument":2,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":3,"phrase":"2 ran (of which 0 constructed an object rather than computing a result; 0 with no instrument failure: 0 honoured, 0 violated, 0 with no contract checked; 2 where Syntology's instrument failed) · 1 unverified","sample_list":"/paper/llm-unlearning-reveals-a-stronger-than#ran","syntology_url":"https://syntology.ai/paper/2504.10185","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.10185"}},"official":{"repos":["optml-group/mu-coreset"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":0,"ran_from_kinds":["official"]}}},{"url":"/paper/the-scalability-of-simplicity-empirical","slug":"the-scalability-of-simplicity-empirical","title":"The Scalability of Simplicity: Empirical Analysis of Vision-Language Learning with a Single Transformer","date":"2025-04-14","arxiv_id":"2504.10462","repositories_listed":1,"syntology":{"n":9,"n_ran":2,"n_constructed":2,"n_ran_checked":2,"n_instrument":0,"n_unverified":7,"n_honours":0,"n_violates":0,"n_no_contract":2,"n_pointer_only":0,"phrase":"2 ran (of which 2 constructed an object rather than computing a result; 2 with no instrument failure: 0 honoured, 0 violated, 2 with no contract checked; 0 where Syntology's instrument failed) · 7 unverified; every one of the 2 samples that ran constructed an object rather than computing a result","sample_list":"/paper/the-scalability-of-simplicity-empirical#ran","syntology_url":"https://syntology.ai/paper/2504.10462","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.10462"}},"official":{"repos":["bytedance/sail"],"state":"official (archive's flag): 2 ran","n_ran":2,"n_constructed":2,"n_ran_no_instrument_failure":2,"n_unverified":7,"ran_from_kinds":["official"]}}},{"url":"/paper/clinicalgpt-r1-pushing-reasoning-capability","slug":"clinicalgpt-r1-pushing-reasoning-capability","title":"ClinicalGPT-R1: Pushing reasoning capability of generalist disease diagnosis with large language model","date":"2025-04-13","arxiv_id":"2504.09421","repositories_listed":1,"syntology":null},{"url":"/paper/fine-tuning-an-large-language-model-for","slug":"fine-tuning-an-large-language-model-for","title":"Fine-tuning a Large Language Model for Automating Computational Fluid Dynamics Simulations","date":"2025-04-13","arxiv_id":"2504.09602","repositories_listed":1,"syntology":null},{"url":"/paper/segearth-r1-geospatial-pixel-reasoning-via","slug":"segearth-r1-geospatial-pixel-reasoning-via","title":"SegEarth-R1: Geospatial Pixel Reasoning via Large Language Model","date":"2025-04-13","arxiv_id":"2504.09644","repositories_listed":1,"syntology":{"n":5,"n_ran":3,"n_constructed":0,"n_ran_checked":3,"n_instrument":0,"n_unverified":2,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":1,"phrase":"3 ran (of which 0 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 2 unverified","sample_list":"/paper/segearth-r1-geospatial-pixel-reasoning-via#ran","syntology_url":"https://syntology.ai/paper/2504.09644","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.09644"}},"official":{"repos":["earth-insights/segearth-r1"],"state":"official (archive's flag): 3 ran","n_ran":3,"n_constructed":0,"n_ran_no_instrument_failure":3,"n_unverified":2,"ran_from_kinds":["official"]}}},{"url":"/paper/ai-university-an-llm-based-platform-for","slug":"ai-university-an-llm-based-platform-for","title":"AI-University: An LLM-based platform for instructional alignment to scientific classrooms","date":"2025-04-11","arxiv_id":"2504.08846","repositories_listed":1,"syntology":null},{"url":"/paper/medrep-medical-concept-representation-for","slug":"medrep-medical-concept-representation-for","title":"MedRep: Medical Concept Representation for General Electronic Health Record Foundation Models","date":"2025-04-11","arxiv_id":"2504.08329","repositories_listed":1,"syntology":null},{"url":"/paper/playpen-an-environment-for-exploring-learning","slug":"playpen-an-environment-for-exploring-learning","title":"Playpen: An Environment for Exploring Learning Through Conversational Interaction","date":"2025-04-11","arxiv_id":"2504.08590","repositories_listed":1,"syntology":null},{"url":"/paper/apt-serve-adaptive-request-scheduling-on","slug":"apt-serve-adaptive-request-scheduling-on","title":"Apt-Serve: Adaptive Request Scheduling on Hybrid Cache for Scalable LLM Inference Serving","date":"2025-04-10","arxiv_id":"2504.07494","repositories_listed":1,"syntology":null},{"url":"/paper/glus-global-local-reasoning-unified-into-a","slug":"glus-global-local-reasoning-unified-into-a","title":"GLUS: Global-Local Reasoning Unified into A Single Large Language Model for Video Segmentation","date":"2025-04-10","arxiv_id":"2504.07962","repositories_listed":1,"syntology":{"n":12,"n_ran":3,"n_constructed":3,"n_ran_checked":3,"n_instrument":0,"n_unverified":9,"n_honours":0,"n_violates":0,"n_no_contract":3,"n_pointer_only":12,"phrase":"3 ran (of which 3 constructed an object rather than computing a result; 3 with no instrument failure: 0 honoured, 0 violated, 3 with no contract checked; 0 where Syntology's instrument failed) · 9 unverified; every one of the 3 samples that ran constructed an object rather than computing a result","sample_list":"/paper/glus-global-local-reasoning-unified-into-a#ran","syntology_url":"https://syntology.ai/paper/2504.07962","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.07962"}},"official":null}},{"url":"/paper/revisiting-llm-evaluation-through-mechanism","slug":"revisiting-llm-evaluation-through-mechanism","title":"Model Utility Law: Evaluating LLMs beyond Performance through Mechanism Interpretable Metric","date":"2025-04-10","arxiv_id":"2504.07440","repositories_listed":1,"syntology":null},{"url":"/paper/payador-a-minimalist-approach-to-grounding","slug":"payador-a-minimalist-approach-to-grounding","title":"PAYADOR: A Minimalist Approach to Grounding Language Models on Structured Data for Interactive Storytelling and Role-playing Games","date":"2025-04-09","arxiv_id":"2504.07304","repositories_listed":1,"syntology":null},{"url":"/paper/ruopinionne-2024-extraction-of-opinion-tuples","slug":"ruopinionne-2024-extraction-of-opinion-tuples","title":"RuOpinionNE-2024: Extraction of Opinion Tuples from Russian News Texts","date":"2025-04-09","arxiv_id":"2504.06947","repositories_listed":1,"syntology":null},{"url":"/paper/are-generative-ai-agents-effective","slug":"are-generative-ai-agents-effective","title":"Are Generative AI Agents Effective Personalized Financial Advisors?","date":"2025-04-08","arxiv_id":"2504.05862","repositories_listed":1,"syntology":null},{"url":"/paper/safechat-a-framework-for-building-trustworthy","slug":"safechat-a-framework-for-building-trustworthy","title":"SafeChat: A Framework for Building Trustworthy Collaborative Assistants and a Case Study of its Usefulness","date":"2025-04-08","arxiv_id":"2504.07995","repositories_listed":1,"syntology":null},{"url":"/paper/collab-rag-boosting-retrieval-augmented","slug":"collab-rag-boosting-retrieval-augmented","title":"Collab-RAG: Boosting Retrieval-Augmented Generation for Complex Question Answering via White-Box and Black-Box LLM Collaboration","date":"2025-04-07","arxiv_id":"2504.04915","repositories_listed":1,"syntology":{"n":1,"n_ran":0,"n_constructed":0,"n_ran_checked":0,"n_instrument":0,"n_unverified":1,"n_honours":0,"n_violates":0,"n_no_contract":0,"n_pointer_only":0,"phrase":"0 ran · 1 unverified","sample_list":"/paper/collab-rag-boosting-retrieval-augmented#ran","syntology_url":"https://syntology.ai/paper/2504.04915","mcp":{"get_harvested_code_for_paper":{"arxiv_id":"2504.04915"}},"official":{"repos":["ritaranx/collab-rag"],"state":"official: harvested, nothing ran","n_ran":0,"n_constructed":0,"n_ran_no_instrument_failure":0,"n_unverified":1,"ran_from_kinds":[]}}}],"record_sha256":"caeb91a5c0ebbd9387eaf5b78ee86802d2c89a30e0b9ab390f00af50c491f899","record_changed_at":"2026-09-28","record_changed_at_basis":"first_hashed"}